| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | using System.Diagnostics; |
| | | 5 | | using System.Globalization; |
| | | 6 | | using System.Runtime.CompilerServices; |
| | | 7 | | using System.Runtime.InteropServices; |
| | | 8 | | using System.Runtime.Intrinsics; |
| | | 9 | | using System.Runtime.Intrinsics.Arm; |
| | | 10 | | using System.Text; |
| | | 11 | | |
| | | 12 | | namespace System.Buffers |
| | | 13 | | { |
| | | 14 | | // Provides implementations for helpers shared across multiple SearchValues<string> implementations, |
| | | 15 | | // such as normalizing and matching values under different case sensitivity rules. |
| | | 16 | | internal static class StringSearchValuesHelper |
| | | 17 | | { |
| | | 18 | | [Conditional("DEBUG")] |
| | | 19 | | public static void ValidateReadPosition(ref char searchSpaceStart, int searchSpaceLength, ref char searchSpace, |
| | | 20 | | { |
| | 0 | 21 | | Debug.Assert(searchSpaceLength >= 0); |
| | | 22 | | |
| | 0 | 23 | | ValidateReadPosition(MemoryMarshal.CreateReadOnlySpan(ref searchSpaceStart, searchSpaceLength), ref searchSp |
| | 0 | 24 | | } |
| | | 25 | | |
| | | 26 | | [Conditional("DEBUG")] |
| | | 27 | | public static void ValidateReadPosition(ReadOnlySpan<char> span, ref char searchSpace, int offset = 0) |
| | | 28 | | { |
| | 0 | 29 | | Debug.Assert(offset >= 0); |
| | | 30 | | |
| | 0 | 31 | | nint currentByteOffset = Unsafe.ByteOffset(ref MemoryMarshal.GetReference(span), ref searchSpace); |
| | 0 | 32 | | Debug.Assert(currentByteOffset >= 0); |
| | 0 | 33 | | Debug.Assert((currentByteOffset & 1) == 0); |
| | | 34 | | |
| | 0 | 35 | | int currentOffset = (int)(currentByteOffset / 2); |
| | 0 | 36 | | int availableLength = span.Length - currentOffset; |
| | 0 | 37 | | Debug.Assert(offset <= availableLength); |
| | 0 | 38 | | } |
| | | 39 | | |
| | | 40 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 41 | | public static bool StartsWith<TCaseSensitivity>(ref char matchStart, int lengthRemaining, string[] candidates) |
| | | 42 | | where TCaseSensitivity : struct, ICaseSensitivity |
| | | 43 | | { |
| | 0 | 44 | | foreach (string candidate in candidates) |
| | | 45 | | { |
| | 0 | 46 | | if (StartsWith<TCaseSensitivity>(ref matchStart, lengthRemaining, candidate)) |
| | | 47 | | { |
| | 0 | 48 | | return true; |
| | | 49 | | } |
| | | 50 | | } |
| | | 51 | | |
| | 0 | 52 | | return false; |
| | | 53 | | } |
| | | 54 | | |
| | | 55 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 56 | | public static bool StartsWith<TCaseSensitivity>(ref char matchStart, int lengthRemaining, string candidate) |
| | | 57 | | where TCaseSensitivity : struct, ICaseSensitivity |
| | | 58 | | { |
| | 0 | 59 | | Debug.Assert(lengthRemaining > 0); |
| | | 60 | | |
| | 0 | 61 | | if (lengthRemaining < candidate.Length) |
| | | 62 | | { |
| | 0 | 63 | | return false; |
| | | 64 | | } |
| | | 65 | | |
| | 0 | 66 | | return UnknownLengthEquals<TCaseSensitivity>(ref matchStart, candidate); |
| | | 67 | | } |
| | | 68 | | |
| | | 69 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 70 | | private static bool UnknownLengthEquals<TCaseSensitivity>(ref char matchStart, string candidate) |
| | | 71 | | where TCaseSensitivity : struct, ICaseSensitivity |
| | | 72 | | { |
| | 0 | 73 | | if (typeof(TCaseSensitivity) == typeof(CaseSensitive)) |
| | | 74 | | { |
| | 0 | 75 | | return SpanHelpers.SequenceEqual( |
| | 0 | 76 | | ref Unsafe.As<char, byte>(ref matchStart), |
| | 0 | 77 | | ref candidate.GetRawStringDataAsUInt8(), |
| | 0 | 78 | | (uint)candidate.Length * sizeof(char)); |
| | | 79 | | } |
| | | 80 | | |
| | 0 | 81 | | if (typeof(TCaseSensitivity) == typeof(CaseInsensitiveAscii) || |
| | 0 | 82 | | typeof(TCaseSensitivity) == typeof(CaseInsensitiveAsciiLetters)) |
| | | 83 | | { |
| | 0 | 84 | | return Ascii.EqualsIgnoreCase(ref matchStart, ref candidate.GetRawStringData(), (uint)candidate.Length); |
| | | 85 | | } |
| | | 86 | | |
| | 0 | 87 | | Debug.Assert(typeof(TCaseSensitivity) == typeof(CaseInsensitiveUnicode)); |
| | 0 | 88 | | return Ordinal.EqualsIgnoreCase(ref matchStart, ref candidate.GetRawStringData(), candidate.Length); |
| | | 89 | | } |
| | | 90 | | |
| | | 91 | | public interface IValueLength { } |
| | | 92 | | |
| | | 93 | | public readonly struct ValueLengthLessThan4 : IValueLength { } |
| | | 94 | | |
| | | 95 | | public readonly struct ValueLength4To8 : IValueLength { } |
| | | 96 | | |
| | | 97 | | public readonly struct ValueLength9To16 : IValueLength { } |
| | | 98 | | |
| | | 99 | | // "Unknown" is currently only used by Teddy when confirming matches. |
| | | 100 | | public readonly struct ValueLengthLongOrUnknown : IValueLength { } |
| | | 101 | | |
| | | 102 | | public readonly struct SingleValueState |
| | | 103 | | { |
| | | 104 | | public readonly string Value; |
| | | 105 | | public readonly nint SecondReadByteOffset; |
| | | 106 | | public readonly Vector256<ushort> Value256; |
| | | 107 | | public readonly Vector256<ushort> ToUpperMask256; |
| | | 108 | | |
| | 0 | 109 | | public readonly ulong Value64_0 => Value256.AsUInt64()[0]; |
| | 0 | 110 | | public readonly ulong Value64_1 => Value256.AsUInt64()[1]; |
| | 0 | 111 | | public readonly uint Value32_0 => Value256.AsUInt32()[0]; |
| | 0 | 112 | | public readonly uint Value32_1 => Value256.AsUInt32()[1]; |
| | | 113 | | |
| | 0 | 114 | | public readonly ulong ToUpperMask64_0 => ToUpperMask256.AsUInt64()[0]; |
| | 0 | 115 | | public readonly ulong ToUpperMask64_1 => ToUpperMask256.AsUInt64()[1]; |
| | 0 | 116 | | public readonly uint ToUpperMask32_0 => ToUpperMask256.AsUInt32()[0]; |
| | 0 | 117 | | public readonly uint ToUpperMask32_1 => ToUpperMask256.AsUInt32()[1]; |
| | | 118 | | |
| | | 119 | | public SingleValueState(string value, bool ignoreCase) |
| | | 120 | | { |
| | 0 | 121 | | Debug.Assert(value.Length >= 2); |
| | | 122 | | |
| | 0 | 123 | | Value = value; |
| | | 124 | | |
| | | 125 | | // We precompute vectors specific to this value to speed up later comparisons. |
| | | 126 | | // We group values depending on their length (2-3, 4-8, 9-16). |
| | | 127 | | // For any of those lengths, we can load the whole value with two overlapped reads (e.g. 2x 8 characters |
| | | 128 | | // For a string "Hello World", we would load |
| | | 129 | | // [Hello Wo] |
| | | 130 | | // [lo World] |
| | | 131 | | // SecondReadByteOffset: 6 bytes (3 characters) |
| | | 132 | | // We then precompute a mask that converts any potential input to the uppercase variant, specific to thi |
| | | 133 | | // We must ensure that the ASCII letter mask only applies to the letters, not the space character. |
| | | 134 | | // Value256: [HELLO WOLO WORLD] (note that the value is already converted to uppercase if we're ig |
| | | 135 | | // ToUpperMask256: [xxxxx xxxx xxxxx] (x = ~0x20 for ASCII letters, 0xFFFF otherwise) |
| | | 136 | | // |
| | | 137 | | // Given a potential match, we can now confirm whether we found a match by loading the candidate in the |
| | | 138 | | // Vector256 input = [Vector128.Load(candidate), Vector128.Load(candidate + 6 bytes)]; |
| | | 139 | | // bool matches = (input & ToUpperMask256) == Value256; |
| | | 140 | | |
| | | 141 | | // The two vectors may overlap completely for Length == 2 or Length == 4, and that's fine. |
| | | 142 | | // The second comparison during validation is redundant in such cases, but the alternative is to introdu |
| | | 143 | | |
| | 0 | 144 | | if (value.Length <= 16) |
| | | 145 | | { |
| | 0 | 146 | | if (value.Length > 8) |
| | | 147 | | { |
| | 0 | 148 | | SecondReadByteOffset = (value.Length - 8) * sizeof(char); |
| | 0 | 149 | | Value256 = Vector256.Create( |
| | 0 | 150 | | Vector128.LoadUnsafe(ref value.GetRawStringDataAsUInt16()), |
| | 0 | 151 | | Vector128.LoadUnsafe(ref Unsafe.AddByteOffset(ref value.GetRawStringDataAsUInt16(), SecondRe |
| | | 152 | | } |
| | 0 | 153 | | else if (value.Length >= 4) |
| | | 154 | | { |
| | 0 | 155 | | SecondReadByteOffset = (value.Length - 4) * sizeof(char); |
| | 0 | 156 | | Value256 = Vector256.Create(Vector128.Create( |
| | 0 | 157 | | Unsafe.ReadUnaligned<ulong>(ref value.GetRawStringDataAsUInt8()), |
| | 0 | 158 | | Unsafe.ReadUnaligned<ulong>(ref Unsafe.Add(ref value.GetRawStringDataAsUInt8(), SecondReadBy |
| | 0 | 159 | | )).AsUInt16(); |
| | | 160 | | } |
| | | 161 | | else |
| | | 162 | | { |
| | 0 | 163 | | Debug.Assert(value.Length is 2 or 3); |
| | | 164 | | |
| | 0 | 165 | | SecondReadByteOffset = (value.Length - 2) * sizeof(char); |
| | 0 | 166 | | Value256 = Vector256.Create(Vector128.Create(Vector64.Create( |
| | 0 | 167 | | Unsafe.ReadUnaligned<uint>(ref value.GetRawStringDataAsUInt8()), |
| | 0 | 168 | | Unsafe.ReadUnaligned<uint>(ref Unsafe.Add(ref value.GetRawStringDataAsUInt8(), SecondReadByt |
| | 0 | 169 | | ))).AsUInt16(); |
| | | 170 | | } |
| | | 171 | | |
| | 0 | 172 | | if (ignoreCase) |
| | | 173 | | { |
| | 0 | 174 | | Vector256<ushort> isAsciiLetter = |
| | 0 | 175 | | Vector256.GreaterThanOrEqual(Value256, Vector256.Create((ushort)'A')) & |
| | 0 | 176 | | Vector256.LessThanOrEqual(Value256, Vector256.Create((ushort)'Z')); |
| | | 177 | | |
| | 0 | 178 | | ToUpperMask256 = Vector256.ConditionalSelect(isAsciiLetter, Vector256.Create(unchecked((ushort)~ |
| | | 179 | | } |
| | | 180 | | } |
| | 0 | 181 | | } |
| | | 182 | | |
| | | 183 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 184 | | public bool MatchesLength9To16_CaseSensitive(ref char matchStart) |
| | | 185 | | { |
| | 0 | 186 | | Debug.Assert(Value.Length is >= 9 and <= 16); |
| | 0 | 187 | | Debug.Assert(ToUpperMask256 == default); |
| | | 188 | | |
| | 0 | 189 | | if (Vector256.IsHardwareAccelerated) |
| | | 190 | | { |
| | 0 | 191 | | Vector256<ushort> input = Vector256.Create( |
| | 0 | 192 | | Vector128.LoadUnsafe(ref matchStart), |
| | 0 | 193 | | Vector128.LoadUnsafe(ref Unsafe.AddByteOffset(ref matchStart, SecondReadByteOffset))); |
| | | 194 | | |
| | 0 | 195 | | return input == Value256; |
| | | 196 | | } |
| | | 197 | | else |
| | | 198 | | { |
| | 0 | 199 | | Vector128<ushort> different = Vector128.LoadUnsafe(ref matchStart) ^ Value256.GetLower(); |
| | 0 | 200 | | different |= Vector128.LoadUnsafe(ref Unsafe.AddByteOffset(ref matchStart, SecondReadByteOffset)) ^ |
| | 0 | 201 | | return different == Vector128<ushort>.Zero; |
| | | 202 | | } |
| | | 203 | | } |
| | | 204 | | |
| | | 205 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 206 | | public bool MatchesLength9To16_CaseInsensitiveAscii(ref char matchStart) |
| | | 207 | | { |
| | 0 | 208 | | Debug.Assert(Value.Length is >= 9 and <= 16); |
| | 0 | 209 | | Debug.Assert(ToUpperMask256 != default); |
| | | 210 | | |
| | 0 | 211 | | if (Vector256.IsHardwareAccelerated) |
| | | 212 | | { |
| | 0 | 213 | | Vector256<ushort> input = Vector256.Create( |
| | 0 | 214 | | Vector128.LoadUnsafe(ref matchStart), |
| | 0 | 215 | | Vector128.LoadUnsafe(ref Unsafe.AddByteOffset(ref matchStart, SecondReadByteOffset))); |
| | | 216 | | |
| | 0 | 217 | | return (input & ToUpperMask256) == Value256; |
| | | 218 | | } |
| | | 219 | | else |
| | | 220 | | { |
| | 0 | 221 | | Vector128<ushort> different = (Vector128.LoadUnsafe(ref matchStart) & ToUpperMask256.GetLower()) ^ V |
| | 0 | 222 | | different |= (Vector128.LoadUnsafe(ref Unsafe.AddByteOffset(ref matchStart, SecondReadByteOffset)) & |
| | 0 | 223 | | return different == Vector128<ushort>.Zero; |
| | | 224 | | } |
| | | 225 | | } |
| | | 226 | | } |
| | | 227 | | |
| | | 228 | | public interface ICaseSensitivity |
| | | 229 | | { |
| | | 230 | | static abstract char TransformInput(char input); |
| | | 231 | | static abstract Vector128<byte> TransformInput(Vector128<byte> input); |
| | | 232 | | static abstract Vector256<byte> TransformInput(Vector256<byte> input); |
| | | 233 | | static abstract Vector512<byte> TransformInput(Vector512<byte> input); |
| | | 234 | | static abstract bool Equals<TValueLength>(ref char matchStart, ref readonly SingleValueState state) where TV |
| | | 235 | | } |
| | | 236 | | |
| | | 237 | | // Performs no case transformations. |
| | | 238 | | public readonly struct CaseSensitive : ICaseSensitivity |
| | | 239 | | { |
| | | 240 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 0 | 241 | | public static char TransformInput(char input) => input; |
| | | 242 | | |
| | | 243 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 0 | 244 | | public static Vector128<byte> TransformInput(Vector128<byte> input) => input; |
| | | 245 | | |
| | | 246 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 0 | 247 | | public static Vector256<byte> TransformInput(Vector256<byte> input) => input; |
| | | 248 | | |
| | | 249 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 0 | 250 | | public static Vector512<byte> TransformInput(Vector512<byte> input) => input; |
| | | 251 | | |
| | | 252 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 253 | | public static bool Equals<TValueLength>(ref char matchStart, ref readonly SingleValueState state) |
| | | 254 | | where TValueLength : struct, IValueLength |
| | | 255 | | { |
| | | 256 | | if (typeof(TValueLength) == typeof(ValueLengthLongOrUnknown)) |
| | | 257 | | { |
| | 0 | 258 | | return UnknownLengthEquals<CaseSensitive>(ref matchStart, state.Value); |
| | | 259 | | } |
| | 0 | 260 | | else if (typeof(TValueLength) == typeof(ValueLength9To16)) |
| | | 261 | | { |
| | 0 | 262 | | return state.MatchesLength9To16_CaseSensitive(ref matchStart); |
| | | 263 | | } |
| | 0 | 264 | | else if (typeof(TValueLength) == typeof(ValueLength4To8)) |
| | | 265 | | { |
| | 0 | 266 | | ref byte matchByteStart = ref Unsafe.As<char, byte>(ref matchStart); |
| | 0 | 267 | | ulong differentBits = Unsafe.ReadUnaligned<ulong>(ref matchByteStart) - state.Value64_0; |
| | 0 | 268 | | differentBits |= Unsafe.ReadUnaligned<ulong>(ref Unsafe.Add(ref matchByteStart, state.SecondReadByte |
| | 0 | 269 | | return differentBits == 0; |
| | | 270 | | } |
| | | 271 | | else |
| | | 272 | | { |
| | 0 | 273 | | Debug.Assert(state.Value.Length is 2 or 3); |
| | | 274 | | |
| | 0 | 275 | | ref byte matchByteStart = ref Unsafe.As<char, byte>(ref matchStart); |
| | | 276 | | |
| | | 277 | | if (AdvSimd.IsSupported) |
| | | 278 | | { |
| | | 279 | | // See comments on SingleStringSearchValuesPackedThreeChars.CanSkipAnchorMatchVerification. |
| | | 280 | | // When running on Arm64, this helper is also used to confirm vectorized anchor matches. |
| | | 281 | | // We do so because we're using UnzipEven when packing inputs, which may produce false positive |
| | | 282 | | // When called from SingleStringSearchValuesThreeChars (non-packed), we could skip to the else b |
| | | 283 | | Debug.Assert(matchStart == state.Value[0] || (matchStart & 0xFF) == state.Value[0]); |
| | | 284 | | |
| | | 285 | | uint differentBits = Unsafe.ReadUnaligned<uint>(ref matchByteStart) - state.Value32_0; |
| | | 286 | | differentBits |= Unsafe.ReadUnaligned<uint>(ref Unsafe.Add(ref matchByteStart, state.SecondReadB |
| | | 287 | | return differentBits == 0; |
| | | 288 | | } |
| | | 289 | | else |
| | | 290 | | { |
| | | 291 | | // Otherwise, this path is not used when confirming vectorized anchor matches. |
| | | 292 | | // It's only used as part of the scalar search loop, which always checks that the first characte |
| | | 293 | | // We know that the candidate is 2 or 3 characters long, and that the first character has alread |
| | | 294 | | // We only have to to check whether the last 2 characters also match. |
| | 0 | 295 | | Debug.Assert(matchStart == state.Value[0], "This should only be called after the first character |
| | | 296 | | |
| | 0 | 297 | | return Unsafe.ReadUnaligned<uint>(ref Unsafe.Add(ref matchByteStart, state.SecondReadByteOffset) |
| | | 298 | | } |
| | | 299 | | } |
| | | 300 | | } |
| | | 301 | | } |
| | | 302 | | |
| | | 303 | | // Transforms inputs to their uppercase variants with the assumption that all input characters are ASCII letters |
| | | 304 | | // These helpers may produce wrong results for other characters, and the callers must account for that. |
| | | 305 | | public readonly struct CaseInsensitiveAsciiLetters : ICaseSensitivity |
| | | 306 | | { |
| | | 307 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 0 | 308 | | public static char TransformInput(char input) => (char)(input & ~0x20); |
| | | 309 | | |
| | | 310 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 0 | 311 | | public static Vector128<byte> TransformInput(Vector128<byte> input) => input & Vector128.Create(unchecked((b |
| | | 312 | | |
| | | 313 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 0 | 314 | | public static Vector256<byte> TransformInput(Vector256<byte> input) => input & Vector256.Create(unchecked((b |
| | | 315 | | |
| | | 316 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 0 | 317 | | public static Vector512<byte> TransformInput(Vector512<byte> input) => input & Vector512.Create(unchecked((b |
| | | 318 | | |
| | | 319 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 320 | | public static bool Equals<TValueLength>(ref char matchStart, ref readonly SingleValueState state) |
| | | 321 | | where TValueLength : struct, IValueLength |
| | | 322 | | { |
| | | 323 | | if (typeof(TValueLength) == typeof(ValueLengthLongOrUnknown)) |
| | | 324 | | { |
| | 0 | 325 | | return UnknownLengthEquals<CaseInsensitiveAsciiLetters>(ref matchStart, state.Value); |
| | | 326 | | } |
| | 0 | 327 | | else if (typeof(TValueLength) == typeof(ValueLength9To16)) |
| | | 328 | | { |
| | 0 | 329 | | return state.MatchesLength9To16_CaseInsensitiveAscii(ref matchStart); |
| | | 330 | | } |
| | 0 | 331 | | else if (typeof(TValueLength) == typeof(ValueLength4To8)) |
| | | 332 | | { |
| | | 333 | | const ulong CaseMask = ~0x20002000200020u; |
| | 0 | 334 | | ref byte matchByteStart = ref Unsafe.As<char, byte>(ref matchStart); |
| | 0 | 335 | | ulong differentBits = (Unsafe.ReadUnaligned<ulong>(ref matchByteStart) & CaseMask) - state.Value64_0 |
| | 0 | 336 | | differentBits |= (Unsafe.ReadUnaligned<ulong>(ref Unsafe.Add(ref matchByteStart, state.SecondReadByt |
| | 0 | 337 | | return differentBits == 0; |
| | | 338 | | } |
| | | 339 | | else |
| | | 340 | | { |
| | 0 | 341 | | Debug.Assert(state.Value.Length is 2 or 3); |
| | | 342 | | |
| | | 343 | | const uint CaseMask = ~0x200020u; |
| | 0 | 344 | | ref byte matchByteStart = ref Unsafe.As<char, byte>(ref matchStart); |
| | | 345 | | |
| | | 346 | | if (AdvSimd.IsSupported) |
| | | 347 | | { |
| | | 348 | | // See comments on SingleStringSearchValuesPackedThreeChars.CanSkipAnchorMatchVerification. |
| | | 349 | | // When running on Arm64, this helper is also used to confirm vectorized anchor matches. |
| | | 350 | | // We do so because we're using UnzipEven when packing inputs, which may produce false positive |
| | | 351 | | // When called from SingleStringSearchValuesThreeChars (non-packed), we could skip to the else b |
| | | 352 | | Debug.Assert(TransformInput((char)(matchStart & 0xFF)) == state.Value[0]); |
| | | 353 | | |
| | | 354 | | uint differentBits = (Unsafe.ReadUnaligned<uint>(ref matchByteStart) & CaseMask) - state.Value32 |
| | | 355 | | differentBits |= (Unsafe.ReadUnaligned<uint>(ref Unsafe.Add(ref matchByteStart, state.SecondRead |
| | | 356 | | return differentBits == 0; |
| | | 357 | | } |
| | | 358 | | else |
| | | 359 | | { |
| | | 360 | | // Otherwise, this path is not used when confirming vectorized anchor matches. |
| | | 361 | | // It's only used as part of the scalar search loop, which always checks that the first characte |
| | | 362 | | // We know that the candidate is 2 or 3 characters long, and that the first character has alread |
| | | 363 | | // We only have to to check whether the last 2 characters also match. |
| | 0 | 364 | | Debug.Assert(TransformInput(matchStart) == state.Value[0], "This should only be called after the |
| | | 365 | | |
| | 0 | 366 | | return (Unsafe.ReadUnaligned<uint>(ref Unsafe.Add(ref matchByteStart, state.SecondReadByteOffset |
| | | 367 | | } |
| | | 368 | | } |
| | | 369 | | } |
| | | 370 | | } |
| | | 371 | | |
| | | 372 | | // Transforms inputs to their uppercase variants with the assumption that all input characters are ASCII. |
| | | 373 | | // These helpers may produce wrong results for non-ASCII inputs, and the callers must account for that. |
| | | 374 | | public readonly struct CaseInsensitiveAscii : ICaseSensitivity |
| | | 375 | | { |
| | | 376 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 0 | 377 | | public static char TransformInput(char input) => TextInfo.ToUpperAsciiInvariant(input); |
| | | 378 | | |
| | | 379 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 380 | | public static Vector128<byte> TransformInput(Vector128<byte> input) |
| | | 381 | | { |
| | 0 | 382 | | Vector128<byte> subtraction = Vector128.Create((byte)(128 + 'a')); |
| | 0 | 383 | | Vector128<byte> comparison = Vector128.Create((byte)(128 + 26)); |
| | 0 | 384 | | Vector128<byte> caseConversion = Vector128.Create((byte)0x20); |
| | | 385 | | |
| | 0 | 386 | | Vector128<byte> matches = Vector128.LessThan((input - subtraction).AsSByte(), comparison.AsSByte()).AsBy |
| | 0 | 387 | | return input ^ (matches & caseConversion); |
| | | 388 | | } |
| | | 389 | | |
| | | 390 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 391 | | public static Vector256<byte> TransformInput(Vector256<byte> input) |
| | | 392 | | { |
| | 0 | 393 | | Vector256<byte> subtraction = Vector256.Create((byte)(128 + 'a')); |
| | 0 | 394 | | Vector256<byte> comparison = Vector256.Create((byte)(128 + 26)); |
| | 0 | 395 | | Vector256<byte> caseConversion = Vector256.Create((byte)0x20); |
| | | 396 | | |
| | 0 | 397 | | Vector256<byte> matches = Vector256.LessThan((input - subtraction).AsSByte(), comparison.AsSByte()).AsBy |
| | 0 | 398 | | return input ^ (matches & caseConversion); |
| | | 399 | | } |
| | | 400 | | |
| | | 401 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 402 | | public static Vector512<byte> TransformInput(Vector512<byte> input) |
| | | 403 | | { |
| | 0 | 404 | | Vector512<byte> subtraction = Vector512.Create((byte)(128 + 'a')); |
| | 0 | 405 | | Vector512<byte> comparison = Vector512.Create((byte)(128 + 26)); |
| | 0 | 406 | | Vector512<byte> caseConversion = Vector512.Create((byte)0x20); |
| | | 407 | | |
| | 0 | 408 | | Vector512<byte> matches = Vector512.LessThan((input - subtraction).AsSByte(), comparison.AsSByte()).AsBy |
| | 0 | 409 | | return input ^ (matches & caseConversion); |
| | | 410 | | } |
| | | 411 | | |
| | | 412 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 413 | | public static bool Equals<TValueLength>(ref char matchStart, ref readonly SingleValueState state) |
| | | 414 | | where TValueLength : struct, IValueLength |
| | | 415 | | { |
| | 0 | 416 | | if (typeof(TValueLength) == typeof(ValueLengthLongOrUnknown)) |
| | | 417 | | { |
| | 0 | 418 | | return UnknownLengthEquals<CaseInsensitiveAscii>(ref matchStart, state.Value); |
| | | 419 | | } |
| | 0 | 420 | | else if (typeof(TValueLength) == typeof(ValueLength9To16)) |
| | | 421 | | { |
| | 0 | 422 | | return state.MatchesLength9To16_CaseInsensitiveAscii(ref matchStart); |
| | | 423 | | } |
| | 0 | 424 | | else if (typeof(TValueLength) == typeof(ValueLength4To8)) |
| | | 425 | | { |
| | 0 | 426 | | ref byte matchByteStart = ref Unsafe.As<char, byte>(ref matchStart); |
| | 0 | 427 | | ulong differentBits = (Unsafe.ReadUnaligned<ulong>(ref matchByteStart) & state.ToUpperMask64_0) - st |
| | 0 | 428 | | differentBits |= (Unsafe.ReadUnaligned<ulong>(ref Unsafe.Add(ref matchByteStart, state.SecondReadByt |
| | 0 | 429 | | return differentBits == 0; |
| | | 430 | | } |
| | | 431 | | else |
| | | 432 | | { |
| | 0 | 433 | | Debug.Assert(state.Value.Length is 2 or 3); |
| | | 434 | | |
| | 0 | 435 | | ref byte matchByteStart = ref Unsafe.As<char, byte>(ref matchStart); |
| | 0 | 436 | | uint differentBits = (Unsafe.ReadUnaligned<uint>(ref matchByteStart) & state.ToUpperMask32_0) - stat |
| | 0 | 437 | | differentBits |= (Unsafe.ReadUnaligned<uint>(ref Unsafe.Add(ref matchByteStart, state.SecondReadByte |
| | 0 | 438 | | return differentBits == 0; |
| | | 439 | | } |
| | | 440 | | } |
| | | 441 | | } |
| | | 442 | | |
| | | 443 | | // We can't efficiently map non-ASCII inputs to their Ordinal uppercase variants, |
| | | 444 | | // so this helper is only used for the verification of the whole input. |
| | | 445 | | public readonly struct CaseInsensitiveUnicode : ICaseSensitivity |
| | | 446 | | { |
| | 0 | 447 | | public static char TransformInput(char input) => throw new UnreachableException(); |
| | 0 | 448 | | public static Vector128<byte> TransformInput(Vector128<byte> input) => throw new UnreachableException(); |
| | 0 | 449 | | public static Vector256<byte> TransformInput(Vector256<byte> input) => throw new UnreachableException(); |
| | 0 | 450 | | public static Vector512<byte> TransformInput(Vector512<byte> input) => throw new UnreachableException(); |
| | | 451 | | |
| | | 452 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 453 | | public static bool Equals<TValueLength>(ref char matchStart, ref readonly SingleValueState state) |
| | | 454 | | where TValueLength : struct, IValueLength |
| | | 455 | | { |
| | 0 | 456 | | if (typeof(TValueLength) == typeof(ValueLengthLongOrUnknown)) |
| | | 457 | | { |
| | 0 | 458 | | return UnknownLengthEquals<CaseInsensitiveUnicode>(ref matchStart, state.Value); |
| | | 459 | | } |
| | | 460 | | else |
| | | 461 | | { |
| | 0 | 462 | | return Ordinal.EqualsIgnoreCase_Scalar(ref matchStart, ref state.Value.GetRawStringData(), state.Val |
| | | 463 | | } |
| | | 464 | | } |
| | | 465 | | } |
| | | 466 | | } |
| | | 467 | | } |
| | | 468 | | |