| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | using System.Collections.Generic; |
| | | 5 | | using System.Diagnostics; |
| | | 6 | | using System.Diagnostics.CodeAnalysis; |
| | | 7 | | using System.Runtime.Intrinsics; |
| | | 8 | | using System.Runtime.Intrinsics.Arm; |
| | | 9 | | using System.Runtime.Intrinsics.Wasm; |
| | | 10 | | using System.Runtime.Intrinsics.X86; |
| | | 11 | | using System.Text; |
| | | 12 | | using System.Text.Unicode; |
| | | 13 | | using static System.Buffers.StringSearchValuesHelper; |
| | | 14 | | |
| | | 15 | | namespace System.Buffers |
| | | 16 | | { |
| | | 17 | | internal static class StringSearchValues |
| | | 18 | | { |
| | | 19 | | private const int TeddyBucketCount = 8; |
| | | 20 | | |
| | 0 | 21 | | private static readonly SearchValues<char> s_asciiLetters = |
| | 0 | 22 | | SearchValues.Create("ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz"); |
| | | 23 | | |
| | | 24 | | public static SearchValues<string> Create(ReadOnlySpan<string> values, bool ignoreCase) |
| | | 25 | | { |
| | 0 | 26 | | if (values.Length == 0) |
| | | 27 | | { |
| | 0 | 28 | | return new EmptySearchValues<string>(); |
| | | 29 | | } |
| | | 30 | | |
| | 0 | 31 | | if (values.Length == 1) |
| | | 32 | | { |
| | | 33 | | // Avoid additional overheads for single-value inputs. |
| | 0 | 34 | | string value = values[0]; |
| | 0 | 35 | | ArgumentNullException.ThrowIfNull(value, nameof(values)); |
| | 0 | 36 | | string normalizedValue = NormalizeIfNeeded(value, ignoreCase); |
| | | 37 | | |
| | 0 | 38 | | AnalyzeValues(new ReadOnlySpan<string>(ref normalizedValue), ref ignoreCase, out bool ascii, out bool as |
| | 0 | 39 | | return CreateForSingleValue(normalizedValue, uniqueValues: null, ignoreCase, ascii, asciiLettersOnly); |
| | | 40 | | } |
| | | 41 | | |
| | 0 | 42 | | var uniqueValues = new HashSet<string>(values.Length, ignoreCase ? StringComparer.OrdinalIgnoreCase : String |
| | | 43 | | |
| | 0 | 44 | | foreach (string value in values) |
| | | 45 | | { |
| | 0 | 46 | | ArgumentNullException.ThrowIfNull(value, nameof(values)); |
| | | 47 | | |
| | 0 | 48 | | uniqueValues.Add(value); |
| | | 49 | | } |
| | | 50 | | |
| | 0 | 51 | | if (uniqueValues.Contains(string.Empty)) |
| | | 52 | | { |
| | 0 | 53 | | return new SingleStringSearchValuesFallback<SearchValues.FalseConst>(string.Empty, uniqueValues); |
| | | 54 | | } |
| | | 55 | | |
| | 0 | 56 | | Span<string> normalizedValues = new string[uniqueValues.Count]; |
| | 0 | 57 | | int i = 0; |
| | 0 | 58 | | foreach (string value in uniqueValues) |
| | | 59 | | { |
| | 0 | 60 | | normalizedValues[i++] = NormalizeIfNeeded(value, ignoreCase); |
| | | 61 | | } |
| | 0 | 62 | | Debug.Assert(i == normalizedValues.Length); |
| | | 63 | | |
| | | 64 | | // Aho-Corasick's ctor expects values to be sorted by length. |
| | 0 | 65 | | normalizedValues.Sort(static (a, b) => a.Length.CompareTo(b.Length)); |
| | | 66 | | |
| | | 67 | | // We may not end up choosing Aho-Corasick as the implementation, but it has a nice property of |
| | | 68 | | // finding all the unreachable values during the construction stage, so we build the trie early. |
| | 0 | 69 | | HashSet<string>? unreachableValues = null; |
| | 0 | 70 | | var ahoCorasickBuilder = new AhoCorasickBuilder(normalizedValues, ignoreCase, ref unreachableValues); |
| | | 71 | | |
| | 0 | 72 | | if (unreachableValues is not null) |
| | | 73 | | { |
| | | 74 | | // Some values are exact prefixes of other values. |
| | | 75 | | // Exclude those values now to reduce the number of buckets and make verification steps cheaper during s |
| | 0 | 76 | | normalizedValues = RemoveUnreachableValues(normalizedValues, unreachableValues); |
| | | 77 | | } |
| | | 78 | | |
| | 0 | 79 | | SearchValues<string> searchValues = CreateFromNormalizedValues(normalizedValues, uniqueValues, ignoreCase, r |
| | 0 | 80 | | ahoCorasickBuilder.Dispose(); |
| | 0 | 81 | | return searchValues; |
| | | 82 | | |
| | | 83 | | static string NormalizeIfNeeded(string value, bool ignoreCase) => |
| | 0 | 84 | | ignoreCase ? value.ToUpperOrdinal() : value; |
| | | 85 | | |
| | | 86 | | static Span<string> RemoveUnreachableValues(Span<string> values, HashSet<string> unreachableValues) |
| | | 87 | | { |
| | 0 | 88 | | int newCount = 0; |
| | 0 | 89 | | foreach (string value in values) |
| | | 90 | | { |
| | 0 | 91 | | if (!unreachableValues.Contains(value)) |
| | | 92 | | { |
| | 0 | 93 | | values[newCount++] = value; |
| | | 94 | | } |
| | | 95 | | } |
| | | 96 | | |
| | 0 | 97 | | Debug.Assert(newCount <= values.Length - unreachableValues.Count); |
| | 0 | 98 | | Debug.Assert(newCount > 0); |
| | | 99 | | |
| | 0 | 100 | | return values.Slice(0, newCount); |
| | | 101 | | } |
| | | 102 | | } |
| | | 103 | | |
| | | 104 | | private static SearchValues<string> CreateFromNormalizedValues( |
| | | 105 | | ReadOnlySpan<string> values, |
| | | 106 | | HashSet<string> uniqueValues, |
| | | 107 | | bool ignoreCase, |
| | | 108 | | ref AhoCorasickBuilder ahoCorasickBuilder) |
| | | 109 | | { |
| | 0 | 110 | | AnalyzeValues(values, ref ignoreCase, out bool allAscii, out bool asciiLettersOnly, out bool nonAsciiAffecte |
| | | 111 | | |
| | 0 | 112 | | if (values.Length == 1) |
| | | 113 | | { |
| | | 114 | | // We may reach this if we've removed unreachable values and ended up with only 1 remaining. |
| | 0 | 115 | | return CreateForSingleValue(values[0], uniqueValues, ignoreCase, allAscii, asciiLettersOnly); |
| | | 116 | | } |
| | | 117 | | |
| | 0 | 118 | | if ((Ssse3.IsSupported || AdvSimd.Arm64.IsSupported || PackedSimd.IsSupported) && |
| | 0 | 119 | | TryGetTeddyAcceleratedValues(values, uniqueValues, ignoreCase, allAscii, asciiLettersOnly, nonAsciiAffec |
| | | 120 | | { |
| | 0 | 121 | | return searchValues; |
| | | 122 | | } |
| | | 123 | | |
| | | 124 | | // Fall back to Aho-Corasick for all other multi-value sets. |
| | 0 | 125 | | AhoCorasick ahoCorasick = ahoCorasickBuilder.Build(); |
| | | 126 | | |
| | 0 | 127 | | if (!ignoreCase) |
| | | 128 | | { |
| | 0 | 129 | | return PickAhoCorasickImplementation<CaseSensitive>(ahoCorasick, uniqueValues); |
| | | 130 | | } |
| | | 131 | | |
| | 0 | 132 | | if (nonAsciiAffectedByCaseConversion) |
| | | 133 | | { |
| | 0 | 134 | | if (ContainsInvalidValues(values)) |
| | | 135 | | { |
| | | 136 | | // Aho-Corasick can't deal with the matching semantics of invalid values. |
| | | 137 | | // We will use a slow but correct O(n * m) fallback implementation. |
| | 0 | 138 | | return new MultiStringIgnoreCaseSearchValuesFallback(uniqueValues); |
| | | 139 | | } |
| | | 140 | | |
| | 0 | 141 | | return PickAhoCorasickImplementation<CaseInsensitiveUnicode>(ahoCorasick, uniqueValues); |
| | | 142 | | } |
| | | 143 | | |
| | 0 | 144 | | if (asciiLettersOnly) |
| | | 145 | | { |
| | 0 | 146 | | return PickAhoCorasickImplementation<CaseInsensitiveAsciiLetters>(ahoCorasick, uniqueValues); |
| | | 147 | | } |
| | | 148 | | |
| | 0 | 149 | | return PickAhoCorasickImplementation<CaseInsensitiveAscii>(ahoCorasick, uniqueValues); |
| | | 150 | | |
| | | 151 | | static SearchValues<string> PickAhoCorasickImplementation<TCaseSensitivity>(AhoCorasick ahoCorasick, HashSet |
| | | 152 | | where TCaseSensitivity : struct, ICaseSensitivity |
| | | 153 | | { |
| | 0 | 154 | | return ahoCorasick.ShouldUseAsciiFastScan |
| | 0 | 155 | | ? new StringSearchValuesAhoCorasick<TCaseSensitivity, AhoCorasick.IndexOfAnyAsciiFastScan>(ahoCorasi |
| | 0 | 156 | | : new StringSearchValuesAhoCorasick<TCaseSensitivity, AhoCorasick.NoFastScan>(ahoCorasick, uniqueVal |
| | | 157 | | } |
| | | 158 | | } |
| | | 159 | | |
| | | 160 | | private static SearchValues<string>? TryGetTeddyAcceleratedValues( |
| | | 161 | | ReadOnlySpan<string> values, |
| | | 162 | | HashSet<string> uniqueValues, |
| | | 163 | | bool ignoreCase, |
| | | 164 | | bool allAscii, |
| | | 165 | | bool asciiLettersOnly, |
| | | 166 | | bool nonAsciiAffectedByCaseConversion, |
| | | 167 | | int minLength) |
| | | 168 | | { |
| | 0 | 169 | | if (minLength == 1) |
| | | 170 | | { |
| | | 171 | | // An 'N=1' implementation is possible, but callers should |
| | | 172 | | // consider using SearchValues<char> instead in such cases. |
| | | 173 | | // It can be added if Regex ends up running into this case. |
| | 0 | 174 | | return null; |
| | | 175 | | } |
| | | 176 | | |
| | 0 | 177 | | if (values.Length > RabinKarp.MaxValues) |
| | | 178 | | { |
| | | 179 | | // The more values we have, the higher the chance of hash/fingerprint collisions. |
| | | 180 | | // To avoid spending too much time in verification steps, fallback to Aho-Corasick which guarantees O(n) |
| | | 181 | | // If it turns out that this limit is commonly exceeded, we can tweak the number of buckets |
| | | 182 | | // in the implementation, or use different variants depending on input. |
| | 0 | 183 | | return null; |
| | | 184 | | } |
| | | 185 | | |
| | 0 | 186 | | int n = minLength == 2 ? 2 : 3; |
| | | 187 | | |
| | 0 | 188 | | if (Ssse3.IsSupported || PackedSimd.IsSupported) |
| | | 189 | | { |
| | 0 | 190 | | foreach (string value in values) |
| | | 191 | | { |
| | 0 | 192 | | if (value.AsSpan(0, n).Contains('\0')) |
| | | 193 | | { |
| | | 194 | | // If we let null chars through here, Teddy would still work correctly, but it |
| | | 195 | | // would hit more false positives that the verification step would have to rule out. |
| | | 196 | | // Ssse3.PackUnsignedSaturate and PackedSimd.ConvertNarrowingSaturateUnsigned both |
| | | 197 | | // treat negative signed-16 values as 0, so we filter out null-containing needles |
| | | 198 | | // for both to avoid that source of false positives. |
| | 0 | 199 | | return null; |
| | | 200 | | } |
| | | 201 | | } |
| | | 202 | | } |
| | | 203 | | |
| | | 204 | | // Even if the values contain non-ASCII chars, we may be able to use Teddy as long as the |
| | | 205 | | // first N characters are ASCII. |
| | 0 | 206 | | if (!allAscii) |
| | | 207 | | { |
| | 0 | 208 | | foreach (string value in values) |
| | | 209 | | { |
| | 0 | 210 | | if (!Ascii.IsValid(value.AsSpan(0, n))) |
| | | 211 | | { |
| | | 212 | | // A vectorized implementation for non-ASCII values is possible. |
| | | 213 | | // It can be added if it turns out to be a common enough scenario. |
| | 0 | 214 | | return null; |
| | | 215 | | } |
| | | 216 | | } |
| | | 217 | | } |
| | | 218 | | |
| | 0 | 219 | | if (!ignoreCase) |
| | | 220 | | { |
| | 0 | 221 | | return PickTeddyImplementation<CaseSensitive, CaseSensitive>(values, uniqueValues, n); |
| | | 222 | | } |
| | | 223 | | |
| | 0 | 224 | | if (asciiLettersOnly) |
| | | 225 | | { |
| | 0 | 226 | | return PickTeddyImplementation<CaseInsensitiveAsciiLetters, CaseInsensitiveAsciiLetters>(values, uniqueV |
| | | 227 | | } |
| | | 228 | | |
| | | 229 | | // Even if the whole value isn't ASCII letters only, we can still use a faster approach |
| | | 230 | | // for the vectorized part as long as the first N characters are. |
| | 0 | 231 | | bool asciiStartLettersOnly = true; |
| | 0 | 232 | | bool asciiStartUnaffectedByCaseConversion = true; |
| | | 233 | | |
| | 0 | 234 | | foreach (string value in values) |
| | | 235 | | { |
| | 0 | 236 | | ReadOnlySpan<char> slice = value.AsSpan(0, n); |
| | 0 | 237 | | asciiStartLettersOnly = asciiStartLettersOnly && !slice.ContainsAnyExcept(s_asciiLetters); |
| | 0 | 238 | | asciiStartUnaffectedByCaseConversion = asciiStartUnaffectedByCaseConversion && !slice.ContainsAny(s_asci |
| | | 239 | | } |
| | | 240 | | |
| | 0 | 241 | | Debug.Assert(!(asciiStartLettersOnly && asciiStartUnaffectedByCaseConversion)); |
| | | 242 | | |
| | | 243 | | // If we still have empty buckets we could use and we're ignoring case, we may be able to |
| | | 244 | | // generate all possible permutations of the first N characters and switch to case-sensitive searching. |
| | | 245 | | // E.g. ["ab", "c!"] => ["ab", "Ab" "aB", "AB", "c!", "C!"]. |
| | | 246 | | // This won't apply to inputs with many letters (e.g. "abc" => 8 permutations on its own). |
| | 0 | 247 | | if (!asciiStartUnaffectedByCaseConversion && |
| | 0 | 248 | | values.Length < TeddyBucketCount && |
| | 0 | 249 | | TryGenerateAllCasePermutationsForPrefixes(values, n, TeddyBucketCount, out string[]? newValues)) |
| | | 250 | | { |
| | 0 | 251 | | asciiStartUnaffectedByCaseConversion = true; |
| | 0 | 252 | | values = newValues; |
| | | 253 | | } |
| | | 254 | | |
| | 0 | 255 | | if (asciiStartUnaffectedByCaseConversion) |
| | | 256 | | { |
| | 0 | 257 | | return nonAsciiAffectedByCaseConversion |
| | 0 | 258 | | ? PickTeddyImplementation<CaseSensitive, CaseInsensitiveUnicode>(values, uniqueValues, n) |
| | 0 | 259 | | : PickTeddyImplementation<CaseSensitive, CaseInsensitiveAscii>(values, uniqueValues, n); |
| | | 260 | | } |
| | | 261 | | |
| | 0 | 262 | | if (nonAsciiAffectedByCaseConversion) |
| | | 263 | | { |
| | 0 | 264 | | return asciiStartLettersOnly |
| | 0 | 265 | | ? PickTeddyImplementation<CaseInsensitiveAsciiLetters, CaseInsensitiveUnicode>(values, uniqueValues, |
| | 0 | 266 | | : PickTeddyImplementation<CaseInsensitiveAscii, CaseInsensitiveUnicode>(values, uniqueValues, n); |
| | | 267 | | } |
| | | 268 | | |
| | 0 | 269 | | return asciiStartLettersOnly |
| | 0 | 270 | | ? PickTeddyImplementation<CaseInsensitiveAsciiLetters, CaseInsensitiveAscii>(values, uniqueValues, n) |
| | 0 | 271 | | : PickTeddyImplementation<CaseInsensitiveAscii, CaseInsensitiveAscii>(values, uniqueValues, n); |
| | | 272 | | } |
| | | 273 | | |
| | | 274 | | private static SearchValues<string> PickTeddyImplementation<TStartCaseSensitivity, TCaseSensitivity>( |
| | | 275 | | ReadOnlySpan<string> values, |
| | | 276 | | HashSet<string> uniqueValues, |
| | | 277 | | int n) |
| | | 278 | | where TStartCaseSensitivity : struct, ICaseSensitivity |
| | | 279 | | where TCaseSensitivity : struct, ICaseSensitivity |
| | | 280 | | { |
| | 0 | 281 | | Debug.Assert(typeof(TStartCaseSensitivity) != typeof(CaseInsensitiveUnicode)); |
| | 0 | 282 | | Debug.Assert(values.Length > 1); |
| | 0 | 283 | | Debug.Assert(n is 2 or 3); |
| | | 284 | | |
| | 0 | 285 | | if (values.Length > TeddyBucketCount) |
| | | 286 | | { |
| | 0 | 287 | | string[][] buckets = TeddyBucketizer.Bucketize(values, TeddyBucketCount, n); |
| | | 288 | | |
| | | 289 | | // Potential optimization: We don't have to pick the first N characters for the fingerprint. |
| | | 290 | | // Different offset selection can noticeably improve throughput (e.g. 2x). |
| | | 291 | | |
| | 0 | 292 | | return n == 2 |
| | 0 | 293 | | ? new AsciiStringSearchValuesTeddyBucketizedN2<TStartCaseSensitivity, TCaseSensitivity>(buckets, val |
| | 0 | 294 | | : new AsciiStringSearchValuesTeddyBucketizedN3<TStartCaseSensitivity, TCaseSensitivity>(buckets, val |
| | | 295 | | } |
| | | 296 | | else |
| | | 297 | | { |
| | 0 | 298 | | return n == 2 |
| | 0 | 299 | | ? new AsciiStringSearchValuesTeddyNonBucketizedN2<TStartCaseSensitivity, TCaseSensitivity>(values, u |
| | 0 | 300 | | : new AsciiStringSearchValuesTeddyNonBucketizedN3<TStartCaseSensitivity, TCaseSensitivity>(values, u |
| | | 301 | | } |
| | | 302 | | } |
| | | 303 | | |
| | | 304 | | private static bool TryGenerateAllCasePermutationsForPrefixes(ReadOnlySpan<string> values, int n, int maxValues, |
| | | 305 | | { |
| | 0 | 306 | | Debug.Assert(n is 2 or 3); |
| | 0 | 307 | | Debug.Assert(values.Length < maxValues); |
| | | 308 | | |
| | | 309 | | // Count how many possible permutations there are. |
| | 0 | 310 | | int newValuesCount = 0; |
| | | 311 | | |
| | 0 | 312 | | foreach (string value in values) |
| | | 313 | | { |
| | 0 | 314 | | int permutations = 1; |
| | | 315 | | |
| | 0 | 316 | | foreach (char c in value.AsSpan(0, n)) |
| | | 317 | | { |
| | 0 | 318 | | Debug.Assert(char.IsAscii(c)); |
| | | 319 | | |
| | 0 | 320 | | if (char.IsAsciiLetter(c)) |
| | | 321 | | { |
| | 0 | 322 | | permutations *= 2; |
| | | 323 | | } |
| | | 324 | | } |
| | | 325 | | |
| | 0 | 326 | | newValuesCount += permutations; |
| | | 327 | | } |
| | | 328 | | |
| | 0 | 329 | | Debug.Assert(newValuesCount > values.Length, "Shouldn't have been called if there were no letters present"); |
| | | 330 | | |
| | 0 | 331 | | if (newValuesCount > maxValues) |
| | | 332 | | { |
| | 0 | 333 | | newValues = null; |
| | 0 | 334 | | return false; |
| | | 335 | | } |
| | | 336 | | |
| | | 337 | | // Generate the permutations. |
| | 0 | 338 | | newValues = new string[newValuesCount]; |
| | 0 | 339 | | newValuesCount = 0; |
| | | 340 | | |
| | 0 | 341 | | foreach (string value in values) |
| | | 342 | | { |
| | 0 | 343 | | int start = newValuesCount; |
| | | 344 | | |
| | 0 | 345 | | newValues[newValuesCount++] = value; |
| | | 346 | | |
| | 0 | 347 | | for (int i = 0; i < n; i++) |
| | | 348 | | { |
| | 0 | 349 | | char c = value[i]; |
| | | 350 | | |
| | 0 | 351 | | if (char.IsAsciiLetter(c)) |
| | | 352 | | { |
| | | 353 | | // Copy all the previous permutations of this value but change the casing of the i-th character. |
| | 0 | 354 | | foreach (string previous in newValues.AsSpan(start, newValuesCount - start)) |
| | | 355 | | { |
| | 0 | 356 | | newValues[newValuesCount++] = $"{previous.AsSpan(0, i)}{(char)(c ^ 0x20)}{previous.AsSpan(i |
| | | 357 | | } |
| | | 358 | | } |
| | | 359 | | } |
| | | 360 | | } |
| | | 361 | | |
| | 0 | 362 | | Debug.Assert(newValuesCount == newValues.Length); |
| | 0 | 363 | | return true; |
| | | 364 | | } |
| | | 365 | | |
| | | 366 | | private static SearchValues<string> CreateForSingleValue( |
| | | 367 | | string value, |
| | | 368 | | HashSet<string>? uniqueValues, |
| | | 369 | | bool ignoreCase, |
| | | 370 | | bool allAscii, |
| | | 371 | | bool asciiLettersOnly) |
| | | 372 | | { |
| | | 373 | | // We make use of optimizations that may overflow on 32bit systems for long values. |
| | | 374 | | int maxLength = IntPtr.Size == 4 ? 1_000_000_000 : int.MaxValue; |
| | | 375 | | |
| | 0 | 376 | | if (Vector128.IsHardwareAccelerated && value.Length > 1 && value.Length <= maxLength) |
| | | 377 | | { |
| | 0 | 378 | | SearchValues<string>? searchValues = value.Length switch |
| | 0 | 379 | | { |
| | 0 | 380 | | < 4 => TryCreateSingleValuesThreeChars<ValueLengthLessThan4>(value, uniqueValues, ignoreCase, allAsc |
| | 0 | 381 | | <= 8 => TryCreateSingleValuesThreeChars<ValueLength4To8>(value, uniqueValues, ignoreCase, allAscii, |
| | 0 | 382 | | <= 16 => TryCreateSingleValuesThreeChars<ValueLength9To16>(value, uniqueValues, ignoreCase, allAscii |
| | 0 | 383 | | _ => TryCreateSingleValuesThreeChars<ValueLengthLongOrUnknown>(value, uniqueValues, ignoreCase, allA |
| | 0 | 384 | | }; |
| | | 385 | | |
| | 0 | 386 | | if (searchValues is not null) |
| | | 387 | | { |
| | 0 | 388 | | return searchValues; |
| | | 389 | | } |
| | | 390 | | } |
| | | 391 | | |
| | 0 | 392 | | uniqueValues ??= new HashSet<string>(1, ignoreCase ? StringComparer.OrdinalIgnoreCase : StringComparer.Ordin |
| | | 393 | | |
| | 0 | 394 | | return ignoreCase |
| | 0 | 395 | | ? new SingleStringSearchValuesFallback<SearchValues.TrueConst>(value, uniqueValues) |
| | 0 | 396 | | : new SingleStringSearchValuesFallback<SearchValues.FalseConst>(value, uniqueValues); |
| | | 397 | | } |
| | | 398 | | |
| | | 399 | | private static SearchValues<string>? TryCreateSingleValuesThreeChars<TValueLength>( |
| | | 400 | | string value, |
| | | 401 | | HashSet<string>? uniqueValues, |
| | | 402 | | bool ignoreCase, |
| | | 403 | | bool allAscii, |
| | | 404 | | bool asciiLettersOnly) |
| | | 405 | | where TValueLength : struct, IValueLength |
| | | 406 | | { |
| | 0 | 407 | | if (!ignoreCase) |
| | | 408 | | { |
| | 0 | 409 | | return CreateSingleValuesThreeChars<TValueLength, CaseSensitive>(value, uniqueValues); |
| | | 410 | | } |
| | | 411 | | |
| | 0 | 412 | | if (asciiLettersOnly) |
| | | 413 | | { |
| | 0 | 414 | | return CreateSingleValuesThreeChars<TValueLength, CaseInsensitiveAsciiLetters>(value, uniqueValues); |
| | | 415 | | } |
| | | 416 | | |
| | 0 | 417 | | if (allAscii) |
| | | 418 | | { |
| | 0 | 419 | | return CreateSingleValuesThreeChars<TValueLength, CaseInsensitiveAscii>(value, uniqueValues); |
| | | 420 | | } |
| | | 421 | | |
| | | 422 | | // SingleStringSearchValuesThreeChars doesn't have logic to handle non-ASCII case conversion, so we require |
| | | 423 | | // Right now we're always selecting the first character as one of the anchors, and we need at least two. |
| | 0 | 424 | | if (char.IsAscii(value[0]) && value.AsSpan(1).ContainsAnyInRange((char)0, (char)127)) |
| | | 425 | | { |
| | 0 | 426 | | return CreateSingleValuesThreeChars<TValueLength, CaseInsensitiveUnicode>(value, uniqueValues); |
| | | 427 | | } |
| | | 428 | | |
| | 0 | 429 | | return null; |
| | | 430 | | } |
| | | 431 | | |
| | | 432 | | private static SearchValues<string> CreateSingleValuesThreeChars<TValueLength, TCaseSensitivity>( |
| | | 433 | | string value, |
| | | 434 | | HashSet<string>? uniqueValues) |
| | | 435 | | where TValueLength : struct, IValueLength |
| | | 436 | | where TCaseSensitivity : struct, ICaseSensitivity |
| | | 437 | | { |
| | 0 | 438 | | CharacterFrequencyHelper.GetSingleStringMultiCharacterOffsets(value, ignoreCase: typeof(TCaseSensitivity) != |
| | | 439 | | |
| | 0 | 440 | | if (CanUsePackedImpl(value[0]) && CanUsePackedImpl(value[ch2Offset]) && CanUsePackedImpl(value[ch3Offset])) |
| | | 441 | | { |
| | 0 | 442 | | return new SingleStringSearchValuesPackedThreeChars<TValueLength, TCaseSensitivity>(uniqueValues, value, |
| | | 443 | | } |
| | | 444 | | |
| | 0 | 445 | | return new SingleStringSearchValuesThreeChars<TValueLength, TCaseSensitivity>(uniqueValues, value, ch2Offset |
| | | 446 | | |
| | | 447 | | // Unlike with PackedSpanHelpers (Sse2 only), we are also using this approach on ARM64. |
| | | 448 | | // We use PackUnsignedSaturate on X86 and UnzipEven on ARM, so the set of allowed characters differs slightl |
| | | 449 | | static bool CanUsePackedImpl(char c) => |
| | | 450 | | PackedSpanHelpers.PackedIndexOfIsSupported ? PackedSpanHelpers.CanUsePackedIndexOf(c) : |
| | | 451 | | (AdvSimd.Arm64.IsSupported && c <= byte.MaxValue); |
| | | 452 | | } |
| | | 453 | | |
| | | 454 | | private static void AnalyzeValues( |
| | | 455 | | ReadOnlySpan<string> values, |
| | | 456 | | ref bool ignoreCase, |
| | | 457 | | out bool allAscii, |
| | | 458 | | out bool asciiLettersOnly, |
| | | 459 | | out bool nonAsciiAffectedByCaseConversion, |
| | | 460 | | out int minLength) |
| | | 461 | | { |
| | 0 | 462 | | allAscii = true; |
| | 0 | 463 | | asciiLettersOnly = true; |
| | 0 | 464 | | minLength = int.MaxValue; |
| | | 465 | | |
| | 0 | 466 | | foreach (string value in values) |
| | | 467 | | { |
| | 0 | 468 | | allAscii = allAscii && Ascii.IsValid(value); |
| | 0 | 469 | | asciiLettersOnly = asciiLettersOnly && !value.AsSpan().ContainsAnyExcept(s_asciiLetters); |
| | 0 | 470 | | minLength = Math.Min(minLength, value.Length); |
| | | 471 | | } |
| | | 472 | | |
| | | 473 | | // Potential optimization: Not all characters participate in Unicode case conversion. |
| | | 474 | | // If we can determine that none of the non-ASCII characters do, we can make searching faster |
| | | 475 | | // by using the same paths as we do for ASCII-only values. |
| | 0 | 476 | | nonAsciiAffectedByCaseConversion = ignoreCase && !allAscii; |
| | | 477 | | |
| | | 478 | | // If all the characters in values are unaffected by casing, we can avoid the ignoreCase overhead. |
| | 0 | 479 | | if (ignoreCase && !nonAsciiAffectedByCaseConversion && !asciiLettersOnly) |
| | | 480 | | { |
| | 0 | 481 | | ignoreCase = false; |
| | | 482 | | |
| | 0 | 483 | | foreach (string value in values) |
| | | 484 | | { |
| | 0 | 485 | | if (value.AsSpan().ContainsAny(s_asciiLetters)) |
| | | 486 | | { |
| | 0 | 487 | | ignoreCase = true; |
| | 0 | 488 | | break; |
| | | 489 | | } |
| | | 490 | | } |
| | | 491 | | } |
| | 0 | 492 | | } |
| | | 493 | | |
| | | 494 | | private static bool ContainsInvalidValues(ReadOnlySpan<string> values) |
| | | 495 | | { |
| | 0 | 496 | | foreach (string value in values) |
| | | 497 | | { |
| | 0 | 498 | | if (!Utf16.IsValid(value)) |
| | | 499 | | { |
| | 0 | 500 | | return true; |
| | | 501 | | } |
| | | 502 | | } |
| | | 503 | | |
| | 0 | 504 | | return false; |
| | | 505 | | } |
| | | 506 | | } |
| | | 507 | | } |
| | | 508 | | |