| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | using System.Buffers; |
| | | 5 | | using System.Collections.Generic; |
| | | 6 | | using System.Diagnostics; |
| | | 7 | | using System.Diagnostics.CodeAnalysis; |
| | | 8 | | using System.Text; |
| | | 9 | | |
| | | 10 | | namespace System |
| | | 11 | | { |
| | | 12 | | internal static class UriHelper |
| | | 13 | | { |
| | | 14 | | public static string SpanToLowerInvariantString(ReadOnlySpan<char> span) |
| | 0 | 15 | | { |
| | 0 | 16 | | return string.Create(span.Length, span, static (buffer, span) => |
| | 0 | 17 | | { |
| | 0 | 18 | | int charsWritten = span.ToLowerInvariant(buffer); |
| | 0 | 19 | | Debug.Assert(charsWritten == buffer.Length); |
| | 0 | 20 | | }); |
| | 0 | 21 | | } |
| | | 22 | | |
| | | 23 | | public static string NormalizeAndConcat(string? start, ReadOnlySpan<char> toNormalize) |
| | 0 | 24 | | { |
| | 0 | 25 | | var vsb = new ValueStringBuilder(stackalloc char[Uri.StackallocThreshold]); |
| | | 26 | | |
| | | 27 | | int charsWritten; |
| | 0 | 28 | | while (!toNormalize.TryNormalize(vsb.RawChars, out charsWritten, NormalizationForm.FormC)) |
| | 0 | 29 | | { |
| | 0 | 30 | | vsb.EnsureCapacity(vsb.Capacity + 1); |
| | 0 | 31 | | } |
| | | 32 | | |
| | 0 | 33 | | string result = string.Concat(start, vsb.RawChars.Slice(0, charsWritten)); |
| | 0 | 34 | | vsb.Dispose(); |
| | 0 | 35 | | return result; |
| | 0 | 36 | | } |
| | | 37 | | |
| | | 38 | | // http://host/Path/Path/File?Query is the base of |
| | | 39 | | // - http://host/Path/Path/File/ ... (those "File" words may be different in semantic but anyway) |
| | | 40 | | // - http://host/Path/Path/#Fragment |
| | | 41 | | // - http://host/Path/Path/?Query |
| | | 42 | | // - http://host/Path/Path/MoreDir/ ... |
| | | 43 | | // - http://host/Path/Path/OtherFile?Query |
| | | 44 | | // - http://host/Path/Path/Fl |
| | | 45 | | // - http://host/Path/Path/ |
| | | 46 | | // |
| | | 47 | | // It is not a base for |
| | | 48 | | // - http://host/Path/Path (that last "Path" is not considered as a directory) |
| | | 49 | | // - http://host/Path/Path?Query |
| | | 50 | | // - http://host/Path/Path#Fragment |
| | | 51 | | // - http://host/Path/Path2/ |
| | | 52 | | // - http://host/Path/Path2/MoreDir |
| | | 53 | | // - http://host/Path/File |
| | | 54 | | // |
| | | 55 | | // ASSUMES that strings like http://host/Path/Path/MoreDir/../../ have been canonicalized before going to this |
| | | 56 | | // ASSUMES that back slashes already have been converted if applicable. |
| | | 57 | | // |
| | | 58 | | internal static bool TestForSubPath(ReadOnlySpan<char> self, ReadOnlySpan<char> other, bool ignoreCase) |
| | 0 | 59 | | { |
| | 0 | 60 | | int i = 0; |
| | | 61 | | char chSelf; |
| | | 62 | | char chOther; |
| | | 63 | | |
| | 0 | 64 | | bool AllSameBeforeSlash = true; |
| | | 65 | | |
| | 0 | 66 | | for (; i < self.Length && i < other.Length; ++i) |
| | 0 | 67 | | { |
| | 0 | 68 | | chSelf = self[i]; |
| | 0 | 69 | | chOther = other[i]; |
| | | 70 | | |
| | 0 | 71 | | if (chSelf == '?' || chSelf == '#') |
| | 0 | 72 | | { |
| | | 73 | | // survived so far and selfPtr does not have any more path segments |
| | 0 | 74 | | return true; |
| | | 75 | | } |
| | | 76 | | |
| | | 77 | | // If selfPtr terminates a path segment, so must otherPtr |
| | 0 | 78 | | if (chSelf == '/') |
| | 0 | 79 | | { |
| | 0 | 80 | | if (chOther != '/') |
| | 0 | 81 | | { |
| | | 82 | | // comparison has failed |
| | 0 | 83 | | return false; |
| | | 84 | | } |
| | | 85 | | // plus the segments must be the same |
| | 0 | 86 | | if (!AllSameBeforeSlash) |
| | 0 | 87 | | { |
| | | 88 | | // comparison has failed |
| | 0 | 89 | | return false; |
| | | 90 | | } |
| | | 91 | | //so far so good |
| | 0 | 92 | | AllSameBeforeSlash = true; |
| | 0 | 93 | | continue; |
| | | 94 | | } |
| | | 95 | | |
| | | 96 | | // if otherPtr terminates then selfPtr must not have any more path segments |
| | 0 | 97 | | if (chOther == '?' || chOther == '#') |
| | 0 | 98 | | { |
| | 0 | 99 | | break; |
| | | 100 | | } |
| | | 101 | | |
| | 0 | 102 | | if (!ignoreCase) |
| | 0 | 103 | | { |
| | 0 | 104 | | if (chSelf != chOther) |
| | 0 | 105 | | { |
| | 0 | 106 | | AllSameBeforeSlash = false; |
| | 0 | 107 | | } |
| | 0 | 108 | | } |
| | | 109 | | else |
| | 0 | 110 | | { |
| | 0 | 111 | | if (char.ToLowerInvariant(chSelf) != char.ToLowerInvariant(chOther)) |
| | 0 | 112 | | { |
| | 0 | 113 | | AllSameBeforeSlash = false; |
| | 0 | 114 | | } |
| | 0 | 115 | | } |
| | 0 | 116 | | } |
| | | 117 | | |
| | | 118 | | // If self is longer then it must not have any more path segments |
| | 0 | 119 | | for (; i < self.Length; ++i) |
| | 0 | 120 | | { |
| | 0 | 121 | | if ((chSelf = self[i]) == '?' || chSelf == '#') |
| | 0 | 122 | | { |
| | 0 | 123 | | return true; |
| | | 124 | | } |
| | 0 | 125 | | if (chSelf == '/') |
| | 0 | 126 | | { |
| | 0 | 127 | | return false; |
| | | 128 | | } |
| | 0 | 129 | | } |
| | | 130 | | //survived by getting to the end of selfPtr |
| | 0 | 131 | | return true; |
| | 0 | 132 | | } |
| | | 133 | | |
| | | 134 | | public static bool TryEscapeDataString(ReadOnlySpan<char> charsToEscape, Span<char> destination, out int charsWr |
| | 0 | 135 | | { |
| | 0 | 136 | | if (destination.Length < charsToEscape.Length) |
| | 0 | 137 | | { |
| | 0 | 138 | | charsWritten = 0; |
| | 0 | 139 | | return false; |
| | | 140 | | } |
| | | 141 | | |
| | 0 | 142 | | int indexOfFirstToEscape = charsToEscape.IndexOfAnyExcept(Unreserved); |
| | 0 | 143 | | if (indexOfFirstToEscape < 0) |
| | 0 | 144 | | { |
| | | 145 | | // Nothing to escape, just copy the original chars. |
| | 0 | 146 | | charsToEscape.CopyTo(destination); |
| | 0 | 147 | | charsWritten = charsToEscape.Length; |
| | 0 | 148 | | return true; |
| | | 149 | | } |
| | | 150 | | |
| | | 151 | | // We may throw for very large inputs (when growing the ValueStringBuilder). |
| | | 152 | | scoped ValueStringBuilder vsb; |
| | | 153 | | |
| | | 154 | | // If the input and destination buffers overlap, we must take care not to overwrite parts of the input befor |
| | 0 | 155 | | bool overlapped = charsToEscape.Overlaps(destination); |
| | | 156 | | |
| | 0 | 157 | | if (overlapped) |
| | 0 | 158 | | { |
| | 0 | 159 | | vsb = new ValueStringBuilder(stackalloc char[Uri.StackallocThreshold]); |
| | 0 | 160 | | vsb.EnsureCapacity(charsToEscape.Length); |
| | 0 | 161 | | } |
| | | 162 | | else |
| | 0 | 163 | | { |
| | 0 | 164 | | vsb = new ValueStringBuilder(destination.Slice(indexOfFirstToEscape)); |
| | 0 | 165 | | } |
| | | 166 | | |
| | 0 | 167 | | EscapeStringToBuilder(charsToEscape.Slice(indexOfFirstToEscape), ref vsb, Unreserved, checkExistingEscaped: |
| | | 168 | | |
| | 0 | 169 | | int newLength = checked(indexOfFirstToEscape + vsb.Length); |
| | 0 | 170 | | Debug.Assert(newLength > charsToEscape.Length); |
| | | 171 | | |
| | 0 | 172 | | if (destination.Length >= newLength) |
| | 0 | 173 | | { |
| | 0 | 174 | | charsToEscape.Slice(0, indexOfFirstToEscape).CopyTo(destination); |
| | | 175 | | |
| | 0 | 176 | | if (overlapped) |
| | 0 | 177 | | { |
| | 0 | 178 | | vsb.AsSpan().CopyTo(destination.Slice(indexOfFirstToEscape)); |
| | 0 | 179 | | vsb.Dispose(); |
| | 0 | 180 | | } |
| | | 181 | | else |
| | 0 | 182 | | { |
| | | 183 | | // We are expecting the builder not to grow if the original span was large enough. |
| | | 184 | | // This means that we MUST NOT over allocate anywhere in EscapeStringToBuilder (e.g. append and then |
| | 0 | 185 | | Debug.Assert(vsb.RawChars.Overlaps(destination)); |
| | 0 | 186 | | } |
| | | 187 | | |
| | 0 | 188 | | charsWritten = newLength; |
| | 0 | 189 | | return true; |
| | | 190 | | } |
| | | 191 | | |
| | 0 | 192 | | vsb.Dispose(); |
| | 0 | 193 | | charsWritten = 0; |
| | 0 | 194 | | return false; |
| | 0 | 195 | | } |
| | | 196 | | |
| | | 197 | | public static string EscapeString(string stringToEscape, bool checkExistingEscaped, SearchValues<char> noEscape) |
| | 0 | 198 | | { |
| | 0 | 199 | | ArgumentNullException.ThrowIfNull(stringToEscape); |
| | | 200 | | |
| | 0 | 201 | | return EscapeString(stringToEscape, checkExistingEscaped, noEscape, stringToEscape); |
| | 0 | 202 | | } |
| | | 203 | | |
| | | 204 | | public static string EscapeString(ReadOnlySpan<char> charsToEscape, bool checkExistingEscaped, SearchValues<char |
| | 0 | 205 | | { |
| | 0 | 206 | | Debug.Assert(!noEscape.Contains('%'), "Need to treat % specially; it should be part of any escaped set"); |
| | 0 | 207 | | Debug.Assert(backingString is null || backingString.Length == charsToEscape.Length); |
| | | 208 | | |
| | 0 | 209 | | int indexOfFirstToEscape = charsToEscape.IndexOfAnyExcept(noEscape); |
| | 0 | 210 | | if (indexOfFirstToEscape < 0) |
| | 0 | 211 | | { |
| | | 212 | | // Nothing to escape, just return the original value. |
| | 0 | 213 | | return backingString ?? charsToEscape.ToString(); |
| | | 214 | | } |
| | | 215 | | |
| | | 216 | | // Otherwise, create a ValueStringBuilder to store the escaped data into, |
| | | 217 | | // escape the rest, and concat the result with the characters we skipped above. |
| | 0 | 218 | | var vsb = new ValueStringBuilder(stackalloc char[Uri.StackallocThreshold]); |
| | | 219 | | |
| | | 220 | | // We may throw for very large inputs (when growing the ValueStringBuilder). |
| | 0 | 221 | | vsb.EnsureCapacity(charsToEscape.Length); |
| | | 222 | | |
| | 0 | 223 | | EscapeStringToBuilder(charsToEscape.Slice(indexOfFirstToEscape), ref vsb, noEscape, checkExistingEscaped); |
| | | 224 | | |
| | 0 | 225 | | string result = string.Concat(charsToEscape.Slice(0, indexOfFirstToEscape), vsb.AsSpan()); |
| | 0 | 226 | | vsb.Dispose(); |
| | 0 | 227 | | return result; |
| | 0 | 228 | | } |
| | | 229 | | |
| | | 230 | | internal static void EscapeString(scoped ReadOnlySpan<char> stringToEscape, ref ValueStringBuilder dest, |
| | | 231 | | bool checkExistingEscaped, SearchValues<char> noEscape) |
| | 0 | 232 | | { |
| | 0 | 233 | | Debug.Assert(!noEscape.Contains('%'), "Need to treat % specially; it should be part of any escaped set"); |
| | | 234 | | |
| | 0 | 235 | | int indexOfFirstToEscape = stringToEscape.IndexOfAnyExcept(noEscape); |
| | 0 | 236 | | if (indexOfFirstToEscape < 0) |
| | 0 | 237 | | { |
| | | 238 | | // Nothing to escape, just copy the whole span. |
| | 0 | 239 | | dest.Append(stringToEscape); |
| | 0 | 240 | | } |
| | | 241 | | else |
| | 0 | 242 | | { |
| | 0 | 243 | | dest.Append(stringToEscape.Slice(0, indexOfFirstToEscape)); |
| | | 244 | | |
| | 0 | 245 | | EscapeStringToBuilder(stringToEscape.Slice(indexOfFirstToEscape), ref dest, noEscape, checkExistingEscap |
| | 0 | 246 | | } |
| | 0 | 247 | | } |
| | | 248 | | |
| | | 249 | | private static void EscapeStringToBuilder( |
| | | 250 | | scoped ReadOnlySpan<char> stringToEscape, ref ValueStringBuilder vsb, |
| | | 251 | | SearchValues<char> noEscape, bool checkExistingEscaped) |
| | 0 | 252 | | { |
| | 0 | 253 | | Debug.Assert(!stringToEscape.IsEmpty && !noEscape.Contains(stringToEscape[0])); |
| | | 254 | | |
| | | 255 | | // Allocate enough stack space to hold any Rune's UTF8 encoding. |
| | 0 | 256 | | Span<byte> utf8Bytes = [0, 0, 0, 0]; |
| | | 257 | | |
| | 0 | 258 | | while (!stringToEscape.IsEmpty) |
| | 0 | 259 | | { |
| | 0 | 260 | | char c = stringToEscape[0]; |
| | | 261 | | |
| | 0 | 262 | | if (!char.IsAscii(c)) |
| | 0 | 263 | | { |
| | 0 | 264 | | if (Rune.DecodeFromUtf16(stringToEscape, out Rune r, out int charsConsumed) != OperationStatus.Done) |
| | 0 | 265 | | { |
| | 0 | 266 | | r = Rune.ReplacementChar; |
| | 0 | 267 | | } |
| | | 268 | | |
| | 0 | 269 | | Debug.Assert(stringToEscape.EnumerateRunes() is { } e && e.MoveNext() && e.Current == r); |
| | 0 | 270 | | Debug.Assert(charsConsumed is 1 or 2); |
| | | 271 | | |
| | 0 | 272 | | stringToEscape = stringToEscape.Slice(charsConsumed); |
| | | 273 | | |
| | | 274 | | // The rune is non-ASCII, so encode it as UTF8, and escape each UTF8 byte. |
| | 0 | 275 | | r.TryEncodeToUtf8(utf8Bytes, out int bytesWritten); |
| | 0 | 276 | | foreach (byte b in utf8Bytes.Slice(0, bytesWritten)) |
| | 0 | 277 | | { |
| | 0 | 278 | | PercentEncodeByte(b, ref vsb); |
| | 0 | 279 | | } |
| | | 280 | | |
| | 0 | 281 | | continue; |
| | | 282 | | } |
| | | 283 | | |
| | 0 | 284 | | if (!noEscape.Contains(c)) |
| | 0 | 285 | | { |
| | | 286 | | // If we're checking for existing escape sequences, then if this is the beginning of |
| | | 287 | | // one, check the next two characters in the sequence. |
| | 0 | 288 | | if (c == '%' && checkExistingEscaped) |
| | 0 | 289 | | { |
| | | 290 | | // If the next two characters are valid escaped ASCII, then just output them as-is. |
| | 0 | 291 | | if (stringToEscape.Length > 2 && char.IsAsciiHexDigit(stringToEscape[1]) && char.IsAsciiHexDigit |
| | 0 | 292 | | { |
| | 0 | 293 | | vsb.Append('%'); |
| | 0 | 294 | | vsb.Append(stringToEscape[1]); |
| | 0 | 295 | | vsb.Append(stringToEscape[2]); |
| | 0 | 296 | | stringToEscape = stringToEscape.Slice(3); |
| | 0 | 297 | | continue; |
| | | 298 | | } |
| | 0 | 299 | | } |
| | | 300 | | |
| | 0 | 301 | | PercentEncodeByte((byte)c, ref vsb); |
| | 0 | 302 | | stringToEscape = stringToEscape.Slice(1); |
| | 0 | 303 | | continue; |
| | | 304 | | } |
| | | 305 | | |
| | | 306 | | // We have a character we don't want to escape. It's likely there are more, do a vectorized search. |
| | 0 | 307 | | int charsToCopy = stringToEscape.IndexOfAnyExcept(noEscape); |
| | 0 | 308 | | if (charsToCopy < 0) |
| | 0 | 309 | | { |
| | 0 | 310 | | charsToCopy = stringToEscape.Length; |
| | 0 | 311 | | } |
| | 0 | 312 | | Debug.Assert(charsToCopy > 0); |
| | | 313 | | |
| | 0 | 314 | | vsb.Append(stringToEscape.Slice(0, charsToCopy)); |
| | 0 | 315 | | stringToEscape = stringToEscape.Slice(charsToCopy); |
| | 0 | 316 | | } |
| | 0 | 317 | | } |
| | | 318 | | |
| | | 319 | | internal static void Unescape(scoped ReadOnlySpan<char> chars, ref ValueStringBuilder dest) |
| | 0 | 320 | | { |
| | 0 | 321 | | for (int i = 0; (uint)i < (uint)chars.Length;) |
| | 0 | 322 | | { |
| | 0 | 323 | | if (chars[i] == '%' && (uint)(i + 2) < (uint)chars.Length) |
| | 0 | 324 | | { |
| | 0 | 325 | | char unescaped = DecodeHexChars(chars[i + 1], chars[i + 2]); |
| | | 326 | | |
| | 0 | 327 | | if (unescaped == Uri.c_DummyChar) |
| | 0 | 328 | | { |
| | 0 | 329 | | i++; |
| | 0 | 330 | | continue; |
| | | 331 | | } |
| | | 332 | | |
| | | 333 | | // Copy previous characters that don't require any transformations. |
| | | 334 | | // Using a loop instead of Append(span) to avoid the call overhead for typically short sections. |
| | 0 | 335 | | foreach (char c in chars.Slice(0, i)) |
| | 0 | 336 | | { |
| | 0 | 337 | | dest.Append(c); |
| | 0 | 338 | | } |
| | | 339 | | |
| | 0 | 340 | | if (char.IsAscii(unescaped)) |
| | 0 | 341 | | { |
| | 0 | 342 | | dest.Append(unescaped); |
| | 0 | 343 | | i += 3; |
| | 0 | 344 | | } |
| | | 345 | | else |
| | 0 | 346 | | { |
| | 0 | 347 | | int charactersRead = PercentEncodingHelper.UnescapePercentEncodedUTF8Sequence( |
| | 0 | 348 | | chars.Slice(i), |
| | 0 | 349 | | ref dest, |
| | 0 | 350 | | isQuery: false, |
| | 0 | 351 | | iriParsing: false); |
| | | 352 | | |
| | 0 | 353 | | Debug.Assert(charactersRead > 0); |
| | 0 | 354 | | i += charactersRead; |
| | 0 | 355 | | } |
| | | 356 | | |
| | 0 | 357 | | chars = chars.Slice(i); |
| | 0 | 358 | | i = 0; |
| | 0 | 359 | | } |
| | | 360 | | else |
| | 0 | 361 | | { |
| | 0 | 362 | | i++; |
| | 0 | 363 | | } |
| | 0 | 364 | | } |
| | | 365 | | |
| | 0 | 366 | | dest.Append(chars); |
| | 0 | 367 | | } |
| | | 368 | | |
| | | 369 | | internal static void UnescapeString(scoped ReadOnlySpan<char> chars, ref ValueStringBuilder dest, |
| | | 370 | | char rsvd1, char rsvd2, char rsvd3, UnescapeMode unescapeMode, UriParser? syntax, bool isQuery) |
| | 0 | 371 | | { |
| | 0 | 372 | | Debug.Assert(unescapeMode != UnescapeMode.None); |
| | | 373 | | |
| | 0 | 374 | | bool escapeReserved = false; |
| | 0 | 375 | | bool iriParsing = Uri.IriParsingStatic(syntax) |
| | 0 | 376 | | && ((unescapeMode & UnescapeMode.EscapeUnescape) == UnescapeMode.EscapeUnescape); |
| | | 377 | | |
| | 0 | 378 | | while (!chars.IsEmpty) |
| | 0 | 379 | | { |
| | | 380 | | int i; |
| | 0 | 381 | | char ch = (char)0; |
| | | 382 | | |
| | 0 | 383 | | for (i = 0; (uint)i < (uint)chars.Length; i++) |
| | 0 | 384 | | { |
| | 0 | 385 | | ch = chars[i]; |
| | | 386 | | |
| | 0 | 387 | | if (ch == '%') |
| | 0 | 388 | | { |
| | 0 | 389 | | if ((unescapeMode & UnescapeMode.Unescape) == 0) |
| | 0 | 390 | | { |
| | | 391 | | // re-escape, don't check anything else |
| | 0 | 392 | | escapeReserved = true; |
| | 0 | 393 | | } |
| | 0 | 394 | | else if ((uint)(i + 2) < (uint)chars.Length) |
| | 0 | 395 | | { |
| | 0 | 396 | | ch = DecodeHexChars(chars[i + 1], chars[i + 2]); |
| | | 397 | | |
| | | 398 | | // re-escape % from an invalid sequence |
| | 0 | 399 | | if (ch == Uri.c_DummyChar) |
| | 0 | 400 | | { |
| | 0 | 401 | | if ((unescapeMode & UnescapeMode.Escape) != 0) |
| | 0 | 402 | | escapeReserved = true; |
| | | 403 | | else |
| | 0 | 404 | | continue; // we should throw instead but since v1.0 would just print '%' |
| | 0 | 405 | | } |
| | | 406 | | // Do not unescape '%' itself unless full unescape is requested |
| | 0 | 407 | | else if (ch == '%') |
| | 0 | 408 | | { |
| | 0 | 409 | | i += 2; |
| | 0 | 410 | | continue; |
| | | 411 | | } |
| | | 412 | | // Do not unescape a reserved char unless full unescape is requested |
| | 0 | 413 | | else if (ch == rsvd1 || ch == rsvd2 || ch == rsvd3) |
| | 0 | 414 | | { |
| | 0 | 415 | | i += 2; |
| | 0 | 416 | | continue; |
| | | 417 | | } |
| | | 418 | | // Do not unescape a dangerous char unless it's V1ToStringFlags mode |
| | 0 | 419 | | else if ((unescapeMode & UnescapeMode.V1ToStringFlag) == 0 && IsNotSafeForUnescape(ch)) |
| | 0 | 420 | | { |
| | 0 | 421 | | i += 2; |
| | 0 | 422 | | continue; |
| | | 423 | | } |
| | 0 | 424 | | else if (iriParsing && (ch <= '\x9F' ? IsNotSafeForUnescape(ch) : !IriHelper.CheckIriUnicode |
| | 0 | 425 | | { |
| | | 426 | | // check if unenscaping gives a char outside iri range |
| | | 427 | | // if it does then keep it escaped |
| | 0 | 428 | | i += 2; |
| | 0 | 429 | | continue; |
| | | 430 | | } |
| | | 431 | | // unescape escaped char or escape % |
| | 0 | 432 | | break; |
| | | 433 | | } |
| | | 434 | | else |
| | 0 | 435 | | { |
| | 0 | 436 | | escapeReserved = true; |
| | 0 | 437 | | } |
| | | 438 | | // escape (escapeReserved==true) or otherwise unescape the sequence |
| | 0 | 439 | | break; |
| | | 440 | | } |
| | 0 | 441 | | else if ((unescapeMode & UnescapeMode.Escape) != 0) |
| | 0 | 442 | | { |
| | | 443 | | // Could actually escape some of the characters |
| | 0 | 444 | | if (ch == rsvd1 || ch == rsvd2 || ch == rsvd3) |
| | 0 | 445 | | { |
| | | 446 | | // found an unescaped reserved character -> escape it |
| | 0 | 447 | | escapeReserved = true; |
| | 0 | 448 | | break; |
| | | 449 | | } |
| | 0 | 450 | | else if ((unescapeMode & UnescapeMode.V1ToStringFlag) == 0 |
| | 0 | 451 | | && (ch <= '\x1F' || (ch >= '\x7F' && ch <= '\x9F'))) |
| | 0 | 452 | | { |
| | | 453 | | // found an unescaped reserved character -> escape it |
| | 0 | 454 | | escapeReserved = true; |
| | 0 | 455 | | break; |
| | | 456 | | } |
| | 0 | 457 | | } |
| | 0 | 458 | | } |
| | | 459 | | |
| | | 460 | | // Copy previous characters that don't require any transformations. |
| | | 461 | | // Using a loop instead of Append(span) to avoid the call overhead for typically short sections. |
| | 0 | 462 | | foreach (char c in chars.Slice(0, i)) |
| | 0 | 463 | | { |
| | 0 | 464 | | dest.Append(c); |
| | 0 | 465 | | } |
| | | 466 | | |
| | 0 | 467 | | if (i < chars.Length) |
| | 0 | 468 | | { |
| | 0 | 469 | | if (escapeReserved) |
| | 0 | 470 | | { |
| | 0 | 471 | | PercentEncodeByte((byte)chars[i], ref dest); |
| | 0 | 472 | | escapeReserved = false; |
| | 0 | 473 | | i++; |
| | 0 | 474 | | } |
| | 0 | 475 | | else if (ch <= 127) |
| | 0 | 476 | | { |
| | 0 | 477 | | dest.Append(ch); |
| | 0 | 478 | | i += 3; |
| | 0 | 479 | | } |
| | | 480 | | else |
| | 0 | 481 | | { |
| | | 482 | | // Unicode |
| | 0 | 483 | | int charactersRead = PercentEncodingHelper.UnescapePercentEncodedUTF8Sequence( |
| | 0 | 484 | | chars.Slice(i), |
| | 0 | 485 | | ref dest, |
| | 0 | 486 | | isQuery, |
| | 0 | 487 | | iriParsing); |
| | | 488 | | |
| | 0 | 489 | | Debug.Assert(charactersRead > 0); |
| | 0 | 490 | | i += charactersRead; |
| | 0 | 491 | | } |
| | 0 | 492 | | } |
| | | 493 | | |
| | 0 | 494 | | chars = chars.Slice(i); |
| | 0 | 495 | | } |
| | 0 | 496 | | } |
| | | 497 | | |
| | | 498 | | internal static void PercentEncodeByte(byte b, ref ValueStringBuilder to) |
| | 0 | 499 | | { |
| | 0 | 500 | | to.Append('%'); |
| | 0 | 501 | | HexConverter.ToCharsBuffer(b, to.AppendSpan(2), 0, HexConverter.Casing.Upper); |
| | 0 | 502 | | } |
| | | 503 | | |
| | | 504 | | /// <summary> |
| | | 505 | | /// Converts 2 hex chars to a byte (returned in a char), e.g, "0a" becomes (char)0x0A. |
| | | 506 | | /// <para>If either char is not hex, returns <see cref="Uri.c_DummyChar"/>.</para> |
| | | 507 | | /// </summary> |
| | | 508 | | internal static char DecodeHexChars(int first, int second) |
| | 0 | 509 | | { |
| | 0 | 510 | | int a = HexConverter.FromChar(first); |
| | 0 | 511 | | int b = HexConverter.FromChar(second); |
| | | 512 | | |
| | 0 | 513 | | if ((a | b) == 0xFF) |
| | 0 | 514 | | { |
| | | 515 | | // either a or b is 0xFF (invalid) |
| | 0 | 516 | | return Uri.c_DummyChar; |
| | | 517 | | } |
| | | 518 | | |
| | 0 | 519 | | return (char)((a << 4) | b); |
| | 0 | 520 | | } |
| | | 521 | | |
| | | 522 | | // When unescaping in safe mode, do not unescape the RFC 3986 reserved set: |
| | | 523 | | // reserved = gen-delims / sub-delims |
| | | 524 | | // gen-delims = ":" / "/" / "?" / "#" / "[" / "]" / "@" |
| | | 525 | | // sub-delims = "!" / "$" / "&" / "'" / "(" / ")" |
| | | 526 | | // / "*" / "+" / "," / ";" / "=" |
| | | 527 | | // |
| | | 528 | | // In addition, do not unescape the following unsafe characters: |
| | | 529 | | // excluded = "%" / "\" |
| | | 530 | | internal static bool IsNotSafeForUnescape(char ch) => |
| | 0 | 531 | | s_notSafeForUnescapeChars.Contains(ch); |
| | | 532 | | |
| | 0 | 533 | | private static readonly SearchValues<char> s_notSafeForUnescapeChars = SearchValues.Create( |
| | 0 | 534 | | "\u0000\u0001\u0002\u0003\u0004\u0005\u0006\u0007\u0008\u0009\u000A\u000B\u000C\u000D\u000E\u000F" + |
| | 0 | 535 | | "\u0010\u0011\u0012\u0013\u0014\u0015\u0016\u0017\u0018\u0019\u001A\u001B\u001C\u001D\u001E\u001F" + |
| | 0 | 536 | | ";/?:@&=+$,#[]!'()*" + "%\\" + "\u007F" + |
| | 0 | 537 | | "\u0080\u0081\u0082\u0083\u0084\u0085\u0086\u0087\u0088\u0089\u008A\u008B\u008C\u008D\u008E\u008F" + |
| | 0 | 538 | | "\u0090\u0091\u0092\u0093\u0094\u0095\u0096\u0097\u0098\u0099\u009A\u009B\u009C\u009D\u009E\u009F"); |
| | | 539 | | |
| | | 540 | | /// <summary>All ASCII letters and digits, as well as the RFC3986 unreserved marks '-', '_', '.', and '~'.</summ |
| | 0 | 541 | | public static readonly SearchValues<char> Unreserved = |
| | 0 | 542 | | SearchValues.Create("-.0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZ_abcdefghijklmnopqrstuvwxyz~"); |
| | | 543 | | |
| | | 544 | | /// <summary>All ASCII letters and digits, as well as the RFC3986 reserved and unreserved marks.</summary> |
| | 0 | 545 | | public static readonly SearchValues<char> UnreservedReserved = |
| | 0 | 546 | | SearchValues.Create("!#$&'()*+,-./0123456789:;=?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[]_abcdefghijklmnopqrstuvwxyz~"); |
| | | 547 | | |
| | 0 | 548 | | public static readonly SearchValues<char> UnreservedReservedExceptHash = |
| | 0 | 549 | | SearchValues.Create("!$&'()*+,-./0123456789:;=?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[]_abcdefghijklmnopqrstuvwxyz~"); |
| | | 550 | | |
| | 0 | 551 | | public static readonly SearchValues<char> UnreservedReservedExceptQuestionMarkHash = |
| | 0 | 552 | | SearchValues.Create("!$&'()*+,-./0123456789:;=@ABCDEFGHIJKLMNOPQRSTUVWXYZ[]_abcdefghijklmnopqrstuvwxyz~"); |
| | | 553 | | |
| | 0 | 554 | | internal static readonly char[] s_WSchars = new char[] { ' ', '\n', '\r', '\t' }; |
| | | 555 | | |
| | | 556 | | internal static bool IsLWS(char ch) |
| | 1322 | 557 | | { |
| | 1322 | 558 | | return (ch <= ' ') && (ch == ' ' || ch == '\n' || ch == '\r' || ch == '\t'); |
| | 1322 | 559 | | } |
| | | 560 | | |
| | | 561 | | // Is this a Bidirectional control char.. These get stripped |
| | | 562 | | internal static bool IsBidiControlCharacter(char ch) => |
| | 0 | 563 | | char.IsBetween(ch, '\u200E', '\u202E') && !char.IsBetween(ch, '\u2010', '\u2029'); |
| | | 564 | | |
| | | 565 | | // Strip Bidirectional control characters from this string |
| | | 566 | | public static string StripBidiControlCharacters(ReadOnlySpan<char> strToClean, string? backingString = null) |
| | 0 | 567 | | { |
| | 0 | 568 | | Debug.Assert(backingString is null || strToClean.Length == backingString.Length); |
| | | 569 | | |
| | 0 | 570 | | if (StripBidiControlCharacters(strToClean, out string? stripped)) |
| | 0 | 571 | | { |
| | 0 | 572 | | return stripped; |
| | | 573 | | } |
| | | 574 | | |
| | 0 | 575 | | return backingString ?? strToClean.ToString(); |
| | 0 | 576 | | } |
| | | 577 | | |
| | | 578 | | public static bool StripBidiControlCharacters(ReadOnlySpan<char> strToClean, [NotNullWhen(true)] out string? str |
| | 0 | 579 | | { |
| | 0 | 580 | | int charsToRemove = 0; |
| | | 581 | | |
| | 0 | 582 | | int indexOfPossibleCharToRemove = strToClean.IndexOfAnyInRange('\u200E', '\u202E'); |
| | 0 | 583 | | if (indexOfPossibleCharToRemove >= 0) |
| | 0 | 584 | | { |
| | | 585 | | // Slow path: Contains chars that fall in the [u200E, u202E] range (so likely Bidi) |
| | 0 | 586 | | foreach (char c in strToClean.Slice(indexOfPossibleCharToRemove)) |
| | 0 | 587 | | { |
| | 0 | 588 | | if (IsBidiControlCharacter(c)) |
| | 0 | 589 | | { |
| | 0 | 590 | | charsToRemove++; |
| | 0 | 591 | | } |
| | 0 | 592 | | } |
| | 0 | 593 | | } |
| | | 594 | | |
| | 0 | 595 | | if (charsToRemove == 0) |
| | 0 | 596 | | { |
| | | 597 | | // Hot path |
| | 0 | 598 | | stripped = null; |
| | 0 | 599 | | return false; |
| | | 600 | | } |
| | | 601 | | |
| | 0 | 602 | | stripped = string.Create(strToClean.Length - charsToRemove, strToClean, static (buffer, strToClean) => |
| | 0 | 603 | | { |
| | 0 | 604 | | int destIndex = 0; |
| | 0 | 605 | | foreach (char c in strToClean) |
| | 0 | 606 | | { |
| | 0 | 607 | | if (!IsBidiControlCharacter(c)) |
| | 0 | 608 | | { |
| | 0 | 609 | | buffer[destIndex++] = c; |
| | 0 | 610 | | } |
| | 0 | 611 | | } |
| | 0 | 612 | | Debug.Assert(buffer.Length == destIndex); |
| | 0 | 613 | | }); |
| | 0 | 614 | | return true; |
| | 0 | 615 | | } |
| | | 616 | | |
| | | 617 | | // This will compress any "\" "/../" "/./" "///" "/..../" /XXX.../, etc found in the input |
| | | 618 | | // |
| | | 619 | | // The passed options control whether to use aggressive compression or the one specified in RFC 2396 |
| | | 620 | | public static int Compress(Span<char> span, bool convertPathSlashes, bool canonicalizeAsFilePath) |
| | 0 | 621 | | { |
| | 0 | 622 | | if (span.IsEmpty) |
| | 0 | 623 | | { |
| | 0 | 624 | | return 0; |
| | | 625 | | } |
| | | 626 | | |
| | 0 | 627 | | if (convertPathSlashes) |
| | 0 | 628 | | { |
| | 0 | 629 | | span.Replace('\\', '/'); |
| | 0 | 630 | | } |
| | | 631 | | |
| | 0 | 632 | | ValueListBuilder<(int Start, int Length)> removedSegments = default; |
| | | 633 | | |
| | 0 | 634 | | int slashCount = 0; |
| | 0 | 635 | | int lastSlash = 0; |
| | 0 | 636 | | int dotCount = 0; |
| | 0 | 637 | | int removeSegments = 0; |
| | | 638 | | |
| | 0 | 639 | | for (int i = span.Length - 1; i >= 0; i--) |
| | 0 | 640 | | { |
| | 0 | 641 | | char ch = span[i]; |
| | | 642 | | |
| | | 643 | | // compress multiple '/' for file URI |
| | 0 | 644 | | if (ch == '/') |
| | 0 | 645 | | { |
| | 0 | 646 | | ++slashCount; |
| | 0 | 647 | | } |
| | | 648 | | else |
| | 0 | 649 | | { |
| | 0 | 650 | | if (slashCount > 1) |
| | 0 | 651 | | { |
| | | 652 | | // else preserve repeated slashes |
| | 0 | 653 | | lastSlash = i + 1; |
| | 0 | 654 | | } |
| | 0 | 655 | | slashCount = 0; |
| | 0 | 656 | | } |
| | | 657 | | |
| | 0 | 658 | | if (ch == '.') |
| | 0 | 659 | | { |
| | 0 | 660 | | ++dotCount; |
| | 0 | 661 | | continue; |
| | | 662 | | } |
| | 0 | 663 | | else if (dotCount != 0) |
| | 0 | 664 | | { |
| | 0 | 665 | | bool skipSegment = canonicalizeAsFilePath && (dotCount > 2 || ch != '/'); |
| | | 666 | | |
| | | 667 | | // Cases: |
| | | 668 | | // /./ = remove this segment |
| | | 669 | | // /../ = remove this segment, mark next for removal |
| | | 670 | | // /....x = DO NOT TOUCH, leave as is |
| | | 671 | | // x.../ = DO NOT TOUCH, leave as is, except for V2 legacy mode |
| | 0 | 672 | | if (!skipSegment && ch == '/') |
| | 0 | 673 | | { |
| | 0 | 674 | | if ((lastSlash == i + dotCount + 1 // "/..../" |
| | 0 | 675 | | || (lastSlash == 0 && i + dotCount + 1 == span.Length)) // "/..." |
| | 0 | 676 | | && (dotCount <= 2)) |
| | 0 | 677 | | { |
| | | 678 | | // /./ or /.<eos> or /../ or /..<eos> |
| | 0 | 679 | | removedSegments.Append((i + 1, dotCount + (lastSlash == 0 ? 0 : 1))); |
| | | 680 | | |
| | 0 | 681 | | lastSlash = i; |
| | 0 | 682 | | if (dotCount == 2) |
| | 0 | 683 | | { |
| | | 684 | | // We have 2 dots in between like /../ or /..<eos>, |
| | | 685 | | // Mark next segment for removal and remove this /../ or /.. |
| | 0 | 686 | | ++removeSegments; |
| | 0 | 687 | | } |
| | 0 | 688 | | dotCount = 0; |
| | 0 | 689 | | continue; |
| | | 690 | | } |
| | 0 | 691 | | } |
| | | 692 | | // .NET 4.5 no longer removes trailing dots in a path segment x.../ or x...<eos> |
| | 0 | 693 | | dotCount = 0; |
| | | 694 | | |
| | | 695 | | // Here all other cases go such as |
| | | 696 | | // x.[..]y or /.[..]x or (/x.[...][/] && removeSegments !=0) |
| | 0 | 697 | | } |
| | | 698 | | |
| | | 699 | | // Now we may want to remove a segment because of previous /../ |
| | 0 | 700 | | if (ch == '/') |
| | 0 | 701 | | { |
| | 0 | 702 | | if (removeSegments != 0) |
| | 0 | 703 | | { |
| | 0 | 704 | | removeSegments--; |
| | 0 | 705 | | removedSegments.Append((i + 1, lastSlash - i)); |
| | 0 | 706 | | } |
| | | 707 | | |
| | 0 | 708 | | lastSlash = i; |
| | 0 | 709 | | } |
| | 0 | 710 | | } |
| | | 711 | | |
| | 0 | 712 | | if (canonicalizeAsFilePath) |
| | 0 | 713 | | { |
| | 0 | 714 | | if (slashCount <= 1) |
| | 0 | 715 | | { |
| | 0 | 716 | | if (removeSegments != 0 && span[0] != '/') |
| | 0 | 717 | | { |
| | | 718 | | // remove first not rooted segment |
| | 0 | 719 | | removedSegments.Append((0, lastSlash + 1)); |
| | 0 | 720 | | } |
| | 0 | 721 | | else if (dotCount != 0) |
| | 0 | 722 | | { |
| | | 723 | | // If final string starts with a segment looking like .[...]/ or .[...]<eos> |
| | | 724 | | // then we remove this first segment |
| | 0 | 725 | | if (lastSlash == dotCount || (lastSlash == 0 && dotCount == span.Length)) |
| | 0 | 726 | | { |
| | 0 | 727 | | removedSegments.Append((0, dotCount + (lastSlash == 0 ? 0 : 1))); |
| | 0 | 728 | | } |
| | 0 | 729 | | } |
| | 0 | 730 | | } |
| | 0 | 731 | | } |
| | | 732 | | |
| | 0 | 733 | | if (removedSegments.Length == 0) |
| | 0 | 734 | | { |
| | 0 | 735 | | return span.Length; |
| | | 736 | | } |
| | | 737 | | |
| | | 738 | | // Merge any remaining segments. |
| | | 739 | | // Write and read offsets are only ever the same for the first segment. |
| | | 740 | | // Copying the first section would no-op anyway, so we start with the first removed segment. |
| | 0 | 741 | | int writeOffset = removedSegments[^1].Start; |
| | 0 | 742 | | int readOffset = writeOffset; |
| | | 743 | | |
| | 0 | 744 | | for (int i = removedSegments.Length - 1; i >= 0; i--) |
| | 0 | 745 | | { |
| | 0 | 746 | | (int start, int length) = removedSegments[i]; |
| | | 747 | | |
| | 0 | 748 | | Debug.Assert(start >= readOffset && length > 0 && start + length <= span.Length); |
| | | 749 | | |
| | 0 | 750 | | if (readOffset != start) |
| | 0 | 751 | | { |
| | 0 | 752 | | Debug.Assert(readOffset > writeOffset); |
| | | 753 | | |
| | 0 | 754 | | int segmentLength = start - readOffset; |
| | 0 | 755 | | span.Slice(readOffset, segmentLength).CopyTo(span.Slice(writeOffset)); |
| | 0 | 756 | | writeOffset += segmentLength; |
| | 0 | 757 | | } |
| | | 758 | | |
| | 0 | 759 | | readOffset = start + length; |
| | 0 | 760 | | } |
| | | 761 | | |
| | 0 | 762 | | if (readOffset != span.Length) |
| | 0 | 763 | | { |
| | 0 | 764 | | Debug.Assert(readOffset > writeOffset); |
| | | 765 | | |
| | 0 | 766 | | span.Slice(readOffset).CopyTo(span.Slice(writeOffset)); |
| | 0 | 767 | | writeOffset += span.Length - readOffset; |
| | 0 | 768 | | } |
| | | 769 | | |
| | 0 | 770 | | removedSegments.Dispose(); |
| | 0 | 771 | | return writeOffset; |
| | 0 | 772 | | } |
| | | 773 | | } |
| | | 774 | | } |
| | | 775 | | |