< Summary

Line coverage
0%
Covered lines: 3
Uncovered lines: 480
Coverable lines: 483
Total lines: 775
Line coverage: 0.6%
Branch coverage
0%
Covered branches: 1
Total branches: 244
Branch coverage: 0.4%
Method coverage

Feature is only available for sponsors

Upgrade to PRO version

Metrics

File(s)

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.Uri/src/System/UriHelper.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Buffers;
 5using System.Collections.Generic;
 6using System.Diagnostics;
 7using System.Diagnostics.CodeAnalysis;
 8using System.Text;
 9
 10namespace System
 11{
 12    internal static class UriHelper
 13    {
 14        public static string SpanToLowerInvariantString(ReadOnlySpan<char> span)
 015        {
 016            return string.Create(span.Length, span, static (buffer, span) =>
 017            {
 018                int charsWritten = span.ToLowerInvariant(buffer);
 019                Debug.Assert(charsWritten == buffer.Length);
 020            });
 021        }
 22
 23        public static string NormalizeAndConcat(string? start, ReadOnlySpan<char> toNormalize)
 024        {
 025            var vsb = new ValueStringBuilder(stackalloc char[Uri.StackallocThreshold]);
 26
 27            int charsWritten;
 028            while (!toNormalize.TryNormalize(vsb.RawChars, out charsWritten, NormalizationForm.FormC))
 029            {
 030                vsb.EnsureCapacity(vsb.Capacity + 1);
 031            }
 32
 033            string result = string.Concat(start, vsb.RawChars.Slice(0, charsWritten));
 034            vsb.Dispose();
 035            return result;
 036        }
 37
 38        // http://host/Path/Path/File?Query is the base of
 39        //      - http://host/Path/Path/File/ ...    (those "File" words may be different in semantic but anyway)
 40        //      - http://host/Path/Path/#Fragment
 41        //      - http://host/Path/Path/?Query
 42        //      - http://host/Path/Path/MoreDir/ ...
 43        //      - http://host/Path/Path/OtherFile?Query
 44        //      - http://host/Path/Path/Fl
 45        //      - http://host/Path/Path/
 46        //
 47        //  It is not a base for
 48        //      - http://host/Path/Path         (that last "Path" is not considered as a directory)
 49        //      - http://host/Path/Path?Query
 50        //      - http://host/Path/Path#Fragment
 51        //      - http://host/Path/Path2/
 52        //      - http://host/Path/Path2/MoreDir
 53        //      - http://host/Path/File
 54        //
 55        // ASSUMES that strings like http://host/Path/Path/MoreDir/../../  have been canonicalized before going to this 
 56        // ASSUMES that back slashes already have been converted if applicable.
 57        //
 58        internal static bool TestForSubPath(ReadOnlySpan<char> self, ReadOnlySpan<char> other, bool ignoreCase)
 059        {
 060            int i = 0;
 61            char chSelf;
 62            char chOther;
 63
 064            bool AllSameBeforeSlash = true;
 65
 066            for (; i < self.Length && i < other.Length; ++i)
 067            {
 068                chSelf = self[i];
 069                chOther = other[i];
 70
 071                if (chSelf == '?' || chSelf == '#')
 072                {
 73                    // survived so far and selfPtr does not have any more path segments
 074                    return true;
 75                }
 76
 77                // If selfPtr terminates a path segment, so must otherPtr
 078                if (chSelf == '/')
 079                {
 080                    if (chOther != '/')
 081                    {
 82                        // comparison has failed
 083                        return false;
 84                    }
 85                    // plus the segments must be the same
 086                    if (!AllSameBeforeSlash)
 087                    {
 88                        // comparison has failed
 089                        return false;
 90                    }
 91                    //so far so good
 092                    AllSameBeforeSlash = true;
 093                    continue;
 94                }
 95
 96                // if otherPtr terminates then selfPtr must not have any more path segments
 097                if (chOther == '?' || chOther == '#')
 098                {
 099                    break;
 100                }
 101
 0102                if (!ignoreCase)
 0103                {
 0104                    if (chSelf != chOther)
 0105                    {
 0106                        AllSameBeforeSlash = false;
 0107                    }
 0108                }
 109                else
 0110                {
 0111                    if (char.ToLowerInvariant(chSelf) != char.ToLowerInvariant(chOther))
 0112                    {
 0113                        AllSameBeforeSlash = false;
 0114                    }
 0115                }
 0116            }
 117
 118            // If self is longer then it must not have any more path segments
 0119            for (; i < self.Length; ++i)
 0120            {
 0121                if ((chSelf = self[i]) == '?' || chSelf == '#')
 0122                {
 0123                    return true;
 124                }
 0125                if (chSelf == '/')
 0126                {
 0127                    return false;
 128                }
 0129            }
 130            //survived by getting to the end of selfPtr
 0131            return true;
 0132        }
 133
 134        public static bool TryEscapeDataString(ReadOnlySpan<char> charsToEscape, Span<char> destination, out int charsWr
 0135        {
 0136            if (destination.Length < charsToEscape.Length)
 0137            {
 0138                charsWritten = 0;
 0139                return false;
 140            }
 141
 0142            int indexOfFirstToEscape = charsToEscape.IndexOfAnyExcept(Unreserved);
 0143            if (indexOfFirstToEscape < 0)
 0144            {
 145                // Nothing to escape, just copy the original chars.
 0146                charsToEscape.CopyTo(destination);
 0147                charsWritten = charsToEscape.Length;
 0148                return true;
 149            }
 150
 151            // We may throw for very large inputs (when growing the ValueStringBuilder).
 152            scoped ValueStringBuilder vsb;
 153
 154            // If the input and destination buffers overlap, we must take care not to overwrite parts of the input befor
 0155            bool overlapped = charsToEscape.Overlaps(destination);
 156
 0157            if (overlapped)
 0158            {
 0159                vsb = new ValueStringBuilder(stackalloc char[Uri.StackallocThreshold]);
 0160                vsb.EnsureCapacity(charsToEscape.Length);
 0161            }
 162            else
 0163            {
 0164                vsb = new ValueStringBuilder(destination.Slice(indexOfFirstToEscape));
 0165            }
 166
 0167            EscapeStringToBuilder(charsToEscape.Slice(indexOfFirstToEscape), ref vsb, Unreserved, checkExistingEscaped: 
 168
 0169            int newLength = checked(indexOfFirstToEscape + vsb.Length);
 0170            Debug.Assert(newLength > charsToEscape.Length);
 171
 0172            if (destination.Length >= newLength)
 0173            {
 0174                charsToEscape.Slice(0, indexOfFirstToEscape).CopyTo(destination);
 175
 0176                if (overlapped)
 0177                {
 0178                    vsb.AsSpan().CopyTo(destination.Slice(indexOfFirstToEscape));
 0179                    vsb.Dispose();
 0180                }
 181                else
 0182                {
 183                    // We are expecting the builder not to grow if the original span was large enough.
 184                    // This means that we MUST NOT over allocate anywhere in EscapeStringToBuilder (e.g. append and then
 0185                    Debug.Assert(vsb.RawChars.Overlaps(destination));
 0186                }
 187
 0188                charsWritten = newLength;
 0189                return true;
 190            }
 191
 0192            vsb.Dispose();
 0193            charsWritten = 0;
 0194            return false;
 0195        }
 196
 197        public static string EscapeString(string stringToEscape, bool checkExistingEscaped, SearchValues<char> noEscape)
 0198        {
 0199            ArgumentNullException.ThrowIfNull(stringToEscape);
 200
 0201            return EscapeString(stringToEscape, checkExistingEscaped, noEscape, stringToEscape);
 0202        }
 203
 204        public static string EscapeString(ReadOnlySpan<char> charsToEscape, bool checkExistingEscaped, SearchValues<char
 0205        {
 0206            Debug.Assert(!noEscape.Contains('%'), "Need to treat % specially; it should be part of any escaped set");
 0207            Debug.Assert(backingString is null || backingString.Length == charsToEscape.Length);
 208
 0209            int indexOfFirstToEscape = charsToEscape.IndexOfAnyExcept(noEscape);
 0210            if (indexOfFirstToEscape < 0)
 0211            {
 212                // Nothing to escape, just return the original value.
 0213                return backingString ?? charsToEscape.ToString();
 214            }
 215
 216            // Otherwise, create a ValueStringBuilder to store the escaped data into,
 217            // escape the rest, and concat the result with the characters we skipped above.
 0218            var vsb = new ValueStringBuilder(stackalloc char[Uri.StackallocThreshold]);
 219
 220            // We may throw for very large inputs (when growing the ValueStringBuilder).
 0221            vsb.EnsureCapacity(charsToEscape.Length);
 222
 0223            EscapeStringToBuilder(charsToEscape.Slice(indexOfFirstToEscape), ref vsb, noEscape, checkExistingEscaped);
 224
 0225            string result = string.Concat(charsToEscape.Slice(0, indexOfFirstToEscape), vsb.AsSpan());
 0226            vsb.Dispose();
 0227            return result;
 0228        }
 229
 230        internal static void EscapeString(scoped ReadOnlySpan<char> stringToEscape, ref ValueStringBuilder dest,
 231            bool checkExistingEscaped, SearchValues<char> noEscape)
 0232        {
 0233            Debug.Assert(!noEscape.Contains('%'), "Need to treat % specially; it should be part of any escaped set");
 234
 0235            int indexOfFirstToEscape = stringToEscape.IndexOfAnyExcept(noEscape);
 0236            if (indexOfFirstToEscape < 0)
 0237            {
 238                // Nothing to escape, just copy the whole span.
 0239                dest.Append(stringToEscape);
 0240            }
 241            else
 0242            {
 0243                dest.Append(stringToEscape.Slice(0, indexOfFirstToEscape));
 244
 0245                EscapeStringToBuilder(stringToEscape.Slice(indexOfFirstToEscape), ref dest, noEscape, checkExistingEscap
 0246            }
 0247        }
 248
 249        private static void EscapeStringToBuilder(
 250            scoped ReadOnlySpan<char> stringToEscape, ref ValueStringBuilder vsb,
 251            SearchValues<char> noEscape, bool checkExistingEscaped)
 0252        {
 0253            Debug.Assert(!stringToEscape.IsEmpty && !noEscape.Contains(stringToEscape[0]));
 254
 255            // Allocate enough stack space to hold any Rune's UTF8 encoding.
 0256            Span<byte> utf8Bytes = [0, 0, 0, 0];
 257
 0258            while (!stringToEscape.IsEmpty)
 0259            {
 0260                char c = stringToEscape[0];
 261
 0262                if (!char.IsAscii(c))
 0263                {
 0264                    if (Rune.DecodeFromUtf16(stringToEscape, out Rune r, out int charsConsumed) != OperationStatus.Done)
 0265                    {
 0266                        r = Rune.ReplacementChar;
 0267                    }
 268
 0269                    Debug.Assert(stringToEscape.EnumerateRunes() is { } e && e.MoveNext() && e.Current == r);
 0270                    Debug.Assert(charsConsumed is 1 or 2);
 271
 0272                    stringToEscape = stringToEscape.Slice(charsConsumed);
 273
 274                    // The rune is non-ASCII, so encode it as UTF8, and escape each UTF8 byte.
 0275                    r.TryEncodeToUtf8(utf8Bytes, out int bytesWritten);
 0276                    foreach (byte b in utf8Bytes.Slice(0, bytesWritten))
 0277                    {
 0278                        PercentEncodeByte(b, ref vsb);
 0279                    }
 280
 0281                    continue;
 282                }
 283
 0284                if (!noEscape.Contains(c))
 0285                {
 286                    // If we're checking for existing escape sequences, then if this is the beginning of
 287                    // one, check the next two characters in the sequence.
 0288                    if (c == '%' && checkExistingEscaped)
 0289                    {
 290                        // If the next two characters are valid escaped ASCII, then just output them as-is.
 0291                        if (stringToEscape.Length > 2 && char.IsAsciiHexDigit(stringToEscape[1]) && char.IsAsciiHexDigit
 0292                        {
 0293                            vsb.Append('%');
 0294                            vsb.Append(stringToEscape[1]);
 0295                            vsb.Append(stringToEscape[2]);
 0296                            stringToEscape = stringToEscape.Slice(3);
 0297                            continue;
 298                        }
 0299                    }
 300
 0301                    PercentEncodeByte((byte)c, ref vsb);
 0302                    stringToEscape = stringToEscape.Slice(1);
 0303                    continue;
 304                }
 305
 306                // We have a character we don't want to escape. It's likely there are more, do a vectorized search.
 0307                int charsToCopy = stringToEscape.IndexOfAnyExcept(noEscape);
 0308                if (charsToCopy < 0)
 0309                {
 0310                    charsToCopy = stringToEscape.Length;
 0311                }
 0312                Debug.Assert(charsToCopy > 0);
 313
 0314                vsb.Append(stringToEscape.Slice(0, charsToCopy));
 0315                stringToEscape = stringToEscape.Slice(charsToCopy);
 0316            }
 0317        }
 318
 319        internal static void Unescape(scoped ReadOnlySpan<char> chars, ref ValueStringBuilder dest)
 0320        {
 0321            for (int i = 0; (uint)i < (uint)chars.Length;)
 0322            {
 0323                if (chars[i] == '%' && (uint)(i + 2) < (uint)chars.Length)
 0324                {
 0325                    char unescaped = DecodeHexChars(chars[i + 1], chars[i + 2]);
 326
 0327                    if (unescaped == Uri.c_DummyChar)
 0328                    {
 0329                        i++;
 0330                        continue;
 331                    }
 332
 333                    // Copy previous characters that don't require any transformations.
 334                    // Using a loop instead of Append(span) to avoid the call overhead for typically short sections.
 0335                    foreach (char c in chars.Slice(0, i))
 0336                    {
 0337                        dest.Append(c);
 0338                    }
 339
 0340                    if (char.IsAscii(unescaped))
 0341                    {
 0342                        dest.Append(unescaped);
 0343                        i += 3;
 0344                    }
 345                    else
 0346                    {
 0347                        int charactersRead = PercentEncodingHelper.UnescapePercentEncodedUTF8Sequence(
 0348                            chars.Slice(i),
 0349                            ref dest,
 0350                            isQuery: false,
 0351                            iriParsing: false);
 352
 0353                        Debug.Assert(charactersRead > 0);
 0354                        i += charactersRead;
 0355                    }
 356
 0357                    chars = chars.Slice(i);
 0358                    i = 0;
 0359                }
 360                else
 0361                {
 0362                    i++;
 0363                }
 0364            }
 365
 0366            dest.Append(chars);
 0367        }
 368
 369        internal static void UnescapeString(scoped ReadOnlySpan<char> chars, ref ValueStringBuilder dest,
 370            char rsvd1, char rsvd2, char rsvd3, UnescapeMode unescapeMode, UriParser? syntax, bool isQuery)
 0371        {
 0372            Debug.Assert(unescapeMode != UnescapeMode.None);
 373
 0374            bool escapeReserved = false;
 0375            bool iriParsing = Uri.IriParsingStatic(syntax)
 0376                                && ((unescapeMode & UnescapeMode.EscapeUnescape) == UnescapeMode.EscapeUnescape);
 377
 0378            while (!chars.IsEmpty)
 0379            {
 380                int i;
 0381                char ch = (char)0;
 382
 0383                for (i = 0; (uint)i < (uint)chars.Length; i++)
 0384                {
 0385                    ch = chars[i];
 386
 0387                    if (ch == '%')
 0388                    {
 0389                        if ((unescapeMode & UnescapeMode.Unescape) == 0)
 0390                        {
 391                            // re-escape, don't check anything else
 0392                            escapeReserved = true;
 0393                        }
 0394                        else if ((uint)(i + 2) < (uint)chars.Length)
 0395                        {
 0396                            ch = DecodeHexChars(chars[i + 1], chars[i + 2]);
 397
 398                            // re-escape % from an invalid sequence
 0399                            if (ch == Uri.c_DummyChar)
 0400                            {
 0401                                if ((unescapeMode & UnescapeMode.Escape) != 0)
 0402                                    escapeReserved = true;
 403                                else
 0404                                    continue;   // we should throw instead but since v1.0 would just print '%'
 0405                            }
 406                            // Do not unescape '%' itself unless full unescape is requested
 0407                            else if (ch == '%')
 0408                            {
 0409                                i += 2;
 0410                                continue;
 411                            }
 412                            // Do not unescape a reserved char unless full unescape is requested
 0413                            else if (ch == rsvd1 || ch == rsvd2 || ch == rsvd3)
 0414                            {
 0415                                i += 2;
 0416                                continue;
 417                            }
 418                            // Do not unescape a dangerous char unless it's V1ToStringFlags mode
 0419                            else if ((unescapeMode & UnescapeMode.V1ToStringFlag) == 0 && IsNotSafeForUnescape(ch))
 0420                            {
 0421                                i += 2;
 0422                                continue;
 423                            }
 0424                            else if (iriParsing && (ch <= '\x9F' ? IsNotSafeForUnescape(ch) : !IriHelper.CheckIriUnicode
 0425                            {
 426                                // check if unenscaping gives a char outside iri range
 427                                // if it does then keep it escaped
 0428                                i += 2;
 0429                                continue;
 430                            }
 431                            // unescape escaped char or escape %
 0432                            break;
 433                        }
 434                        else
 0435                        {
 0436                            escapeReserved = true;
 0437                        }
 438                        // escape (escapeReserved==true) or otherwise unescape the sequence
 0439                        break;
 440                    }
 0441                    else if ((unescapeMode & UnescapeMode.Escape) != 0)
 0442                    {
 443                        // Could actually escape some of the characters
 0444                        if (ch == rsvd1 || ch == rsvd2 || ch == rsvd3)
 0445                        {
 446                            // found an unescaped reserved character -> escape it
 0447                            escapeReserved = true;
 0448                            break;
 449                        }
 0450                        else if ((unescapeMode & UnescapeMode.V1ToStringFlag) == 0
 0451                            && (ch <= '\x1F' || (ch >= '\x7F' && ch <= '\x9F')))
 0452                        {
 453                            // found an unescaped reserved character -> escape it
 0454                            escapeReserved = true;
 0455                            break;
 456                        }
 0457                    }
 0458                }
 459
 460                // Copy previous characters that don't require any transformations.
 461                // Using a loop instead of Append(span) to avoid the call overhead for typically short sections.
 0462                foreach (char c in chars.Slice(0, i))
 0463                {
 0464                    dest.Append(c);
 0465                }
 466
 0467                if (i < chars.Length)
 0468                {
 0469                    if (escapeReserved)
 0470                    {
 0471                        PercentEncodeByte((byte)chars[i], ref dest);
 0472                        escapeReserved = false;
 0473                        i++;
 0474                    }
 0475                    else if (ch <= 127)
 0476                    {
 0477                        dest.Append(ch);
 0478                        i += 3;
 0479                    }
 480                    else
 0481                    {
 482                        // Unicode
 0483                        int charactersRead = PercentEncodingHelper.UnescapePercentEncodedUTF8Sequence(
 0484                            chars.Slice(i),
 0485                            ref dest,
 0486                            isQuery,
 0487                            iriParsing);
 488
 0489                        Debug.Assert(charactersRead > 0);
 0490                        i += charactersRead;
 0491                    }
 0492                }
 493
 0494                chars = chars.Slice(i);
 0495            }
 0496        }
 497
 498        internal static void PercentEncodeByte(byte b, ref ValueStringBuilder to)
 0499        {
 0500            to.Append('%');
 0501            HexConverter.ToCharsBuffer(b, to.AppendSpan(2), 0, HexConverter.Casing.Upper);
 0502        }
 503
 504        /// <summary>
 505        /// Converts 2 hex chars to a byte (returned in a char), e.g, "0a" becomes (char)0x0A.
 506        /// <para>If either char is not hex, returns <see cref="Uri.c_DummyChar"/>.</para>
 507        /// </summary>
 508        internal static char DecodeHexChars(int first, int second)
 0509        {
 0510            int a = HexConverter.FromChar(first);
 0511            int b = HexConverter.FromChar(second);
 512
 0513            if ((a | b) == 0xFF)
 0514            {
 515                // either a or b is 0xFF (invalid)
 0516                return Uri.c_DummyChar;
 517            }
 518
 0519            return (char)((a << 4) | b);
 0520        }
 521
 522        // When unescaping in safe mode, do not unescape the RFC 3986 reserved set:
 523        // reserved    = gen-delims / sub-delims
 524        // gen-delims  = ":" / "/" / "?" / "#" / "[" / "]" / "@"
 525        // sub-delims  = "!" / "$" / "&" / "'" / "(" / ")"
 526        //             / "*" / "+" / "," / ";" / "="
 527        //
 528        // In addition, do not unescape the following unsafe characters:
 529        // excluded    = "%" / "\"
 530        internal static bool IsNotSafeForUnescape(char ch) =>
 0531            s_notSafeForUnescapeChars.Contains(ch);
 532
 0533        private static readonly SearchValues<char> s_notSafeForUnescapeChars = SearchValues.Create(
 0534            "\u0000\u0001\u0002\u0003\u0004\u0005\u0006\u0007\u0008\u0009\u000A\u000B\u000C\u000D\u000E\u000F" +
 0535            "\u0010\u0011\u0012\u0013\u0014\u0015\u0016\u0017\u0018\u0019\u001A\u001B\u001C\u001D\u001E\u001F" +
 0536            ";/?:@&=+$,#[]!'()*" + "%\\" + "\u007F" +
 0537            "\u0080\u0081\u0082\u0083\u0084\u0085\u0086\u0087\u0088\u0089\u008A\u008B\u008C\u008D\u008E\u008F" +
 0538            "\u0090\u0091\u0092\u0093\u0094\u0095\u0096\u0097\u0098\u0099\u009A\u009B\u009C\u009D\u009E\u009F");
 539
 540        /// <summary>All ASCII letters and digits, as well as the RFC3986 unreserved marks '-', '_', '.', and '~'.</summ
 0541        public static readonly SearchValues<char> Unreserved =
 0542            SearchValues.Create("-.0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZ_abcdefghijklmnopqrstuvwxyz~");
 543
 544        /// <summary>All ASCII letters and digits, as well as the RFC3986 reserved and unreserved marks.</summary>
 0545        public static readonly SearchValues<char> UnreservedReserved =
 0546            SearchValues.Create("!#$&'()*+,-./0123456789:;=?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[]_abcdefghijklmnopqrstuvwxyz~");
 547
 0548        public static readonly SearchValues<char> UnreservedReservedExceptHash =
 0549            SearchValues.Create("!$&'()*+,-./0123456789:;=?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[]_abcdefghijklmnopqrstuvwxyz~");
 550
 0551        public static readonly SearchValues<char> UnreservedReservedExceptQuestionMarkHash =
 0552            SearchValues.Create("!$&'()*+,-./0123456789:;=@ABCDEFGHIJKLMNOPQRSTUVWXYZ[]_abcdefghijklmnopqrstuvwxyz~");
 553
 0554        internal static readonly char[] s_WSchars = new char[] { ' ', '\n', '\r', '\t' };
 555
 556        internal static bool IsLWS(char ch)
 1322557        {
 1322558            return (ch <= ' ') && (ch == ' ' || ch == '\n' || ch == '\r' || ch == '\t');
 1322559        }
 560
 561        // Is this a Bidirectional control char.. These get stripped
 562        internal static bool IsBidiControlCharacter(char ch) =>
 0563            char.IsBetween(ch, '\u200E', '\u202E') && !char.IsBetween(ch, '\u2010', '\u2029');
 564
 565        // Strip Bidirectional control characters from this string
 566        public static string StripBidiControlCharacters(ReadOnlySpan<char> strToClean, string? backingString = null)
 0567        {
 0568            Debug.Assert(backingString is null || strToClean.Length == backingString.Length);
 569
 0570            if (StripBidiControlCharacters(strToClean, out string? stripped))
 0571            {
 0572                return stripped;
 573            }
 574
 0575            return backingString ?? strToClean.ToString();
 0576        }
 577
 578        public static bool StripBidiControlCharacters(ReadOnlySpan<char> strToClean, [NotNullWhen(true)] out string? str
 0579        {
 0580            int charsToRemove = 0;
 581
 0582            int indexOfPossibleCharToRemove = strToClean.IndexOfAnyInRange('\u200E', '\u202E');
 0583            if (indexOfPossibleCharToRemove >= 0)
 0584            {
 585                // Slow path: Contains chars that fall in the [u200E, u202E] range (so likely Bidi)
 0586                foreach (char c in strToClean.Slice(indexOfPossibleCharToRemove))
 0587                {
 0588                    if (IsBidiControlCharacter(c))
 0589                    {
 0590                        charsToRemove++;
 0591                    }
 0592                }
 0593            }
 594
 0595            if (charsToRemove == 0)
 0596            {
 597                // Hot path
 0598                stripped = null;
 0599                return false;
 600            }
 601
 0602            stripped = string.Create(strToClean.Length - charsToRemove, strToClean, static (buffer, strToClean) =>
 0603            {
 0604                int destIndex = 0;
 0605                foreach (char c in strToClean)
 0606                {
 0607                    if (!IsBidiControlCharacter(c))
 0608                    {
 0609                        buffer[destIndex++] = c;
 0610                    }
 0611                }
 0612                Debug.Assert(buffer.Length == destIndex);
 0613            });
 0614            return true;
 0615        }
 616
 617        // This will compress any "\" "/../" "/./" "///" "/..../" /XXX.../, etc found in the input
 618        //
 619        // The passed options control whether to use aggressive compression or the one specified in RFC 2396
 620        public static int Compress(Span<char> span, bool convertPathSlashes, bool canonicalizeAsFilePath)
 0621        {
 0622            if (span.IsEmpty)
 0623            {
 0624                return 0;
 625            }
 626
 0627            if (convertPathSlashes)
 0628            {
 0629                span.Replace('\\', '/');
 0630            }
 631
 0632            ValueListBuilder<(int Start, int Length)> removedSegments = default;
 633
 0634            int slashCount = 0;
 0635            int lastSlash = 0;
 0636            int dotCount = 0;
 0637            int removeSegments = 0;
 638
 0639            for (int i = span.Length - 1; i >= 0; i--)
 0640            {
 0641                char ch = span[i];
 642
 643                // compress multiple '/' for file URI
 0644                if (ch == '/')
 0645                {
 0646                    ++slashCount;
 0647                }
 648                else
 0649                {
 0650                    if (slashCount > 1)
 0651                    {
 652                        // else preserve repeated slashes
 0653                        lastSlash = i + 1;
 0654                    }
 0655                    slashCount = 0;
 0656                }
 657
 0658                if (ch == '.')
 0659                {
 0660                    ++dotCount;
 0661                    continue;
 662                }
 0663                else if (dotCount != 0)
 0664                {
 0665                    bool skipSegment = canonicalizeAsFilePath && (dotCount > 2 || ch != '/');
 666
 667                    // Cases:
 668                    // /./                  = remove this segment
 669                    // /../                 = remove this segment, mark next for removal
 670                    // /....x               = DO NOT TOUCH, leave as is
 671                    // x.../                = DO NOT TOUCH, leave as is, except for V2 legacy mode
 0672                    if (!skipSegment && ch == '/')
 0673                    {
 0674                        if ((lastSlash == i + dotCount + 1 // "/..../"
 0675                                || (lastSlash == 0 && i + dotCount + 1 == span.Length)) // "/..."
 0676                            && (dotCount <= 2))
 0677                        {
 678                            //  /./ or /.<eos> or /../ or /..<eos>
 0679                            removedSegments.Append((i + 1, dotCount + (lastSlash == 0 ? 0 : 1)));
 680
 0681                            lastSlash = i;
 0682                            if (dotCount == 2)
 0683                            {
 684                                // We have 2 dots in between like /../ or /..<eos>,
 685                                // Mark next segment for removal and remove this /../ or /..
 0686                                ++removeSegments;
 0687                            }
 0688                            dotCount = 0;
 0689                            continue;
 690                        }
 0691                    }
 692                    // .NET 4.5 no longer removes trailing dots in a path segment x.../  or  x...<eos>
 0693                    dotCount = 0;
 694
 695                    // Here all other cases go such as
 696                    // x.[..]y or /.[..]x or (/x.[...][/] && removeSegments !=0)
 0697                }
 698
 699                // Now we may want to remove a segment because of previous /../
 0700                if (ch == '/')
 0701                {
 0702                    if (removeSegments != 0)
 0703                    {
 0704                        removeSegments--;
 0705                        removedSegments.Append((i + 1, lastSlash - i));
 0706                    }
 707
 0708                    lastSlash = i;
 0709                }
 0710            }
 711
 0712            if (canonicalizeAsFilePath)
 0713            {
 0714                if (slashCount <= 1)
 0715                {
 0716                    if (removeSegments != 0 && span[0] != '/')
 0717                    {
 718                        // remove first not rooted segment
 0719                        removedSegments.Append((0, lastSlash + 1));
 0720                    }
 0721                    else if (dotCount != 0)
 0722                    {
 723                        // If final string starts with a segment looking like .[...]/ or .[...]<eos>
 724                        // then we remove this first segment
 0725                        if (lastSlash == dotCount || (lastSlash == 0 && dotCount == span.Length))
 0726                        {
 0727                            removedSegments.Append((0, dotCount + (lastSlash == 0 ? 0 : 1)));
 0728                        }
 0729                    }
 0730                }
 0731            }
 732
 0733            if (removedSegments.Length == 0)
 0734            {
 0735                return span.Length;
 736            }
 737
 738            // Merge any remaining segments.
 739            // Write and read offsets are only ever the same for the first segment.
 740            // Copying the first section would no-op anyway, so we start with the first removed segment.
 0741            int writeOffset = removedSegments[^1].Start;
 0742            int readOffset = writeOffset;
 743
 0744            for (int i = removedSegments.Length - 1; i >= 0; i--)
 0745            {
 0746                (int start, int length) = removedSegments[i];
 747
 0748                Debug.Assert(start >= readOffset && length > 0 && start + length <= span.Length);
 749
 0750                if (readOffset != start)
 0751                {
 0752                    Debug.Assert(readOffset > writeOffset);
 753
 0754                    int segmentLength = start - readOffset;
 0755                    span.Slice(readOffset, segmentLength).CopyTo(span.Slice(writeOffset));
 0756                    writeOffset += segmentLength;
 0757                }
 758
 0759                readOffset = start + length;
 0760            }
 761
 0762            if (readOffset != span.Length)
 0763            {
 0764                Debug.Assert(readOffset > writeOffset);
 765
 0766                span.Slice(readOffset).CopyTo(span.Slice(writeOffset));
 0767                writeOffset += span.Length - readOffset;
 0768            }
 769
 0770            removedSegments.Dispose();
 0771            return writeOffset;
 0772        }
 773    }
 774}
 775

Methods/Properties

SpanToLowerInvariantString(System.ReadOnlySpan`1<System.Char>)
NormalizeAndConcat(System.String,System.ReadOnlySpan`1<System.Char>)
TestForSubPath(System.ReadOnlySpan`1<System.Char>,System.ReadOnlySpan`1<System.Char>,System.Boolean)
TryEscapeDataString(System.ReadOnlySpan`1<System.Char>,System.Span`1<System.Char>,System.Int32&)
EscapeString(System.String,System.Boolean,System.Buffers.SearchValues`1<System.Char>)
EscapeString(System.ReadOnlySpan`1<System.Char>,System.Boolean,System.Buffers.SearchValues`1<System.Char>,System.String)
EscapeString(System.ReadOnlySpan`1<System.Char>,System.Text.ValueStringBuilder&,System.Boolean,System.Buffers.SearchValues`1<System.Char>)
EscapeStringToBuilder(System.ReadOnlySpan`1<System.Char>,System.Text.ValueStringBuilder&,System.Buffers.SearchValues`1<System.Char>,System.Boolean)
Unescape(System.ReadOnlySpan`1<System.Char>,System.Text.ValueStringBuilder&)
UnescapeString(System.ReadOnlySpan`1<System.Char>,System.Text.ValueStringBuilder&,System.Char,System.Char,System.Char,System.UnescapeMode,System.UriParser,System.Boolean)
PercentEncodeByte(System.Byte,System.Text.ValueStringBuilder&)
DecodeHexChars(System.Int32,System.Int32)
IsNotSafeForUnescape(System.Char)
.cctor()
IsLWS(System.Char)
IsBidiControlCharacter(System.Char)
StripBidiControlCharacters(System.ReadOnlySpan`1<System.Char>,System.String)
StripBidiControlCharacters(System.ReadOnlySpan`1<System.Char>,System.String&)
Compress(System.Span`1<System.Char>,System.Boolean,System.Boolean)