< Summary

Line coverage
0%
Covered lines: 0
Uncovered lines: 167
Coverable lines: 167
Total lines: 312
Line coverage: 0%
Branch coverage
0%
Covered branches: 0
Total branches: 76
Branch coverage: 0%
Method coverage

Feature is only available for sponsors

Upgrade to PRO version

Metrics

MethodBranch coverage Cyclomatic complexity NPath complexity Sequence coverage
.cctor()100%110%
ParseCanonicalName(...)0%16160%
IsValid(...)0%42420%
IdnEquivalent(...)0%440%
TryGetUnicodeEquivalent(...)0%12120%
AppendIdnUnicode(...)0%220%

File(s)

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.Uri/src/System/DomainNameHelper.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Buffers;
 5using System.Diagnostics;
 6using System.Globalization;
 7using System.Runtime.CompilerServices;
 8using System.Text;
 9
 10namespace System
 11{
 12    // The class designed as to keep working set of Uri class as minimal.
 13    // The idea is to stay with static helper methods and strings
 14    internal static class DomainNameHelper
 15    {
 16        // Regular ascii dot '.'
 17        // IDEOGRAPHIC FULL STOP '\u3002'
 18        // FULLWIDTH FULL STOP '\uFF0E'
 19        // HALFWIDTH IDEOGRAPHIC FULL STOP '\uFF61'
 20        // Using SearchValues isn't beneficial here as it would defer to IndexOfAny(char, char, char, char) anyway
 21        private const string IriDotCharacters = ".\u3002\uFF0E\uFF61";
 22
 23        // The Unicode specification allows certain code points to be normalized not to
 24        // punycode, but to ASCII representations that retain the same meaning. For example,
 25        // the codepoint U+00BC "Vulgar Fraction One Quarter" is normalized to '1/4' rather
 26        // than being punycoded.
 27        //
 28        // This means that a host containing Unicode characters can be normalized to contain
 29        // URI reserved characters, changing the meaning of a URI only when certain properties
 30        // such as IdnHost are accessed. To be safe, disallow control characters in normalized hosts.
 031        private static readonly SearchValues<char> s_unsafeForNormalizedHostChars =
 032            SearchValues.Create(@"\/?@#:[]");
 33
 34        // Takes into account the additional legal domain name characters '-' and '_'
 35        // Note that '_' char is formally invalid but is historically in use, especially on corpnets
 036        private static readonly SearchValues<char> s_validChars =
 037            SearchValues.Create("-0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZ_abcdefghijklmnopqrstuvwxyz.");
 38
 39        // For IRI, we're accepting anything non-ascii (except 0x80-0x9F), so invert the condition to search for invalid
 040        private static readonly SearchValues<char> s_iriInvalidChars = SearchValues.Create(
 041            "\u0000\u0001\u0002\u0003\u0004\u0005\u0006\u0007\u0008\u0009\u000A\u000B\u000C\u000D\u000E\u000F" +
 042            "\u0010\u0011\u0012\u0013\u0014\u0015\u0016\u0017\u0018\u0019\u001A\u001B\u001C\u001D\u001E\u001F" +
 043            " !\"#$%&'()*+,/:;<=>?@[\\]^`{|}~\u007F" +
 044            "\u0080\u0081\u0082\u0083\u0084\u0085\u0086\u0087\u0088\u0089\u008A\u008B\u008C\u008D\u008E\u008F" +
 045            "\u0090\u0091\u0092\u0093\u0094\u0095\u0096\u0097\u0098\u0099\u009A\u009B\u009C\u009D\u009E\u009F");
 46
 047        private static readonly SearchValues<char> s_asciiLetterUpperOrColonChars =
 048            SearchValues.Create("ABCDEFGHIJKLMNOPQRSTUVWXYZ:");
 49
 050        private static readonly IdnMapping s_idnMapping = new IdnMapping();
 51
 52        private const string Localhost = "localhost";
 53        private const string Loopback = "loopback";
 54
 55        internal static string ParseCanonicalName(string str, int start, int end, ref bool loopback)
 056        {
 57            // Do a quick search for the colon or uppercase letters
 058            int index = str.AsSpan(start, end - start).LastIndexOfAny(s_asciiLetterUpperOrColonChars);
 059            if (index >= 0)
 060            {
 061                Debug.Assert(!str.AsSpan(start, index).Contains(':'),
 062                    "A colon should appear at most once, and must never be followed by letters.");
 63
 064                if (str[start + index] == ':')
 065                {
 66                    // Shrink the slice to only include chars before the colon
 067                    end = start + index;
 68
 69                    // Look for uppercase letters again.
 70                    // The index value doesn't matter anymore (nor does the search direction), just whether we've found 
 071                    index = str.AsSpan(start, index).IndexOfAnyInRange('A', 'Z');
 072                }
 073            }
 74
 075            Debug.Assert(index == -1 || char.IsAsciiLetterUpper(str[start + index]));
 76
 077            ReadOnlySpan<char> span = str.AsSpan(start, end - start);
 078            if (index >= 0)
 079            {
 080                if (span.Equals(Localhost, StringComparison.OrdinalIgnoreCase) ||
 081                    span.Equals(Loopback, StringComparison.OrdinalIgnoreCase))
 082                {
 083                    loopback = true;
 084                    return Localhost;
 85                }
 86
 87                // We saw uppercase letters. Avoid allocating both the substring and the lower-cased variant.
 088                return UriHelper.SpanToLowerInvariantString(span);
 89            }
 90
 091            if (span is Localhost or Loopback)
 092            {
 093                loopback = true;
 094                return Localhost;
 95            }
 96
 097            return str.Substring(start, end - start);
 098        }
 99
 100        public static bool IsValid(ReadOnlySpan<char> hostname, bool iri, bool notImplicitFile, out int length)
 0101        {
 0102            int invalidCharOrDelimiterIndex = iri
 0103                ? hostname.IndexOfAny(s_iriInvalidChars)
 0104                : hostname.IndexOfAnyExcept(s_validChars);
 105
 0106            if (invalidCharOrDelimiterIndex >= 0)
 0107            {
 0108                char c = hostname[invalidCharOrDelimiterIndex];
 109
 0110                if (c is '/' or '\\' || (notImplicitFile && (c is ':' or '?' or '#')))
 0111                {
 0112                    hostname = hostname.Slice(0, invalidCharOrDelimiterIndex);
 0113                }
 114                else
 0115                {
 0116                    length = 0;
 0117                    return false;
 118                }
 0119            }
 120
 0121            length = hostname.Length;
 122
 0123            if (length == 0)
 0124            {
 0125                return false;
 126            }
 127
 128            //  Determines whether a string is a valid domain name label. In keeping
 129            //  with RFC 1123, section 2.1, the requirement that the first character
 130            //  of a label be alphabetic is dropped. Therefore, Domain names are
 131            //  formed as:
 132            //
 133            //      <label> -> <alphanum> [<alphanum> | <hyphen> | <underscore>] * 62
 134
 135            // We already verified the content, now verify the lengths of individual labels
 0136            while (true)
 0137            {
 0138                char firstChar = hostname[0];
 0139                if ((!iri || firstChar < 0xA0) && !char.IsAsciiLetterOrDigit(firstChar))
 0140                {
 0141                    return false;
 142                }
 143
 0144                int dotIndex = iri
 0145                    ? hostname.IndexOfAny(IriDotCharacters)
 0146                    : hostname.IndexOf('.');
 147
 0148                int labelLength = dotIndex < 0 ? hostname.Length : dotIndex;
 149
 0150                if (iri)
 0151                {
 0152                    ReadOnlySpan<char> label = hostname.Slice(0, labelLength);
 0153                    if (!Ascii.IsValid(label))
 0154                    {
 155                        // Account for the ACE prefix ("xn--")
 0156                        labelLength += 4;
 157
 0158                        foreach (char c in label)
 0159                        {
 0160                            if (c > 0xFF)
 0161                            {
 162                                // counts for two octets
 0163                                labelLength++;
 0164                            }
 0165                        }
 0166                    }
 0167                }
 168
 0169                if (!IriHelper.IsInInclusiveRange((uint)labelLength, 1, 63))
 0170                {
 0171                    return false;
 172                }
 173
 0174                if (dotIndex < 0)
 0175                {
 176                    // We validated the last label
 0177                    return true;
 178                }
 179
 0180                hostname = hostname.Slice(dotIndex + 1);
 181
 0182                if (hostname.IsEmpty)
 0183                {
 184                    // Hostname ended with a dot
 0185                    return true;
 186                }
 0187            }
 0188        }
 189
 190        /// <summary>Converts a host name into its idn equivalent.</summary>
 191        public static string IdnEquivalent(string hostname)
 0192        {
 193            // check if only ascii chars
 194            // special case since idnmapping will not lowercase if only ascii present
 0195            if (Ascii.IsValid(hostname))
 0196            {
 197                // just lowercase for ascii
 0198                return hostname.ToLowerInvariant();
 199            }
 200
 0201            string bidiStrippedHost = UriHelper.StripBidiControlCharacters(hostname, hostname);
 202
 203            try
 0204            {
 0205                string asciiForm = s_idnMapping.GetAscii(bidiStrippedHost);
 0206                if (asciiForm.AsSpan().ContainsAny(s_unsafeForNormalizedHostChars))
 0207                {
 0208                    throw new UriFormatException(SR.net_uri_BadUnicodeHostForIdn);
 209                }
 0210                return asciiForm;
 211            }
 0212            catch (ArgumentException)
 0213            {
 0214                throw new UriFormatException(SR.net_uri_BadUnicodeHostForIdn);
 215            }
 0216        }
 217
 218        public static unsafe bool TryGetUnicodeEquivalent(string hostname, ref ValueStringBuilder dest)
 0219        {
 0220            Debug.Assert(ReferenceEquals(hostname, UriHelper.StripBidiControlCharacters(hostname, hostname)));
 221
 222            // We run a loop where for every label
 223            // a) if label is ascii and no ace then we lowercase it
 224            // b) if label is ascii and ace and not valid idn then just lowercase it
 225            // c) if label is ascii and ace and is valid idn then get its unicode eqvl
 226            // d) if label is unicode then clean it by running it through idnmapping
 227
 228            // Buffer for intermediate ASCII form when processing non-ASCII labels
 229            // Max label length is 63 chars, but punycode can expand - 256 is a safe upper bound
 0230            Span<char> asciiBuffer = stackalloc char[256];
 231
 0232            for (int i = 0; i < hostname.Length; i++)
 0233            {
 0234                if (i != 0)
 0235                {
 0236                    dest.Append('.');
 0237                }
 238
 0239                ReadOnlySpan<char> label = hostname.AsSpan(i);
 240
 0241                int dotIndex = label.IndexOfAny(IriDotCharacters);
 0242                if (dotIndex >= 0)
 0243                {
 0244                    label = label.Slice(0, dotIndex);
 0245                }
 246
 0247                if (!Ascii.IsValid(label))
 0248                {
 249                    // For non-ASCII labels, first convert to ASCII (punycode), then to normalized Unicode
 250                    // Use span-based APIs to avoid intermediate string allocations
 251                    try
 0252                    {
 0253                        bool asciiSuccess = s_idnMapping.TryGetAscii(label, asciiBuffer, out int asciiWritten);
 0254                        Debug.Assert(asciiSuccess, "TryGetAscii should always succeed with a 255-char buffer for valid I
 255
 256                        // Now convert the ASCII form to Unicode and append directly to dest
 0257                        AppendIdnUnicode(asciiBuffer.Slice(0, asciiWritten), ref dest);
 0258                    }
 0259                    catch (ArgumentException)
 0260                    {
 0261                        return false;
 262                    }
 0263                }
 264                else
 0265                {
 0266                    bool aceValid = false;
 267
 0268                    if (label.StartsWith("xn--", StringComparison.Ordinal))
 0269                    {
 270                        // Check ace validity - use span-based API to avoid string allocation
 271                        try
 0272                        {
 0273                            AppendIdnUnicode(label, ref dest);
 0274                            aceValid = true;
 0275                        }
 0276                        catch (ArgumentException)
 0277                        {
 278                            // not valid ace so treat it as a normal ascii label
 0279                        }
 0280                    }
 281
 0282                    if (!aceValid)
 0283                    {
 284                        // for invalid aces we just lowercase the label
 0285                        int charsWritten = label.ToLowerInvariant(dest.AppendSpan(label.Length));
 0286                        Debug.Assert(charsWritten == label.Length);
 0287                    }
 0288                }
 289
 0290                i += label.Length;
 0291            }
 292
 0293            return true;
 0294        }
 295
 296        /// <summary>
 297        /// Converts ASCII (punycode) to Unicode and appends directly to the ValueStringBuilder.
 298        /// </summary>
 299        private static void AppendIdnUnicode(scoped ReadOnlySpan<char> ascii, ref ValueStringBuilder dest)
 0300        {
 301            int charsWritten;
 302
 0303            while (!s_idnMapping.TryGetUnicode(ascii, dest.RawChars.Slice(dest.Length), out charsWritten))
 0304            {
 0305                dest.EnsureCapacity(dest.Capacity + 1);
 0306            }
 307
 0308            dest.Length += charsWritten;
 0309        }
 310    }
 311}
 312