| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | using System.Buffers; |
| | | 5 | | using System.Diagnostics; |
| | | 6 | | using System.Globalization; |
| | | 7 | | using System.Runtime.CompilerServices; |
| | | 8 | | using System.Text; |
| | | 9 | | |
| | | 10 | | namespace System |
| | | 11 | | { |
| | | 12 | | // The class designed as to keep working set of Uri class as minimal. |
| | | 13 | | // The idea is to stay with static helper methods and strings |
| | | 14 | | internal static class DomainNameHelper |
| | | 15 | | { |
| | | 16 | | // Regular ascii dot '.' |
| | | 17 | | // IDEOGRAPHIC FULL STOP '\u3002' |
| | | 18 | | // FULLWIDTH FULL STOP '\uFF0E' |
| | | 19 | | // HALFWIDTH IDEOGRAPHIC FULL STOP '\uFF61' |
| | | 20 | | // Using SearchValues isn't beneficial here as it would defer to IndexOfAny(char, char, char, char) anyway |
| | | 21 | | private const string IriDotCharacters = ".\u3002\uFF0E\uFF61"; |
| | | 22 | | |
| | | 23 | | // The Unicode specification allows certain code points to be normalized not to |
| | | 24 | | // punycode, but to ASCII representations that retain the same meaning. For example, |
| | | 25 | | // the codepoint U+00BC "Vulgar Fraction One Quarter" is normalized to '1/4' rather |
| | | 26 | | // than being punycoded. |
| | | 27 | | // |
| | | 28 | | // This means that a host containing Unicode characters can be normalized to contain |
| | | 29 | | // URI reserved characters, changing the meaning of a URI only when certain properties |
| | | 30 | | // such as IdnHost are accessed. To be safe, disallow control characters in normalized hosts. |
| | 0 | 31 | | private static readonly SearchValues<char> s_unsafeForNormalizedHostChars = |
| | 0 | 32 | | SearchValues.Create(@"\/?@#:[]"); |
| | | 33 | | |
| | | 34 | | // Takes into account the additional legal domain name characters '-' and '_' |
| | | 35 | | // Note that '_' char is formally invalid but is historically in use, especially on corpnets |
| | 0 | 36 | | private static readonly SearchValues<char> s_validChars = |
| | 0 | 37 | | SearchValues.Create("-0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZ_abcdefghijklmnopqrstuvwxyz."); |
| | | 38 | | |
| | | 39 | | // For IRI, we're accepting anything non-ascii (except 0x80-0x9F), so invert the condition to search for invalid |
| | 0 | 40 | | private static readonly SearchValues<char> s_iriInvalidChars = SearchValues.Create( |
| | 0 | 41 | | "\u0000\u0001\u0002\u0003\u0004\u0005\u0006\u0007\u0008\u0009\u000A\u000B\u000C\u000D\u000E\u000F" + |
| | 0 | 42 | | "\u0010\u0011\u0012\u0013\u0014\u0015\u0016\u0017\u0018\u0019\u001A\u001B\u001C\u001D\u001E\u001F" + |
| | 0 | 43 | | " !\"#$%&'()*+,/:;<=>?@[\\]^`{|}~\u007F" + |
| | 0 | 44 | | "\u0080\u0081\u0082\u0083\u0084\u0085\u0086\u0087\u0088\u0089\u008A\u008B\u008C\u008D\u008E\u008F" + |
| | 0 | 45 | | "\u0090\u0091\u0092\u0093\u0094\u0095\u0096\u0097\u0098\u0099\u009A\u009B\u009C\u009D\u009E\u009F"); |
| | | 46 | | |
| | 0 | 47 | | private static readonly SearchValues<char> s_asciiLetterUpperOrColonChars = |
| | 0 | 48 | | SearchValues.Create("ABCDEFGHIJKLMNOPQRSTUVWXYZ:"); |
| | | 49 | | |
| | 0 | 50 | | private static readonly IdnMapping s_idnMapping = new IdnMapping(); |
| | | 51 | | |
| | | 52 | | private const string Localhost = "localhost"; |
| | | 53 | | private const string Loopback = "loopback"; |
| | | 54 | | |
| | | 55 | | internal static string ParseCanonicalName(string str, int start, int end, ref bool loopback) |
| | 0 | 56 | | { |
| | | 57 | | // Do a quick search for the colon or uppercase letters |
| | 0 | 58 | | int index = str.AsSpan(start, end - start).LastIndexOfAny(s_asciiLetterUpperOrColonChars); |
| | 0 | 59 | | if (index >= 0) |
| | 0 | 60 | | { |
| | 0 | 61 | | Debug.Assert(!str.AsSpan(start, index).Contains(':'), |
| | 0 | 62 | | "A colon should appear at most once, and must never be followed by letters."); |
| | | 63 | | |
| | 0 | 64 | | if (str[start + index] == ':') |
| | 0 | 65 | | { |
| | | 66 | | // Shrink the slice to only include chars before the colon |
| | 0 | 67 | | end = start + index; |
| | | 68 | | |
| | | 69 | | // Look for uppercase letters again. |
| | | 70 | | // The index value doesn't matter anymore (nor does the search direction), just whether we've found |
| | 0 | 71 | | index = str.AsSpan(start, index).IndexOfAnyInRange('A', 'Z'); |
| | 0 | 72 | | } |
| | 0 | 73 | | } |
| | | 74 | | |
| | 0 | 75 | | Debug.Assert(index == -1 || char.IsAsciiLetterUpper(str[start + index])); |
| | | 76 | | |
| | 0 | 77 | | ReadOnlySpan<char> span = str.AsSpan(start, end - start); |
| | 0 | 78 | | if (index >= 0) |
| | 0 | 79 | | { |
| | 0 | 80 | | if (span.Equals(Localhost, StringComparison.OrdinalIgnoreCase) || |
| | 0 | 81 | | span.Equals(Loopback, StringComparison.OrdinalIgnoreCase)) |
| | 0 | 82 | | { |
| | 0 | 83 | | loopback = true; |
| | 0 | 84 | | return Localhost; |
| | | 85 | | } |
| | | 86 | | |
| | | 87 | | // We saw uppercase letters. Avoid allocating both the substring and the lower-cased variant. |
| | 0 | 88 | | return UriHelper.SpanToLowerInvariantString(span); |
| | | 89 | | } |
| | | 90 | | |
| | 0 | 91 | | if (span is Localhost or Loopback) |
| | 0 | 92 | | { |
| | 0 | 93 | | loopback = true; |
| | 0 | 94 | | return Localhost; |
| | | 95 | | } |
| | | 96 | | |
| | 0 | 97 | | return str.Substring(start, end - start); |
| | 0 | 98 | | } |
| | | 99 | | |
| | | 100 | | public static bool IsValid(ReadOnlySpan<char> hostname, bool iri, bool notImplicitFile, out int length) |
| | 0 | 101 | | { |
| | 0 | 102 | | int invalidCharOrDelimiterIndex = iri |
| | 0 | 103 | | ? hostname.IndexOfAny(s_iriInvalidChars) |
| | 0 | 104 | | : hostname.IndexOfAnyExcept(s_validChars); |
| | | 105 | | |
| | 0 | 106 | | if (invalidCharOrDelimiterIndex >= 0) |
| | 0 | 107 | | { |
| | 0 | 108 | | char c = hostname[invalidCharOrDelimiterIndex]; |
| | | 109 | | |
| | 0 | 110 | | if (c is '/' or '\\' || (notImplicitFile && (c is ':' or '?' or '#'))) |
| | 0 | 111 | | { |
| | 0 | 112 | | hostname = hostname.Slice(0, invalidCharOrDelimiterIndex); |
| | 0 | 113 | | } |
| | | 114 | | else |
| | 0 | 115 | | { |
| | 0 | 116 | | length = 0; |
| | 0 | 117 | | return false; |
| | | 118 | | } |
| | 0 | 119 | | } |
| | | 120 | | |
| | 0 | 121 | | length = hostname.Length; |
| | | 122 | | |
| | 0 | 123 | | if (length == 0) |
| | 0 | 124 | | { |
| | 0 | 125 | | return false; |
| | | 126 | | } |
| | | 127 | | |
| | | 128 | | // Determines whether a string is a valid domain name label. In keeping |
| | | 129 | | // with RFC 1123, section 2.1, the requirement that the first character |
| | | 130 | | // of a label be alphabetic is dropped. Therefore, Domain names are |
| | | 131 | | // formed as: |
| | | 132 | | // |
| | | 133 | | // <label> -> <alphanum> [<alphanum> | <hyphen> | <underscore>] * 62 |
| | | 134 | | |
| | | 135 | | // We already verified the content, now verify the lengths of individual labels |
| | 0 | 136 | | while (true) |
| | 0 | 137 | | { |
| | 0 | 138 | | char firstChar = hostname[0]; |
| | 0 | 139 | | if ((!iri || firstChar < 0xA0) && !char.IsAsciiLetterOrDigit(firstChar)) |
| | 0 | 140 | | { |
| | 0 | 141 | | return false; |
| | | 142 | | } |
| | | 143 | | |
| | 0 | 144 | | int dotIndex = iri |
| | 0 | 145 | | ? hostname.IndexOfAny(IriDotCharacters) |
| | 0 | 146 | | : hostname.IndexOf('.'); |
| | | 147 | | |
| | 0 | 148 | | int labelLength = dotIndex < 0 ? hostname.Length : dotIndex; |
| | | 149 | | |
| | 0 | 150 | | if (iri) |
| | 0 | 151 | | { |
| | 0 | 152 | | ReadOnlySpan<char> label = hostname.Slice(0, labelLength); |
| | 0 | 153 | | if (!Ascii.IsValid(label)) |
| | 0 | 154 | | { |
| | | 155 | | // Account for the ACE prefix ("xn--") |
| | 0 | 156 | | labelLength += 4; |
| | | 157 | | |
| | 0 | 158 | | foreach (char c in label) |
| | 0 | 159 | | { |
| | 0 | 160 | | if (c > 0xFF) |
| | 0 | 161 | | { |
| | | 162 | | // counts for two octets |
| | 0 | 163 | | labelLength++; |
| | 0 | 164 | | } |
| | 0 | 165 | | } |
| | 0 | 166 | | } |
| | 0 | 167 | | } |
| | | 168 | | |
| | 0 | 169 | | if (!IriHelper.IsInInclusiveRange((uint)labelLength, 1, 63)) |
| | 0 | 170 | | { |
| | 0 | 171 | | return false; |
| | | 172 | | } |
| | | 173 | | |
| | 0 | 174 | | if (dotIndex < 0) |
| | 0 | 175 | | { |
| | | 176 | | // We validated the last label |
| | 0 | 177 | | return true; |
| | | 178 | | } |
| | | 179 | | |
| | 0 | 180 | | hostname = hostname.Slice(dotIndex + 1); |
| | | 181 | | |
| | 0 | 182 | | if (hostname.IsEmpty) |
| | 0 | 183 | | { |
| | | 184 | | // Hostname ended with a dot |
| | 0 | 185 | | return true; |
| | | 186 | | } |
| | 0 | 187 | | } |
| | 0 | 188 | | } |
| | | 189 | | |
| | | 190 | | /// <summary>Converts a host name into its idn equivalent.</summary> |
| | | 191 | | public static string IdnEquivalent(string hostname) |
| | 0 | 192 | | { |
| | | 193 | | // check if only ascii chars |
| | | 194 | | // special case since idnmapping will not lowercase if only ascii present |
| | 0 | 195 | | if (Ascii.IsValid(hostname)) |
| | 0 | 196 | | { |
| | | 197 | | // just lowercase for ascii |
| | 0 | 198 | | return hostname.ToLowerInvariant(); |
| | | 199 | | } |
| | | 200 | | |
| | 0 | 201 | | string bidiStrippedHost = UriHelper.StripBidiControlCharacters(hostname, hostname); |
| | | 202 | | |
| | | 203 | | try |
| | 0 | 204 | | { |
| | 0 | 205 | | string asciiForm = s_idnMapping.GetAscii(bidiStrippedHost); |
| | 0 | 206 | | if (asciiForm.AsSpan().ContainsAny(s_unsafeForNormalizedHostChars)) |
| | 0 | 207 | | { |
| | 0 | 208 | | throw new UriFormatException(SR.net_uri_BadUnicodeHostForIdn); |
| | | 209 | | } |
| | 0 | 210 | | return asciiForm; |
| | | 211 | | } |
| | 0 | 212 | | catch (ArgumentException) |
| | 0 | 213 | | { |
| | 0 | 214 | | throw new UriFormatException(SR.net_uri_BadUnicodeHostForIdn); |
| | | 215 | | } |
| | 0 | 216 | | } |
| | | 217 | | |
| | | 218 | | public static unsafe bool TryGetUnicodeEquivalent(string hostname, ref ValueStringBuilder dest) |
| | 0 | 219 | | { |
| | 0 | 220 | | Debug.Assert(ReferenceEquals(hostname, UriHelper.StripBidiControlCharacters(hostname, hostname))); |
| | | 221 | | |
| | | 222 | | // We run a loop where for every label |
| | | 223 | | // a) if label is ascii and no ace then we lowercase it |
| | | 224 | | // b) if label is ascii and ace and not valid idn then just lowercase it |
| | | 225 | | // c) if label is ascii and ace and is valid idn then get its unicode eqvl |
| | | 226 | | // d) if label is unicode then clean it by running it through idnmapping |
| | | 227 | | |
| | | 228 | | // Buffer for intermediate ASCII form when processing non-ASCII labels |
| | | 229 | | // Max label length is 63 chars, but punycode can expand - 256 is a safe upper bound |
| | 0 | 230 | | Span<char> asciiBuffer = stackalloc char[256]; |
| | | 231 | | |
| | 0 | 232 | | for (int i = 0; i < hostname.Length; i++) |
| | 0 | 233 | | { |
| | 0 | 234 | | if (i != 0) |
| | 0 | 235 | | { |
| | 0 | 236 | | dest.Append('.'); |
| | 0 | 237 | | } |
| | | 238 | | |
| | 0 | 239 | | ReadOnlySpan<char> label = hostname.AsSpan(i); |
| | | 240 | | |
| | 0 | 241 | | int dotIndex = label.IndexOfAny(IriDotCharacters); |
| | 0 | 242 | | if (dotIndex >= 0) |
| | 0 | 243 | | { |
| | 0 | 244 | | label = label.Slice(0, dotIndex); |
| | 0 | 245 | | } |
| | | 246 | | |
| | 0 | 247 | | if (!Ascii.IsValid(label)) |
| | 0 | 248 | | { |
| | | 249 | | // For non-ASCII labels, first convert to ASCII (punycode), then to normalized Unicode |
| | | 250 | | // Use span-based APIs to avoid intermediate string allocations |
| | | 251 | | try |
| | 0 | 252 | | { |
| | 0 | 253 | | bool asciiSuccess = s_idnMapping.TryGetAscii(label, asciiBuffer, out int asciiWritten); |
| | 0 | 254 | | Debug.Assert(asciiSuccess, "TryGetAscii should always succeed with a 255-char buffer for valid I |
| | | 255 | | |
| | | 256 | | // Now convert the ASCII form to Unicode and append directly to dest |
| | 0 | 257 | | AppendIdnUnicode(asciiBuffer.Slice(0, asciiWritten), ref dest); |
| | 0 | 258 | | } |
| | 0 | 259 | | catch (ArgumentException) |
| | 0 | 260 | | { |
| | 0 | 261 | | return false; |
| | | 262 | | } |
| | 0 | 263 | | } |
| | | 264 | | else |
| | 0 | 265 | | { |
| | 0 | 266 | | bool aceValid = false; |
| | | 267 | | |
| | 0 | 268 | | if (label.StartsWith("xn--", StringComparison.Ordinal)) |
| | 0 | 269 | | { |
| | | 270 | | // Check ace validity - use span-based API to avoid string allocation |
| | | 271 | | try |
| | 0 | 272 | | { |
| | 0 | 273 | | AppendIdnUnicode(label, ref dest); |
| | 0 | 274 | | aceValid = true; |
| | 0 | 275 | | } |
| | 0 | 276 | | catch (ArgumentException) |
| | 0 | 277 | | { |
| | | 278 | | // not valid ace so treat it as a normal ascii label |
| | 0 | 279 | | } |
| | 0 | 280 | | } |
| | | 281 | | |
| | 0 | 282 | | if (!aceValid) |
| | 0 | 283 | | { |
| | | 284 | | // for invalid aces we just lowercase the label |
| | 0 | 285 | | int charsWritten = label.ToLowerInvariant(dest.AppendSpan(label.Length)); |
| | 0 | 286 | | Debug.Assert(charsWritten == label.Length); |
| | 0 | 287 | | } |
| | 0 | 288 | | } |
| | | 289 | | |
| | 0 | 290 | | i += label.Length; |
| | 0 | 291 | | } |
| | | 292 | | |
| | 0 | 293 | | return true; |
| | 0 | 294 | | } |
| | | 295 | | |
| | | 296 | | /// <summary> |
| | | 297 | | /// Converts ASCII (punycode) to Unicode and appends directly to the ValueStringBuilder. |
| | | 298 | | /// </summary> |
| | | 299 | | private static void AppendIdnUnicode(scoped ReadOnlySpan<char> ascii, ref ValueStringBuilder dest) |
| | 0 | 300 | | { |
| | | 301 | | int charsWritten; |
| | | 302 | | |
| | 0 | 303 | | while (!s_idnMapping.TryGetUnicode(ascii, dest.RawChars.Slice(dest.Length), out charsWritten)) |
| | 0 | 304 | | { |
| | 0 | 305 | | dest.EnsureCapacity(dest.Capacity + 1); |
| | 0 | 306 | | } |
| | | 307 | | |
| | 0 | 308 | | dest.Length += charsWritten; |
| | 0 | 309 | | } |
| | | 310 | | } |
| | | 311 | | } |
| | | 312 | | |