| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | // This file contains the IDN functions and implementation. |
| | | 5 | | // |
| | | 6 | | // This allows encoding of non-ASCII domain names in a "punycode" form, |
| | | 7 | | // for example: |
| | | 8 | | // |
| | | 9 | | // \u5B89\u5BA4\u5948\u7F8E\u6075-with-SUPER-MONKEYS |
| | | 10 | | // |
| | | 11 | | // is encoded as: |
| | | 12 | | // |
| | | 13 | | // xn---with-SUPER-MONKEYS-pc58ag80a8qai00g7n9n |
| | | 14 | | // |
| | | 15 | | // Additional options are provided to allow unassigned IDN characters and |
| | | 16 | | // to validate according to the Std3ASCII Rules (like DNS names). |
| | | 17 | | // |
| | | 18 | | // There are also rules regarding bidirectionality of text and the length |
| | | 19 | | // of segments. |
| | | 20 | | // |
| | | 21 | | // For additional rules see also: |
| | | 22 | | // RFC 3490 - Internationalizing Domain Names in Applications (IDNA) |
| | | 23 | | // RFC 3491 - Nameprep: A Stringprep Profile for Internationalized Domain Names (IDN) |
| | | 24 | | // RFC 3492 - Punycode: A Bootstring encoding of Unicode for Internationalized Domain Names in Applications (IDNA) |
| | | 25 | | |
| | | 26 | | using System.Diagnostics; |
| | | 27 | | using System.Diagnostics.CodeAnalysis; |
| | | 28 | | using System.Runtime.CompilerServices; |
| | | 29 | | using System.Runtime.InteropServices; |
| | | 30 | | using System.Text; |
| | | 31 | | |
| | | 32 | | namespace System.Globalization |
| | | 33 | | { |
| | | 34 | | // IdnMapping class used to map names to Punycode |
| | | 35 | | public sealed partial class IdnMapping |
| | | 36 | | { |
| | | 37 | | private bool _allowUnassigned; |
| | | 38 | | private bool _useStd3AsciiRules; |
| | | 39 | | |
| | 0 | 40 | | public IdnMapping() |
| | | 41 | | { |
| | 0 | 42 | | } |
| | | 43 | | |
| | | 44 | | public bool AllowUnassigned |
| | | 45 | | { |
| | 0 | 46 | | get => _allowUnassigned; |
| | 0 | 47 | | set => _allowUnassigned = value; |
| | | 48 | | } |
| | | 49 | | |
| | | 50 | | public bool UseStd3AsciiRules |
| | | 51 | | { |
| | 0 | 52 | | get => _useStd3AsciiRules; |
| | 0 | 53 | | set => _useStd3AsciiRules = value; |
| | | 54 | | } |
| | | 55 | | |
| | | 56 | | // Gets ASCII (Punycode) version of the string |
| | | 57 | | public string GetAscii(string unicode) => |
| | 0 | 58 | | GetAscii(unicode, 0); |
| | | 59 | | |
| | | 60 | | public string GetAscii(string unicode, int index) |
| | | 61 | | { |
| | 0 | 62 | | ArgumentNullException.ThrowIfNull(unicode); |
| | | 63 | | |
| | 0 | 64 | | return GetAscii(unicode, index, unicode.Length - index); |
| | | 65 | | } |
| | | 66 | | |
| | | 67 | | public string GetAscii(string unicode, int index, int count) |
| | | 68 | | { |
| | 0 | 69 | | ArgumentNullException.ThrowIfNull(unicode); |
| | | 70 | | |
| | 0 | 71 | | ArgumentOutOfRangeException.ThrowIfNegative(index); |
| | 0 | 72 | | ArgumentOutOfRangeException.ThrowIfNegative(count); |
| | 0 | 73 | | if (index > unicode.Length) |
| | 0 | 74 | | throw new ArgumentOutOfRangeException(nameof(index), SR.ArgumentOutOfRange_IndexMustBeLessOrEqual); |
| | 0 | 75 | | if (index > unicode.Length - count) |
| | 0 | 76 | | throw new ArgumentOutOfRangeException(nameof(unicode), SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 77 | | |
| | 0 | 78 | | if (count == 0) |
| | | 79 | | { |
| | 0 | 80 | | throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode)); |
| | | 81 | | } |
| | 0 | 82 | | if (unicode[index + count - 1] == 0) |
| | | 83 | | { |
| | 0 | 84 | | throw new ArgumentException(SR.Format(SR.Argument_InvalidCharSequence, index + count - 1), nameof(unicod |
| | | 85 | | } |
| | | 86 | | |
| | 0 | 87 | | if (GlobalizationMode.Invariant) |
| | | 88 | | { |
| | 0 | 89 | | return GetAsciiInvariant(unicode, index, count); |
| | | 90 | | } |
| | | 91 | | |
| | 0 | 92 | | return GlobalizationMode.UseNls ? |
| | 0 | 93 | | NlsGetAsciiCore(unicode, index, count) : |
| | 0 | 94 | | IcuGetAsciiCore(unicode, index, count); |
| | | 95 | | } |
| | | 96 | | |
| | | 97 | | /// <summary> |
| | | 98 | | /// Encodes a Unicode domain name to its ASCII (Punycode) equivalent. |
| | | 99 | | /// </summary> |
| | | 100 | | /// <param name="unicode">The Unicode domain name to convert.</param> |
| | | 101 | | /// <param name="destination">The buffer to write the ASCII result to. This buffer must not overlap with <paramr |
| | | 102 | | /// <param name="charsWritten">When this method returns, contains the number of characters that were written to |
| | | 103 | | /// <returns><see langword="true"/> if the conversion was successful and the result was written to <paramref nam |
| | | 104 | | /// <exception cref="ArgumentException"><paramref name="unicode"/> is invalid based on the <see cref="AllowUnass |
| | | 105 | | public bool TryGetAscii(ReadOnlySpan<char> unicode, Span<char> destination, out int charsWritten) |
| | | 106 | | { |
| | 0 | 107 | | if (unicode.Length == 0) |
| | | 108 | | { |
| | 0 | 109 | | throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode)); |
| | | 110 | | } |
| | | 111 | | |
| | 0 | 112 | | if (unicode[^1] == 0) |
| | | 113 | | { |
| | 0 | 114 | | throw new ArgumentException(SR.Format(SR.Argument_InvalidCharSequence, unicode.Length - 1), nameof(unico |
| | | 115 | | } |
| | | 116 | | |
| | 0 | 117 | | if (unicode.Overlaps(destination)) |
| | | 118 | | { |
| | 0 | 119 | | ThrowHelper.ThrowArgumentException(ExceptionResource.InvalidOperation_SpanOverlappedOperation); |
| | | 120 | | } |
| | | 121 | | |
| | 0 | 122 | | if (GlobalizationMode.Invariant) |
| | | 123 | | { |
| | 0 | 124 | | return TryGetAsciiInvariant(unicode, destination, out charsWritten); |
| | | 125 | | } |
| | | 126 | | |
| | 0 | 127 | | return GlobalizationMode.UseNls ? |
| | 0 | 128 | | NlsTryGetAsciiCore(unicode, destination, out charsWritten) : |
| | 0 | 129 | | IcuTryGetAsciiCore(unicode, destination, out charsWritten); |
| | | 130 | | } |
| | | 131 | | |
| | | 132 | | // Gets Unicode version of the string. Normalized and limited to IDNA characters. |
| | | 133 | | public string GetUnicode(string ascii) => |
| | 0 | 134 | | GetUnicode(ascii, 0); |
| | | 135 | | |
| | | 136 | | public string GetUnicode(string ascii, int index) |
| | | 137 | | { |
| | 0 | 138 | | ArgumentNullException.ThrowIfNull(ascii); |
| | | 139 | | |
| | 0 | 140 | | return GetUnicode(ascii, index, ascii.Length - index); |
| | | 141 | | } |
| | | 142 | | |
| | | 143 | | public string GetUnicode(string ascii, int index, int count) |
| | | 144 | | { |
| | 0 | 145 | | ArgumentNullException.ThrowIfNull(ascii); |
| | | 146 | | |
| | 0 | 147 | | ArgumentOutOfRangeException.ThrowIfNegative(index); |
| | 0 | 148 | | ArgumentOutOfRangeException.ThrowIfNegative(count); |
| | 0 | 149 | | if (index > ascii.Length) |
| | 0 | 150 | | throw new ArgumentOutOfRangeException(nameof(index), SR.ArgumentOutOfRange_IndexMustBeLessOrEqual); |
| | 0 | 151 | | if (index > ascii.Length - count) |
| | 0 | 152 | | throw new ArgumentOutOfRangeException(nameof(ascii), SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 153 | | |
| | | 154 | | // This is a case (i.e. explicitly null-terminated input) where behavior in .NET and Win32 intentionally dif |
| | | 155 | | // The .NET APIs should (and did in v4.0 and earlier) throw an ArgumentException on input that includes a te |
| | | 156 | | // The Win32 APIs fail on an embedded null, but not on a terminating null. |
| | 0 | 157 | | if (count > 0 && ascii[index + count - 1] == (char)0) |
| | 0 | 158 | | throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii)); |
| | | 159 | | |
| | 0 | 160 | | if (GlobalizationMode.Invariant) |
| | | 161 | | { |
| | 0 | 162 | | return GetUnicodeInvariant(ascii, index, count); |
| | | 163 | | } |
| | | 164 | | |
| | 0 | 165 | | return GlobalizationMode.UseNls ? |
| | 0 | 166 | | NlsGetUnicodeCore(ascii, index, count) : |
| | 0 | 167 | | IcuGetUnicodeCore(ascii, index, count); |
| | | 168 | | } |
| | | 169 | | |
| | | 170 | | /// <summary> |
| | | 171 | | /// Decodes one or more encoded domain name labels to a string of Unicode characters. |
| | | 172 | | /// </summary> |
| | | 173 | | /// <param name="ascii">The ASCII domain name to convert. The string may contain one or more labels, where each |
| | | 174 | | /// <param name="destination">The buffer to write the Unicode result to. This buffer must not overlap with <para |
| | | 175 | | /// <param name="charsWritten">When this method returns, contains the number of characters that were written to |
| | | 176 | | /// <returns><see langword="true"/> if the conversion was successful and the result was written to <paramref nam |
| | | 177 | | /// <exception cref="ArgumentException"><paramref name="ascii"/> is invalid based on the <see cref="AllowUnassig |
| | | 178 | | public bool TryGetUnicode(ReadOnlySpan<char> ascii, Span<char> destination, out int charsWritten) |
| | | 179 | | { |
| | | 180 | | // This is a case (i.e. explicitly null-terminated input) where behavior in .NET and Win32 intentionally dif |
| | | 181 | | // The .NET APIs should (and did in v4.0 and earlier) throw an ArgumentException on input that includes a te |
| | | 182 | | // The Win32 APIs fail on an embedded null, but not on a terminating null. |
| | 0 | 183 | | if (ascii.Length > 0 && ascii[^1] == (char)0) |
| | | 184 | | { |
| | 0 | 185 | | throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii)); |
| | | 186 | | } |
| | | 187 | | |
| | 0 | 188 | | if (ascii.Overlaps(destination)) |
| | | 189 | | { |
| | 0 | 190 | | ThrowHelper.ThrowArgumentException(ExceptionResource.InvalidOperation_SpanOverlappedOperation); |
| | | 191 | | } |
| | | 192 | | |
| | 0 | 193 | | if (GlobalizationMode.Invariant) |
| | | 194 | | { |
| | 0 | 195 | | return TryGetUnicodeInvariant(ascii, destination, out charsWritten); |
| | | 196 | | } |
| | | 197 | | |
| | 0 | 198 | | return GlobalizationMode.UseNls ? |
| | 0 | 199 | | NlsTryGetUnicodeCore(ascii, destination, out charsWritten) : |
| | 0 | 200 | | IcuTryGetUnicodeCore(ascii, destination, out charsWritten); |
| | | 201 | | } |
| | | 202 | | |
| | | 203 | | public override bool Equals([NotNullWhen(true)] object? obj) => |
| | 0 | 204 | | obj is IdnMapping that && |
| | 0 | 205 | | _allowUnassigned == that._allowUnassigned && |
| | 0 | 206 | | _useStd3AsciiRules == that._useStd3AsciiRules; |
| | | 207 | | |
| | | 208 | | public override int GetHashCode() => |
| | 0 | 209 | | (_allowUnassigned ? 100 : 200) + (_useStd3AsciiRules ? 1000 : 2000); |
| | | 210 | | |
| | | 211 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 212 | | private static string GetStringForOutput(string? originalString, ReadOnlySpan<char> input, ReadOnlySpan<char> ou |
| | | 213 | | { |
| | 0 | 214 | | Debug.Assert(input.Length > 0); |
| | | 215 | | |
| | 0 | 216 | | if (originalString is not null && |
| | 0 | 217 | | originalString.Length == input.Length && |
| | 0 | 218 | | input.Length == output.Length && |
| | 0 | 219 | | Ordinal.EqualsIgnoreCase(ref MemoryMarshal.GetReference(input), ref MemoryMarshal.GetReference(output), |
| | | 220 | | { |
| | 0 | 221 | | return originalString; |
| | | 222 | | } |
| | | 223 | | |
| | 0 | 224 | | return output.ToString(); |
| | | 225 | | } |
| | | 226 | | |
| | | 227 | | // |
| | | 228 | | // Invariant implementation |
| | | 229 | | // |
| | | 230 | | |
| | | 231 | | private const char c_delimiter = '-'; |
| | | 232 | | private const string c_strAcePrefix = "xn--"; |
| | | 233 | | private const int c_labelLimit = 63; // Not including dots |
| | | 234 | | private const int c_defaultNameLimit = 255; // Including dots |
| | | 235 | | private const int c_initialN = 0x80; |
| | | 236 | | private const int c_maxint = 0x7ffffff; |
| | | 237 | | private const int c_initialBias = 72; |
| | | 238 | | private const int c_punycodeBase = 36; |
| | | 239 | | private const int c_tmin = 1; |
| | | 240 | | private const int c_tmax = 26; |
| | | 241 | | private const int c_skew = 38; |
| | | 242 | | private const int c_damp = 700; |
| | | 243 | | |
| | | 244 | | private string GetAsciiInvariant(string unicodeString, int index, int count) |
| | | 245 | | { |
| | 0 | 246 | | ReadOnlySpan<char> unicode = unicodeString.AsSpan(index, count); |
| | | 247 | | |
| | | 248 | | // Check for ASCII only string, which will be unchanged |
| | 0 | 249 | | if (ValidateStd3AndAscii(unicode, UseStd3AsciiRules, true)) |
| | | 250 | | { |
| | | 251 | | // Return original string if the entire string was requested and it doesn't need modification |
| | 0 | 252 | | if (index == 0 && count == unicodeString.Length) |
| | | 253 | | { |
| | 0 | 254 | | return unicodeString; |
| | | 255 | | } |
| | | 256 | | |
| | 0 | 257 | | return unicode.ToString(); |
| | | 258 | | } |
| | | 259 | | |
| | | 260 | | // Cannot be null terminated (normalization won't help us with this one, and |
| | | 261 | | // may have returned false before checking the whole string above) |
| | 0 | 262 | | Debug.Assert(unicode.Length >= 1, "[IdnMapping.GetAscii] Expected 0 length strings to fail before now."); |
| | 0 | 263 | | if (unicode[^1] <= 0x1f) |
| | | 264 | | { |
| | 0 | 265 | | throw new ArgumentException(SR.Format(SR.Argument_InvalidCharSequence, unicode.Length - 1), nameof(unico |
| | | 266 | | } |
| | | 267 | | |
| | | 268 | | // May need to check Std3 rules again for non-ascii |
| | 0 | 269 | | if (UseStd3AsciiRules) |
| | | 270 | | { |
| | 0 | 271 | | ValidateStd3AndAscii(unicode, true, false); |
| | | 272 | | } |
| | | 273 | | |
| | | 274 | | // Go ahead and encode it |
| | 0 | 275 | | return PunycodeEncode(unicode); |
| | | 276 | | } |
| | | 277 | | |
| | | 278 | | private bool TryGetAsciiInvariant(ReadOnlySpan<char> unicode, Span<char> destination, out int charsWritten) |
| | | 279 | | { |
| | | 280 | | // Check for ASCII only string, which will be unchanged |
| | 0 | 281 | | if (ValidateStd3AndAscii(unicode, UseStd3AsciiRules, true)) |
| | | 282 | | { |
| | 0 | 283 | | if (unicode.Length <= destination.Length) |
| | | 284 | | { |
| | 0 | 285 | | unicode.CopyTo(destination); |
| | 0 | 286 | | charsWritten = unicode.Length; |
| | 0 | 287 | | return true; |
| | | 288 | | } |
| | | 289 | | |
| | 0 | 290 | | charsWritten = 0; |
| | 0 | 291 | | return false; |
| | | 292 | | } |
| | | 293 | | |
| | | 294 | | // Cannot be null terminated (normalization won't help us with this one, and |
| | | 295 | | // may have returned false before checking the whole string above) |
| | 0 | 296 | | Debug.Assert(unicode.Length >= 1, "[IdnMapping.GetAscii] Expected 0 length strings to fail before now."); |
| | 0 | 297 | | if (unicode[^1] <= 0x1f) |
| | | 298 | | { |
| | 0 | 299 | | throw new ArgumentException(SR.Format(SR.Argument_InvalidCharSequence, unicode.Length - 1), nameof(unico |
| | | 300 | | } |
| | | 301 | | |
| | | 302 | | // May need to check Std3 rules again for non-ascii |
| | 0 | 303 | | if (UseStd3AsciiRules) |
| | | 304 | | { |
| | 0 | 305 | | ValidateStd3AndAscii(unicode, true, false); |
| | | 306 | | } |
| | | 307 | | |
| | | 308 | | // Go ahead and encode it |
| | 0 | 309 | | string result = PunycodeEncode(unicode); |
| | 0 | 310 | | if (result.Length <= destination.Length) |
| | | 311 | | { |
| | 0 | 312 | | result.CopyTo(destination); |
| | 0 | 313 | | charsWritten = result.Length; |
| | 0 | 314 | | return true; |
| | | 315 | | } |
| | | 316 | | |
| | 0 | 317 | | charsWritten = 0; |
| | 0 | 318 | | return false; |
| | | 319 | | } |
| | | 320 | | |
| | | 321 | | // See if we're only ASCII |
| | | 322 | | private static bool ValidateStd3AndAscii(ReadOnlySpan<char> unicode, bool bUseStd3, bool bCheckAscii) |
| | | 323 | | { |
| | | 324 | | // If its empty, then its too small |
| | 0 | 325 | | if (unicode.Length == 0) |
| | 0 | 326 | | throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode)); |
| | | 327 | | |
| | 0 | 328 | | int iLastDot = -1; |
| | | 329 | | |
| | | 330 | | // Loop the whole string |
| | 0 | 331 | | for (int i = 0; i < unicode.Length; i++) |
| | | 332 | | { |
| | | 333 | | // Aren't allowing control chars (or 7f, but idn tables catch that, they don't catch \0 at end though) |
| | 0 | 334 | | if (unicode[i] <= 0x1f) |
| | | 335 | | { |
| | 0 | 336 | | throw new ArgumentException(SR.Format(SR.Argument_InvalidCharSequence, i), nameof(unicode)); |
| | | 337 | | } |
| | | 338 | | |
| | | 339 | | // If its Unicode or a control character, return false (non-ascii) |
| | 0 | 340 | | if (bCheckAscii && unicode[i] >= 0x7f) |
| | 0 | 341 | | return false; |
| | | 342 | | |
| | | 343 | | // Check for dots |
| | 0 | 344 | | if (IsDot(unicode[i])) |
| | | 345 | | { |
| | | 346 | | // Can't have 2 dots in a row |
| | 0 | 347 | | if (i == iLastDot + 1) |
| | 0 | 348 | | throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode)); |
| | | 349 | | |
| | | 350 | | // If its too far between dots then fail |
| | 0 | 351 | | if (i - iLastDot > c_labelLimit + 1) |
| | 0 | 352 | | throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode)); |
| | | 353 | | |
| | | 354 | | // If validating Std3, then char before dot can't be - char |
| | 0 | 355 | | if (bUseStd3 && i > 0) |
| | 0 | 356 | | ValidateStd3(unicode[i - 1], true); |
| | | 357 | | |
| | | 358 | | // Remember where the last dot is |
| | 0 | 359 | | iLastDot = i; |
| | 0 | 360 | | continue; |
| | | 361 | | } |
| | | 362 | | |
| | | 363 | | // If necessary, make sure its a valid std3 character |
| | 0 | 364 | | if (bUseStd3) |
| | | 365 | | { |
| | 0 | 366 | | ValidateStd3(unicode[i], i == iLastDot + 1); |
| | | 367 | | } |
| | | 368 | | } |
| | | 369 | | |
| | | 370 | | // If we never had a dot, then we need to be shorter than the label limit |
| | 0 | 371 | | if (iLastDot == -1 && unicode.Length > c_labelLimit) |
| | 0 | 372 | | throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode)); |
| | | 373 | | |
| | | 374 | | // Need to validate entire string length, 1 shorter if last char wasn't a dot |
| | 0 | 375 | | if (unicode.Length > c_defaultNameLimit - (IsDot(unicode[^1]) ? 0 : 1)) |
| | 0 | 376 | | throw new ArgumentException(SR.Format(SR.Argument_IdnBadNameSize, |
| | 0 | 377 | | c_defaultNameLimit - (IsDot(unicode[^1]) ? 0 : 1)), nameof(unico |
| | | 378 | | |
| | | 379 | | // If last char wasn't a dot we need to check for trailing - |
| | 0 | 380 | | if (bUseStd3 && !IsDot(unicode[^1])) |
| | 0 | 381 | | ValidateStd3(unicode[^1], true); |
| | | 382 | | |
| | 0 | 383 | | return true; |
| | | 384 | | } |
| | | 385 | | |
| | | 386 | | /* PunycodeEncode() converts Unicode to Punycode. The input */ |
| | | 387 | | /* is represented as an array of Unicode code points (not code */ |
| | | 388 | | /* units; surrogate pairs are not allowed), and the output */ |
| | | 389 | | /* will be represented as an array of ASCII code points. The */ |
| | | 390 | | /* output string is *not* null-terminated; it will contain */ |
| | | 391 | | /* zeros if and only if the input contains zeros. (Of course */ |
| | | 392 | | /* the caller can leave room for a terminator and add one if */ |
| | | 393 | | /* needed.) The input_length is the number of code points in */ |
| | | 394 | | /* the input. The output_length is an in/out argument: the */ |
| | | 395 | | /* caller passes in the maximum number of code points that it */ |
| | | 396 | | |
| | | 397 | | /* can receive, and on successful return it will contain the */ |
| | | 398 | | /* number of code points actually output. The case_flags array */ |
| | | 399 | | /* holds input_length boolean values, where nonzero suggests that */ |
| | | 400 | | /* the corresponding Unicode character be forced to uppercase */ |
| | | 401 | | /* after being decoded (if possible), and zero suggests that */ |
| | | 402 | | /* it be forced to lowercase (if possible). ASCII code points */ |
| | | 403 | | /* are encoded literally, except that ASCII letters are forced */ |
| | | 404 | | /* to uppercase or lowercase according to the corresponding */ |
| | | 405 | | /* uppercase flags. If case_flags is a null pointer then ASCII */ |
| | | 406 | | /* letters are left as they are, and other code points are */ |
| | | 407 | | /* treated as if their uppercase flags were zero. The return */ |
| | | 408 | | /* value can be any of the punycode_status values defined above */ |
| | | 409 | | /* except punycode_bad_input; if not punycode_success, then */ |
| | | 410 | | /* output_size and output might contain garbage. */ |
| | | 411 | | private static string PunycodeEncode(ReadOnlySpan<char> unicode) |
| | | 412 | | { |
| | | 413 | | // 0 length strings aren't allowed |
| | 0 | 414 | | if (unicode.Length == 0) |
| | 0 | 415 | | throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode)); |
| | | 416 | | |
| | 0 | 417 | | StringBuilder output = new StringBuilder(unicode.Length); |
| | 0 | 418 | | int iNextDot = 0; |
| | 0 | 419 | | int iAfterLastDot = 0; |
| | 0 | 420 | | int iOutputAfterLastDot = 0; |
| | | 421 | | |
| | | 422 | | // Find the next dot |
| | 0 | 423 | | while (iNextDot < unicode.Length) |
| | | 424 | | { |
| | | 425 | | // Legal "dot" separators (i.e: . in www.microsoft.com) |
| | | 426 | | const string DotSeparators = ".\u3002\uFF0E\uFF61"; |
| | | 427 | | |
| | | 428 | | // Find end of this segment |
| | 0 | 429 | | iNextDot = unicode.Slice(iAfterLastDot).IndexOfAny(DotSeparators); |
| | 0 | 430 | | iNextDot = iNextDot < 0 ? unicode.Length : iNextDot + iAfterLastDot; |
| | | 431 | | |
| | | 432 | | // Only allowed to have empty . section at end (www.microsoft.com.) |
| | 0 | 433 | | if (iNextDot == iAfterLastDot) |
| | | 434 | | { |
| | | 435 | | // Only allowed to have empty sections as trailing . |
| | 0 | 436 | | if (iNextDot != unicode.Length) |
| | 0 | 437 | | throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode)); |
| | | 438 | | // Last dot, stop |
| | | 439 | | break; |
| | | 440 | | } |
| | | 441 | | |
| | | 442 | | // We'll need an Ace prefix |
| | 0 | 443 | | output.Append(c_strAcePrefix); |
| | | 444 | | |
| | | 445 | | // Everything resets every segment. |
| | 0 | 446 | | bool bRightToLeft = false; |
| | | 447 | | |
| | | 448 | | // Check for RTL. If right-to-left, then 1st & last chars must be RTL |
| | 0 | 449 | | StrongBidiCategory eBidi = CharUnicodeInfo.GetBidiCategory(unicode, iAfterLastDot); |
| | 0 | 450 | | if (eBidi == StrongBidiCategory.StrongRightToLeft) |
| | | 451 | | { |
| | | 452 | | // It has to be right to left. |
| | 0 | 453 | | bRightToLeft = true; |
| | | 454 | | |
| | | 455 | | // Check last char |
| | 0 | 456 | | int iTest = iNextDot - 1; |
| | 0 | 457 | | if (char.IsLowSurrogate(unicode[iTest])) |
| | | 458 | | { |
| | 0 | 459 | | iTest--; |
| | | 460 | | } |
| | | 461 | | |
| | 0 | 462 | | eBidi = CharUnicodeInfo.GetBidiCategory(unicode, iTest); |
| | 0 | 463 | | if (eBidi != StrongBidiCategory.StrongRightToLeft) |
| | | 464 | | { |
| | | 465 | | // Oops, last wasn't RTL, last should be RTL if first is RTL |
| | 0 | 466 | | throw new ArgumentException(SR.Argument_IdnBadBidi, nameof(unicode)); |
| | | 467 | | } |
| | | 468 | | } |
| | | 469 | | |
| | | 470 | | // Handle the basic code points |
| | | 471 | | int basicCount; |
| | 0 | 472 | | int numProcessed = 0; // Num code points that have been processed so far (this segment) |
| | 0 | 473 | | for (basicCount = iAfterLastDot; basicCount < iNextDot; basicCount++) |
| | | 474 | | { |
| | | 475 | | // Can't be lonely surrogate because it would've thrown in normalization |
| | 0 | 476 | | Debug.Assert(!char.IsLowSurrogate(unicode[basicCount]), "[IdnMapping.punycode_encode]Unexpected low |
| | | 477 | | |
| | | 478 | | // Double check our bidi rules |
| | 0 | 479 | | StrongBidiCategory testBidi = CharUnicodeInfo.GetBidiCategory(unicode, basicCount); |
| | | 480 | | |
| | | 481 | | // If we're RTL, we can't have LTR chars |
| | 0 | 482 | | if (bRightToLeft && testBidi == StrongBidiCategory.StrongLeftToRight) |
| | | 483 | | { |
| | | 484 | | // Oops, throw error |
| | 0 | 485 | | throw new ArgumentException(SR.Argument_IdnBadBidi, nameof(unicode)); |
| | | 486 | | } |
| | | 487 | | |
| | | 488 | | // If we're not RTL we can't have RTL chars |
| | 0 | 489 | | if (!bRightToLeft && testBidi == StrongBidiCategory.StrongRightToLeft) |
| | | 490 | | { |
| | | 491 | | // Oops, throw error |
| | 0 | 492 | | throw new ArgumentException(SR.Argument_IdnBadBidi, nameof(unicode)); |
| | | 493 | | } |
| | | 494 | | |
| | | 495 | | // If its basic then add it |
| | 0 | 496 | | if (Basic(unicode[basicCount])) |
| | | 497 | | { |
| | 0 | 498 | | output.Append(EncodeBasic(unicode[basicCount])); |
| | 0 | 499 | | numProcessed++; |
| | | 500 | | } |
| | | 501 | | // If its a surrogate, skip the next since our bidi category tester doesn't handle it. |
| | 0 | 502 | | else if (basicCount + 1 < iNextDot && char.IsSurrogatePair(unicode[basicCount], unicode[basicCount + |
| | 0 | 503 | | basicCount++; |
| | | 504 | | } |
| | | 505 | | |
| | 0 | 506 | | int numBasicCodePoints = numProcessed; // number of basic code points |
| | | 507 | | |
| | | 508 | | // Stop if we ONLY had basic code points |
| | 0 | 509 | | if (numBasicCodePoints == iNextDot - iAfterLastDot) |
| | | 510 | | { |
| | | 511 | | // Get rid of xn-- and this segments done |
| | 0 | 512 | | output.Remove(iOutputAfterLastDot, c_strAcePrefix.Length); |
| | | 513 | | } |
| | | 514 | | else |
| | | 515 | | { |
| | | 516 | | // If it has some non-basic code points the input cannot start with xn-- |
| | 0 | 517 | | if (unicode.Slice(iAfterLastDot).StartsWith(c_strAcePrefix, StringComparison.OrdinalIgnoreCase)) |
| | 0 | 518 | | throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(unicode)); |
| | | 519 | | |
| | | 520 | | // Need to do ACE encoding |
| | 0 | 521 | | int numSurrogatePairs = 0; // number of surrogate pairs so far |
| | | 522 | | |
| | | 523 | | // Add a delimiter (-) if we had any basic code points (between basic and encoded pieces) |
| | 0 | 524 | | if (numBasicCodePoints > 0) |
| | | 525 | | { |
| | 0 | 526 | | output.Append(c_delimiter); |
| | | 527 | | } |
| | | 528 | | |
| | | 529 | | // Initialize the state |
| | 0 | 530 | | int n = c_initialN; |
| | 0 | 531 | | int delta = 0; |
| | 0 | 532 | | int bias = c_initialBias; |
| | | 533 | | |
| | | 534 | | // Main loop |
| | 0 | 535 | | while (numProcessed < (iNextDot - iAfterLastDot)) |
| | | 536 | | { |
| | | 537 | | /* All non-basic code points < n have been */ |
| | | 538 | | /* handled already. Find the next larger one: */ |
| | | 539 | | int j; |
| | | 540 | | int m; |
| | | 541 | | int test; |
| | 0 | 542 | | for (m = c_maxint, j = iAfterLastDot; |
| | 0 | 543 | | j < iNextDot; |
| | 0 | 544 | | j += IsSupplementary(test) ? 2 : 1) |
| | | 545 | | { |
| | 0 | 546 | | test = GetCodePoint(unicode, j); |
| | 0 | 547 | | if (test >= n && test < m) m = test; |
| | | 548 | | } |
| | | 549 | | |
| | | 550 | | /* Increase delta enough to advance the decoder's */ |
| | | 551 | | /* <n,i> state to <m,0>, but guard against overflow: */ |
| | 0 | 552 | | delta += (int)((m - n) * ((numProcessed - numSurrogatePairs) + 1)); |
| | 0 | 553 | | Debug.Assert(delta > 0, "[IdnMapping.cs]1 punycode_encode - delta overflowed int"); |
| | 0 | 554 | | n = m; |
| | | 555 | | |
| | 0 | 556 | | for (j = iAfterLastDot; j < iNextDot; j += IsSupplementary(test) ? 2 : 1) |
| | | 557 | | { |
| | | 558 | | // Make sure we're aware of surrogates |
| | 0 | 559 | | test = GetCodePoint(unicode, j); |
| | | 560 | | |
| | | 561 | | // Adjust for character position (only the chars in our string already, some |
| | | 562 | | // haven't been processed. |
| | | 563 | | |
| | 0 | 564 | | if (test < n) |
| | | 565 | | { |
| | 0 | 566 | | delta++; |
| | 0 | 567 | | Debug.Assert(delta > 0, "[IdnMapping.cs]2 punycode_encode - delta overflowed int"); |
| | | 568 | | } |
| | | 569 | | |
| | 0 | 570 | | if (test == n) |
| | | 571 | | { |
| | | 572 | | // Represent delta as a generalized variable-length integer: |
| | | 573 | | int q, k; |
| | 0 | 574 | | for (q = delta, k = c_punycodeBase; ; k += c_punycodeBase) |
| | | 575 | | { |
| | 0 | 576 | | int t = k <= bias ? c_tmin : k >= bias + c_tmax ? c_tmax : k - bias; |
| | 0 | 577 | | if (q < t) break; |
| | 0 | 578 | | Debug.Assert(c_punycodeBase != t, "[IdnMapping.punycode_encode]Expected c_punycodeBa |
| | 0 | 579 | | output.Append(EncodeDigit(t + (q - t) % (c_punycodeBase - t))); |
| | 0 | 580 | | q = (q - t) / (c_punycodeBase - t); |
| | | 581 | | } |
| | | 582 | | |
| | 0 | 583 | | output.Append(EncodeDigit(q)); |
| | 0 | 584 | | bias = Adapt(delta, (numProcessed - numSurrogatePairs) + 1, numProcessed == numBasicCode |
| | 0 | 585 | | delta = 0; |
| | 0 | 586 | | numProcessed++; |
| | | 587 | | |
| | 0 | 588 | | if (IsSupplementary(m)) |
| | | 589 | | { |
| | 0 | 590 | | numProcessed++; |
| | 0 | 591 | | numSurrogatePairs++; |
| | | 592 | | } |
| | | 593 | | } |
| | | 594 | | } |
| | 0 | 595 | | ++delta; |
| | 0 | 596 | | ++n; |
| | 0 | 597 | | Debug.Assert(delta > 0, "[IdnMapping.cs]3 punycode_encode - delta overflowed int"); |
| | | 598 | | } |
| | | 599 | | } |
| | | 600 | | |
| | | 601 | | // Make sure its not too big |
| | 0 | 602 | | if (output.Length - iOutputAfterLastDot > c_labelLimit) |
| | 0 | 603 | | throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode)); |
| | | 604 | | |
| | | 605 | | // Done with this segment, add dot if necessary |
| | 0 | 606 | | if (iNextDot != unicode.Length) |
| | 0 | 607 | | output.Append('.'); |
| | | 608 | | |
| | 0 | 609 | | iAfterLastDot = iNextDot + 1; |
| | 0 | 610 | | iOutputAfterLastDot = output.Length; |
| | | 611 | | } |
| | | 612 | | |
| | | 613 | | // Throw if we're too long |
| | 0 | 614 | | if (output.Length > c_defaultNameLimit - (IsDot(unicode[^1]) ? 0 : 1)) |
| | 0 | 615 | | throw new ArgumentException(SR.Format(SR.Argument_IdnBadNameSize, |
| | 0 | 616 | | c_defaultNameLimit - (IsDot(unicode[^1]) ? 0 : 1)), nameof(unicode)); |
| | | 617 | | // Return our output string |
| | 0 | 618 | | return output.ToString(); |
| | | 619 | | } |
| | | 620 | | |
| | | 621 | | // Is it a dot? |
| | | 622 | | // are we U+002E (., full stop), U+3002 (ideographic full stop), U+FF0E (fullwidth full stop), or |
| | | 623 | | // U+FF61 (halfwidth ideographic full stop). |
| | | 624 | | // Note: IDNA Normalization gets rid of dots now, but testing for last dot is before normalization |
| | | 625 | | private static bool IsDot(char c) => |
| | 0 | 626 | | c == '.' || c == '\u3002' || c == '\uFF0E' || c == '\uFF61'; |
| | | 627 | | |
| | | 628 | | private static bool IsSupplementary(int cTest) => |
| | 0 | 629 | | cTest >= 0x10000; |
| | | 630 | | |
| | | 631 | | private static bool Basic(uint cp) => |
| | | 632 | | // Is it in ASCII range? |
| | 0 | 633 | | cp < 0x80; |
| | | 634 | | |
| | | 635 | | private static int GetCodePoint(ReadOnlySpan<char> s, int index) |
| | | 636 | | { |
| | | 637 | | // Check if the character at index is a high surrogate. |
| | 0 | 638 | | if (char.IsHighSurrogate(s[index]) && index + 1 < s.Length && char.IsLowSurrogate(s[index + 1])) |
| | | 639 | | { |
| | 0 | 640 | | return char.ConvertToUtf32(s[index], s[index + 1]); |
| | | 641 | | } |
| | | 642 | | |
| | 0 | 643 | | return s[index]; |
| | | 644 | | } |
| | | 645 | | |
| | | 646 | | // Validate Std3 rules for a character |
| | | 647 | | private static void ValidateStd3(char c, bool bNextToDot) |
| | | 648 | | { |
| | | 649 | | // Check for illegal characters |
| | 0 | 650 | | if (c <= ',' || c == '/' || (c >= ':' && c <= '@') || // Lots of characters not allowed |
| | 0 | 651 | | (c >= '[' && c <= '`') || (c >= '{' && c <= (char)0x7F) || |
| | 0 | 652 | | (c == '-' && bNextToDot)) |
| | 0 | 653 | | throw new ArgumentException(SR.Format(SR.Argument_IdnBadStd3, c), nameof(c)); |
| | 0 | 654 | | } |
| | | 655 | | |
| | | 656 | | private string GetUnicodeInvariant(string ascii, int index, int count) |
| | | 657 | | { |
| | | 658 | | // Convert Punycode to Unicode |
| | 0 | 659 | | string asciiSlice = ascii.Substring(index, count); |
| | 0 | 660 | | string strUnicode = PunycodeDecode(asciiSlice); |
| | | 661 | | |
| | | 662 | | // Output name MUST obey IDNA rules & round trip (casing differences are allowed) |
| | 0 | 663 | | string asciiRoundtrip = GetAscii(strUnicode); |
| | 0 | 664 | | if (!asciiRoundtrip.Equals(asciiSlice, StringComparison.OrdinalIgnoreCase)) |
| | | 665 | | { |
| | 0 | 666 | | throw new ArgumentException(SR.Argument_IdnIllegalName, nameof(ascii)); |
| | | 667 | | } |
| | | 668 | | |
| | | 669 | | // If the ASCII round-trip equals the original string, return it as-is (no allocation) |
| | 0 | 670 | | if (index == 0 && count == ascii.Length && strUnicode.Equals(ascii, StringComparison.OrdinalIgnoreCase)) |
| | | 671 | | { |
| | 0 | 672 | | return ascii; |
| | | 673 | | } |
| | | 674 | | |
| | 0 | 675 | | return strUnicode; |
| | | 676 | | } |
| | | 677 | | |
| | | 678 | | private bool TryGetUnicodeInvariant(ReadOnlySpan<char> ascii, Span<char> destination, out int charsWritten) |
| | | 679 | | { |
| | | 680 | | // Convert the span to a string for PunycodeDecode since it uses string operations extensively |
| | 0 | 681 | | string asciiString = ascii.ToString(); |
| | | 682 | | |
| | | 683 | | // Convert Punycode to Unicode |
| | 0 | 684 | | string strUnicode = PunycodeDecode(asciiString); |
| | | 685 | | |
| | | 686 | | // Output name MUST obey IDNA rules & round trip (casing differences are allowed) |
| | 0 | 687 | | if (!asciiString.Equals(GetAscii(strUnicode), StringComparison.OrdinalIgnoreCase)) |
| | | 688 | | { |
| | 0 | 689 | | throw new ArgumentException(SR.Argument_IdnIllegalName, nameof(ascii)); |
| | | 690 | | } |
| | | 691 | | |
| | 0 | 692 | | if (strUnicode.Length <= destination.Length) |
| | | 693 | | { |
| | 0 | 694 | | strUnicode.CopyTo(destination); |
| | 0 | 695 | | charsWritten = strUnicode.Length; |
| | 0 | 696 | | return true; |
| | | 697 | | } |
| | | 698 | | |
| | 0 | 699 | | charsWritten = 0; |
| | 0 | 700 | | return false; |
| | | 701 | | } |
| | | 702 | | |
| | | 703 | | /* PunycodeDecode() converts Punycode to Unicode. The input is */ |
| | | 704 | | /* represented as an array of ASCII code points, and the output */ |
| | | 705 | | /* will be represented as an array of Unicode code points. The */ |
| | | 706 | | /* input_length is the number of code points in the input. The */ |
| | | 707 | | /* output_length is an in/out argument: the caller passes in */ |
| | | 708 | | /* the maximum number of code points that it can receive, and */ |
| | | 709 | | /* on successful return it will contain the actual number of */ |
| | | 710 | | /* code points output. The case_flags array needs room for at */ |
| | | 711 | | /* least output_length values, or it can be a null pointer if the */ |
| | | 712 | | /* case information is not needed. A nonzero flag suggests that */ |
| | | 713 | | /* the corresponding Unicode character be forced to uppercase */ |
| | | 714 | | /* by the caller (if possible), while zero suggests that it be */ |
| | | 715 | | /* forced to lowercase (if possible). ASCII code points are */ |
| | | 716 | | /* output already in the proper case, but their flags will be set */ |
| | | 717 | | /* appropriately so that applying the flags would be harmless. */ |
| | | 718 | | /* The return value can be any of the punycode_status values */ |
| | | 719 | | /* defined above; if not punycode_success, then output_length, */ |
| | | 720 | | /* output, and case_flags might contain garbage. On success, the */ |
| | | 721 | | /* decoder will never need to write an output_length greater than */ |
| | | 722 | | /* input_length, because of how the encoding is defined. */ |
| | | 723 | | |
| | | 724 | | private static string PunycodeDecode(string ascii) |
| | | 725 | | { |
| | | 726 | | // 0 length strings aren't allowed |
| | 0 | 727 | | if (ascii.Length == 0) |
| | 0 | 728 | | throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(ascii)); |
| | | 729 | | |
| | | 730 | | // Throw if we're too long |
| | 0 | 731 | | if (ascii.Length > c_defaultNameLimit - (IsDot(ascii[^1]) ? 0 : 1)) |
| | 0 | 732 | | throw new ArgumentException(SR.Format(SR.Argument_IdnBadNameSize, |
| | 0 | 733 | | c_defaultNameLimit - (IsDot(ascii[^1]) ? 0 : 1)), nameof(ascii)); |
| | | 734 | | |
| | | 735 | | // output stringbuilder |
| | 0 | 736 | | StringBuilder output = new StringBuilder(ascii.Length); |
| | | 737 | | |
| | | 738 | | // Dot searching |
| | 0 | 739 | | int iNextDot = 0; |
| | 0 | 740 | | int iAfterLastDot = 0; |
| | 0 | 741 | | int iOutputAfterLastDot = 0; |
| | | 742 | | |
| | 0 | 743 | | while (iNextDot < ascii.Length) |
| | | 744 | | { |
| | | 745 | | // Find end of this segment |
| | 0 | 746 | | iNextDot = ascii.IndexOf('.', iAfterLastDot); |
| | 0 | 747 | | if (iNextDot < 0 || iNextDot > ascii.Length) |
| | 0 | 748 | | iNextDot = ascii.Length; |
| | | 749 | | |
| | | 750 | | // Only allowed to have empty . section at end (www.microsoft.com.) |
| | 0 | 751 | | if (iNextDot == iAfterLastDot) |
| | | 752 | | { |
| | | 753 | | // Only allowed to have empty sections as trailing . |
| | 0 | 754 | | if (iNextDot != ascii.Length) |
| | 0 | 755 | | throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(ascii)); |
| | | 756 | | |
| | | 757 | | // Last dot, stop |
| | | 758 | | break; |
| | | 759 | | } |
| | | 760 | | |
| | | 761 | | // In either case it can't be bigger than segment size |
| | 0 | 762 | | if (iNextDot - iAfterLastDot > c_labelLimit) |
| | 0 | 763 | | throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(ascii)); |
| | | 764 | | |
| | | 765 | | // See if this section's ASCII or ACE |
| | 0 | 766 | | if (!ascii.AsSpan(iAfterLastDot).StartsWith(c_strAcePrefix, StringComparison.OrdinalIgnoreCase)) |
| | | 767 | | { |
| | | 768 | | // Its ASCII, copy it |
| | 0 | 769 | | output.Append(ascii, iAfterLastDot, iNextDot - iAfterLastDot); |
| | | 770 | | } |
| | | 771 | | else |
| | | 772 | | { |
| | | 773 | | // Not ASCII, bump up iAfterLastDot to be after ACE Prefix |
| | 0 | 774 | | iAfterLastDot += c_strAcePrefix.Length; |
| | | 775 | | |
| | | 776 | | // Get number of basic code points (where delimiter is) |
| | | 777 | | // numBasicCodePoints < 0 if there're no basic code points |
| | 0 | 778 | | int iTemp = ascii.LastIndexOf(c_delimiter, iNextDot - 1); |
| | | 779 | | |
| | | 780 | | // Trailing - not allowed |
| | 0 | 781 | | if (iTemp == iNextDot - 1) |
| | 0 | 782 | | throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii)); |
| | | 783 | | |
| | | 784 | | int numBasicCodePoints; |
| | 0 | 785 | | if (iTemp <= iAfterLastDot) |
| | 0 | 786 | | numBasicCodePoints = 0; |
| | | 787 | | else |
| | | 788 | | { |
| | 0 | 789 | | numBasicCodePoints = iTemp - iAfterLastDot; |
| | | 790 | | |
| | | 791 | | // Copy all the basic code points, making sure they're all in the allowed range, |
| | | 792 | | // and losing the casing for all of them. |
| | 0 | 793 | | for (int copyAscii = iAfterLastDot; copyAscii < iAfterLastDot + numBasicCodePoints; copyAscii++) |
| | | 794 | | { |
| | | 795 | | // Make sure we don't allow unicode in the ascii part |
| | 0 | 796 | | if (ascii[copyAscii] > 0x7f) |
| | 0 | 797 | | throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii)); |
| | | 798 | | |
| | | 799 | | // When appending make sure they get lower cased |
| | 0 | 800 | | output.Append((char)(char.IsAsciiLetterUpper(ascii[copyAscii]) ? ascii[copyAscii] - 'A' + 'a |
| | | 801 | | } |
| | | 802 | | } |
| | | 803 | | |
| | | 804 | | // Get ready for main loop. Start at beginning if we didn't have any |
| | | 805 | | // basic code points, otherwise start after the -. |
| | | 806 | | // asciiIndex will be next character to read from ascii |
| | 0 | 807 | | int asciiIndex = iAfterLastDot + (numBasicCodePoints > 0 ? numBasicCodePoints + 1 : 0); |
| | | 808 | | |
| | | 809 | | // initialize our state |
| | 0 | 810 | | int n = c_initialN; |
| | 0 | 811 | | int bias = c_initialBias; |
| | 0 | 812 | | int i = 0; |
| | | 813 | | |
| | | 814 | | int w, k; |
| | | 815 | | |
| | | 816 | | // no Supplementary characters yet |
| | 0 | 817 | | int numSurrogatePairs = 0; |
| | | 818 | | |
| | | 819 | | // Main loop, read rest of ascii |
| | 0 | 820 | | while (asciiIndex < iNextDot) |
| | | 821 | | { |
| | | 822 | | /* Decode a generalized variable-length integer into delta, */ |
| | | 823 | | /* which gets added to i. The overflow checking is easier */ |
| | | 824 | | /* if we increase i as we go, then subtract off its starting */ |
| | | 825 | | /* value at the end to obtain delta. */ |
| | 0 | 826 | | int oldi = i; |
| | | 827 | | |
| | 0 | 828 | | for (w = 1, k = c_punycodeBase; ; k += c_punycodeBase) |
| | | 829 | | { |
| | | 830 | | // Check to make sure we aren't overrunning our ascii string |
| | 0 | 831 | | if (asciiIndex >= iNextDot) |
| | 0 | 832 | | throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii)); |
| | | 833 | | |
| | | 834 | | // decode the digit from the next char |
| | 0 | 835 | | int digit = DecodeDigit(ascii[asciiIndex++]); |
| | | 836 | | |
| | 0 | 837 | | Debug.Assert(w > 0, "[IdnMapping.punycode_decode]Expected w > 0"); |
| | 0 | 838 | | if (digit > (c_maxint - i) / w) |
| | 0 | 839 | | throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii)); |
| | | 840 | | |
| | 0 | 841 | | i += (int)(digit * w); |
| | 0 | 842 | | int t = k <= bias ? c_tmin : k >= bias + c_tmax ? c_tmax : k - bias; |
| | 0 | 843 | | if (digit < t) |
| | | 844 | | break; |
| | 0 | 845 | | Debug.Assert(c_punycodeBase != t, "[IdnMapping.punycode_decode]Expected t != c_punycodeBase |
| | 0 | 846 | | if (w > c_maxint / (c_punycodeBase - t)) |
| | 0 | 847 | | throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii)); |
| | 0 | 848 | | w *= (c_punycodeBase - t); |
| | | 849 | | } |
| | | 850 | | |
| | 0 | 851 | | bias = Adapt(i - oldi, (output.Length - iOutputAfterLastDot - numSurrogatePairs) + 1, oldi == 0) |
| | | 852 | | |
| | | 853 | | /* i was supposed to wrap around from output.Length to 0, */ |
| | | 854 | | /* incrementing n each time, so we'll fix that now: */ |
| | 0 | 855 | | Debug.Assert((output.Length - iOutputAfterLastDot - numSurrogatePairs) + 1 > 0, |
| | 0 | 856 | | "[IdnMapping.punycode_decode]Expected to have added > 0 characters this segment"); |
| | 0 | 857 | | if (i / ((output.Length - iOutputAfterLastDot - numSurrogatePairs) + 1) > c_maxint - n) |
| | 0 | 858 | | throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii)); |
| | 0 | 859 | | n += (int)(i / (output.Length - iOutputAfterLastDot - numSurrogatePairs + 1)); |
| | 0 | 860 | | i %= (output.Length - iOutputAfterLastDot - numSurrogatePairs + 1); |
| | | 861 | | |
| | | 862 | | // Make sure n is legal |
| | 0 | 863 | | if (n < 0 || n > 0x10ffff || (n >= 0xD800 && n <= 0xDFFF)) |
| | 0 | 864 | | throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii)); |
| | | 865 | | |
| | | 866 | | // insert n at position i of the output: Really tricky if we have surrogates |
| | | 867 | | int iUseInsertLocation; |
| | 0 | 868 | | string strTemp = char.ConvertFromUtf32(n); |
| | | 869 | | |
| | | 870 | | // If we have supplimentary characters |
| | 0 | 871 | | if (numSurrogatePairs > 0) |
| | | 872 | | { |
| | | 873 | | // Hard way, we have supplimentary characters |
| | | 874 | | int iCount; |
| | 0 | 875 | | for (iCount = i, iUseInsertLocation = iOutputAfterLastDot; iCount > 0; iCount--, iUseInsertL |
| | | 876 | | { |
| | | 877 | | // If its a surrogate, we have to go one more |
| | 0 | 878 | | if (iUseInsertLocation >= output.Length) |
| | 0 | 879 | | throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii)); |
| | 0 | 880 | | if (char.IsSurrogate(output[iUseInsertLocation])) |
| | 0 | 881 | | iUseInsertLocation++; |
| | | 882 | | } |
| | | 883 | | } |
| | | 884 | | else |
| | | 885 | | { |
| | | 886 | | // No Supplementary chars yet, just add i |
| | 0 | 887 | | iUseInsertLocation = iOutputAfterLastDot + i; |
| | | 888 | | } |
| | | 889 | | |
| | | 890 | | // Insert it |
| | 0 | 891 | | output.Insert(iUseInsertLocation, strTemp); |
| | | 892 | | |
| | | 893 | | // If it was a surrogate increment our counter |
| | 0 | 894 | | if (IsSupplementary(n)) |
| | 0 | 895 | | numSurrogatePairs++; |
| | | 896 | | |
| | | 897 | | // Index gets updated |
| | 0 | 898 | | i++; |
| | | 899 | | } |
| | | 900 | | |
| | | 901 | | // Do BIDI testing |
| | 0 | 902 | | bool bRightToLeft = false; |
| | | 903 | | |
| | | 904 | | // Check for RTL. If right-to-left, then 1st & last chars must be RTL |
| | 0 | 905 | | StrongBidiCategory eBidi = CharUnicodeInfo.GetBidiCategory(output, iOutputAfterLastDot); |
| | 0 | 906 | | if (eBidi == StrongBidiCategory.StrongRightToLeft) |
| | | 907 | | { |
| | | 908 | | // It has to be right to left. |
| | 0 | 909 | | bRightToLeft = true; |
| | | 910 | | } |
| | | 911 | | |
| | | 912 | | // Check the rest of them to make sure RTL/LTR is consistent |
| | 0 | 913 | | for (int iTest = iOutputAfterLastDot; iTest < output.Length; iTest++) |
| | | 914 | | { |
| | | 915 | | // This might happen if we run into a pair |
| | 0 | 916 | | if (char.IsLowSurrogate(output[iTest])) |
| | | 917 | | continue; |
| | | 918 | | |
| | | 919 | | // Check to see if its LTR |
| | 0 | 920 | | eBidi = CharUnicodeInfo.GetBidiCategory(output, iTest); |
| | 0 | 921 | | if ((bRightToLeft && eBidi == StrongBidiCategory.StrongLeftToRight) || |
| | 0 | 922 | | (!bRightToLeft && eBidi == StrongBidiCategory.StrongRightToLeft)) |
| | 0 | 923 | | throw new ArgumentException(SR.Argument_IdnBadBidi, nameof(ascii)); |
| | | 924 | | } |
| | | 925 | | |
| | | 926 | | // Its also a requirement that the last one be RTL if 1st is RTL |
| | 0 | 927 | | if (bRightToLeft && eBidi != StrongBidiCategory.StrongRightToLeft) |
| | | 928 | | { |
| | | 929 | | // Oops, last wasn't RTL, last should be RTL if first is RTL |
| | 0 | 930 | | throw new ArgumentException(SR.Argument_IdnBadBidi, nameof(ascii)); |
| | | 931 | | } |
| | | 932 | | } |
| | | 933 | | |
| | | 934 | | // See if this label was too long |
| | 0 | 935 | | if (iNextDot - iAfterLastDot > c_labelLimit) |
| | 0 | 936 | | throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(ascii)); |
| | | 937 | | |
| | | 938 | | // Done with this segment, add dot if necessary |
| | 0 | 939 | | if (iNextDot != ascii.Length) |
| | 0 | 940 | | output.Append('.'); |
| | | 941 | | |
| | 0 | 942 | | iAfterLastDot = iNextDot + 1; |
| | 0 | 943 | | iOutputAfterLastDot = output.Length; |
| | | 944 | | } |
| | | 945 | | |
| | | 946 | | // Throw if we're too long |
| | 0 | 947 | | if (output.Length > c_defaultNameLimit - (IsDot(output[^1]) ? 0 : 1)) |
| | 0 | 948 | | throw new ArgumentException(SR.Format(SR.Argument_IdnBadNameSize, c_defaultNameLimit - (IsDot(output[^1] |
| | | 949 | | |
| | | 950 | | // Return our output string |
| | 0 | 951 | | return output.ToString(); |
| | | 952 | | } |
| | | 953 | | |
| | | 954 | | // DecodeDigit(cp) returns the numeric value of a basic code */ |
| | | 955 | | // point (for use in representing integers) in the range 0 to */ |
| | | 956 | | // c_punycodeBase-1, or <0 if cp is does not represent a value. */ |
| | | 957 | | |
| | | 958 | | private static int DecodeDigit(char cp) |
| | | 959 | | { |
| | 0 | 960 | | if (char.IsAsciiDigit(cp)) |
| | 0 | 961 | | return cp - '0' + 26; |
| | | 962 | | |
| | | 963 | | // Two flavors for case differences |
| | 0 | 964 | | if (char.IsAsciiLetterLower(cp)) |
| | 0 | 965 | | return cp - 'a'; |
| | | 966 | | |
| | 0 | 967 | | if (char.IsAsciiLetterUpper(cp)) |
| | 0 | 968 | | return cp - 'A'; |
| | | 969 | | |
| | | 970 | | // Expected 0-9, A-Z or a-z, everything else is illegal |
| | 0 | 971 | | throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(cp)); |
| | | 972 | | } |
| | | 973 | | |
| | | 974 | | private static int Adapt(int delta, int numpoints, bool firsttime) |
| | | 975 | | { |
| | | 976 | | uint k; |
| | | 977 | | |
| | 0 | 978 | | delta = firsttime ? delta / c_damp : delta / 2; |
| | 0 | 979 | | Debug.Assert(numpoints != 0, "[IdnMapping.adapt]Expected non-zero numpoints."); |
| | 0 | 980 | | delta += delta / numpoints; |
| | | 981 | | |
| | 0 | 982 | | for (k = 0; delta > ((c_punycodeBase - c_tmin) * c_tmax) / 2; k += c_punycodeBase) |
| | | 983 | | { |
| | 0 | 984 | | delta /= c_punycodeBase - c_tmin; |
| | | 985 | | } |
| | | 986 | | |
| | 0 | 987 | | Debug.Assert(delta + c_skew != 0, "[IdnMapping.adapt]Expected non-zero delta+skew."); |
| | 0 | 988 | | return (int)(k + (c_punycodeBase - c_tmin + 1) * delta / (delta + c_skew)); |
| | | 989 | | } |
| | | 990 | | |
| | | 991 | | /* EncodeBasic(bcp,flag) forces a basic code point to lowercase */ |
| | | 992 | | /* if flag is false, uppercase if flag is true, and returns */ |
| | | 993 | | /* the resulting code point. The code point is unchanged if it */ |
| | | 994 | | /* is caseless. The behavior is undefined if bcp is not a basic */ |
| | | 995 | | /* code point. */ |
| | | 996 | | |
| | | 997 | | private static char EncodeBasic(char bcp) |
| | | 998 | | { |
| | 0 | 999 | | if (char.IsAsciiLetterUpper(bcp)) |
| | 0 | 1000 | | bcp += (char)('a' - 'A'); |
| | | 1001 | | |
| | 0 | 1002 | | return bcp; |
| | | 1003 | | } |
| | | 1004 | | |
| | | 1005 | | /* EncodeDigit(d,flag) returns the basic code point whose value */ |
| | | 1006 | | /* (when used for representing integers) is d, which needs to be in */ |
| | | 1007 | | /* the range 0 to punycodeBase-1. The lowercase form is used unless flag is */ |
| | | 1008 | | /* true, in which case the uppercase form is used. */ |
| | | 1009 | | |
| | | 1010 | | private static char EncodeDigit(int d) |
| | | 1011 | | { |
| | 0 | 1012 | | Debug.Assert(d >= 0 && d < c_punycodeBase, "[IdnMapping.encode_digit]Expected 0 <= d < punycodeBase"); |
| | | 1013 | | // 26-35 map to ASCII 0-9 |
| | 0 | 1014 | | if (d > 25) return (char)(d - 26 + '0'); |
| | | 1015 | | |
| | | 1016 | | // 0-25 map to a-z or A-Z |
| | 0 | 1017 | | return (char)(d + 'a'); |
| | | 1018 | | } |
| | | 1019 | | } |
| | | 1020 | | } |
| | | 1021 | | |