< Summary

Line coverage
0%
Covered lines: 0
Uncovered lines: 510
Coverable lines: 510
Total lines: 1389
Line coverage: 0%
Branch coverage
0%
Covered branches: 0
Total branches: 398
Branch coverage: 0%
Method coverage

Feature is only available for sponsors

Upgrade to PRO version

Metrics

File(s)

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.CoreLib/src/System/Globalization/IdnMapping.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4// This file contains the IDN functions and implementation.
 5//
 6// This allows encoding of non-ASCII domain names in a "punycode" form,
 7// for example:
 8//
 9//     \u5B89\u5BA4\u5948\u7F8E\u6075-with-SUPER-MONKEYS
 10//
 11// is encoded as:
 12//
 13//     xn---with-SUPER-MONKEYS-pc58ag80a8qai00g7n9n
 14//
 15// Additional options are provided to allow unassigned IDN characters and
 16// to validate according to the Std3ASCII Rules (like DNS names).
 17//
 18// There are also rules regarding bidirectionality of text and the length
 19// of segments.
 20//
 21// For additional rules see also:
 22//  RFC 3490 - Internationalizing Domain Names in Applications (IDNA)
 23//  RFC 3491 - Nameprep: A Stringprep Profile for Internationalized Domain Names (IDN)
 24//  RFC 3492 - Punycode: A Bootstring encoding of Unicode for Internationalized Domain Names in Applications (IDNA)
 25
 26using System.Diagnostics;
 27using System.Diagnostics.CodeAnalysis;
 28using System.Runtime.CompilerServices;
 29using System.Runtime.InteropServices;
 30using System.Text;
 31
 32namespace System.Globalization
 33{
 34    // IdnMapping class used to map names to Punycode
 35    public sealed partial class IdnMapping
 36    {
 37        private bool _allowUnassigned;
 38        private bool _useStd3AsciiRules;
 39
 040        public IdnMapping()
 41        {
 042        }
 43
 44        public bool AllowUnassigned
 45        {
 046            get => _allowUnassigned;
 047            set => _allowUnassigned = value;
 48        }
 49
 50        public bool UseStd3AsciiRules
 51        {
 052            get => _useStd3AsciiRules;
 053            set => _useStd3AsciiRules = value;
 54        }
 55
 56        // Gets ASCII (Punycode) version of the string
 57        public string GetAscii(string unicode) =>
 058            GetAscii(unicode, 0);
 59
 60        public string GetAscii(string unicode, int index)
 61        {
 062            ArgumentNullException.ThrowIfNull(unicode);
 63
 064            return GetAscii(unicode, index, unicode.Length - index);
 65        }
 66
 67        public string GetAscii(string unicode, int index, int count)
 68        {
 069            ArgumentNullException.ThrowIfNull(unicode);
 70
 071            ArgumentOutOfRangeException.ThrowIfNegative(index);
 072            ArgumentOutOfRangeException.ThrowIfNegative(count);
 073            if (index > unicode.Length)
 074                throw new ArgumentOutOfRangeException(nameof(index), SR.ArgumentOutOfRange_IndexMustBeLessOrEqual);
 075            if (index > unicode.Length - count)
 076                throw new ArgumentOutOfRangeException(nameof(unicode), SR.ArgumentOutOfRange_IndexCountBuffer);
 77
 078            if (count == 0)
 79            {
 080                throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode));
 81            }
 082            if (unicode[index + count - 1] == 0)
 83            {
 084                throw new ArgumentException(SR.Format(SR.Argument_InvalidCharSequence, index + count - 1), nameof(unicod
 85            }
 86
 087            if (GlobalizationMode.Invariant)
 88            {
 089                return GetAsciiInvariant(unicode, index, count);
 90            }
 91
 092            return GlobalizationMode.UseNls ?
 093                NlsGetAsciiCore(unicode, index, count) :
 094                IcuGetAsciiCore(unicode, index, count);
 95        }
 96
 97        /// <summary>
 98        /// Encodes a Unicode domain name to its ASCII (Punycode) equivalent.
 99        /// </summary>
 100        /// <param name="unicode">The Unicode domain name to convert.</param>
 101        /// <param name="destination">The buffer to write the ASCII result to. This buffer must not overlap with <paramr
 102        /// <param name="charsWritten">When this method returns, contains the number of characters that were written to 
 103        /// <returns><see langword="true"/> if the conversion was successful and the result was written to <paramref nam
 104        /// <exception cref="ArgumentException"><paramref name="unicode"/> is invalid based on the <see cref="AllowUnass
 105        public bool TryGetAscii(ReadOnlySpan<char> unicode, Span<char> destination, out int charsWritten)
 106        {
 0107            if (unicode.Length == 0)
 108            {
 0109                throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode));
 110            }
 111
 0112            if (unicode[^1] == 0)
 113            {
 0114                throw new ArgumentException(SR.Format(SR.Argument_InvalidCharSequence, unicode.Length - 1), nameof(unico
 115            }
 116
 0117            if (unicode.Overlaps(destination))
 118            {
 0119                ThrowHelper.ThrowArgumentException(ExceptionResource.InvalidOperation_SpanOverlappedOperation);
 120            }
 121
 0122            if (GlobalizationMode.Invariant)
 123            {
 0124                return TryGetAsciiInvariant(unicode, destination, out charsWritten);
 125            }
 126
 0127            return GlobalizationMode.UseNls ?
 0128                NlsTryGetAsciiCore(unicode, destination, out charsWritten) :
 0129                IcuTryGetAsciiCore(unicode, destination, out charsWritten);
 130        }
 131
 132        // Gets Unicode version of the string.  Normalized and limited to IDNA characters.
 133        public string GetUnicode(string ascii) =>
 0134            GetUnicode(ascii, 0);
 135
 136        public string GetUnicode(string ascii, int index)
 137        {
 0138            ArgumentNullException.ThrowIfNull(ascii);
 139
 0140            return GetUnicode(ascii, index, ascii.Length - index);
 141        }
 142
 143        public string GetUnicode(string ascii, int index, int count)
 144        {
 0145            ArgumentNullException.ThrowIfNull(ascii);
 146
 0147            ArgumentOutOfRangeException.ThrowIfNegative(index);
 0148            ArgumentOutOfRangeException.ThrowIfNegative(count);
 0149            if (index > ascii.Length)
 0150                throw new ArgumentOutOfRangeException(nameof(index), SR.ArgumentOutOfRange_IndexMustBeLessOrEqual);
 0151            if (index > ascii.Length - count)
 0152                throw new ArgumentOutOfRangeException(nameof(ascii), SR.ArgumentOutOfRange_IndexCountBuffer);
 153
 154            // This is a case (i.e. explicitly null-terminated input) where behavior in .NET and Win32 intentionally dif
 155            // The .NET APIs should (and did in v4.0 and earlier) throw an ArgumentException on input that includes a te
 156            // The Win32 APIs fail on an embedded null, but not on a terminating null.
 0157            if (count > 0 && ascii[index + count - 1] == (char)0)
 0158                throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii));
 159
 0160            if (GlobalizationMode.Invariant)
 161            {
 0162                return GetUnicodeInvariant(ascii, index, count);
 163            }
 164
 0165            return GlobalizationMode.UseNls ?
 0166                NlsGetUnicodeCore(ascii, index, count) :
 0167                IcuGetUnicodeCore(ascii, index, count);
 168        }
 169
 170        /// <summary>
 171        /// Decodes one or more encoded domain name labels to a string of Unicode characters.
 172        /// </summary>
 173        /// <param name="ascii">The ASCII domain name to convert. The string may contain one or more labels, where each 
 174        /// <param name="destination">The buffer to write the Unicode result to. This buffer must not overlap with <para
 175        /// <param name="charsWritten">When this method returns, contains the number of characters that were written to 
 176        /// <returns><see langword="true"/> if the conversion was successful and the result was written to <paramref nam
 177        /// <exception cref="ArgumentException"><paramref name="ascii"/> is invalid based on the <see cref="AllowUnassig
 178        public bool TryGetUnicode(ReadOnlySpan<char> ascii, Span<char> destination, out int charsWritten)
 179        {
 180            // This is a case (i.e. explicitly null-terminated input) where behavior in .NET and Win32 intentionally dif
 181            // The .NET APIs should (and did in v4.0 and earlier) throw an ArgumentException on input that includes a te
 182            // The Win32 APIs fail on an embedded null, but not on a terminating null.
 0183            if (ascii.Length > 0 && ascii[^1] == (char)0)
 184            {
 0185                throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii));
 186            }
 187
 0188            if (ascii.Overlaps(destination))
 189            {
 0190                ThrowHelper.ThrowArgumentException(ExceptionResource.InvalidOperation_SpanOverlappedOperation);
 191            }
 192
 0193            if (GlobalizationMode.Invariant)
 194            {
 0195                return TryGetUnicodeInvariant(ascii, destination, out charsWritten);
 196            }
 197
 0198            return GlobalizationMode.UseNls ?
 0199                NlsTryGetUnicodeCore(ascii, destination, out charsWritten) :
 0200                IcuTryGetUnicodeCore(ascii, destination, out charsWritten);
 201        }
 202
 203        public override bool Equals([NotNullWhen(true)] object? obj) =>
 0204            obj is IdnMapping that &&
 0205            _allowUnassigned == that._allowUnassigned &&
 0206            _useStd3AsciiRules == that._useStd3AsciiRules;
 207
 208        public override int GetHashCode() =>
 0209            (_allowUnassigned ? 100 : 200) + (_useStd3AsciiRules ? 1000 : 2000);
 210
 211        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 212        private static string GetStringForOutput(string? originalString, ReadOnlySpan<char> input, ReadOnlySpan<char> ou
 213        {
 0214            Debug.Assert(input.Length > 0);
 215
 0216            if (originalString is not null &&
 0217                originalString.Length == input.Length &&
 0218                input.Length == output.Length &&
 0219                Ordinal.EqualsIgnoreCase(ref MemoryMarshal.GetReference(input), ref MemoryMarshal.GetReference(output), 
 220            {
 0221                return originalString;
 222            }
 223
 0224            return output.ToString();
 225        }
 226
 227        //
 228        // Invariant implementation
 229        //
 230
 231        private const char c_delimiter = '-';
 232        private const string c_strAcePrefix = "xn--";
 233        private const int c_labelLimit = 63;          // Not including dots
 234        private const int c_defaultNameLimit = 255;   // Including dots
 235        private const int c_initialN = 0x80;
 236        private const int c_maxint = 0x7ffffff;
 237        private const int c_initialBias = 72;
 238        private const int c_punycodeBase = 36;
 239        private const int c_tmin = 1;
 240        private const int c_tmax = 26;
 241        private const int c_skew = 38;
 242        private const int c_damp = 700;
 243
 244        private string GetAsciiInvariant(string unicodeString, int index, int count)
 245        {
 0246            ReadOnlySpan<char> unicode = unicodeString.AsSpan(index, count);
 247
 248            // Check for ASCII only string, which will be unchanged
 0249            if (ValidateStd3AndAscii(unicode, UseStd3AsciiRules, true))
 250            {
 251                // Return original string if the entire string was requested and it doesn't need modification
 0252                if (index == 0 && count == unicodeString.Length)
 253                {
 0254                    return unicodeString;
 255                }
 256
 0257                return unicode.ToString();
 258            }
 259
 260            // Cannot be null terminated (normalization won't help us with this one, and
 261            // may have returned false before checking the whole string above)
 0262            Debug.Assert(unicode.Length >= 1, "[IdnMapping.GetAscii] Expected 0 length strings to fail before now.");
 0263            if (unicode[^1] <= 0x1f)
 264            {
 0265                throw new ArgumentException(SR.Format(SR.Argument_InvalidCharSequence, unicode.Length - 1), nameof(unico
 266            }
 267
 268            // May need to check Std3 rules again for non-ascii
 0269            if (UseStd3AsciiRules)
 270            {
 0271                ValidateStd3AndAscii(unicode, true, false);
 272            }
 273
 274            // Go ahead and encode it
 0275            return PunycodeEncode(unicode);
 276        }
 277
 278        private bool TryGetAsciiInvariant(ReadOnlySpan<char> unicode, Span<char> destination, out int charsWritten)
 279        {
 280            // Check for ASCII only string, which will be unchanged
 0281            if (ValidateStd3AndAscii(unicode, UseStd3AsciiRules, true))
 282            {
 0283                if (unicode.Length <= destination.Length)
 284                {
 0285                    unicode.CopyTo(destination);
 0286                    charsWritten = unicode.Length;
 0287                    return true;
 288                }
 289
 0290                charsWritten = 0;
 0291                return false;
 292            }
 293
 294            // Cannot be null terminated (normalization won't help us with this one, and
 295            // may have returned false before checking the whole string above)
 0296            Debug.Assert(unicode.Length >= 1, "[IdnMapping.GetAscii] Expected 0 length strings to fail before now.");
 0297            if (unicode[^1] <= 0x1f)
 298            {
 0299                throw new ArgumentException(SR.Format(SR.Argument_InvalidCharSequence, unicode.Length - 1), nameof(unico
 300            }
 301
 302            // May need to check Std3 rules again for non-ascii
 0303            if (UseStd3AsciiRules)
 304            {
 0305                ValidateStd3AndAscii(unicode, true, false);
 306            }
 307
 308            // Go ahead and encode it
 0309            string result = PunycodeEncode(unicode);
 0310            if (result.Length <= destination.Length)
 311            {
 0312                result.CopyTo(destination);
 0313                charsWritten = result.Length;
 0314                return true;
 315            }
 316
 0317            charsWritten = 0;
 0318            return false;
 319        }
 320
 321        // See if we're only ASCII
 322        private static bool ValidateStd3AndAscii(ReadOnlySpan<char> unicode, bool bUseStd3, bool bCheckAscii)
 323        {
 324            // If its empty, then its too small
 0325            if (unicode.Length == 0)
 0326                throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode));
 327
 0328            int iLastDot = -1;
 329
 330            // Loop the whole string
 0331            for (int i = 0; i < unicode.Length; i++)
 332            {
 333                // Aren't allowing control chars (or 7f, but idn tables catch that, they don't catch \0 at end though)
 0334                if (unicode[i] <= 0x1f)
 335                {
 0336                    throw new ArgumentException(SR.Format(SR.Argument_InvalidCharSequence, i), nameof(unicode));
 337                }
 338
 339                // If its Unicode or a control character, return false (non-ascii)
 0340                if (bCheckAscii && unicode[i] >= 0x7f)
 0341                    return false;
 342
 343                // Check for dots
 0344                if (IsDot(unicode[i]))
 345                {
 346                    // Can't have 2 dots in a row
 0347                    if (i == iLastDot + 1)
 0348                        throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode));
 349
 350                    // If its too far between dots then fail
 0351                    if (i - iLastDot > c_labelLimit + 1)
 0352                        throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode));
 353
 354                    // If validating Std3, then char before dot can't be - char
 0355                    if (bUseStd3 && i > 0)
 0356                        ValidateStd3(unicode[i - 1], true);
 357
 358                    // Remember where the last dot is
 0359                    iLastDot = i;
 0360                    continue;
 361                }
 362
 363                // If necessary, make sure its a valid std3 character
 0364                if (bUseStd3)
 365                {
 0366                    ValidateStd3(unicode[i], i == iLastDot + 1);
 367                }
 368            }
 369
 370            // If we never had a dot, then we need to be shorter than the label limit
 0371            if (iLastDot == -1 && unicode.Length > c_labelLimit)
 0372                throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode));
 373
 374            // Need to validate entire string length, 1 shorter if last char wasn't a dot
 0375            if (unicode.Length > c_defaultNameLimit - (IsDot(unicode[^1]) ? 0 : 1))
 0376                throw new ArgumentException(SR.Format(SR.Argument_IdnBadNameSize,
 0377                                                        c_defaultNameLimit - (IsDot(unicode[^1]) ? 0 : 1)), nameof(unico
 378
 379            // If last char wasn't a dot we need to check for trailing -
 0380            if (bUseStd3 && !IsDot(unicode[^1]))
 0381                ValidateStd3(unicode[^1], true);
 382
 0383            return true;
 384        }
 385
 386        /* PunycodeEncode() converts Unicode to Punycode.  The input     */
 387        /* is represented as an array of Unicode code points (not code    */
 388        /* units; surrogate pairs are not allowed), and the output        */
 389        /* will be represented as an array of ASCII code points.  The     */
 390        /* output string is *not* null-terminated; it will contain        */
 391        /* zeros if and only if the input contains zeros.  (Of course     */
 392        /* the caller can leave room for a terminator and add one if      */
 393        /* needed.)  The input_length is the number of code points in     */
 394        /* the input.  The output_length is an in/out argument: the       */
 395        /* caller passes in the maximum number of code points that it     */
 396
 397        /* can receive, and on successful return it will contain the      */
 398        /* number of code points actually output.  The case_flags array   */
 399        /* holds input_length boolean values, where nonzero suggests that */
 400        /* the corresponding Unicode character be forced to uppercase     */
 401        /* after being decoded (if possible), and zero suggests that      */
 402        /* it be forced to lowercase (if possible).  ASCII code points    */
 403        /* are encoded literally, except that ASCII letters are forced    */
 404        /* to uppercase or lowercase according to the corresponding       */
 405        /* uppercase flags.  If case_flags is a null pointer then ASCII   */
 406        /* letters are left as they are, and other code points are        */
 407        /* treated as if their uppercase flags were zero.  The return     */
 408        /* value can be any of the punycode_status values defined above   */
 409        /* except punycode_bad_input; if not punycode_success, then       */
 410        /* output_size and output might contain garbage.                  */
 411        private static string PunycodeEncode(ReadOnlySpan<char> unicode)
 412        {
 413            // 0 length strings aren't allowed
 0414            if (unicode.Length == 0)
 0415                throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode));
 416
 0417            StringBuilder output = new StringBuilder(unicode.Length);
 0418            int iNextDot = 0;
 0419            int iAfterLastDot = 0;
 0420            int iOutputAfterLastDot = 0;
 421
 422            // Find the next dot
 0423            while (iNextDot < unicode.Length)
 424            {
 425                // Legal "dot" separators (i.e: . in www.microsoft.com)
 426                const string DotSeparators = ".\u3002\uFF0E\uFF61";
 427
 428                // Find end of this segment
 0429                iNextDot = unicode.Slice(iAfterLastDot).IndexOfAny(DotSeparators);
 0430                iNextDot = iNextDot < 0 ? unicode.Length : iNextDot + iAfterLastDot;
 431
 432                // Only allowed to have empty . section at end (www.microsoft.com.)
 0433                if (iNextDot == iAfterLastDot)
 434                {
 435                    // Only allowed to have empty sections as trailing .
 0436                    if (iNextDot != unicode.Length)
 0437                        throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode));
 438                    // Last dot, stop
 439                    break;
 440                }
 441
 442                // We'll need an Ace prefix
 0443                output.Append(c_strAcePrefix);
 444
 445                // Everything resets every segment.
 0446                bool bRightToLeft = false;
 447
 448                // Check for RTL.  If right-to-left, then 1st & last chars must be RTL
 0449                StrongBidiCategory eBidi = CharUnicodeInfo.GetBidiCategory(unicode, iAfterLastDot);
 0450                if (eBidi == StrongBidiCategory.StrongRightToLeft)
 451                {
 452                    // It has to be right to left.
 0453                    bRightToLeft = true;
 454
 455                    // Check last char
 0456                    int iTest = iNextDot - 1;
 0457                    if (char.IsLowSurrogate(unicode[iTest]))
 458                    {
 0459                        iTest--;
 460                    }
 461
 0462                    eBidi = CharUnicodeInfo.GetBidiCategory(unicode, iTest);
 0463                    if (eBidi != StrongBidiCategory.StrongRightToLeft)
 464                    {
 465                        // Oops, last wasn't RTL, last should be RTL if first is RTL
 0466                        throw new ArgumentException(SR.Argument_IdnBadBidi, nameof(unicode));
 467                    }
 468                }
 469
 470                // Handle the basic code points
 471                int basicCount;
 0472                int numProcessed = 0;           // Num code points that have been processed so far (this segment)
 0473                for (basicCount = iAfterLastDot; basicCount < iNextDot; basicCount++)
 474                {
 475                    // Can't be lonely surrogate because it would've thrown in normalization
 0476                    Debug.Assert(!char.IsLowSurrogate(unicode[basicCount]), "[IdnMapping.punycode_encode]Unexpected low 
 477
 478                    // Double check our bidi rules
 0479                    StrongBidiCategory testBidi = CharUnicodeInfo.GetBidiCategory(unicode, basicCount);
 480
 481                    // If we're RTL, we can't have LTR chars
 0482                    if (bRightToLeft && testBidi == StrongBidiCategory.StrongLeftToRight)
 483                    {
 484                        // Oops, throw error
 0485                        throw new ArgumentException(SR.Argument_IdnBadBidi, nameof(unicode));
 486                    }
 487
 488                    // If we're not RTL we can't have RTL chars
 0489                    if (!bRightToLeft && testBidi == StrongBidiCategory.StrongRightToLeft)
 490                    {
 491                        // Oops, throw error
 0492                        throw new ArgumentException(SR.Argument_IdnBadBidi, nameof(unicode));
 493                    }
 494
 495                    // If its basic then add it
 0496                    if (Basic(unicode[basicCount]))
 497                    {
 0498                        output.Append(EncodeBasic(unicode[basicCount]));
 0499                        numProcessed++;
 500                    }
 501                    // If its a surrogate, skip the next since our bidi category tester doesn't handle it.
 0502                    else if (basicCount + 1 < iNextDot && char.IsSurrogatePair(unicode[basicCount], unicode[basicCount +
 0503                        basicCount++;
 504                }
 505
 0506                int numBasicCodePoints = numProcessed;     // number of basic code points
 507
 508                // Stop if we ONLY had basic code points
 0509                if (numBasicCodePoints == iNextDot - iAfterLastDot)
 510                {
 511                    // Get rid of xn-- and this segments done
 0512                    output.Remove(iOutputAfterLastDot, c_strAcePrefix.Length);
 513                }
 514                else
 515                {
 516                    // If it has some non-basic code points the input cannot start with xn--
 0517                    if (unicode.Slice(iAfterLastDot).StartsWith(c_strAcePrefix, StringComparison.OrdinalIgnoreCase))
 0518                        throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(unicode));
 519
 520                    // Need to do ACE encoding
 0521                    int numSurrogatePairs = 0;            // number of surrogate pairs so far
 522
 523                    // Add a delimiter (-) if we had any basic code points (between basic and encoded pieces)
 0524                    if (numBasicCodePoints > 0)
 525                    {
 0526                        output.Append(c_delimiter);
 527                    }
 528
 529                    // Initialize the state
 0530                    int n = c_initialN;
 0531                    int delta = 0;
 0532                    int bias = c_initialBias;
 533
 534                    // Main loop
 0535                    while (numProcessed < (iNextDot - iAfterLastDot))
 536                    {
 537                        /* All non-basic code points < n have been     */
 538                        /* handled already.  Find the next larger one: */
 539                        int j;
 540                        int m;
 541                        int test;
 0542                        for (m = c_maxint, j = iAfterLastDot;
 0543                             j < iNextDot;
 0544                             j += IsSupplementary(test) ? 2 : 1)
 545                        {
 0546                            test = GetCodePoint(unicode, j);
 0547                            if (test >= n && test < m) m = test;
 548                        }
 549
 550                        /* Increase delta enough to advance the decoder's    */
 551                        /* <n,i> state to <m,0>, but guard against overflow: */
 0552                        delta += (int)((m - n) * ((numProcessed - numSurrogatePairs) + 1));
 0553                        Debug.Assert(delta > 0, "[IdnMapping.cs]1 punycode_encode - delta overflowed int");
 0554                        n = m;
 555
 0556                        for (j = iAfterLastDot; j < iNextDot; j += IsSupplementary(test) ? 2 : 1)
 557                        {
 558                            // Make sure we're aware of surrogates
 0559                            test = GetCodePoint(unicode, j);
 560
 561                            // Adjust for character position (only the chars in our string already, some
 562                            // haven't been processed.
 563
 0564                            if (test < n)
 565                            {
 0566                                delta++;
 0567                                Debug.Assert(delta > 0, "[IdnMapping.cs]2 punycode_encode - delta overflowed int");
 568                            }
 569
 0570                            if (test == n)
 571                            {
 572                                // Represent delta as a generalized variable-length integer:
 573                                int q, k;
 0574                                for (q = delta, k = c_punycodeBase; ; k += c_punycodeBase)
 575                                {
 0576                                    int t = k <= bias ? c_tmin : k >= bias + c_tmax ? c_tmax : k - bias;
 0577                                    if (q < t) break;
 0578                                    Debug.Assert(c_punycodeBase != t, "[IdnMapping.punycode_encode]Expected c_punycodeBa
 0579                                    output.Append(EncodeDigit(t + (q - t) % (c_punycodeBase - t)));
 0580                                    q = (q - t) / (c_punycodeBase - t);
 581                                }
 582
 0583                                output.Append(EncodeDigit(q));
 0584                                bias = Adapt(delta, (numProcessed - numSurrogatePairs) + 1, numProcessed == numBasicCode
 0585                                delta = 0;
 0586                                numProcessed++;
 587
 0588                                if (IsSupplementary(m))
 589                                {
 0590                                    numProcessed++;
 0591                                    numSurrogatePairs++;
 592                                }
 593                            }
 594                        }
 0595                        ++delta;
 0596                        ++n;
 0597                        Debug.Assert(delta > 0, "[IdnMapping.cs]3 punycode_encode - delta overflowed int");
 598                    }
 599                }
 600
 601                // Make sure its not too big
 0602                if (output.Length - iOutputAfterLastDot > c_labelLimit)
 0603                    throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(unicode));
 604
 605                // Done with this segment, add dot if necessary
 0606                if (iNextDot != unicode.Length)
 0607                    output.Append('.');
 608
 0609                iAfterLastDot = iNextDot + 1;
 0610                iOutputAfterLastDot = output.Length;
 611            }
 612
 613            // Throw if we're too long
 0614            if (output.Length > c_defaultNameLimit - (IsDot(unicode[^1]) ? 0 : 1))
 0615                throw new ArgumentException(SR.Format(SR.Argument_IdnBadNameSize,
 0616                                                c_defaultNameLimit - (IsDot(unicode[^1]) ? 0 : 1)), nameof(unicode));
 617            // Return our output string
 0618            return output.ToString();
 619        }
 620
 621        // Is it a dot?
 622        // are we U+002E (., full stop), U+3002 (ideographic full stop), U+FF0E (fullwidth full stop), or
 623        // U+FF61 (halfwidth ideographic full stop).
 624        // Note: IDNA Normalization gets rid of dots now, but testing for last dot is before normalization
 625        private static bool IsDot(char c) =>
 0626            c == '.' || c == '\u3002' || c == '\uFF0E' || c == '\uFF61';
 627
 628        private static bool IsSupplementary(int cTest) =>
 0629            cTest >= 0x10000;
 630
 631        private static bool Basic(uint cp) =>
 632            // Is it in ASCII range?
 0633            cp < 0x80;
 634
 635        private static int GetCodePoint(ReadOnlySpan<char> s, int index)
 636        {
 637            // Check if the character at index is a high surrogate.
 0638            if (char.IsHighSurrogate(s[index]) && index + 1 < s.Length && char.IsLowSurrogate(s[index + 1]))
 639            {
 0640                return char.ConvertToUtf32(s[index], s[index + 1]);
 641            }
 642
 0643            return s[index];
 644        }
 645
 646        // Validate Std3 rules for a character
 647        private static void ValidateStd3(char c, bool bNextToDot)
 648        {
 649            // Check for illegal characters
 0650            if (c <= ',' || c == '/' || (c >= ':' && c <= '@') ||      // Lots of characters not allowed
 0651                (c >= '[' && c <= '`') || (c >= '{' && c <= (char)0x7F) ||
 0652                (c == '-' && bNextToDot))
 0653                throw new ArgumentException(SR.Format(SR.Argument_IdnBadStd3, c), nameof(c));
 0654        }
 655
 656        private string GetUnicodeInvariant(string ascii, int index, int count)
 657        {
 658            // Convert Punycode to Unicode
 0659            string asciiSlice = ascii.Substring(index, count);
 0660            string strUnicode = PunycodeDecode(asciiSlice);
 661
 662            // Output name MUST obey IDNA rules & round trip (casing differences are allowed)
 0663            string asciiRoundtrip = GetAscii(strUnicode);
 0664            if (!asciiRoundtrip.Equals(asciiSlice, StringComparison.OrdinalIgnoreCase))
 665            {
 0666                throw new ArgumentException(SR.Argument_IdnIllegalName, nameof(ascii));
 667            }
 668
 669            // If the ASCII round-trip equals the original string, return it as-is (no allocation)
 0670            if (index == 0 && count == ascii.Length && strUnicode.Equals(ascii, StringComparison.OrdinalIgnoreCase))
 671            {
 0672                return ascii;
 673            }
 674
 0675            return strUnicode;
 676        }
 677
 678        private bool TryGetUnicodeInvariant(ReadOnlySpan<char> ascii, Span<char> destination, out int charsWritten)
 679        {
 680            // Convert the span to a string for PunycodeDecode since it uses string operations extensively
 0681            string asciiString = ascii.ToString();
 682
 683            // Convert Punycode to Unicode
 0684            string strUnicode = PunycodeDecode(asciiString);
 685
 686            // Output name MUST obey IDNA rules & round trip (casing differences are allowed)
 0687            if (!asciiString.Equals(GetAscii(strUnicode), StringComparison.OrdinalIgnoreCase))
 688            {
 0689                throw new ArgumentException(SR.Argument_IdnIllegalName, nameof(ascii));
 690            }
 691
 0692            if (strUnicode.Length <= destination.Length)
 693            {
 0694                strUnicode.CopyTo(destination);
 0695                charsWritten = strUnicode.Length;
 0696                return true;
 697            }
 698
 0699            charsWritten = 0;
 0700            return false;
 701        }
 702
 703        /* PunycodeDecode() converts Punycode to Unicode.  The input is  */
 704        /* represented as an array of ASCII code points, and the output   */
 705        /* will be represented as an array of Unicode code points.  The   */
 706        /* input_length is the number of code points in the input.  The   */
 707        /* output_length is an in/out argument: the caller passes in      */
 708        /* the maximum number of code points that it can receive, and     */
 709        /* on successful return it will contain the actual number of      */
 710        /* code points output.  The case_flags array needs room for at    */
 711        /* least output_length values, or it can be a null pointer if the */
 712        /* case information is not needed.  A nonzero flag suggests that  */
 713        /* the corresponding Unicode character be forced to uppercase     */
 714        /* by the caller (if possible), while zero suggests that it be    */
 715        /* forced to lowercase (if possible).  ASCII code points are      */
 716        /* output already in the proper case, but their flags will be set */
 717        /* appropriately so that applying the flags would be harmless.    */
 718        /* The return value can be any of the punycode_status values      */
 719        /* defined above; if not punycode_success, then output_length,    */
 720        /* output, and case_flags might contain garbage.  On success, the */
 721        /* decoder will never need to write an output_length greater than */
 722        /* input_length, because of how the encoding is defined.          */
 723
 724        private static string PunycodeDecode(string ascii)
 725        {
 726            // 0 length strings aren't allowed
 0727            if (ascii.Length == 0)
 0728                throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(ascii));
 729
 730            // Throw if we're too long
 0731            if (ascii.Length > c_defaultNameLimit - (IsDot(ascii[^1]) ? 0 : 1))
 0732                throw new ArgumentException(SR.Format(SR.Argument_IdnBadNameSize,
 0733                                            c_defaultNameLimit - (IsDot(ascii[^1]) ? 0 : 1)), nameof(ascii));
 734
 735            // output stringbuilder
 0736            StringBuilder output = new StringBuilder(ascii.Length);
 737
 738            // Dot searching
 0739            int iNextDot = 0;
 0740            int iAfterLastDot = 0;
 0741            int iOutputAfterLastDot = 0;
 742
 0743            while (iNextDot < ascii.Length)
 744            {
 745                // Find end of this segment
 0746                iNextDot = ascii.IndexOf('.', iAfterLastDot);
 0747                if (iNextDot < 0 || iNextDot > ascii.Length)
 0748                    iNextDot = ascii.Length;
 749
 750                // Only allowed to have empty . section at end (www.microsoft.com.)
 0751                if (iNextDot == iAfterLastDot)
 752                {
 753                    // Only allowed to have empty sections as trailing .
 0754                    if (iNextDot != ascii.Length)
 0755                        throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(ascii));
 756
 757                    // Last dot, stop
 758                    break;
 759                }
 760
 761                // In either case it can't be bigger than segment size
 0762                if (iNextDot - iAfterLastDot > c_labelLimit)
 0763                    throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(ascii));
 764
 765                // See if this section's ASCII or ACE
 0766                if (!ascii.AsSpan(iAfterLastDot).StartsWith(c_strAcePrefix, StringComparison.OrdinalIgnoreCase))
 767                {
 768                    // Its ASCII, copy it
 0769                    output.Append(ascii, iAfterLastDot, iNextDot - iAfterLastDot);
 770                }
 771                else
 772                {
 773                    // Not ASCII, bump up iAfterLastDot to be after ACE Prefix
 0774                    iAfterLastDot += c_strAcePrefix.Length;
 775
 776                    // Get number of basic code points (where delimiter is)
 777                    // numBasicCodePoints < 0 if there're no basic code points
 0778                    int iTemp = ascii.LastIndexOf(c_delimiter, iNextDot - 1);
 779
 780                    // Trailing - not allowed
 0781                    if (iTemp == iNextDot - 1)
 0782                        throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii));
 783
 784                    int numBasicCodePoints;
 0785                    if (iTemp <= iAfterLastDot)
 0786                        numBasicCodePoints = 0;
 787                    else
 788                    {
 0789                        numBasicCodePoints = iTemp - iAfterLastDot;
 790
 791                        // Copy all the basic code points, making sure they're all in the allowed range,
 792                        // and losing the casing for all of them.
 0793                        for (int copyAscii = iAfterLastDot; copyAscii < iAfterLastDot + numBasicCodePoints; copyAscii++)
 794                        {
 795                            // Make sure we don't allow unicode in the ascii part
 0796                            if (ascii[copyAscii] > 0x7f)
 0797                                throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii));
 798
 799                            // When appending make sure they get lower cased
 0800                            output.Append((char)(char.IsAsciiLetterUpper(ascii[copyAscii]) ? ascii[copyAscii] - 'A' + 'a
 801                        }
 802                    }
 803
 804                    // Get ready for main loop.  Start at beginning if we didn't have any
 805                    // basic code points, otherwise start after the -.
 806                    // asciiIndex will be next character to read from ascii
 0807                    int asciiIndex = iAfterLastDot + (numBasicCodePoints > 0 ? numBasicCodePoints + 1 : 0);
 808
 809                    // initialize our state
 0810                    int n = c_initialN;
 0811                    int bias = c_initialBias;
 0812                    int i = 0;
 813
 814                    int w, k;
 815
 816                    // no Supplementary characters yet
 0817                    int numSurrogatePairs = 0;
 818
 819                    // Main loop, read rest of ascii
 0820                    while (asciiIndex < iNextDot)
 821                    {
 822                        /* Decode a generalized variable-length integer into delta,  */
 823                        /* which gets added to i.  The overflow checking is easier   */
 824                        /* if we increase i as we go, then subtract off its starting */
 825                        /* value at the end to obtain delta.                         */
 0826                        int oldi = i;
 827
 0828                        for (w = 1, k = c_punycodeBase; ; k += c_punycodeBase)
 829                        {
 830                            // Check to make sure we aren't overrunning our ascii string
 0831                            if (asciiIndex >= iNextDot)
 0832                                throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii));
 833
 834                            // decode the digit from the next char
 0835                            int digit = DecodeDigit(ascii[asciiIndex++]);
 836
 0837                            Debug.Assert(w > 0, "[IdnMapping.punycode_decode]Expected w > 0");
 0838                            if (digit > (c_maxint - i) / w)
 0839                                throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii));
 840
 0841                            i += (int)(digit * w);
 0842                            int t = k <= bias ? c_tmin : k >= bias + c_tmax ? c_tmax : k - bias;
 0843                            if (digit < t)
 844                                break;
 0845                            Debug.Assert(c_punycodeBase != t, "[IdnMapping.punycode_decode]Expected t != c_punycodeBase 
 0846                            if (w > c_maxint / (c_punycodeBase - t))
 0847                                throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii));
 0848                            w *= (c_punycodeBase - t);
 849                        }
 850
 0851                        bias = Adapt(i - oldi, (output.Length - iOutputAfterLastDot - numSurrogatePairs) + 1, oldi == 0)
 852
 853                        /* i was supposed to wrap around from output.Length to 0,   */
 854                        /* incrementing n each time, so we'll fix that now: */
 0855                        Debug.Assert((output.Length - iOutputAfterLastDot - numSurrogatePairs) + 1 > 0,
 0856                            "[IdnMapping.punycode_decode]Expected to have added > 0 characters this segment");
 0857                        if (i / ((output.Length - iOutputAfterLastDot - numSurrogatePairs) + 1) > c_maxint - n)
 0858                            throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii));
 0859                        n += (int)(i / (output.Length - iOutputAfterLastDot - numSurrogatePairs + 1));
 0860                        i %= (output.Length - iOutputAfterLastDot - numSurrogatePairs + 1);
 861
 862                        // Make sure n is legal
 0863                        if (n < 0 || n > 0x10ffff || (n >= 0xD800 && n <= 0xDFFF))
 0864                            throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii));
 865
 866                        // insert n at position i of the output:  Really tricky if we have surrogates
 867                        int iUseInsertLocation;
 0868                        string strTemp = char.ConvertFromUtf32(n);
 869
 870                        // If we have supplimentary characters
 0871                        if (numSurrogatePairs > 0)
 872                        {
 873                            // Hard way, we have supplimentary characters
 874                            int iCount;
 0875                            for (iCount = i, iUseInsertLocation = iOutputAfterLastDot; iCount > 0; iCount--, iUseInsertL
 876                            {
 877                                // If its a surrogate, we have to go one more
 0878                                if (iUseInsertLocation >= output.Length)
 0879                                    throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(ascii));
 0880                                if (char.IsSurrogate(output[iUseInsertLocation]))
 0881                                    iUseInsertLocation++;
 882                            }
 883                        }
 884                        else
 885                        {
 886                            // No Supplementary chars yet, just add i
 0887                            iUseInsertLocation = iOutputAfterLastDot + i;
 888                        }
 889
 890                        // Insert it
 0891                        output.Insert(iUseInsertLocation, strTemp);
 892
 893                        // If it was a surrogate increment our counter
 0894                        if (IsSupplementary(n))
 0895                            numSurrogatePairs++;
 896
 897                        // Index gets updated
 0898                        i++;
 899                    }
 900
 901                    // Do BIDI testing
 0902                    bool bRightToLeft = false;
 903
 904                    // Check for RTL.  If right-to-left, then 1st & last chars must be RTL
 0905                    StrongBidiCategory eBidi = CharUnicodeInfo.GetBidiCategory(output, iOutputAfterLastDot);
 0906                    if (eBidi == StrongBidiCategory.StrongRightToLeft)
 907                    {
 908                        // It has to be right to left.
 0909                        bRightToLeft = true;
 910                    }
 911
 912                    // Check the rest of them to make sure RTL/LTR is consistent
 0913                    for (int iTest = iOutputAfterLastDot; iTest < output.Length; iTest++)
 914                    {
 915                        // This might happen if we run into a pair
 0916                        if (char.IsLowSurrogate(output[iTest]))
 917                            continue;
 918
 919                        // Check to see if its LTR
 0920                        eBidi = CharUnicodeInfo.GetBidiCategory(output, iTest);
 0921                        if ((bRightToLeft && eBidi == StrongBidiCategory.StrongLeftToRight) ||
 0922                            (!bRightToLeft && eBidi == StrongBidiCategory.StrongRightToLeft))
 0923                            throw new ArgumentException(SR.Argument_IdnBadBidi, nameof(ascii));
 924                    }
 925
 926                    // Its also a requirement that the last one be RTL if 1st is RTL
 0927                    if (bRightToLeft && eBidi != StrongBidiCategory.StrongRightToLeft)
 928                    {
 929                        // Oops, last wasn't RTL, last should be RTL if first is RTL
 0930                        throw new ArgumentException(SR.Argument_IdnBadBidi, nameof(ascii));
 931                    }
 932                }
 933
 934                // See if this label was too long
 0935                if (iNextDot - iAfterLastDot > c_labelLimit)
 0936                    throw new ArgumentException(SR.Argument_IdnBadLabelSize, nameof(ascii));
 937
 938                // Done with this segment, add dot if necessary
 0939                if (iNextDot != ascii.Length)
 0940                    output.Append('.');
 941
 0942                iAfterLastDot = iNextDot + 1;
 0943                iOutputAfterLastDot = output.Length;
 944            }
 945
 946            // Throw if we're too long
 0947            if (output.Length > c_defaultNameLimit - (IsDot(output[^1]) ? 0 : 1))
 0948                throw new ArgumentException(SR.Format(SR.Argument_IdnBadNameSize, c_defaultNameLimit - (IsDot(output[^1]
 949
 950            // Return our output string
 0951            return output.ToString();
 952        }
 953
 954        // DecodeDigit(cp) returns the numeric value of a basic code */
 955        // point (for use in representing integers) in the range 0 to */
 956        // c_punycodeBase-1, or <0 if cp is does not represent a value. */
 957
 958        private static int DecodeDigit(char cp)
 959        {
 0960            if (char.IsAsciiDigit(cp))
 0961                return cp - '0' + 26;
 962
 963            // Two flavors for case differences
 0964            if (char.IsAsciiLetterLower(cp))
 0965                return cp - 'a';
 966
 0967            if (char.IsAsciiLetterUpper(cp))
 0968                return cp - 'A';
 969
 970            // Expected 0-9, A-Z or a-z, everything else is illegal
 0971            throw new ArgumentException(SR.Argument_IdnBadPunycode, nameof(cp));
 972        }
 973
 974        private static int Adapt(int delta, int numpoints, bool firsttime)
 975        {
 976            uint k;
 977
 0978            delta = firsttime ? delta / c_damp : delta / 2;
 0979            Debug.Assert(numpoints != 0, "[IdnMapping.adapt]Expected non-zero numpoints.");
 0980            delta += delta / numpoints;
 981
 0982            for (k = 0; delta > ((c_punycodeBase - c_tmin) * c_tmax) / 2; k += c_punycodeBase)
 983            {
 0984                delta /= c_punycodeBase - c_tmin;
 985            }
 986
 0987            Debug.Assert(delta + c_skew != 0, "[IdnMapping.adapt]Expected non-zero delta+skew.");
 0988            return (int)(k + (c_punycodeBase - c_tmin + 1) * delta / (delta + c_skew));
 989        }
 990
 991        /* EncodeBasic(bcp,flag) forces a basic code point to lowercase */
 992        /* if flag is false, uppercase if flag is true, and returns    */
 993        /* the resulting code point.  The code point is unchanged if it  */
 994        /* is caseless.  The behavior is undefined if bcp is not a basic */
 995        /* code point.                                                   */
 996
 997        private static char EncodeBasic(char bcp)
 998        {
 0999            if (char.IsAsciiLetterUpper(bcp))
 01000                bcp += (char)('a' - 'A');
 1001
 01002            return bcp;
 1003        }
 1004
 1005        /* EncodeDigit(d,flag) returns the basic code point whose value      */
 1006        /* (when used for representing integers) is d, which needs to be in   */
 1007        /* the range 0 to punycodeBase-1.  The lowercase form is used unless flag is  */
 1008        /* true, in which case the uppercase form is used. */
 1009
 1010        private static char EncodeDigit(int d)
 1011        {
 01012            Debug.Assert(d >= 0 && d < c_punycodeBase, "[IdnMapping.encode_digit]Expected 0 <= d < punycodeBase");
 1013            // 26-35 map to ASCII 0-9
 01014            if (d > 25) return (char)(d - 26 + '0');
 1015
 1016            // 0-25 map to a-z or A-Z
 01017            return (char)(d + 'a');
 1018        }
 1019    }
 1020}
 1021

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.CoreLib/src/System/Globalization/IdnMapping.Icu.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Diagnostics;
 5
 6namespace System.Globalization
 7{
 8    public sealed partial class IdnMapping
 9    {
 10        private unsafe string IcuGetAsciiCore(string unicodeString, int index, int count)
 11        {
 012            Debug.Assert(!GlobalizationMode.Invariant);
 013            Debug.Assert(!GlobalizationMode.UseNls);
 14
 015            ReadOnlySpan<char> unicode = unicodeString.AsSpan(index, count);
 016            uint flags = IcuFlags;
 017            CheckInvalidIdnCharacters(unicode, flags, nameof(unicode));
 18
 19            const int StackallocThreshold = 512;
 20            // Each unicode character is represented by up to 3 ASCII chars
 21            // and the whole string is prefixed by "xn--" (length 4)
 022            int estimatedLength = (int)Math.Min(checked(count * 3L + 4), StackallocThreshold);
 23            int actualLength;
 024            if ((uint)estimatedLength < StackallocThreshold)
 25            {
 026                Span<char> outputStack = stackalloc char[estimatedLength];
 027                actualLength = Interop.Globalization.ToAscii(flags, unicode, count, outputStack, estimatedLength);
 028                if (actualLength > 0 && actualLength <= estimatedLength)
 29                {
 030                    return GetStringForOutput(unicodeString, unicode, outputStack.Slice(0, actualLength));
 31                }
 32            }
 33            else
 34            {
 035                actualLength = Interop.Globalization.ToAscii(flags, unicode, count, Span<char>.Empty, 0);
 36            }
 037            if (actualLength == 0)
 38            {
 039                throw new ArgumentException(SR.Argument_IdnIllegalName, nameof(unicode));
 40            }
 41
 042            char[] outputHeap = new char[actualLength];
 043            actualLength = Interop.Globalization.ToAscii(flags, unicode, count, outputHeap, actualLength);
 044            if (actualLength == 0 || actualLength > outputHeap.Length)
 45            {
 046                throw new ArgumentException(SR.Argument_IdnIllegalName, nameof(unicode));
 47            }
 48
 049            return GetStringForOutput(unicodeString, unicode, outputHeap.AsSpan(0, actualLength));
 50        }
 51
 52        private bool IcuTryGetAsciiCore(ReadOnlySpan<char> unicode, Span<char> destination, out int charsWritten)
 53        {
 054            Debug.Assert(!GlobalizationMode.Invariant);
 055            Debug.Assert(!GlobalizationMode.UseNls);
 56
 057            uint flags = IcuFlags;
 058            CheckInvalidIdnCharacters(unicode, flags, nameof(unicode));
 59
 060            int actualLength = Interop.Globalization.ToAscii(flags, unicode, unicode.Length, destination, destination.Le
 61
 062            if (actualLength <= destination.Length)
 63            {
 064                if (actualLength == 0)
 65                {
 066                    throw new ArgumentException(SR.Argument_IdnIllegalName, nameof(unicode));
 67                }
 68
 069                charsWritten = actualLength;
 070                return true;
 71            }
 72
 073            charsWritten = 0;
 074            return false;
 75        }
 76
 77        private unsafe string IcuGetUnicodeCore(string asciiString, int index, int count)
 78        {
 079            Debug.Assert(!GlobalizationMode.Invariant);
 080            Debug.Assert(!GlobalizationMode.UseNls);
 81
 082            ReadOnlySpan<char> ascii = asciiString.AsSpan(index, count);
 083            uint flags = IcuFlags;
 084            CheckInvalidIdnCharacters(ascii, flags, nameof(ascii));
 85
 86            const int StackAllocThreshold = 512;
 087            if ((uint)count < StackAllocThreshold)
 88            {
 089                Span<char> output = stackalloc char[count];
 090                return IcuGetUnicodeCore(asciiString, ascii, flags, output, reattempt: true);
 91            }
 92            else
 93            {
 094                char[] output = new char[count];
 095                return IcuGetUnicodeCore(asciiString, ascii, flags, output, reattempt: true);
 96            }
 97        }
 98
 99        private static string IcuGetUnicodeCore(string asciiString, ReadOnlySpan<char> ascii, uint flags, Span<char> out
 100        {
 0101            Debug.Assert(!GlobalizationMode.Invariant);
 0102            Debug.Assert(!GlobalizationMode.UseNls);
 103
 0104            int realLen = Interop.Globalization.ToUnicode(flags, ascii, ascii.Length, output, output.Length);
 105
 0106            if (realLen == 0)
 107            {
 0108                throw new ArgumentException(SR.Argument_IdnIllegalName, nameof(ascii));
 109            }
 0110            else if (realLen <= output.Length)
 111            {
 0112                return GetStringForOutput(asciiString, ascii, output.Slice(0, realLen));
 113            }
 0114            else if (reattempt)
 115            {
 0116                char[] newOutput = new char[realLen];
 0117                return IcuGetUnicodeCore(asciiString, ascii, flags, newOutput, reattempt: false);
 118            }
 119
 0120            throw new ArgumentException(SR.Argument_IdnIllegalName, nameof(ascii));
 121        }
 122
 123        private bool IcuTryGetUnicodeCore(ReadOnlySpan<char> ascii, Span<char> destination, out int charsWritten)
 124        {
 0125            Debug.Assert(!GlobalizationMode.Invariant);
 0126            Debug.Assert(!GlobalizationMode.UseNls);
 127
 0128            uint flags = IcuFlags;
 0129            CheckInvalidIdnCharacters(ascii, flags, nameof(ascii));
 130
 0131            int actualLength = Interop.Globalization.ToUnicode(flags, ascii, ascii.Length, destination, destination.Leng
 132
 0133            if (actualLength <= destination.Length)
 134            {
 0135                if (actualLength == 0)
 136                {
 0137                    throw new ArgumentException(SR.Argument_IdnIllegalName, nameof(ascii));
 138                }
 139
 0140                charsWritten = actualLength;
 0141                return true;
 142            }
 143
 0144            charsWritten = 0;
 0145            return false;
 146        }
 147
 148        private uint IcuFlags
 149        {
 150            get
 151            {
 0152                int flags =
 0153                    (AllowUnassigned ? Interop.Globalization.AllowUnassigned : 0) |
 0154                    (UseStd3AsciiRules ? Interop.Globalization.UseStd3AsciiRules : 0);
 0155                return (uint)flags;
 156            }
 157        }
 158
 159        /// <summary>
 160        /// ICU doesn't check for invalid characters unless the STD3 rules option
 161        /// is enabled.
 162        ///
 163        /// To match Windows behavior, we walk the string ourselves looking for these
 164        /// bad characters so we can continue to throw ArgumentException in these cases.
 165        /// </summary>
 166        private static void CheckInvalidIdnCharacters(ReadOnlySpan<char> s, uint flags, string paramName)
 167        {
 0168            if ((flags & Interop.Globalization.UseStd3AsciiRules) == 0)
 169            {
 0170                for (int i = 0; i < s.Length; i++)
 171                {
 0172                    char c = s[i];
 173
 174                    // These characters are prohibited regardless of the UseStd3AsciiRules property.
 175                    // See https://msdn.microsoft.com/en-us/library/system.globalization.idnmapping.usestd3asciirules(v=
 0176                    if (c <= 0x1F || c == 0x7F)
 177                    {
 0178                        throw new ArgumentException(SR.Argument_IdnIllegalName, paramName);
 179                    }
 180                }
 181            }
 0182        }
 183    }
 184}
 185

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.CoreLib/src/System/Globalization/IdnMapping.Nls.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Diagnostics;
 5using System.Diagnostics.CodeAnalysis;
 6using System.Runtime.InteropServices;
 7
 8namespace System.Globalization
 9{
 10    public sealed partial class IdnMapping
 11    {
 12        private unsafe string NlsGetAsciiCore(string unicodeString, int index, int count)
 13        {
 014            Debug.Assert(!GlobalizationMode.Invariant);
 015            Debug.Assert(GlobalizationMode.UseNls);
 16
 017            ReadOnlySpan<char> unicode = unicodeString.AsSpan(index, count);
 018            uint flags = NlsFlags;
 19
 20            // Determine the required length
 021            int length = Interop.Normaliz.IdnToAscii(flags, unicode, count, Span<char>.Empty, 0);
 022            if (length == 0)
 23            {
 024                ThrowForZeroLength(unicode: true);
 25            }
 26
 27            // Do the conversion
 28            const int StackAllocThreshold = 512; // arbitrary limit to switch from stack to heap allocation
 029            if ((uint)length < StackAllocThreshold)
 30            {
 031                Span<char> output = stackalloc char[length];
 032                return NlsGetAsciiCore(unicodeString, unicode, flags, output);
 33            }
 34            else
 35            {
 036                char[] output = new char[length];
 037                return NlsGetAsciiCore(unicodeString, unicode, flags, output);
 38            }
 39        }
 40
 41        private static string NlsGetAsciiCore(string unicodeString, ReadOnlySpan<char> unicode, uint flags, Span<char> o
 42        {
 043            Debug.Assert(!GlobalizationMode.Invariant);
 044            Debug.Assert(GlobalizationMode.UseNls);
 45
 046            int length = Interop.Normaliz.IdnToAscii(flags, unicode, unicode.Length, output, output.Length);
 047            if (length == 0)
 48            {
 049                ThrowForZeroLength(unicode: true);
 50            }
 051            Debug.Assert(length == output.Length);
 052            return GetStringForOutput(unicodeString, unicode, output.Slice(0, length));
 53        }
 54
 55        private bool NlsTryGetAsciiCore(ReadOnlySpan<char> unicode, Span<char> destination, out int charsWritten)
 56        {
 057            Debug.Assert(!GlobalizationMode.Invariant);
 058            Debug.Assert(GlobalizationMode.UseNls);
 59
 060            uint flags = NlsFlags;
 61
 62            // Determine the required length
 063            int length = Interop.Normaliz.IdnToAscii(flags, unicode, unicode.Length, Span<char>.Empty, 0);
 064            if (length == 0)
 65            {
 066                ThrowForZeroLength(unicode: true);
 67            }
 68
 069            if (length > destination.Length)
 70            {
 071                charsWritten = 0;
 072                return false;
 73            }
 74
 75            // Do the conversion
 076            int actualLength = Interop.Normaliz.IdnToAscii(flags, unicode, unicode.Length, destination, destination.Leng
 077            if (actualLength == 0)
 78            {
 079                ThrowForZeroLength(unicode: true);
 80            }
 81
 082            charsWritten = actualLength;
 083            return true;
 84        }
 85
 86        private unsafe string NlsGetUnicodeCore(string asciiString, int index, int count)
 87        {
 088            Debug.Assert(!GlobalizationMode.Invariant);
 089            Debug.Assert(GlobalizationMode.UseNls);
 90
 091            ReadOnlySpan<char> ascii = asciiString.AsSpan(index, count);
 092            uint flags = NlsFlags;
 93
 94            // Determine the required length
 095            int length = Interop.Normaliz.IdnToUnicode(flags, ascii, count, Span<char>.Empty, 0);
 096            if (length == 0)
 97            {
 098                ThrowForZeroLength(unicode: false);
 99            }
 100
 101            // Do the conversion
 102            const int StackAllocThreshold = 512; // arbitrary limit to switch from stack to heap allocation
 0103            if ((uint)length < StackAllocThreshold)
 104            {
 0105                Span<char> output = stackalloc char[length];
 0106                return NlsGetUnicodeCore(asciiString, ascii, flags, output);
 107            }
 108            else
 109            {
 0110                char[] output = new char[length];
 0111                return NlsGetUnicodeCore(asciiString, ascii, flags, output);
 112            }
 113        }
 114
 115        private static string NlsGetUnicodeCore(string asciiString, ReadOnlySpan<char> ascii, uint flags, Span<char> out
 116        {
 0117            Debug.Assert(!GlobalizationMode.Invariant);
 0118            Debug.Assert(GlobalizationMode.UseNls);
 119
 0120            int length = Interop.Normaliz.IdnToUnicode(flags, ascii, ascii.Length, output, output.Length);
 0121            if (length == 0)
 122            {
 0123                ThrowForZeroLength(unicode: false);
 124            }
 0125            Debug.Assert(length == output.Length);
 0126            return GetStringForOutput(asciiString, ascii, output.Slice(0, length));
 127        }
 128
 129        private bool NlsTryGetUnicodeCore(ReadOnlySpan<char> ascii, Span<char> destination, out int charsWritten)
 130        {
 0131            Debug.Assert(!GlobalizationMode.Invariant);
 0132            Debug.Assert(GlobalizationMode.UseNls);
 133
 0134            uint flags = NlsFlags;
 135
 136            // Determine the required length
 0137            int length = Interop.Normaliz.IdnToUnicode(flags, ascii, ascii.Length, Span<char>.Empty, 0);
 0138            if (length == 0)
 139            {
 0140                ThrowForZeroLength(unicode: false);
 141            }
 142
 0143            if (length > destination.Length)
 144            {
 0145                charsWritten = 0;
 0146                return false;
 147            }
 148
 149            // Do the conversion
 0150            int actualLength = Interop.Normaliz.IdnToUnicode(flags, ascii, ascii.Length, destination, destination.Length
 0151            if (actualLength == 0)
 152            {
 0153                ThrowForZeroLength(unicode: false);
 154            }
 155
 0156            charsWritten = actualLength;
 0157            return true;
 158        }
 159
 160        private uint NlsFlags
 161        {
 162            get
 163            {
 0164                int flags =
 0165                    (AllowUnassigned ? Interop.Normaliz.IDN_ALLOW_UNASSIGNED : 0) |
 0166                    (UseStd3AsciiRules ? Interop.Normaliz.IDN_USE_STD3_ASCII_RULES : 0);
 0167                return (uint)flags;
 168            }
 169        }
 170
 171        [DoesNotReturn]
 172        private static void ThrowForZeroLength(bool unicode)
 173        {
 0174            int lastError = Marshal.GetLastPInvokeError();
 175
 0176            throw new ArgumentException(
 0177                lastError == Interop.Errors.ERROR_INVALID_NAME ? SR.Argument_IdnIllegalName :
 0178                    (unicode ? SR.Argument_InvalidCharSequenceNoIndex : SR.Argument_IdnBadPunycode),
 0179                unicode ? "unicode" : "ascii");
 180        }
 181    }
 182}
 183

Methods/Properties

.ctor()
AllowUnassigned()
AllowUnassigned(System.Boolean)
UseStd3AsciiRules()
UseStd3AsciiRules(System.Boolean)
GetAscii(System.String)
GetAscii(System.String,System.Int32)
GetAscii(System.String,System.Int32,System.Int32)
TryGetAscii(System.ReadOnlySpan`1<System.Char>,System.Span`1<System.Char>,System.Int32&)
GetUnicode(System.String)
GetUnicode(System.String,System.Int32)
GetUnicode(System.String,System.Int32,System.Int32)
TryGetUnicode(System.ReadOnlySpan`1<System.Char>,System.Span`1<System.Char>,System.Int32&)
Equals(System.Object)
GetHashCode()
GetStringForOutput(System.String,System.ReadOnlySpan`1<System.Char>,System.ReadOnlySpan`1<System.Char>)
GetAsciiInvariant(System.String,System.Int32,System.Int32)
TryGetAsciiInvariant(System.ReadOnlySpan`1<System.Char>,System.Span`1<System.Char>,System.Int32&)
ValidateStd3AndAscii(System.ReadOnlySpan`1<System.Char>,System.Boolean,System.Boolean)
PunycodeEncode(System.ReadOnlySpan`1<System.Char>)
IsDot(System.Char)
IsSupplementary(System.Int32)
Basic(System.UInt32)
GetCodePoint(System.ReadOnlySpan`1<System.Char>,System.Int32)
ValidateStd3(System.Char,System.Boolean)
GetUnicodeInvariant(System.String,System.Int32,System.Int32)
TryGetUnicodeInvariant(System.ReadOnlySpan`1<System.Char>,System.Span`1<System.Char>,System.Int32&)
PunycodeDecode(System.String)
DecodeDigit(System.Char)
Adapt(System.Int32,System.Int32,System.Boolean)
EncodeBasic(System.Char)
EncodeDigit(System.Int32)
IcuGetAsciiCore(System.String,System.Int32,System.Int32)
IcuTryGetAsciiCore(System.ReadOnlySpan`1<System.Char>,System.Span`1<System.Char>,System.Int32&)
IcuGetUnicodeCore(System.String,System.Int32,System.Int32)
IcuGetUnicodeCore(System.String,System.ReadOnlySpan`1<System.Char>,System.UInt32,System.Span`1<System.Char>,System.Boolean)
IcuTryGetUnicodeCore(System.ReadOnlySpan`1<System.Char>,System.Span`1<System.Char>,System.Int32&)
IcuFlags()
CheckInvalidIdnCharacters(System.ReadOnlySpan`1<System.Char>,System.UInt32,System.String)
NlsGetAsciiCore(System.String,System.Int32,System.Int32)
NlsGetAsciiCore(System.String,System.ReadOnlySpan`1<System.Char>,System.UInt32,System.Span`1<System.Char>)
NlsTryGetAsciiCore(System.ReadOnlySpan`1<System.Char>,System.Span`1<System.Char>,System.Int32&)
NlsGetUnicodeCore(System.String,System.Int32,System.Int32)
NlsGetUnicodeCore(System.String,System.ReadOnlySpan`1<System.Char>,System.UInt32,System.Span`1<System.Char>)
NlsTryGetUnicodeCore(System.ReadOnlySpan`1<System.Char>,System.Span`1<System.Char>,System.Int32&)
NlsFlags()
ThrowForZeroLength(System.Boolean)