< Summary

Line coverage
18%
Covered lines: 75
Uncovered lines: 339
Coverable lines: 414
Total lines: 1665
Line coverage: 18.1%
Branch coverage
16%
Covered branches: 36
Total branches: 224
Branch coverage: 16%
Method coverage

Feature is only available for sponsors

Upgrade to PRO version

Metrics

MethodBranch coverage Cyclomatic complexity NPath complexity Sequence coverage
.ctor(...)0%220%
.ctor(...)100%110%
.ctor(...)100%110%
.ctor(...)0%220%
.ctor(...)100%11100%
op_Equality(...)100%110%
op_Inequality(...)100%110%
op_LessThan(...)100%110%
op_LessThanOrEqual(...)100%110%
op_GreaterThan(...)100%110%
op_GreaterThanOrEqual(...)100%110%
op_Explicit(...)100%110%
op_Explicit(...)100%110%
op_Explicit(...)100%110%
ChangeCaseCultureAware(...)0%440%
CompareTo(...)100%110%
AsSpan(...)100%110%
DecodeFromUtf16(...)0%10100%
DecodeFromUtf8(...)88.23%343494.44%
DecodeLastFromUtf16(...)0%10100%
DecodeLastFromUtf8(...)0%14140%
EncodeToUtf16(...)0%220%
EncodeToUtf8(...)0%220%
Equals(...)0%220%
Equals(...)100%110%
Equals(...)0%220%
GetHashCode()100%110%
GetRuneAt(...)0%220%
IsValid(...)100%110%
IsValid(...)100%110%
ReadFirstRuneFromUtf16Buffer(...)0%10100%
ReadRuneFromString(...)0%12120%
ToString()0%220%
System.ISpanFormattable.TryFormat(...)100%110%
System.IUtf8SpanFormattable.TryFormat(...)100%110%
System.IUtf8SpanParsable<System.Text.Rune>.TryParse(...)0%440%
System.IUtf8SpanParsable<System.Text.Rune>.Parse(...)0%440%
System.IParsable<System.Text.Rune>.Parse(...)0%440%
System.IParsable<System.Text.Rune>.TryParse(...)0%440%
System.ISpanParsable<System.Text.Rune>.Parse(...)0%440%
System.ISpanParsable<System.Text.Rune>.TryParse(...)0%440%
System.IFormattable.ToString(...)100%110%
TryCreate(...)50%2266.66%
TryCreate(...)0%220%
TryCreate(...)100%110%
TryCreate(...)0%220%
TryEncodeToUtf16(...)100%11100%
TryEncodeToUtf16(...)66.66%6681.81%
TryEncodeToUtf8(...)100%110%
TryEncodeToUtf8(...)0%14140%
TryGetRuneAt(...)0%220%
UnsafeCreate(...)100%11100%
GetNumericValue(...)0%440%
GetUnicodeCategory(...)0%220%
GetUnicodeCategoryNonAscii(...)100%110%
IsCategoryLetter(...)100%110%
IsCategoryLetterOrDecimalDigit(...)0%220%
IsCategoryNumber(...)100%110%
IsCategoryPunctuation(...)100%110%
IsCategorySeparator(...)100%110%
IsCategorySymbol(...)100%110%
IsControl(...)100%110%
IsDigit(...)0%220%
IsLetter(...)0%220%
IsLetterOrDigit(...)0%220%
IsLower(...)0%220%
IsNumber(...)0%220%
IsPunctuation(...)100%110%
IsSeparator(...)100%110%
IsSymbol(...)100%110%
IsUpper(...)0%220%
IsWhiteSpace(...)0%440%
ToLower(...)0%440%
ToLowerInvariant(...)0%440%
ToUpper(...)0%440%
ToUpperInvariant(...)0%440%
ToUpperOrdinal(...)0%660%
ToLowerOrdinal(...)0%660%
System.IComparable.CompareTo(...)0%440%

File(s)

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.CoreLib/src/System/Text/Rune.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Buffers;
 5using System.Diagnostics;
 6using System.Diagnostics.CodeAnalysis;
 7using System.Globalization;
 8using System.Runtime.CompilerServices;
 9using System.Text.Unicode;
 10
 11#if !SYSTEM_PRIVATE_CORELIB
 12#pragma warning disable CS3019 // CLS compliance checking will not be performed because it is not visible from outside t
 13#endif
 14
 15namespace System.Text
 16{
 17    /// <summary>
 18    /// Represents a Unicode scalar value ([ U+0000..U+D7FF ], inclusive; or [ U+E000..U+10FFFF ], inclusive).
 19    /// </summary>
 20    /// <remarks>
 21    /// This type's constructors and conversion operators validate the input, so consumers can call the APIs
 22    /// assuming that the underlying <see cref="Rune"/> instance is well-formed.
 23    /// </remarks>
 24    [DebuggerDisplay("{DebuggerDisplay,nq}")]
 25#if SYSTEM_PRIVATE_CORELIB
 26    public
 27#else
 28    internal
 29#endif
 30    readonly struct Rune : IComparable, IComparable<Rune>, IEquatable<Rune>
 31#if SYSTEM_PRIVATE_CORELIB
 32#pragma warning disable SA1001 // Commas should be spaced correctly
 33        , ISpanFormattable
 34        , IUtf8SpanFormattable
 35        , IParsable<Rune>
 36        , ISpanParsable<Rune>
 37        , IUtf8SpanParsable<Rune>
 38#pragma warning restore SA1001
 39#endif
 40    {
 41        internal const int MaxUtf16CharsPerRune = 2; // supplementary plane code points are encoded as 2 UTF-16 code uni
 42        internal const int MaxUtf8BytesPerRune = 4; // supplementary plane code points are encoded as 4 UTF-8 code units
 43
 44        private const char HighSurrogateStart = '\ud800';
 45        private const char LowSurrogateStart = '\udc00';
 46        private const int HighSurrogateRange = 0x3FF;
 47
 48        private const byte IsWhiteSpaceFlag = 0x80;
 49        private const byte IsLetterOrDigitFlag = 0x40;
 50        private const byte UnicodeCategoryMask = 0x1F;
 51
 52        // Contains information about the ASCII character range [ U+0000..U+007F ], with:
 53        // - 0x80 bit if set means 'is whitespace'
 54        // - 0x40 bit if set means 'is letter or digit'
 55        // - 0x20 bit is reserved for future use
 56        // - bottom 5 bits are the UnicodeCategory of the character
 57        private static ReadOnlySpan<byte> AsciiCharInfo =>
 058        [
 059            0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x8E, 0x8E, 0x8E, 0x8E, 0x8E, 0x0E, 0x0E, // U+0000..U
 060            0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, // U+0010..U
 061            0x8B, 0x18, 0x18, 0x18, 0x1A, 0x18, 0x18, 0x18, 0x14, 0x15, 0x18, 0x19, 0x18, 0x13, 0x18, 0x18, // U+0020..U
 062            0x48, 0x48, 0x48, 0x48, 0x48, 0x48, 0x48, 0x48, 0x48, 0x48, 0x18, 0x18, 0x19, 0x19, 0x19, 0x18, // U+0030..U
 063            0x18, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, // U+0040..U
 064            0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x14, 0x18, 0x15, 0x1B, 0x12, // U+0050..U
 065            0x1B, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, // U+0060..U
 066            0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x14, 0x19, 0x15, 0x19, 0x0E, // U+0070..U
 067        ];
 68
 69        private readonly uint _value;
 70
 71        /// <summary>
 72        /// Creates a <see cref="Rune"/> from the provided UTF-16 code unit.
 73        /// </summary>
 74        /// <exception cref="ArgumentOutOfRangeException">
 75        /// If <paramref name="ch"/> represents a UTF-16 surrogate code point
 76        /// U+D800..U+DFFF, inclusive.
 77        /// </exception>
 78        public Rune(char ch)
 79        {
 080            uint expanded = ch;
 081            if (UnicodeUtility.IsSurrogateCodePoint(expanded))
 82            {
 083                ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.ch);
 84            }
 085            _value = expanded;
 086        }
 87
 88        /// <summary>
 89        /// Creates a <see cref="Rune"/> from the provided UTF-16 surrogate pair.
 90        /// </summary>
 91        /// <exception cref="ArgumentOutOfRangeException">
 92        /// If <paramref name="highSurrogate"/> does not represent a UTF-16 high surrogate code point
 93        /// or <paramref name="lowSurrogate"/> does not represent a UTF-16 low surrogate code point.
 94        /// </exception>
 95        public Rune(char highSurrogate, char lowSurrogate)
 096            : this((uint)char.ConvertToUtf32(highSurrogate, lowSurrogate), false)
 97        {
 098        }
 99
 100        /// <summary>
 101        /// Creates a <see cref="Rune"/> from the provided Unicode scalar value.
 102        /// </summary>
 103        /// <exception cref="ArgumentOutOfRangeException">
 104        /// If <paramref name="value"/> does not represent a value Unicode scalar value.
 105        /// </exception>
 106        public Rune(int value)
 0107            : this((uint)value)
 108        {
 0109        }
 110
 111        /// <summary>
 112        /// Creates a <see cref="Rune"/> from the provided Unicode scalar value.
 113        /// </summary>
 114        /// <exception cref="ArgumentOutOfRangeException">
 115        /// If <paramref name="value"/> does not represent a value Unicode scalar value.
 116        /// </exception>
 117        [CLSCompliant(false)]
 118        public Rune(uint value)
 119        {
 0120            if (!UnicodeUtility.IsValidUnicodeScalar(value))
 121            {
 0122                ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.value);
 123            }
 0124            _value = value;
 0125        }
 126
 127        // non-validating ctor
 128        private Rune(uint scalarValue, bool _)
 129        {
 4445191130            UnicodeDebug.AssertIsValidScalar(scalarValue);
 4445191131            _value = scalarValue;
 4445191132        }
 133
 0134        public static bool operator ==(Rune left, Rune right) => left._value == right._value;
 135
 0136        public static bool operator !=(Rune left, Rune right) => left._value != right._value;
 137
 0138        public static bool operator <(Rune left, Rune right) => left._value < right._value;
 139
 0140        public static bool operator <=(Rune left, Rune right) => left._value <= right._value;
 141
 0142        public static bool operator >(Rune left, Rune right) => left._value > right._value;
 143
 0144        public static bool operator >=(Rune left, Rune right) => left._value >= right._value;
 145
 146        // Operators below are explicit because they may throw.
 147
 0148        public static explicit operator Rune(char ch) => new Rune(ch);
 149
 150        [CLSCompliant(false)]
 0151        public static explicit operator Rune(uint value) => new Rune(value);
 152
 0153        public static explicit operator Rune(int value) => new Rune(value);
 154
 155        // Displayed as "'<char>' (U+XXXX)"; e.g., "'e' (U+0065)"
 156        private string DebuggerDisplay =>
 157#if SYSTEM_PRIVATE_CORELIB
 0158            string.Create(
 0159                CultureInfo.InvariantCulture,
 0160#else
 0161            FormattableString.Invariant(
 0162#endif
 0163                $"U+{_value:X4} '{(IsValid(_value) ? ToString() : "\uFFFD")}'");
 164
 165        /// <summary>
 166        /// Returns true if and only if this scalar value is ASCII ([ U+0000..U+007F ])
 167        /// and therefore representable by a single UTF-8 code unit.
 168        /// </summary>
 0169        public bool IsAscii => UnicodeUtility.IsAsciiCodePoint(_value);
 170
 171        /// <summary>
 172        /// Returns true if and only if this scalar value is within the BMP ([ U+0000..U+FFFF ])
 173        /// and therefore representable by a single UTF-16 code unit.
 174        /// </summary>
 175880175        public bool IsBmp => UnicodeUtility.IsBmpCodePoint(_value);
 176
 177        /// <summary>
 178        /// Returns the Unicode plane (0 to 16, inclusive) which contains this scalar.
 179        /// </summary>
 0180        public int Plane => UnicodeUtility.GetPlane(_value);
 181
 182        /// <summary>
 183        /// A <see cref="Rune"/> instance that represents the Unicode replacement character U+FFFD.
 184        /// </summary>
 2524547185        public static Rune ReplacementChar => UnsafeCreate(UnicodeUtility.ReplacementChar);
 186
 187        /// <summary>
 188        /// Returns the length in code units (<see cref="char"/>) of the
 189        /// UTF-16 sequence required to represent this scalar value.
 190        /// </summary>
 191        /// <remarks>
 192        /// The return value will be 1 or 2.
 193        /// </remarks>
 194        public int Utf16SequenceLength
 195        {
 196            get
 197            {
 811866198                int codeUnitCount = UnicodeUtility.GetUtf16SequenceLength(_value);
 811866199                Debug.Assert(codeUnitCount > 0 && codeUnitCount <= MaxUtf16CharsPerRune);
 811866200                return codeUnitCount;
 201            }
 202        }
 203
 204        /// <summary>
 205        /// Returns the length in code units of the
 206        /// UTF-8 sequence required to represent this scalar value.
 207        /// </summary>
 208        /// <remarks>
 209        /// The return value will be 1 through 4, inclusive.
 210        /// </remarks>
 211        public int Utf8SequenceLength
 212        {
 213            get
 214            {
 0215                int codeUnitCount = UnicodeUtility.GetUtf8SequenceLength(_value);
 0216                Debug.Assert(codeUnitCount > 0 && codeUnitCount <= MaxUtf8BytesPerRune);
 0217                return codeUnitCount;
 218            }
 219        }
 220
 221        /// <summary>
 222        /// Returns the Unicode scalar value as an integer.
 223        /// </summary>
 1865796224        public int Value => (int)_value;
 225
 226#if SYSTEM_PRIVATE_CORELIB
 227        private static unsafe Rune ChangeCaseCultureAware(Rune rune, TextInfo textInfo, bool toUpper)
 228        {
 0229            Debug.Assert(!GlobalizationMode.Invariant, "This should've been checked by the caller.");
 0230            Debug.Assert(textInfo != null, "This should've been checked by the caller.");
 231
 0232            Span<char> original = stackalloc char[MaxUtf16CharsPerRune];
 0233            Span<char> modified = stackalloc char[MaxUtf16CharsPerRune];
 234
 0235            int charCount = rune.EncodeToUtf16(original);
 0236            original = original.Slice(0, charCount);
 0237            modified = modified.Slice(0, charCount);
 238
 0239            if (toUpper)
 240            {
 0241                textInfo.ChangeCaseToUpper(original, modified);
 242            }
 243            else
 244            {
 0245                textInfo.ChangeCaseToLower(original, modified);
 246            }
 247
 248            // We use simple case folding rules, which disallows moving between the BMP and supplementary
 249            // planes when performing a case conversion. The helper methods which reconstruct a Rune
 250            // contain debug asserts for this condition.
 251
 0252            if (rune.IsBmp)
 253            {
 0254                return UnsafeCreate(modified[0]);
 255            }
 256            else
 257            {
 0258                return UnsafeCreate(UnicodeUtility.GetScalarFromUtf16SurrogatePair(modified[0], modified[1]));
 259            }
 260        }
 261#else
 262        private static unsafe Rune ChangeCaseCultureAware(Rune rune, CultureInfo culture, bool toUpper)
 263        {
 264            Debug.Assert(culture != null, "This should've been checked by the caller.");
 265
 266            Span<char> original = stackalloc char[MaxUtf16CharsPerRune]; // worst case scenario = 2 code units (for a su
 267            Span<char> modified = stackalloc char[MaxUtf16CharsPerRune]; // case change should preserve UTF-16 code unit
 268
 269            int charCount = rune.EncodeToUtf16(original);
 270            original = original.Slice(0, charCount);
 271            modified = modified.Slice(0, charCount);
 272
 273            if (toUpper)
 274            {
 275                MemoryExtensions.ToUpper(original, modified, culture);
 276            }
 277            else
 278            {
 279                MemoryExtensions.ToLower(original, modified, culture);
 280            }
 281
 282            // We use simple case folding rules, which disallows moving between the BMP and supplementary
 283            // planes when performing a case conversion. The helper methods which reconstruct a Rune
 284            // contain debug asserts for this condition.
 285
 286            if (rune.IsBmp)
 287            {
 288                return UnsafeCreate(modified[0]);
 289            }
 290            else
 291            {
 292                return UnsafeCreate(UnicodeUtility.GetScalarFromUtf16SurrogatePair(modified[0], modified[1]));
 293            }
 294        }
 295#endif
 296
 0297        public int CompareTo(Rune other) => this.Value - other.Value; // values don't span entire 32-bit domain; won't i
 298
 299        internal ReadOnlySpan<char> AsSpan(Span<char> buffer)
 300        {
 0301            Debug.Assert(buffer.Length >= MaxUtf16CharsPerRune);
 0302            int charsWritten = EncodeToUtf16(buffer);
 0303            return buffer.Slice(0, charsWritten);
 304        }
 305
 306        /// <summary>
 307        /// Decodes the <see cref="Rune"/> at the beginning of the provided UTF-16 source buffer.
 308        /// </summary>
 309        /// <returns>
 310        /// <para>
 311        /// If the source buffer begins with a valid UTF-16 encoded scalar value, returns <see cref="OperationStatus.Don
 312        /// and outs via <paramref name="result"/> the decoded <see cref="Rune"/> and via <paramref name="charsConsumed"
 313        /// number of <see langword="char"/>s used in the input buffer to encode the <see cref="Rune"/>.
 314        /// </para>
 315        /// <para>
 316        /// If the source buffer is empty or contains only a standalone UTF-16 high surrogate character, returns <see cr
 317        /// and outs via <paramref name="result"/> <see cref="ReplacementChar"/> and via <paramref name="charsConsumed"/
 318        /// </para>
 319        /// <para>
 320        /// If the source buffer begins with an ill-formed UTF-16 encoded scalar value, returns <see cref="OperationStat
 321        /// and outs via <paramref name="result"/> <see cref="ReplacementChar"/> and via <paramref name="charsConsumed"/
 322        /// <see langword="char"/>s used in the input buffer to encode the ill-formed sequence.
 323        /// </para>
 324        /// </returns>
 325        /// <remarks>
 326        /// The general calling convention is to call this method in a loop, slicing the <paramref name="source"/> buffe
 327        /// <paramref name="charsConsumed"/> elements on each iteration of the loop. On each iteration of the loop <para
 328        /// will contain the real scalar value if successfully decoded, or it will contain <see cref="ReplacementChar"/>
 329        /// the data could not be successfully decoded. This pattern provides convenient automatic U+FFFD substitution o
 330        /// invalid sequences while iterating through the loop.
 331        /// </remarks>
 332        public static OperationStatus DecodeFromUtf16(ReadOnlySpan<char> source, out Rune result, out int charsConsumed)
 333        {
 0334            if (!source.IsEmpty)
 335            {
 336                // First, check for the common case of a BMP scalar value.
 337                // If this is correct, return immediately.
 338
 0339                char firstChar = source[0];
 0340                if (TryCreate(firstChar, out result))
 341                {
 0342                    charsConsumed = 1;
 0343                    return OperationStatus.Done;
 344                }
 345
 346                // First thing we saw was a UTF-16 surrogate code point.
 347                // Let's optimistically assume for now it's a high surrogate and hope
 348                // that combining it with the next char yields useful results.
 349
 0350                if (source.Length > 1)
 351                {
 0352                    char secondChar = source[1];
 0353                    if (TryCreate(firstChar, secondChar, out result))
 354                    {
 355                        // Success! Formed a supplementary scalar value.
 0356                        charsConsumed = 2;
 0357                        return OperationStatus.Done;
 358                    }
 359                    else
 360                    {
 361                        // Either the first character was a low surrogate, or the second
 362                        // character was not a low surrogate. This is an error.
 363                        goto InvalidData;
 364                    }
 365                }
 0366                else if (!char.IsHighSurrogate(firstChar))
 367                {
 368                    // Quick check to make sure we're not going to report NeedMoreData for
 369                    // a single-element buffer where the data is a standalone low surrogate
 370                    // character. Since no additional data will ever make this valid, we'll
 371                    // report an error immediately.
 372                    goto InvalidData;
 373                }
 374            }
 375
 376            // If we got to this point, the input buffer was empty, or the buffer
 377            // was a single element in length and that element was a high surrogate char.
 378
 0379            charsConsumed = source.Length;
 0380            result = ReplacementChar;
 0381            return OperationStatus.NeedMoreData;
 382
 383        InvalidData:
 384
 0385            charsConsumed = 1; // maximal invalid subsequence for UTF-16 is always a single code unit in length
 0386            result = ReplacementChar;
 0387            return OperationStatus.InvalidData;
 388        }
 389
 390        /// <summary>
 391        /// Decodes the <see cref="Rune"/> at the beginning of the provided UTF-8 source buffer.
 392        /// </summary>
 393        /// <returns>
 394        /// <para>
 395        /// If the source buffer begins with a valid UTF-8 encoded scalar value, returns <see cref="OperationStatus.Done
 396        /// and outs via <paramref name="result"/> the decoded <see cref="Rune"/> and via <paramref name="bytesConsumed"
 397        /// number of <see langword="byte"/>s used in the input buffer to encode the <see cref="Rune"/>.
 398        /// </para>
 399        /// <para>
 400        /// If the source buffer is empty or contains only a partial UTF-8 subsequence, returns <see cref="OperationStat
 401        /// and outs via <paramref name="result"/> <see cref="ReplacementChar"/> and via <paramref name="bytesConsumed"/
 402        /// </para>
 403        /// <para>
 404        /// If the source buffer begins with an ill-formed UTF-8 encoded scalar value, returns <see cref="OperationStatu
 405        /// and outs via <paramref name="result"/> <see cref="ReplacementChar"/> and via <paramref name="bytesConsumed"/
 406        /// <see langword="char"/>s used in the input buffer to encode the ill-formed sequence.
 407        /// </para>
 408        /// </returns>
 409        /// <remarks>
 410        /// The general calling convention is to call this method in a loop, slicing the <paramref name="source"/> buffe
 411        /// <paramref name="bytesConsumed"/> elements on each iteration of the loop. On each iteration of the loop <para
 412        /// will contain the real scalar value if successfully decoded, or it will contain <see cref="ReplacementChar"/>
 413        /// the data could not be successfully decoded. This pattern provides convenient automatic U+FFFD substitution o
 414        /// invalid sequences while iterating through the loop.
 415        /// </remarks>
 416        public static OperationStatus DecodeFromUtf8(ReadOnlySpan<byte> source, out Rune result, out int bytesConsumed)
 417        {
 418            // This method follows the Unicode Standard's recommendation for detecting
 419            // the maximal subpart of an ill-formed subsequence. See The Unicode Standard,
 420            // Ch. 3.9 for more details. In summary, when reporting an invalid subsequence,
 421            // it tries to consume as many code units as possible as long as those code
 422            // units constitute the beginning of a longer well-formed subsequence per Table 3-7.
 423
 424            // Try reading source[0].
 425
 2579395426            int index = 0;
 2579395427            if (source.IsEmpty)
 428            {
 429                goto NeedsMoreData;
 430            }
 431
 2579395432            uint tempValue = source[0];
 2579395433            if (UnicodeUtility.IsAsciiCodePoint(tempValue))
 434            {
 0435                bytesConsumed = 1;
 0436                result = UnsafeCreate(tempValue);
 0437                return OperationStatus.Done;
 438            }
 439
 440            // Per Table 3-7, the beginning of a multibyte sequence must be a code unit in
 441            // the range [C2..F4]. If it's outside of that range, it's either a standalone
 442            // continuation byte, or it's an overlong two-byte sequence, or it's an out-of-range
 443            // four-byte sequence.
 444
 445            // Try reading source[1].
 446
 2579395447            index = 1;
 2579395448            if (!UnicodeUtility.IsInRangeInclusive(tempValue, 0xC2, 0xF4))
 449            {
 450                goto Invalid;
 451            }
 452
 1537997453            tempValue = (tempValue - 0xC2) << 6;
 454
 1537997455            if (source.Length <= 1)
 456            {
 457                goto NeedsMoreData;
 458            }
 459
 460            // Continuation bytes are of the form [10xxxxxx], which means that their two's
 461            // complement representation is in the range [-65..-128]. This allows us to
 462            // perform a single comparison to see if a byte is a continuation byte.
 463
 1350540464            int thisByteSignExtended = (sbyte)source[1];
 1350540465            if (thisByteSignExtended >= -64)
 466            {
 467                goto Invalid;
 468            }
 469
 193792470            tempValue += (uint)thisByteSignExtended;
 193792471            tempValue += 0x80; // remove the continuation byte marker
 193792472            tempValue += (0xC2 - 0xC0) << 6; // remove the leading byte marker
 473
 193792474            if (tempValue < 0x0800)
 475            {
 36737476                Debug.Assert(UnicodeUtility.IsInRangeInclusive(tempValue, 0x0080, 0x07FF));
 36737477                goto Finish; // this is a valid 2-byte sequence
 478            }
 479
 480            // This appears to be a 3- or 4-byte sequence. Since per Table 3-7 we now have
 481            // enough information (from just two code units) to detect overlong or surrogate
 482            // sequences, we need to perform these checks now.
 483
 157055484            if (!UnicodeUtility.IsInRangeInclusive(tempValue, ((0xE0 - 0xC0) << 6) + (0xA0 - 0x80), ((0xF4 - 0xC0) << 6)
 485            {
 486                // The first two bytes were not in the range [[E0 A0]..[F4 8F]].
 487                // This is an overlong 3-byte sequence or an out-of-range 4-byte sequence.
 488                goto Invalid;
 489            }
 490
 130319491            if (UnicodeUtility.IsInRangeInclusive(tempValue, ((0xED - 0xC0) << 6) + (0xA0 - 0x80), ((0xED - 0xC0) << 6) 
 492            {
 493                // This is a UTF-16 surrogate code point, which is invalid in UTF-8.
 494                goto Invalid;
 495            }
 496
 115524497            if (UnicodeUtility.IsInRangeInclusive(tempValue, ((0xF0 - 0xC0) << 6) + (0x80 - 0x80), ((0xF0 - 0xC0) << 6) 
 498            {
 499                // This is an overlong 4-byte sequence.
 500                goto Invalid;
 501            }
 502
 503            // The first two bytes were just fine. We don't need to perform any other checks
 504            // on the remaining bytes other than to see that they're valid continuation bytes.
 505
 506            // Try reading source[2].
 507
 115470508            index = 2;
 115470509            if (source.Length <= 2)
 510            {
 511                goto NeedsMoreData;
 512            }
 513
 96288514            thisByteSignExtended = (sbyte)source[2];
 96288515            if (thisByteSignExtended >= -64)
 516            {
 517                goto Invalid; // this byte is not a UTF-8 continuation byte
 518            }
 519
 51417520            tempValue <<= 6;
 51417521            tempValue += (uint)thisByteSignExtended;
 51417522            tempValue += 0x80; // remove the continuation byte marker
 51417523            tempValue -= (0xE0 - 0xC0) << 12; // remove the leading byte marker
 524
 51417525            if (tempValue <= 0xFFFF)
 526            {
 14921527                Debug.Assert(UnicodeUtility.IsInRangeInclusive(tempValue, 0x0800, 0xFFFF));
 14921528                goto Finish; // this is a valid 3-byte sequence
 529            }
 530
 531            // Try reading source[3].
 532
 36496533            index = 3;
 36496534            if (source.Length <= 3)
 535            {
 536                goto NeedsMoreData;
 537            }
 538
 31139539            thisByteSignExtended = (sbyte)source[3];
 31139540            if (thisByteSignExtended >= -64)
 541            {
 542                goto Invalid; // this byte is not a UTF-8 continuation byte
 543            }
 544
 3190545            tempValue <<= 6;
 3190546            tempValue += (uint)thisByteSignExtended;
 3190547            tempValue += 0x80; // remove the continuation byte marker
 3190548            tempValue -= (0xF0 - 0xE0) << 18; // remove the leading byte marker
 549
 550            // Valid 4-byte sequence
 3190551            UnicodeDebug.AssertIsValidSupplementaryPlaneScalar(tempValue);
 552
 553        Finish:
 554
 54848555            bytesConsumed = index + 1;
 54848556            Debug.Assert(1 <= bytesConsumed && bytesConsumed <= 4); // Valid subsequences are always length [1..4]
 54848557            result = UnsafeCreate(tempValue);
 54848558            return OperationStatus.Done;
 559
 560        NeedsMoreData:
 561
 211996562            Debug.Assert(0 <= index && index <= 3); // Incomplete subsequences are always length 0..3
 211996563            bytesConsumed = index;
 211996564            result = ReplacementChar;
 211996565            return OperationStatus.NeedMoreData;
 566
 567        Invalid:
 568
 2312551569            Debug.Assert(1 <= index && index <= 3); // Invalid subsequences are always length 1..3
 2312551570            bytesConsumed = index;
 2312551571            result = ReplacementChar;
 2312551572            return OperationStatus.InvalidData;
 573        }
 574
 575        /// <summary>
 576        /// Decodes the <see cref="Rune"/> at the end of the provided UTF-16 source buffer.
 577        /// </summary>
 578        /// <remarks>
 579        /// This method is very similar to <see cref="DecodeFromUtf16(ReadOnlySpan{char}, out Rune, out int)"/>, but it 
 580        /// the caller to loop backward instead of forward. The typical calling convention is that on each iteration
 581        /// of the loop, the caller should slice off the final <paramref name="charsConsumed"/> elements of
 582        /// the <paramref name="source"/> buffer.
 583        /// </remarks>
 584        public static OperationStatus DecodeLastFromUtf16(ReadOnlySpan<char> source, out Rune result, out int charsConsu
 585        {
 0586            int index = source.Length - 1;
 0587            if ((uint)index < (uint)source.Length)
 588            {
 589                // First, check for the common case of a BMP scalar value.
 590                // If this is correct, return immediately.
 591
 0592                char finalChar = source[index];
 0593                if (TryCreate(finalChar, out result))
 594                {
 0595                    charsConsumed = 1;
 0596                    return OperationStatus.Done;
 597                }
 598
 0599                if (char.IsLowSurrogate(finalChar))
 600                {
 601                    // The final character was a UTF-16 low surrogate code point.
 602                    // This must be preceded by a UTF-16 high surrogate code point, otherwise
 603                    // we have a standalone low surrogate, which is always invalid.
 604
 0605                    index--;
 0606                    if ((uint)index < (uint)source.Length)
 607                    {
 0608                        char penultimateChar = source[index];
 0609                        if (TryCreate(penultimateChar, finalChar, out result))
 610                        {
 611                            // Success! Formed a supplementary scalar value.
 0612                            charsConsumed = 2;
 0613                            return OperationStatus.Done;
 614                        }
 615                    }
 616
 617                    // If we got to this point, we saw a standalone low surrogate
 618                    // and must report an error.
 619
 0620                    charsConsumed = 1; // standalone surrogate
 0621                    result = ReplacementChar;
 0622                    return OperationStatus.InvalidData;
 623                }
 624            }
 625
 626            // If we got this far, the source buffer was empty, or the source buffer ended
 627            // with a UTF-16 high surrogate code point. These aren't errors since they could
 628            // be valid given more input data.
 629
 0630            charsConsumed = (int)((uint)(-source.Length) >> 31); // 0 -> 0, all other lengths -> 1
 0631            result = ReplacementChar;
 0632            return OperationStatus.NeedMoreData;
 633        }
 634
 635        /// <summary>
 636        /// Decodes the <see cref="Rune"/> at the end of the provided UTF-8 source buffer.
 637        /// </summary>
 638        /// <remarks>
 639        /// This method is very similar to <see cref="DecodeFromUtf8(ReadOnlySpan{byte}, out Rune, out int)"/>, but it a
 640        /// the caller to loop backward instead of forward. The typical calling convention is that on each iteration
 641        /// of the loop, the caller should slice off the final <paramref name="bytesConsumed"/> elements of
 642        /// the <paramref name="source"/> buffer.
 643        /// </remarks>
 644        public static OperationStatus DecodeLastFromUtf8(ReadOnlySpan<byte> source, out Rune value, out int bytesConsume
 645        {
 0646            int index = source.Length - 1;
 0647            if ((uint)index < (uint)source.Length)
 648            {
 649                // The buffer contains at least one byte. Let's check the fast case where the
 650                // buffer ends with an ASCII byte.
 651
 0652                uint tempValue = source[index];
 0653                if (UnicodeUtility.IsAsciiCodePoint(tempValue))
 654                {
 0655                    bytesConsumed = 1;
 0656                    value = UnsafeCreate(tempValue);
 0657                    return OperationStatus.Done;
 658                }
 659
 660                // If the final byte is not an ASCII byte, we may be beginning or in the middle of
 661                // a UTF-8 multi-code unit sequence. We need to back up until we see the start of
 662                // the multi-code unit sequence; we can detect the leading byte because all multi-byte
 663                // sequences begin with a byte whose 0x40 bit is set. Since all multi-byte sequences
 664                // are no greater than 4 code units in length, we only need to search back a maximum
 665                // of four bytes.
 666
 0667                if (((byte)tempValue & 0x40) != 0)
 668                {
 669                    // This is a UTF-8 leading byte. We'll do a forward read from here.
 670                    // It'll return invalid (if given C0, F5, etc.) or incomplete. Both are fine.
 671
 0672                    return DecodeFromUtf8(source.Slice(index), out value, out bytesConsumed);
 673                }
 674
 675                // If we got to this point, the final byte was a UTF-8 continuation byte.
 676                // Let's check the three bytes immediately preceding this, looking for the starting byte.
 677
 0678                for (int i = 3; i > 0; i--)
 679                {
 0680                    index--;
 0681                    if ((uint)index >= (uint)source.Length)
 682                    {
 683                        goto Invalid; // out of data
 684                    }
 685
 686                    // The check below will get hit for ASCII (values 00..7F) and for UTF-8 starting bytes
 687                    // (bits 0xC0 set, values C0..FF). In two's complement this is the range [-64..127].
 688                    // It's just a fast way for us to terminate the search.
 689
 0690                    if ((sbyte)source[index] >= -64)
 691                    {
 692                        goto ForwardDecode;
 693                    }
 694                }
 695
 696            Invalid:
 697
 698                // If we got to this point, either:
 699                // - the last 4 bytes of the input buffer are continuation bytes;
 700                // - the entire input buffer (if fewer than 4 bytes) consists only of continuation bytes; or
 701                // - there's no UTF-8 leading byte between the final continuation byte of the buffer and
 702                //   the previous well-formed subsequence or maximal invalid subsequence.
 703                //
 704                // In all of these cases, the final byte must be a maximal invalid subsequence of length 1.
 705                // See comment near the end of this method for more information.
 706
 0707                value = ReplacementChar;
 0708                bytesConsumed = 1;
 0709                return OperationStatus.InvalidData;
 710
 711            ForwardDecode:
 712
 713                // If we got to this point, we found an ASCII byte or a UTF-8 starting byte at position source[index].
 714                // Technically this could also mean we found an invalid byte like C0 or F5 at this position, but that's
 715                // fine since it'll be handled by the forward read. From this position, we'll perform a forward read
 716                // and see if we consumed the entirety of the buffer.
 717
 0718                source = source.Slice(index);
 0719                Debug.Assert(!source.IsEmpty, "Shouldn't reach this for empty inputs.");
 720
 0721                OperationStatus operationStatus = DecodeFromUtf8(source, out Rune tempRune, out int tempBytesConsumed);
 0722                if (tempBytesConsumed == source.Length)
 723                {
 724                    // If this forward read consumed the entirety of the end of the input buffer, we can return it
 725                    // as the result of this function. It could be well-formed, incomplete, or invalid. If it's
 726                    // invalid and we consumed the remainder of the buffer, we know we've found the maximal invalid
 727                    // subsequence, which is what we wanted anyway.
 728
 0729                    bytesConsumed = tempBytesConsumed;
 0730                    value = tempRune;
 0731                    return operationStatus;
 732                }
 733
 734                // If we got to this point, we know that the final continuation byte wasn't consumed by the forward
 735                // read that we just performed above. This means that the continuation byte has to be part of an
 736                // invalid subsequence since there's no UTF-8 leading byte between what we just consumed and the
 737                // continuation byte at the end of the input. Furthermore, since any maximal invalid subsequence
 738                // of length > 1 must have a UTF-8 leading byte as its first code unit, this implies that the
 739                // continuation byte at the end of the buffer is itself a maximal invalid subsequence of length 1.
 740
 741                goto Invalid;
 742            }
 743            else
 744            {
 745                // Source buffer was empty.
 0746                value = ReplacementChar;
 0747                bytesConsumed = 0;
 0748                return OperationStatus.NeedMoreData;
 749            }
 750        }
 751
 752        /// <summary>
 753        /// Encodes this <see cref="Rune"/> to a UTF-16 destination buffer.
 754        /// </summary>
 755        /// <param name="destination">The buffer to which to write this value as UTF-16.</param>
 756        /// <returns>The number of <see cref="char"/>s written to <paramref name="destination"/>.</returns>
 757        /// <exception cref="ArgumentException">
 758        /// If <paramref name="destination"/> is not large enough to hold the output.
 759        /// </exception>
 760        public int EncodeToUtf16(Span<char> destination)
 761        {
 0762            if (!TryEncodeToUtf16(destination, out int charsWritten))
 763            {
 0764                ThrowHelper.ThrowArgumentException_DestinationTooShort();
 765            }
 766
 0767            return charsWritten;
 768        }
 769
 770        /// <summary>
 771        /// Encodes this <see cref="Rune"/> to a UTF-8 destination buffer.
 772        /// </summary>
 773        /// <param name="destination">The buffer to which to write this value as UTF-8.</param>
 774        /// <returns>The number of <see cref="byte"/>s written to <paramref name="destination"/>.</returns>
 775        /// <exception cref="ArgumentException">
 776        /// If <paramref name="destination"/> is not large enough to hold the output.
 777        /// </exception>
 778        public int EncodeToUtf8(Span<byte> destination)
 779        {
 0780            if (!TryEncodeToUtf8(destination, out int bytesWritten))
 781            {
 0782                ThrowHelper.ThrowArgumentException_DestinationTooShort();
 783            }
 784
 0785            return bytesWritten;
 786        }
 787
 0788        public override bool Equals([NotNullWhen(true)] object? obj) => (obj is Rune other) && Equals(other);
 789
 0790        public bool Equals(Rune other) => this == other;
 791
 792        /// <summary>
 793        /// Returns a value that indicates whether the current instance and a specified rune are equal using the specifi
 794        /// </summary>
 795        /// <param name="other">The rune to compare with the current instance.</param>
 796        /// <param name="comparisonType">One of the enumeration values that specifies the rules to use in the comparison
 797        /// <returns><see langword="true"/> if the current instance and <paramref name="other"/> are equal; otherwise, <
 798        public unsafe bool Equals(Rune other, StringComparison comparisonType)
 799        {
 0800            if (comparisonType is StringComparison.Ordinal)
 801            {
 0802                return this == other;
 803            }
 804
 805            // Convert this to span
 0806            ReadOnlySpan<char> thisChars = AsSpan(stackalloc char[MaxUtf16CharsPerRune]);
 807
 808            // Convert other to span
 0809            ReadOnlySpan<char> otherChars = other.AsSpan(stackalloc char[MaxUtf16CharsPerRune]);
 810
 811            // Compare span equality
 0812            return thisChars.Equals(otherChars, comparisonType);
 813        }
 814
 0815        public override int GetHashCode() => Value;
 816
 817#if SYSTEM_PRIVATE_CORELIB
 818        /// <summary>
 819        /// Gets the <see cref="Rune"/> which begins at index <paramref name="index"/> in
 820        /// string <paramref name="input"/>.
 821        /// </summary>
 822        /// <remarks>
 823        /// Throws if <paramref name="input"/> is null, if <paramref name="index"/> is out of range, or
 824        /// if <paramref name="index"/> does not reference the start of a valid scalar value within <paramref name="inpu
 825        /// </remarks>
 826        public static Rune GetRuneAt(string input, int index)
 827        {
 0828            int runeValue = ReadRuneFromString(input, index);
 0829            if (runeValue < 0)
 830            {
 0831                ThrowHelper.ThrowArgumentException_CannotExtractScalar(ExceptionArgument.index);
 832            }
 833
 0834            return UnsafeCreate((uint)runeValue);
 835        }
 836#endif
 837
 838        /// <summary>
 839        /// Returns <see langword="true"/> iff <paramref name="value"/> is a valid Unicode scalar
 840        /// value, i.e., is in [ U+0000..U+D7FF ], inclusive; or [ U+E000..U+10FFFF ], inclusive.
 841        /// </summary>
 0842        public static bool IsValid(int value) => IsValid((uint)value);
 843
 844        /// <summary>
 845        /// Returns <see langword="true"/> iff <paramref name="value"/> is a valid Unicode scalar
 846        /// value, i.e., is in [ U+0000..U+D7FF ], inclusive; or [ U+E000..U+10FFFF ], inclusive.
 847        /// </summary>
 848        [CLSCompliant(false)]
 0849        public static bool IsValid(uint value) => UnicodeUtility.IsValidUnicodeScalar(value);
 850
 851        // returns a negative number on failure
 852        internal static int ReadFirstRuneFromUtf16Buffer(ReadOnlySpan<char> input)
 853        {
 0854            if (input.IsEmpty)
 855            {
 0856                return -1;
 857            }
 858
 859            // Optimistically assume input is within BMP.
 860
 0861            uint returnValue = input[0];
 0862            if (UnicodeUtility.IsSurrogateCodePoint(returnValue))
 863            {
 0864                if (!UnicodeUtility.IsHighSurrogateCodePoint(returnValue))
 865                {
 0866                    return -1;
 867                }
 868
 869                // Treat 'returnValue' as the high surrogate.
 870
 0871                if (input.Length <= 1)
 872                {
 0873                    return -1; // not an argument exception - just a "bad data" failure
 874                }
 875
 0876                uint potentialLowSurrogate = input[1];
 0877                if (!UnicodeUtility.IsLowSurrogateCodePoint(potentialLowSurrogate))
 878                {
 0879                    return -1;
 880                }
 881
 0882                returnValue = UnicodeUtility.GetScalarFromUtf16SurrogatePair(returnValue, potentialLowSurrogate);
 883            }
 884
 0885            return (int)returnValue;
 886        }
 887
 888#if SYSTEM_PRIVATE_CORELIB
 889        // returns a negative number on failure
 890        private static int ReadRuneFromString(string input, int index)
 891        {
 0892            if (input is null)
 893            {
 0894                ThrowHelper.ThrowArgumentNullException(ExceptionArgument.input);
 895            }
 896
 0897            if ((uint)index >= (uint)input.Length)
 898            {
 0899                ThrowHelper.ThrowArgumentOutOfRange_IndexMustBeLessException();
 900            }
 901
 902            // Optimistically assume input is within BMP.
 903
 0904            uint returnValue = input[index];
 0905            if (UnicodeUtility.IsSurrogateCodePoint(returnValue))
 906            {
 0907                if (!UnicodeUtility.IsHighSurrogateCodePoint(returnValue))
 908                {
 0909                    return -1;
 910                }
 911
 912                // Treat 'returnValue' as the high surrogate.
 913                //
 914                // If this becomes a hot code path, we can skip the below bounds check by reading
 915                // off the end of the string using unsafe code. Since strings are null-terminated,
 916                // we're guaranteed not to read a valid low surrogate, so we'll fail correctly if
 917                // the string terminates unexpectedly.
 918
 0919                index++;
 0920                if ((uint)index >= (uint)input.Length)
 921                {
 0922                    return -1; // not an argument exception - just a "bad data" failure
 923                }
 924
 0925                uint potentialLowSurrogate = input[index];
 0926                if (!UnicodeUtility.IsLowSurrogateCodePoint(potentialLowSurrogate))
 927                {
 0928                    return -1;
 929                }
 930
 0931                returnValue = UnicodeUtility.GetScalarFromUtf16SurrogatePair(returnValue, potentialLowSurrogate);
 932            }
 933
 0934            return (int)returnValue;
 935        }
 936#endif
 937
 938        /// <summary>
 939        /// Returns a <see cref="string"/> representation of this <see cref="Rune"/> instance.
 940        /// </summary>
 941        public override unsafe string ToString()
 942        {
 943#if SYSTEM_PRIVATE_CORELIB
 0944            if (IsBmp)
 945            {
 0946                return string.CreateFromChar((char)_value);
 947            }
 948            else
 949            {
 0950                UnicodeUtility.GetUtf16SurrogatesFromSupplementaryPlaneScalar(_value, out char high, out char low);
 0951                return string.CreateFromChar(high, low);
 952            }
 953#else
 954            if (IsBmp)
 955            {
 956                return ((char)_value).ToString();
 957            }
 958            else
 959            {
 960                Span<char> buffer = stackalloc char[MaxUtf16CharsPerRune];
 961                UnicodeUtility.GetUtf16SurrogatesFromSupplementaryPlaneScalar(_value, out buffer[0], out buffer[1]);
 962                return buffer.ToString();
 963            }
 964#endif
 965        }
 966
 967#if SYSTEM_PRIVATE_CORELIB
 968        bool ISpanFormattable.TryFormat(Span<char> destination, out int charsWritten, ReadOnlySpan<char> format, IFormat
 0969            TryEncodeToUtf16(destination, out charsWritten);
 970
 971        bool IUtf8SpanFormattable.TryFormat(Span<byte> utf8Destination, out int bytesWritten, ReadOnlySpan<char> format,
 0972            TryEncodeToUtf8(utf8Destination, out bytesWritten);
 973
 974        /// <inheritdoc cref="IUtf8SpanParsable{TSelf}.TryParse(ReadOnlySpan{byte}, IFormatProvider?, out TSelf)" />
 975        static bool IUtf8SpanParsable<Rune>.TryParse(ReadOnlySpan<byte> utf8Text, IFormatProvider? provider, out Rune re
 976        {
 0977            if (DecodeFromUtf8(utf8Text, out result, out int bytesConsumed) == OperationStatus.Done)
 978            {
 0979                if (bytesConsumed == utf8Text.Length)
 980                {
 0981                    return true;
 982                }
 983
 0984                result = ReplacementChar;
 985            }
 986
 0987            return false;
 988        }
 989
 990        /// <inheritdoc cref="IUtf8SpanParsable{TSelf}.Parse(ReadOnlySpan{byte}, IFormatProvider?)" />
 991        static Rune IUtf8SpanParsable<Rune>.Parse(ReadOnlySpan<byte> utf8Text, System.IFormatProvider? provider)
 992        {
 0993            if (DecodeFromUtf8(utf8Text, out Rune result, out int bytesConsumed) != OperationStatus.Done || bytesConsume
 994            {
 0995                ThrowHelper.ThrowFormatInvalidString();
 996            }
 997
 0998            return result;
 999        }
 1000
 1001        /// <inheritdoc cref="IParsable{TSelf}.Parse(string, IFormatProvider?)" />
 1002        static Rune IParsable<Rune>.Parse(string s, IFormatProvider? provider)
 1003        {
 01004            ArgumentNullException.ThrowIfNull(s);
 1005
 01006            if (DecodeFromUtf16(s, out Rune result, out int charsConsumed) != OperationStatus.Done || charsConsumed != s
 1007            {
 01008                ThrowHelper.ThrowFormatInvalidString();
 1009            }
 1010
 01011            return result;
 1012        }
 1013
 1014        /// <inheritdoc cref="IParsable{TSelf}.TryParse(string?, IFormatProvider?, out TSelf)" />
 1015        static bool IParsable<Rune>.TryParse([NotNullWhen(true)] string? s, IFormatProvider? provider, out Rune result)
 1016        {
 01017            if (DecodeFromUtf16(s, out result, out int charsConsumed) != OperationStatus.Done || charsConsumed != s!.Len
 1018            {
 01019                result = ReplacementChar;
 01020                return false;
 1021            }
 1022
 01023            return true;
 1024        }
 1025
 1026        /// <inheritdoc cref="ISpanParsable{TSelf}.Parse(ReadOnlySpan{char}, IFormatProvider?)" />
 1027        static Rune ISpanParsable<Rune>.Parse(ReadOnlySpan<char> s, IFormatProvider? provider)
 1028        {
 01029            if (DecodeFromUtf16(s, out Rune result, out int charsConsumed) != OperationStatus.Done || charsConsumed != s
 1030            {
 01031                ThrowHelper.ThrowFormatInvalidString();
 1032            }
 1033
 01034            return result;
 1035        }
 1036
 1037        /// <inheritdoc cref="ISpanParsable{TSelf}.TryParse(ReadOnlySpan{char}, IFormatProvider?, out TSelf)" />
 1038        static bool ISpanParsable<Rune>.TryParse(ReadOnlySpan<char> s, IFormatProvider? provider, out Rune result)
 1039        {
 01040            if (DecodeFromUtf16(s, out result, out int charsConsumed) != OperationStatus.Done || charsConsumed != s.Leng
 1041            {
 01042                result = ReplacementChar;
 01043                return false;
 1044            }
 1045
 01046            return true;
 1047        }
 1048
 01049        string IFormattable.ToString(string? format, IFormatProvider? formatProvider) => ToString();
 1050#endif
 1051
 1052        /// <summary>
 1053        /// Attempts to create a <see cref="Rune"/> from the provided input value.
 1054        /// </summary>
 1055        public static bool TryCreate(char ch, out Rune result)
 1056        {
 18657961057            uint extendedValue = ch;
 18657961058            if (!UnicodeUtility.IsSurrogateCodePoint(extendedValue))
 1059            {
 18657961060                result = UnsafeCreate(extendedValue);
 18657961061                return true;
 1062            }
 1063            else
 1064            {
 01065                result = default;
 01066                return false;
 1067            }
 1068        }
 1069
 1070        /// <summary>
 1071        /// Attempts to create a <see cref="Rune"/> from the provided UTF-16 surrogate pair.
 1072        /// Returns <see langword="false"/> if the input values don't represent a well-formed UTF-16surrogate pair.
 1073        /// </summary>
 1074        public static bool TryCreate(char highSurrogate, char lowSurrogate, out Rune result)
 1075        {
 1076            // First, extend both to 32 bits, then calculate the offset of
 1077            // each candidate surrogate char from the start of its range.
 1078
 01079            uint highSurrogateOffset = (uint)highSurrogate - HighSurrogateStart;
 01080            uint lowSurrogateOffset = (uint)lowSurrogate - LowSurrogateStart;
 1081
 1082            // This is a single comparison which allows us to check both for validity at once since
 1083            // both the high surrogate range and the low surrogate range are the same length.
 1084            // If the comparison fails, we call to a helper method to throw the correct exception message.
 1085
 01086            if ((highSurrogateOffset | lowSurrogateOffset) <= HighSurrogateRange)
 1087            {
 1088                // The 0x40u << 10 below is to account for uuuuu = wwww + 1 in the surrogate encoding.
 01089                result = UnsafeCreate((highSurrogateOffset << 10) + ((uint)lowSurrogate - LowSurrogateStart) + (0x40u <<
 01090                return true;
 1091            }
 1092            else
 1093            {
 1094                // Didn't have a high surrogate followed by a low surrogate.
 01095                result = default;
 01096                return false;
 1097            }
 1098        }
 1099
 1100        /// <summary>
 1101        /// Attempts to create a <see cref="Rune"/> from the provided input value.
 1102        /// </summary>
 01103        public static bool TryCreate(int value, out Rune result) => TryCreate((uint)value, out result);
 1104
 1105        /// <summary>
 1106        /// Attempts to create a <see cref="Rune"/> from the provided input value.
 1107        /// </summary>
 1108        [CLSCompliant(false)]
 1109        public static bool TryCreate(uint value, out Rune result)
 1110        {
 01111            if (UnicodeUtility.IsValidUnicodeScalar(value))
 1112            {
 01113                result = UnsafeCreate(value);
 01114                return true;
 1115            }
 1116            else
 1117            {
 01118                result = default;
 01119                return false;
 1120            }
 1121        }
 1122
 1123        /// <summary>
 1124        /// Encodes this <see cref="Rune"/> to a UTF-16 destination buffer.
 1125        /// </summary>
 1126        /// <param name="destination">The buffer to which to write this value as UTF-16.</param>
 1127        /// <param name="charsWritten">
 1128        /// The number of <see cref="char"/>s written to <paramref name="destination"/>,
 1129        /// or 0 if the destination buffer is not large enough to contain the output.</param>
 1130        /// <returns>True if the value was written to the buffer; otherwise, false.</returns>
 1131        /// <remarks>
 1132        /// The <see cref="Utf16SequenceLength"/> property can be queried ahead of time to determine
 1133        /// the required size of the <paramref name="destination"/> buffer.
 1134        /// </remarks>
 1135        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 1136        public bool TryEncodeToUtf16(Span<char> destination, out int charsWritten)
 1137        {
 1138            // The Rune type fits cleanly into a register, so pass byval rather than byref
 1139            // to avoid stack-spilling the 'this' parameter.
 1758801140            return TryEncodeToUtf16(this, destination, out charsWritten);
 1141        }
 1142
 1143        private static bool TryEncodeToUtf16(Rune value, Span<char> destination, out int charsWritten)
 1144        {
 1758801145            if (!destination.IsEmpty)
 1146            {
 1758801147                if (value.IsBmp)
 1148                {
 1726901149                    destination[0] = (char)value._value;
 1726901150                    charsWritten = 1;
 1726901151                    return true;
 1152                }
 31901153                else if (destination.Length > 1)
 1154                {
 31901155                    UnicodeUtility.GetUtf16SurrogatesFromSupplementaryPlaneScalar((uint)value._value, out destination[0]
 31901156                    charsWritten = 2;
 31901157                    return true;
 1158                }
 1159            }
 1160
 1161            // Destination buffer not large enough
 1162
 01163            charsWritten = default;
 01164            return false;
 1165        }
 1166
 1167        /// <summary>
 1168        /// Encodes this <see cref="Rune"/> to a destination buffer as UTF-8 bytes.
 1169        /// </summary>
 1170        /// <param name="destination">The buffer to which to write this value as UTF-8.</param>
 1171        /// <param name="bytesWritten">
 1172        /// The number of <see cref="byte"/>s written to <paramref name="destination"/>,
 1173        /// or 0 if the destination buffer is not large enough to contain the output.</param>
 1174        /// <returns>True if the value was written to the buffer; otherwise, false.</returns>
 1175        /// <remarks>
 1176        /// The <see cref="Utf8SequenceLength"/> property can be queried ahead of time to determine
 1177        /// the required size of the <paramref name="destination"/> buffer.
 1178        /// </remarks>
 1179        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 1180        public bool TryEncodeToUtf8(Span<byte> destination, out int bytesWritten)
 1181        {
 1182            // The Rune type fits cleanly into a register, so pass byval rather than byref
 1183            // to avoid stack-spilling the 'this' parameter.
 01184            return TryEncodeToUtf8(this, destination, out bytesWritten);
 1185        }
 1186
 1187        private static bool TryEncodeToUtf8(Rune value, Span<byte> destination, out int bytesWritten)
 1188        {
 1189            // The bit patterns below come from the Unicode Standard, Table 3-6.
 1190
 01191            if (!destination.IsEmpty)
 1192            {
 01193                if (value.IsAscii)
 1194                {
 01195                    destination[0] = (byte)value._value;
 01196                    bytesWritten = 1;
 01197                    return true;
 1198                }
 1199
 01200                if (destination.Length > 1)
 1201                {
 01202                    if (value.Value <= 0x7FFu)
 1203                    {
 1204                        // Scalar 00000yyy yyxxxxxx -> bytes [ 110yyyyy 10xxxxxx ]
 01205                        destination[0] = (byte)((value._value + (0b110u << 11)) >> 6);
 01206                        destination[1] = (byte)((value._value & 0x3Fu) + 0x80u);
 01207                        bytesWritten = 2;
 01208                        return true;
 1209                    }
 1210
 01211                    if (destination.Length > 2)
 1212                    {
 01213                        if (value.Value <= 0xFFFFu)
 1214                        {
 1215                            // Scalar zzzzyyyy yyxxxxxx -> bytes [ 1110zzzz 10yyyyyy 10xxxxxx ]
 01216                            destination[0] = (byte)((value._value + (0b1110 << 16)) >> 12);
 01217                            destination[1] = (byte)(((value._value & (0x3Fu << 6)) >> 6) + 0x80u);
 01218                            destination[2] = (byte)((value._value & 0x3Fu) + 0x80u);
 01219                            bytesWritten = 3;
 01220                            return true;
 1221                        }
 1222
 01223                        if (destination.Length > 3)
 1224                        {
 1225                            // Scalar 000uuuuu zzzzyyyy yyxxxxxx -> bytes [ 11110uuu 10uuzzzz 10yyyyyy 10xxxxxx ]
 01226                            destination[0] = (byte)((value._value + (0b11110 << 21)) >> 18);
 01227                            destination[1] = (byte)(((value._value & (0x3Fu << 12)) >> 12) + 0x80u);
 01228                            destination[2] = (byte)(((value._value & (0x3Fu << 6)) >> 6) + 0x80u);
 01229                            destination[3] = (byte)((value._value & 0x3Fu) + 0x80u);
 01230                            bytesWritten = 4;
 01231                            return true;
 1232                        }
 1233                    }
 1234                }
 1235            }
 1236
 1237            // Destination buffer not large enough
 1238
 01239            bytesWritten = default;
 01240            return false;
 1241        }
 1242
 1243#if SYSTEM_PRIVATE_CORELIB
 1244        /// <summary>
 1245        /// Attempts to get the <see cref="Rune"/> which begins at index <paramref name="index"/> in
 1246        /// string <paramref name="input"/>.
 1247        /// </summary>
 1248        /// <returns><see langword="true"/> if a scalar value was successfully extracted from the specified index,
 1249        /// <see langword="false"/> if a value could not be extracted due to invalid data.</returns>
 1250        /// <remarks>
 1251        /// Throws only if <paramref name="input"/> is null or <paramref name="index"/> is out of range.
 1252        /// </remarks>
 1253        public static bool TryGetRuneAt(string input, int index, out Rune value)
 1254        {
 01255            int runeValue = ReadRuneFromString(input, index);
 01256            if (runeValue >= 0)
 1257            {
 01258                value = UnsafeCreate((uint)runeValue);
 01259                return true;
 1260            }
 1261            else
 1262            {
 01263                value = default;
 01264                return false;
 1265            }
 1266        }
 1267#endif
 1268
 1269        // Allows constructing a Unicode scalar value from an arbitrary 32-bit integer without
 1270        // validation. It is the caller's responsibility to have performed manual validation
 1271        // before calling this method. If a Rune instance is forcibly constructed
 1272        // from invalid input, the APIs on this type have undefined behavior, potentially including
 1273        // introducing a security hole in the consuming application.
 1274        //
 1275        // An example of a security hole resulting from an invalid Rune value, which could result
 1276        // in a stack overflow.
 1277        //
 1278        // public int GetMarvin32HashCode(Rune r) {
 1279        //   Span<char> buffer = stackalloc char[r.Utf16SequenceLength];
 1280        //   r.TryEncode(buffer, ...);
 1281        //   return Marvin32.ComputeHash(buffer.AsBytes());
 1282        // }
 1283
 1284        /// <summary>
 1285        /// Creates a <see cref="Rune"/> without performing validation on the input.
 1286        /// </summary>
 1287        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 44451911288        internal static Rune UnsafeCreate(uint scalarValue) => new Rune(scalarValue, false);
 1289
 1290        // These are analogs of APIs on System.Char
 1291
 1292        public static double GetNumericValue(Rune value)
 1293        {
 01294            if (value.IsAscii)
 1295            {
 01296                uint baseNum = value._value - '0';
 01297                return (baseNum <= 9) ? (double)baseNum : -1;
 1298            }
 1299            else
 1300            {
 1301                // not an ASCII char; fall back to globalization table
 1302#if SYSTEM_PRIVATE_CORELIB
 01303                return CharUnicodeInfo.GetNumericValue(value.Value);
 1304#else
 1305                if (value.IsBmp)
 1306                {
 1307                    return CharUnicodeInfo.GetNumericValue((char)value._value);
 1308                }
 1309                return CharUnicodeInfo.GetNumericValue(value.ToString(), 0);
 1310#endif
 1311            }
 1312        }
 1313
 1314        public static UnicodeCategory GetUnicodeCategory(Rune value)
 1315        {
 01316            if (value.IsAscii)
 1317            {
 01318                return (UnicodeCategory)(AsciiCharInfo[value.Value] & UnicodeCategoryMask);
 1319            }
 1320            else
 1321            {
 01322                return GetUnicodeCategoryNonAscii(value);
 1323            }
 1324        }
 1325
 1326        private static UnicodeCategory GetUnicodeCategoryNonAscii(Rune value)
 1327        {
 01328            Debug.Assert(!value.IsAscii, "Shouldn't use this non-optimized code path for ASCII characters.");
 1329#if (!NETSTANDARD2_0 && !NETFRAMEWORK)
 01330            return CharUnicodeInfo.GetUnicodeCategory(value.Value);
 1331#else
 1332            if (value.IsBmp)
 1333            {
 1334                return CharUnicodeInfo.GetUnicodeCategory((char)value._value);
 1335            }
 1336            return CharUnicodeInfo.GetUnicodeCategory(value.ToString(), 0);
 1337#endif
 1338        }
 1339
 1340        // Returns true iff this Unicode category represents a letter
 1341        private static bool IsCategoryLetter(UnicodeCategory category)
 1342        {
 01343            return UnicodeUtility.IsInRangeInclusive((uint)category, (uint)UnicodeCategory.UppercaseLetter, (uint)Unicod
 1344        }
 1345
 1346        // Returns true iff this Unicode category represents a letter or a decimal digit
 1347        private static bool IsCategoryLetterOrDecimalDigit(UnicodeCategory category)
 1348        {
 01349            return UnicodeUtility.IsInRangeInclusive((uint)category, (uint)UnicodeCategory.UppercaseLetter, (uint)Unicod
 01350                || (category == UnicodeCategory.DecimalDigitNumber);
 1351        }
 1352
 1353        // Returns true iff this Unicode category represents a number
 1354        private static bool IsCategoryNumber(UnicodeCategory category)
 1355        {
 01356            return UnicodeUtility.IsInRangeInclusive((uint)category, (uint)UnicodeCategory.DecimalDigitNumber, (uint)Uni
 1357        }
 1358
 1359        // Returns true iff this Unicode category represents a punctuation mark
 1360        private static bool IsCategoryPunctuation(UnicodeCategory category)
 1361        {
 01362            return UnicodeUtility.IsInRangeInclusive((uint)category, (uint)UnicodeCategory.ConnectorPunctuation, (uint)U
 1363        }
 1364
 1365        // Returns true iff this Unicode category represents a separator
 1366        private static bool IsCategorySeparator(UnicodeCategory category)
 1367        {
 01368            return UnicodeUtility.IsInRangeInclusive((uint)category, (uint)UnicodeCategory.SpaceSeparator, (uint)Unicode
 1369        }
 1370
 1371        // Returns true iff this Unicode category represents a symbol
 1372        private static bool IsCategorySymbol(UnicodeCategory category)
 1373        {
 01374            return UnicodeUtility.IsInRangeInclusive((uint)category, (uint)UnicodeCategory.MathSymbol, (uint)UnicodeCate
 1375        }
 1376
 1377        public static bool IsControl(Rune value)
 1378        {
 1379            // Per the Unicode stability policy, the set of control characters
 1380            // is forever fixed at [ U+0000..U+001F ], [ U+007F..U+009F ]. No
 1381            // characters will ever be added to or removed from the "control characters"
 1382            // group. See https://www.unicode.org/policies/stability_policy.html.
 1383
 1384            // Logic below depends on Rune.Value never being -1 (since Rune is a validating type)
 1385            // 00..1F (+1) => 01..20 (&~80) => 01..20
 1386            // 7F..9F (+1) => 80..A0 (&~80) => 00..20
 1387
 01388            return ((value._value + 1) & ~0x80u) <= 0x20u;
 1389        }
 1390
 1391        public static bool IsDigit(Rune value)
 1392        {
 01393            if (value.IsAscii)
 1394            {
 01395                return UnicodeUtility.IsInRangeInclusive(value._value, '0', '9');
 1396            }
 1397            else
 1398            {
 01399                return GetUnicodeCategoryNonAscii(value) == UnicodeCategory.DecimalDigitNumber;
 1400            }
 1401        }
 1402
 1403        public static bool IsLetter(Rune value)
 1404        {
 01405            if (value.IsAscii)
 1406            {
 01407                return ((value._value - 'A') & ~0x20u) <= (uint)('Z' - 'A'); // [A-Za-z]
 1408            }
 1409            else
 1410            {
 01411                return IsCategoryLetter(GetUnicodeCategoryNonAscii(value));
 1412            }
 1413        }
 1414
 1415        public static bool IsLetterOrDigit(Rune value)
 1416        {
 01417            if (value.IsAscii)
 1418            {
 01419                return (AsciiCharInfo[value.Value] & IsLetterOrDigitFlag) != 0;
 1420            }
 1421            else
 1422            {
 01423                return IsCategoryLetterOrDecimalDigit(GetUnicodeCategoryNonAscii(value));
 1424            }
 1425        }
 1426
 1427        public static bool IsLower(Rune value)
 1428        {
 01429            if (value.IsAscii)
 1430            {
 01431                return UnicodeUtility.IsInRangeInclusive(value._value, 'a', 'z');
 1432            }
 1433            else
 1434            {
 01435                return GetUnicodeCategoryNonAscii(value) == UnicodeCategory.LowercaseLetter;
 1436            }
 1437        }
 1438
 1439        public static bool IsNumber(Rune value)
 1440        {
 01441            if (value.IsAscii)
 1442            {
 01443                return UnicodeUtility.IsInRangeInclusive(value._value, '0', '9');
 1444            }
 1445            else
 1446            {
 01447                return IsCategoryNumber(GetUnicodeCategoryNonAscii(value));
 1448            }
 1449        }
 1450
 1451        public static bool IsPunctuation(Rune value)
 1452        {
 01453            return IsCategoryPunctuation(GetUnicodeCategory(value));
 1454        }
 1455
 1456        public static bool IsSeparator(Rune value)
 1457        {
 01458            return IsCategorySeparator(GetUnicodeCategory(value));
 1459        }
 1460
 1461        public static bool IsSymbol(Rune value)
 1462        {
 01463            return IsCategorySymbol(GetUnicodeCategory(value));
 1464        }
 1465
 1466        public static bool IsUpper(Rune value)
 1467        {
 01468            if (value.IsAscii)
 1469            {
 01470                return UnicodeUtility.IsInRangeInclusive(value._value, 'A', 'Z');
 1471            }
 1472            else
 1473            {
 01474                return GetUnicodeCategoryNonAscii(value) == UnicodeCategory.UppercaseLetter;
 1475            }
 1476        }
 1477
 1478        public static bool IsWhiteSpace(Rune value)
 1479        {
 01480            if (value.IsAscii)
 1481            {
 01482                return (AsciiCharInfo[value.Value] & IsWhiteSpaceFlag) != 0;
 1483            }
 1484
 1485            // Only BMP code points can be white space, so only call into CharUnicodeInfo
 1486            // if the incoming value is within the BMP.
 1487
 01488            return value.IsBmp &&
 01489#if SYSTEM_PRIVATE_CORELIB
 01490                CharUnicodeInfo.GetIsWhiteSpace((char)value._value);
 1491#else
 1492                char.IsWhiteSpace((char)value._value);
 1493#endif
 1494        }
 1495
 1496        public static Rune ToLower(Rune value, CultureInfo culture)
 1497        {
 01498            if (culture is null)
 1499            {
 01500                ThrowHelper.ThrowArgumentNullException(ExceptionArgument.culture);
 1501            }
 1502
 1503            // We don't want to special-case ASCII here since the specified culture might handle
 1504            // ASCII characters differently than the invariant culture (e.g., Turkish I). Instead
 1505            // we'll just jump straight to the globalization tables if they're available.
 1506
 1507#if SYSTEM_PRIVATE_CORELIB
 01508            if (GlobalizationMode.Invariant)
 1509            {
 01510                return ToLowerInvariant(value);
 1511            }
 1512
 01513            return ChangeCaseCultureAware(value, culture.TextInfo, toUpper: false);
 1514#else
 1515            return ChangeCaseCultureAware(value, culture, toUpper: false);
 1516#endif
 1517        }
 1518
 1519        public static Rune ToLowerInvariant(Rune value)
 1520        {
 1521            // Handle the most common case (ASCII data) first. Within the common case, we expect
 1522            // that there'll be a mix of lowercase & uppercase chars, so make the conversion branchless.
 1523
 01524            if (value.IsAscii)
 1525            {
 1526                // It's ok for us to use the UTF-16 conversion utility for this since the high
 1527                // 16 bits of the value will never be set so will be left unchanged.
 01528                return UnsafeCreate(Utf16Utility.ConvertAllAsciiCharsInUInt32ToLowercase(value._value));
 1529            }
 1530
 1531#if SYSTEM_PRIVATE_CORELIB
 01532            if (GlobalizationMode.Invariant)
 1533            {
 01534                return UnsafeCreate(CharUnicodeInfo.ToLower(value._value));
 1535            }
 1536
 1537            // Non-ASCII data requires going through the case folding tables.
 1538
 01539            return ChangeCaseCultureAware(value, TextInfo.Invariant, toUpper: false);
 1540#else
 1541            return ChangeCaseCultureAware(value, CultureInfo.InvariantCulture, toUpper: false);
 1542#endif
 1543        }
 1544
 1545        public static Rune ToUpper(Rune value, CultureInfo culture)
 1546        {
 01547            if (culture is null)
 1548            {
 01549                ThrowHelper.ThrowArgumentNullException(ExceptionArgument.culture);
 1550            }
 1551
 1552            // We don't want to special-case ASCII here since the specified culture might handle
 1553            // ASCII characters differently than the invariant culture (e.g., Turkish I). Instead
 1554            // we'll just jump straight to the globalization tables if they're available.
 1555
 1556#if SYSTEM_PRIVATE_CORELIB
 01557            if (GlobalizationMode.Invariant)
 1558            {
 01559                return ToUpperInvariant(value);
 1560            }
 1561
 01562            return ChangeCaseCultureAware(value, culture.TextInfo, toUpper: true);
 1563#else
 1564            return ChangeCaseCultureAware(value, culture, toUpper: true);
 1565#endif
 1566        }
 1567
 1568        public static Rune ToUpperInvariant(Rune value)
 1569        {
 1570            // Handle the most common case (ASCII data) first. Within the common case, we expect
 1571            // that there'll be a mix of lowercase & uppercase chars, so make the conversion branchless.
 1572
 01573            if (value.IsAscii)
 1574            {
 1575                // It's ok for us to use the UTF-16 conversion utility for this since the high
 1576                // 16 bits of the value will never be set so will be left unchanged.
 01577                return UnsafeCreate(Utf16Utility.ConvertAllAsciiCharsInUInt32ToUppercase(value._value));
 1578            }
 1579
 1580#if SYSTEM_PRIVATE_CORELIB
 01581            if (GlobalizationMode.Invariant)
 1582            {
 01583                return UnsafeCreate(CharUnicodeInfo.ToUpper(value._value));
 1584            }
 1585
 1586            // Non-ASCII data requires going through the case folding tables.
 1587
 01588            return ChangeCaseCultureAware(value, TextInfo.Invariant, toUpper: true);
 1589#else
 1590            return ChangeCaseCultureAware(value, CultureInfo.InvariantCulture, toUpper: true);
 1591#endif
 1592        }
 1593
 1594#if SYSTEM_PRIVATE_CORELIB
 1595        /// <summary>
 1596        /// Returns a copy of <paramref name="value"/> converted to uppercase using the casing rules used by
 1597        /// <see cref="StringComparison.OrdinalIgnoreCase"/> comparisons.
 1598        /// </summary>
 1599        /// <param name="value">The character to convert.</param>
 1600        /// <returns>The uppercase equivalent of <paramref name="value"/>.</returns>
 1601        public static Rune ToUpperOrdinal(Rune value)
 1602        {
 01603            if (value.IsAscii)
 1604            {
 01605                return UnsafeCreate(Utf16Utility.ConvertAllAsciiCharsInUInt32ToUppercase(value._value));
 1606            }
 1607
 01608            if (value.IsBmp)
 1609            {
 01610                return UnsafeCreate(TextInfo.ToUpperOrdinal((char)value._value));
 1611            }
 1612
 1613            // Supplementary characters use the same simple scalar mapping as OrdinalIgnoreCase comparisons.
 01614            uint upper = CharUnicodeInfo.ToUpper(value._value);
 01615            return UnsafeCreate(GlobalizationMode.UseNls
 01616                ? Ordinal.PreserveNlsOrdinalCasingClass(value._value, upper)
 01617                : upper);
 1618        }
 1619
 1620        /// <summary>
 1621        /// Returns a copy of <paramref name="value"/> converted to lowercase using ordinal (simple, one-to-one) casing 
 1622        /// </summary>
 1623        /// <param name="value">The character to convert.</param>
 1624        /// <returns>The lowercase equivalent of <paramref name="value"/>.</returns>
 1625        public static Rune ToLowerOrdinal(Rune value)
 1626        {
 01627            if (value.IsAscii)
 1628            {
 01629                return UnsafeCreate(Utf16Utility.ConvertAllAsciiCharsInUInt32ToLowercase(value._value));
 1630            }
 1631
 01632            if (value.IsBmp)
 1633            {
 01634                return UnsafeCreate(TextInfo.ToLowerOrdinal((char)value._value));
 1635            }
 1636
 01637            uint lower = CharUnicodeInfo.ToLower(value._value);
 01638            return UnsafeCreate(GlobalizationMode.UseNls
 01639                ? Ordinal.PreserveNlsOrdinalCasingClass(value._value, lower)
 01640                : lower);
 1641        }
 1642#endif
 1643
 1644        /// <inheritdoc cref="IComparable.CompareTo" />
 1645        int IComparable.CompareTo(object? obj)
 1646        {
 01647            if (obj is null)
 1648            {
 01649                return 1; // non-null ("this") always sorts after null
 1650            }
 1651
 01652            if (obj is Rune other)
 1653            {
 01654                return this.CompareTo(other);
 1655            }
 1656
 1657#if SYSTEM_PRIVATE_CORELIB
 01658            throw new ArgumentException(SR.Arg_MustBeRune);
 1659#else
 1660            throw new ArgumentException();
 1661#endif
 1662        }
 1663    }
 1664}
 1665

Methods/Properties

AsciiCharInfo()
.ctor(System.Char)
.ctor(System.Char,System.Char)
.ctor(System.Int32)
.ctor(System.UInt32)
.ctor(System.UInt32,System.Boolean)
op_Equality(System.Text.Rune,System.Text.Rune)
op_Inequality(System.Text.Rune,System.Text.Rune)
op_LessThan(System.Text.Rune,System.Text.Rune)
op_LessThanOrEqual(System.Text.Rune,System.Text.Rune)
op_GreaterThan(System.Text.Rune,System.Text.Rune)
op_GreaterThanOrEqual(System.Text.Rune,System.Text.Rune)
op_Explicit(System.Char)
op_Explicit(System.UInt32)
op_Explicit(System.Int32)
DebuggerDisplay()
IsAscii()
IsBmp()
Plane()
ReplacementChar()
Utf16SequenceLength()
Utf8SequenceLength()
Value()
ChangeCaseCultureAware(System.Text.Rune,System.Globalization.TextInfo,System.Boolean)
CompareTo(System.Text.Rune)
AsSpan(System.Span`1<System.Char>)
DecodeFromUtf16(System.ReadOnlySpan`1<System.Char>,System.Text.Rune&,System.Int32&)
DecodeFromUtf8(System.ReadOnlySpan`1<System.Byte>,System.Text.Rune&,System.Int32&)
DecodeLastFromUtf16(System.ReadOnlySpan`1<System.Char>,System.Text.Rune&,System.Int32&)
DecodeLastFromUtf8(System.ReadOnlySpan`1<System.Byte>,System.Text.Rune&,System.Int32&)
EncodeToUtf16(System.Span`1<System.Char>)
EncodeToUtf8(System.Span`1<System.Byte>)
Equals(System.Object)
Equals(System.Text.Rune)
Equals(System.Text.Rune,System.StringComparison)
GetHashCode()
GetRuneAt(System.String,System.Int32)
IsValid(System.Int32)
IsValid(System.UInt32)
ReadFirstRuneFromUtf16Buffer(System.ReadOnlySpan`1<System.Char>)
ReadRuneFromString(System.String,System.Int32)
ToString()
System.ISpanFormattable.TryFormat(System.Span`1<System.Char>,System.Int32&,System.ReadOnlySpan`1<System.Char>,System.IFormatProvider)
System.IUtf8SpanFormattable.TryFormat(System.Span`1<System.Byte>,System.Int32&,System.ReadOnlySpan`1<System.Char>,System.IFormatProvider)
System.IUtf8SpanParsable<System.Text.Rune>.TryParse(System.ReadOnlySpan`1<System.Byte>,System.IFormatProvider,System.Text.Rune&)
System.IUtf8SpanParsable<System.Text.Rune>.Parse(System.ReadOnlySpan`1<System.Byte>,System.IFormatProvider)
System.IParsable<System.Text.Rune>.Parse(System.String,System.IFormatProvider)
System.IParsable<System.Text.Rune>.TryParse(System.String,System.IFormatProvider,System.Text.Rune&)
System.ISpanParsable<System.Text.Rune>.Parse(System.ReadOnlySpan`1<System.Char>,System.IFormatProvider)
System.ISpanParsable<System.Text.Rune>.TryParse(System.ReadOnlySpan`1<System.Char>,System.IFormatProvider,System.Text.Rune&)
System.IFormattable.ToString(System.String,System.IFormatProvider)
TryCreate(System.Char,System.Text.Rune&)
TryCreate(System.Char,System.Char,System.Text.Rune&)
TryCreate(System.Int32,System.Text.Rune&)
TryCreate(System.UInt32,System.Text.Rune&)
TryEncodeToUtf16(System.Span`1<System.Char>,System.Int32&)
TryEncodeToUtf16(System.Text.Rune,System.Span`1<System.Char>,System.Int32&)
TryEncodeToUtf8(System.Span`1<System.Byte>,System.Int32&)
TryEncodeToUtf8(System.Text.Rune,System.Span`1<System.Byte>,System.Int32&)
TryGetRuneAt(System.String,System.Int32,System.Text.Rune&)
UnsafeCreate(System.UInt32)
GetNumericValue(System.Text.Rune)
GetUnicodeCategory(System.Text.Rune)
GetUnicodeCategoryNonAscii(System.Text.Rune)
IsCategoryLetter(System.Globalization.UnicodeCategory)
IsCategoryLetterOrDecimalDigit(System.Globalization.UnicodeCategory)
IsCategoryNumber(System.Globalization.UnicodeCategory)
IsCategoryPunctuation(System.Globalization.UnicodeCategory)
IsCategorySeparator(System.Globalization.UnicodeCategory)
IsCategorySymbol(System.Globalization.UnicodeCategory)
IsControl(System.Text.Rune)
IsDigit(System.Text.Rune)
IsLetter(System.Text.Rune)
IsLetterOrDigit(System.Text.Rune)
IsLower(System.Text.Rune)
IsNumber(System.Text.Rune)
IsPunctuation(System.Text.Rune)
IsSeparator(System.Text.Rune)
IsSymbol(System.Text.Rune)
IsUpper(System.Text.Rune)
IsWhiteSpace(System.Text.Rune)
ToLower(System.Text.Rune,System.Globalization.CultureInfo)
ToLowerInvariant(System.Text.Rune)
ToUpper(System.Text.Rune,System.Globalization.CultureInfo)
ToUpperInvariant(System.Text.Rune)
ToUpperOrdinal(System.Text.Rune)
ToLowerOrdinal(System.Text.Rune)
System.IComparable.CompareTo(System.Object)