| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | using System.Buffers; |
| | | 5 | | using System.Diagnostics; |
| | | 6 | | using System.Diagnostics.CodeAnalysis; |
| | | 7 | | using System.Globalization; |
| | | 8 | | using System.Runtime.CompilerServices; |
| | | 9 | | using System.Text.Unicode; |
| | | 10 | | |
| | | 11 | | #if !SYSTEM_PRIVATE_CORELIB |
| | | 12 | | #pragma warning disable CS3019 // CLS compliance checking will not be performed because it is not visible from outside t |
| | | 13 | | #endif |
| | | 14 | | |
| | | 15 | | namespace System.Text |
| | | 16 | | { |
| | | 17 | | /// <summary> |
| | | 18 | | /// Represents a Unicode scalar value ([ U+0000..U+D7FF ], inclusive; or [ U+E000..U+10FFFF ], inclusive). |
| | | 19 | | /// </summary> |
| | | 20 | | /// <remarks> |
| | | 21 | | /// This type's constructors and conversion operators validate the input, so consumers can call the APIs |
| | | 22 | | /// assuming that the underlying <see cref="Rune"/> instance is well-formed. |
| | | 23 | | /// </remarks> |
| | | 24 | | [DebuggerDisplay("{DebuggerDisplay,nq}")] |
| | | 25 | | #if SYSTEM_PRIVATE_CORELIB |
| | | 26 | | public |
| | | 27 | | #else |
| | | 28 | | internal |
| | | 29 | | #endif |
| | | 30 | | readonly struct Rune : IComparable, IComparable<Rune>, IEquatable<Rune> |
| | | 31 | | #if SYSTEM_PRIVATE_CORELIB |
| | | 32 | | #pragma warning disable SA1001 // Commas should be spaced correctly |
| | | 33 | | , ISpanFormattable |
| | | 34 | | , IUtf8SpanFormattable |
| | | 35 | | , IParsable<Rune> |
| | | 36 | | , ISpanParsable<Rune> |
| | | 37 | | , IUtf8SpanParsable<Rune> |
| | | 38 | | #pragma warning restore SA1001 |
| | | 39 | | #endif |
| | | 40 | | { |
| | | 41 | | internal const int MaxUtf16CharsPerRune = 2; // supplementary plane code points are encoded as 2 UTF-16 code uni |
| | | 42 | | internal const int MaxUtf8BytesPerRune = 4; // supplementary plane code points are encoded as 4 UTF-8 code units |
| | | 43 | | |
| | | 44 | | private const char HighSurrogateStart = '\ud800'; |
| | | 45 | | private const char LowSurrogateStart = '\udc00'; |
| | | 46 | | private const int HighSurrogateRange = 0x3FF; |
| | | 47 | | |
| | | 48 | | private const byte IsWhiteSpaceFlag = 0x80; |
| | | 49 | | private const byte IsLetterOrDigitFlag = 0x40; |
| | | 50 | | private const byte UnicodeCategoryMask = 0x1F; |
| | | 51 | | |
| | | 52 | | // Contains information about the ASCII character range [ U+0000..U+007F ], with: |
| | | 53 | | // - 0x80 bit if set means 'is whitespace' |
| | | 54 | | // - 0x40 bit if set means 'is letter or digit' |
| | | 55 | | // - 0x20 bit is reserved for future use |
| | | 56 | | // - bottom 5 bits are the UnicodeCategory of the character |
| | | 57 | | private static ReadOnlySpan<byte> AsciiCharInfo => |
| | 0 | 58 | | [ |
| | 0 | 59 | | 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x8E, 0x8E, 0x8E, 0x8E, 0x8E, 0x0E, 0x0E, // U+0000..U |
| | 0 | 60 | | 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, 0x0E, // U+0010..U |
| | 0 | 61 | | 0x8B, 0x18, 0x18, 0x18, 0x1A, 0x18, 0x18, 0x18, 0x14, 0x15, 0x18, 0x19, 0x18, 0x13, 0x18, 0x18, // U+0020..U |
| | 0 | 62 | | 0x48, 0x48, 0x48, 0x48, 0x48, 0x48, 0x48, 0x48, 0x48, 0x48, 0x18, 0x18, 0x19, 0x19, 0x19, 0x18, // U+0030..U |
| | 0 | 63 | | 0x18, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, // U+0040..U |
| | 0 | 64 | | 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x40, 0x14, 0x18, 0x15, 0x1B, 0x12, // U+0050..U |
| | 0 | 65 | | 0x1B, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, // U+0060..U |
| | 0 | 66 | | 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x41, 0x14, 0x19, 0x15, 0x19, 0x0E, // U+0070..U |
| | 0 | 67 | | ]; |
| | | 68 | | |
| | | 69 | | private readonly uint _value; |
| | | 70 | | |
| | | 71 | | /// <summary> |
| | | 72 | | /// Creates a <see cref="Rune"/> from the provided UTF-16 code unit. |
| | | 73 | | /// </summary> |
| | | 74 | | /// <exception cref="ArgumentOutOfRangeException"> |
| | | 75 | | /// If <paramref name="ch"/> represents a UTF-16 surrogate code point |
| | | 76 | | /// U+D800..U+DFFF, inclusive. |
| | | 77 | | /// </exception> |
| | | 78 | | public Rune(char ch) |
| | | 79 | | { |
| | 0 | 80 | | uint expanded = ch; |
| | 0 | 81 | | if (UnicodeUtility.IsSurrogateCodePoint(expanded)) |
| | | 82 | | { |
| | 0 | 83 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.ch); |
| | | 84 | | } |
| | 0 | 85 | | _value = expanded; |
| | 0 | 86 | | } |
| | | 87 | | |
| | | 88 | | /// <summary> |
| | | 89 | | /// Creates a <see cref="Rune"/> from the provided UTF-16 surrogate pair. |
| | | 90 | | /// </summary> |
| | | 91 | | /// <exception cref="ArgumentOutOfRangeException"> |
| | | 92 | | /// If <paramref name="highSurrogate"/> does not represent a UTF-16 high surrogate code point |
| | | 93 | | /// or <paramref name="lowSurrogate"/> does not represent a UTF-16 low surrogate code point. |
| | | 94 | | /// </exception> |
| | | 95 | | public Rune(char highSurrogate, char lowSurrogate) |
| | 0 | 96 | | : this((uint)char.ConvertToUtf32(highSurrogate, lowSurrogate), false) |
| | | 97 | | { |
| | 0 | 98 | | } |
| | | 99 | | |
| | | 100 | | /// <summary> |
| | | 101 | | /// Creates a <see cref="Rune"/> from the provided Unicode scalar value. |
| | | 102 | | /// </summary> |
| | | 103 | | /// <exception cref="ArgumentOutOfRangeException"> |
| | | 104 | | /// If <paramref name="value"/> does not represent a value Unicode scalar value. |
| | | 105 | | /// </exception> |
| | | 106 | | public Rune(int value) |
| | 0 | 107 | | : this((uint)value) |
| | | 108 | | { |
| | 0 | 109 | | } |
| | | 110 | | |
| | | 111 | | /// <summary> |
| | | 112 | | /// Creates a <see cref="Rune"/> from the provided Unicode scalar value. |
| | | 113 | | /// </summary> |
| | | 114 | | /// <exception cref="ArgumentOutOfRangeException"> |
| | | 115 | | /// If <paramref name="value"/> does not represent a value Unicode scalar value. |
| | | 116 | | /// </exception> |
| | | 117 | | [CLSCompliant(false)] |
| | | 118 | | public Rune(uint value) |
| | | 119 | | { |
| | 0 | 120 | | if (!UnicodeUtility.IsValidUnicodeScalar(value)) |
| | | 121 | | { |
| | 0 | 122 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.value); |
| | | 123 | | } |
| | 0 | 124 | | _value = value; |
| | 0 | 125 | | } |
| | | 126 | | |
| | | 127 | | // non-validating ctor |
| | | 128 | | private Rune(uint scalarValue, bool _) |
| | | 129 | | { |
| | 4445191 | 130 | | UnicodeDebug.AssertIsValidScalar(scalarValue); |
| | 4445191 | 131 | | _value = scalarValue; |
| | 4445191 | 132 | | } |
| | | 133 | | |
| | 0 | 134 | | public static bool operator ==(Rune left, Rune right) => left._value == right._value; |
| | | 135 | | |
| | 0 | 136 | | public static bool operator !=(Rune left, Rune right) => left._value != right._value; |
| | | 137 | | |
| | 0 | 138 | | public static bool operator <(Rune left, Rune right) => left._value < right._value; |
| | | 139 | | |
| | 0 | 140 | | public static bool operator <=(Rune left, Rune right) => left._value <= right._value; |
| | | 141 | | |
| | 0 | 142 | | public static bool operator >(Rune left, Rune right) => left._value > right._value; |
| | | 143 | | |
| | 0 | 144 | | public static bool operator >=(Rune left, Rune right) => left._value >= right._value; |
| | | 145 | | |
| | | 146 | | // Operators below are explicit because they may throw. |
| | | 147 | | |
| | 0 | 148 | | public static explicit operator Rune(char ch) => new Rune(ch); |
| | | 149 | | |
| | | 150 | | [CLSCompliant(false)] |
| | 0 | 151 | | public static explicit operator Rune(uint value) => new Rune(value); |
| | | 152 | | |
| | 0 | 153 | | public static explicit operator Rune(int value) => new Rune(value); |
| | | 154 | | |
| | | 155 | | // Displayed as "'<char>' (U+XXXX)"; e.g., "'e' (U+0065)" |
| | | 156 | | private string DebuggerDisplay => |
| | | 157 | | #if SYSTEM_PRIVATE_CORELIB |
| | 0 | 158 | | string.Create( |
| | 0 | 159 | | CultureInfo.InvariantCulture, |
| | 0 | 160 | | #else |
| | 0 | 161 | | FormattableString.Invariant( |
| | 0 | 162 | | #endif |
| | 0 | 163 | | $"U+{_value:X4} '{(IsValid(_value) ? ToString() : "\uFFFD")}'"); |
| | | 164 | | |
| | | 165 | | /// <summary> |
| | | 166 | | /// Returns true if and only if this scalar value is ASCII ([ U+0000..U+007F ]) |
| | | 167 | | /// and therefore representable by a single UTF-8 code unit. |
| | | 168 | | /// </summary> |
| | 0 | 169 | | public bool IsAscii => UnicodeUtility.IsAsciiCodePoint(_value); |
| | | 170 | | |
| | | 171 | | /// <summary> |
| | | 172 | | /// Returns true if and only if this scalar value is within the BMP ([ U+0000..U+FFFF ]) |
| | | 173 | | /// and therefore representable by a single UTF-16 code unit. |
| | | 174 | | /// </summary> |
| | 175880 | 175 | | public bool IsBmp => UnicodeUtility.IsBmpCodePoint(_value); |
| | | 176 | | |
| | | 177 | | /// <summary> |
| | | 178 | | /// Returns the Unicode plane (0 to 16, inclusive) which contains this scalar. |
| | | 179 | | /// </summary> |
| | 0 | 180 | | public int Plane => UnicodeUtility.GetPlane(_value); |
| | | 181 | | |
| | | 182 | | /// <summary> |
| | | 183 | | /// A <see cref="Rune"/> instance that represents the Unicode replacement character U+FFFD. |
| | | 184 | | /// </summary> |
| | 2524547 | 185 | | public static Rune ReplacementChar => UnsafeCreate(UnicodeUtility.ReplacementChar); |
| | | 186 | | |
| | | 187 | | /// <summary> |
| | | 188 | | /// Returns the length in code units (<see cref="char"/>) of the |
| | | 189 | | /// UTF-16 sequence required to represent this scalar value. |
| | | 190 | | /// </summary> |
| | | 191 | | /// <remarks> |
| | | 192 | | /// The return value will be 1 or 2. |
| | | 193 | | /// </remarks> |
| | | 194 | | public int Utf16SequenceLength |
| | | 195 | | { |
| | | 196 | | get |
| | | 197 | | { |
| | 811866 | 198 | | int codeUnitCount = UnicodeUtility.GetUtf16SequenceLength(_value); |
| | 811866 | 199 | | Debug.Assert(codeUnitCount > 0 && codeUnitCount <= MaxUtf16CharsPerRune); |
| | 811866 | 200 | | return codeUnitCount; |
| | | 201 | | } |
| | | 202 | | } |
| | | 203 | | |
| | | 204 | | /// <summary> |
| | | 205 | | /// Returns the length in code units of the |
| | | 206 | | /// UTF-8 sequence required to represent this scalar value. |
| | | 207 | | /// </summary> |
| | | 208 | | /// <remarks> |
| | | 209 | | /// The return value will be 1 through 4, inclusive. |
| | | 210 | | /// </remarks> |
| | | 211 | | public int Utf8SequenceLength |
| | | 212 | | { |
| | | 213 | | get |
| | | 214 | | { |
| | 0 | 215 | | int codeUnitCount = UnicodeUtility.GetUtf8SequenceLength(_value); |
| | 0 | 216 | | Debug.Assert(codeUnitCount > 0 && codeUnitCount <= MaxUtf8BytesPerRune); |
| | 0 | 217 | | return codeUnitCount; |
| | | 218 | | } |
| | | 219 | | } |
| | | 220 | | |
| | | 221 | | /// <summary> |
| | | 222 | | /// Returns the Unicode scalar value as an integer. |
| | | 223 | | /// </summary> |
| | 1865796 | 224 | | public int Value => (int)_value; |
| | | 225 | | |
| | | 226 | | #if SYSTEM_PRIVATE_CORELIB |
| | | 227 | | private static unsafe Rune ChangeCaseCultureAware(Rune rune, TextInfo textInfo, bool toUpper) |
| | | 228 | | { |
| | 0 | 229 | | Debug.Assert(!GlobalizationMode.Invariant, "This should've been checked by the caller."); |
| | 0 | 230 | | Debug.Assert(textInfo != null, "This should've been checked by the caller."); |
| | | 231 | | |
| | 0 | 232 | | Span<char> original = stackalloc char[MaxUtf16CharsPerRune]; |
| | 0 | 233 | | Span<char> modified = stackalloc char[MaxUtf16CharsPerRune]; |
| | | 234 | | |
| | 0 | 235 | | int charCount = rune.EncodeToUtf16(original); |
| | 0 | 236 | | original = original.Slice(0, charCount); |
| | 0 | 237 | | modified = modified.Slice(0, charCount); |
| | | 238 | | |
| | 0 | 239 | | if (toUpper) |
| | | 240 | | { |
| | 0 | 241 | | textInfo.ChangeCaseToUpper(original, modified); |
| | | 242 | | } |
| | | 243 | | else |
| | | 244 | | { |
| | 0 | 245 | | textInfo.ChangeCaseToLower(original, modified); |
| | | 246 | | } |
| | | 247 | | |
| | | 248 | | // We use simple case folding rules, which disallows moving between the BMP and supplementary |
| | | 249 | | // planes when performing a case conversion. The helper methods which reconstruct a Rune |
| | | 250 | | // contain debug asserts for this condition. |
| | | 251 | | |
| | 0 | 252 | | if (rune.IsBmp) |
| | | 253 | | { |
| | 0 | 254 | | return UnsafeCreate(modified[0]); |
| | | 255 | | } |
| | | 256 | | else |
| | | 257 | | { |
| | 0 | 258 | | return UnsafeCreate(UnicodeUtility.GetScalarFromUtf16SurrogatePair(modified[0], modified[1])); |
| | | 259 | | } |
| | | 260 | | } |
| | | 261 | | #else |
| | | 262 | | private static unsafe Rune ChangeCaseCultureAware(Rune rune, CultureInfo culture, bool toUpper) |
| | | 263 | | { |
| | | 264 | | Debug.Assert(culture != null, "This should've been checked by the caller."); |
| | | 265 | | |
| | | 266 | | Span<char> original = stackalloc char[MaxUtf16CharsPerRune]; // worst case scenario = 2 code units (for a su |
| | | 267 | | Span<char> modified = stackalloc char[MaxUtf16CharsPerRune]; // case change should preserve UTF-16 code unit |
| | | 268 | | |
| | | 269 | | int charCount = rune.EncodeToUtf16(original); |
| | | 270 | | original = original.Slice(0, charCount); |
| | | 271 | | modified = modified.Slice(0, charCount); |
| | | 272 | | |
| | | 273 | | if (toUpper) |
| | | 274 | | { |
| | | 275 | | MemoryExtensions.ToUpper(original, modified, culture); |
| | | 276 | | } |
| | | 277 | | else |
| | | 278 | | { |
| | | 279 | | MemoryExtensions.ToLower(original, modified, culture); |
| | | 280 | | } |
| | | 281 | | |
| | | 282 | | // We use simple case folding rules, which disallows moving between the BMP and supplementary |
| | | 283 | | // planes when performing a case conversion. The helper methods which reconstruct a Rune |
| | | 284 | | // contain debug asserts for this condition. |
| | | 285 | | |
| | | 286 | | if (rune.IsBmp) |
| | | 287 | | { |
| | | 288 | | return UnsafeCreate(modified[0]); |
| | | 289 | | } |
| | | 290 | | else |
| | | 291 | | { |
| | | 292 | | return UnsafeCreate(UnicodeUtility.GetScalarFromUtf16SurrogatePair(modified[0], modified[1])); |
| | | 293 | | } |
| | | 294 | | } |
| | | 295 | | #endif |
| | | 296 | | |
| | 0 | 297 | | public int CompareTo(Rune other) => this.Value - other.Value; // values don't span entire 32-bit domain; won't i |
| | | 298 | | |
| | | 299 | | internal ReadOnlySpan<char> AsSpan(Span<char> buffer) |
| | | 300 | | { |
| | 0 | 301 | | Debug.Assert(buffer.Length >= MaxUtf16CharsPerRune); |
| | 0 | 302 | | int charsWritten = EncodeToUtf16(buffer); |
| | 0 | 303 | | return buffer.Slice(0, charsWritten); |
| | | 304 | | } |
| | | 305 | | |
| | | 306 | | /// <summary> |
| | | 307 | | /// Decodes the <see cref="Rune"/> at the beginning of the provided UTF-16 source buffer. |
| | | 308 | | /// </summary> |
| | | 309 | | /// <returns> |
| | | 310 | | /// <para> |
| | | 311 | | /// If the source buffer begins with a valid UTF-16 encoded scalar value, returns <see cref="OperationStatus.Don |
| | | 312 | | /// and outs via <paramref name="result"/> the decoded <see cref="Rune"/> and via <paramref name="charsConsumed" |
| | | 313 | | /// number of <see langword="char"/>s used in the input buffer to encode the <see cref="Rune"/>. |
| | | 314 | | /// </para> |
| | | 315 | | /// <para> |
| | | 316 | | /// If the source buffer is empty or contains only a standalone UTF-16 high surrogate character, returns <see cr |
| | | 317 | | /// and outs via <paramref name="result"/> <see cref="ReplacementChar"/> and via <paramref name="charsConsumed"/ |
| | | 318 | | /// </para> |
| | | 319 | | /// <para> |
| | | 320 | | /// If the source buffer begins with an ill-formed UTF-16 encoded scalar value, returns <see cref="OperationStat |
| | | 321 | | /// and outs via <paramref name="result"/> <see cref="ReplacementChar"/> and via <paramref name="charsConsumed"/ |
| | | 322 | | /// <see langword="char"/>s used in the input buffer to encode the ill-formed sequence. |
| | | 323 | | /// </para> |
| | | 324 | | /// </returns> |
| | | 325 | | /// <remarks> |
| | | 326 | | /// The general calling convention is to call this method in a loop, slicing the <paramref name="source"/> buffe |
| | | 327 | | /// <paramref name="charsConsumed"/> elements on each iteration of the loop. On each iteration of the loop <para |
| | | 328 | | /// will contain the real scalar value if successfully decoded, or it will contain <see cref="ReplacementChar"/> |
| | | 329 | | /// the data could not be successfully decoded. This pattern provides convenient automatic U+FFFD substitution o |
| | | 330 | | /// invalid sequences while iterating through the loop. |
| | | 331 | | /// </remarks> |
| | | 332 | | public static OperationStatus DecodeFromUtf16(ReadOnlySpan<char> source, out Rune result, out int charsConsumed) |
| | | 333 | | { |
| | 0 | 334 | | if (!source.IsEmpty) |
| | | 335 | | { |
| | | 336 | | // First, check for the common case of a BMP scalar value. |
| | | 337 | | // If this is correct, return immediately. |
| | | 338 | | |
| | 0 | 339 | | char firstChar = source[0]; |
| | 0 | 340 | | if (TryCreate(firstChar, out result)) |
| | | 341 | | { |
| | 0 | 342 | | charsConsumed = 1; |
| | 0 | 343 | | return OperationStatus.Done; |
| | | 344 | | } |
| | | 345 | | |
| | | 346 | | // First thing we saw was a UTF-16 surrogate code point. |
| | | 347 | | // Let's optimistically assume for now it's a high surrogate and hope |
| | | 348 | | // that combining it with the next char yields useful results. |
| | | 349 | | |
| | 0 | 350 | | if (source.Length > 1) |
| | | 351 | | { |
| | 0 | 352 | | char secondChar = source[1]; |
| | 0 | 353 | | if (TryCreate(firstChar, secondChar, out result)) |
| | | 354 | | { |
| | | 355 | | // Success! Formed a supplementary scalar value. |
| | 0 | 356 | | charsConsumed = 2; |
| | 0 | 357 | | return OperationStatus.Done; |
| | | 358 | | } |
| | | 359 | | else |
| | | 360 | | { |
| | | 361 | | // Either the first character was a low surrogate, or the second |
| | | 362 | | // character was not a low surrogate. This is an error. |
| | | 363 | | goto InvalidData; |
| | | 364 | | } |
| | | 365 | | } |
| | 0 | 366 | | else if (!char.IsHighSurrogate(firstChar)) |
| | | 367 | | { |
| | | 368 | | // Quick check to make sure we're not going to report NeedMoreData for |
| | | 369 | | // a single-element buffer where the data is a standalone low surrogate |
| | | 370 | | // character. Since no additional data will ever make this valid, we'll |
| | | 371 | | // report an error immediately. |
| | | 372 | | goto InvalidData; |
| | | 373 | | } |
| | | 374 | | } |
| | | 375 | | |
| | | 376 | | // If we got to this point, the input buffer was empty, or the buffer |
| | | 377 | | // was a single element in length and that element was a high surrogate char. |
| | | 378 | | |
| | 0 | 379 | | charsConsumed = source.Length; |
| | 0 | 380 | | result = ReplacementChar; |
| | 0 | 381 | | return OperationStatus.NeedMoreData; |
| | | 382 | | |
| | | 383 | | InvalidData: |
| | | 384 | | |
| | 0 | 385 | | charsConsumed = 1; // maximal invalid subsequence for UTF-16 is always a single code unit in length |
| | 0 | 386 | | result = ReplacementChar; |
| | 0 | 387 | | return OperationStatus.InvalidData; |
| | | 388 | | } |
| | | 389 | | |
| | | 390 | | /// <summary> |
| | | 391 | | /// Decodes the <see cref="Rune"/> at the beginning of the provided UTF-8 source buffer. |
| | | 392 | | /// </summary> |
| | | 393 | | /// <returns> |
| | | 394 | | /// <para> |
| | | 395 | | /// If the source buffer begins with a valid UTF-8 encoded scalar value, returns <see cref="OperationStatus.Done |
| | | 396 | | /// and outs via <paramref name="result"/> the decoded <see cref="Rune"/> and via <paramref name="bytesConsumed" |
| | | 397 | | /// number of <see langword="byte"/>s used in the input buffer to encode the <see cref="Rune"/>. |
| | | 398 | | /// </para> |
| | | 399 | | /// <para> |
| | | 400 | | /// If the source buffer is empty or contains only a partial UTF-8 subsequence, returns <see cref="OperationStat |
| | | 401 | | /// and outs via <paramref name="result"/> <see cref="ReplacementChar"/> and via <paramref name="bytesConsumed"/ |
| | | 402 | | /// </para> |
| | | 403 | | /// <para> |
| | | 404 | | /// If the source buffer begins with an ill-formed UTF-8 encoded scalar value, returns <see cref="OperationStatu |
| | | 405 | | /// and outs via <paramref name="result"/> <see cref="ReplacementChar"/> and via <paramref name="bytesConsumed"/ |
| | | 406 | | /// <see langword="char"/>s used in the input buffer to encode the ill-formed sequence. |
| | | 407 | | /// </para> |
| | | 408 | | /// </returns> |
| | | 409 | | /// <remarks> |
| | | 410 | | /// The general calling convention is to call this method in a loop, slicing the <paramref name="source"/> buffe |
| | | 411 | | /// <paramref name="bytesConsumed"/> elements on each iteration of the loop. On each iteration of the loop <para |
| | | 412 | | /// will contain the real scalar value if successfully decoded, or it will contain <see cref="ReplacementChar"/> |
| | | 413 | | /// the data could not be successfully decoded. This pattern provides convenient automatic U+FFFD substitution o |
| | | 414 | | /// invalid sequences while iterating through the loop. |
| | | 415 | | /// </remarks> |
| | | 416 | | public static OperationStatus DecodeFromUtf8(ReadOnlySpan<byte> source, out Rune result, out int bytesConsumed) |
| | | 417 | | { |
| | | 418 | | // This method follows the Unicode Standard's recommendation for detecting |
| | | 419 | | // the maximal subpart of an ill-formed subsequence. See The Unicode Standard, |
| | | 420 | | // Ch. 3.9 for more details. In summary, when reporting an invalid subsequence, |
| | | 421 | | // it tries to consume as many code units as possible as long as those code |
| | | 422 | | // units constitute the beginning of a longer well-formed subsequence per Table 3-7. |
| | | 423 | | |
| | | 424 | | // Try reading source[0]. |
| | | 425 | | |
| | 2579395 | 426 | | int index = 0; |
| | 2579395 | 427 | | if (source.IsEmpty) |
| | | 428 | | { |
| | | 429 | | goto NeedsMoreData; |
| | | 430 | | } |
| | | 431 | | |
| | 2579395 | 432 | | uint tempValue = source[0]; |
| | 2579395 | 433 | | if (UnicodeUtility.IsAsciiCodePoint(tempValue)) |
| | | 434 | | { |
| | 0 | 435 | | bytesConsumed = 1; |
| | 0 | 436 | | result = UnsafeCreate(tempValue); |
| | 0 | 437 | | return OperationStatus.Done; |
| | | 438 | | } |
| | | 439 | | |
| | | 440 | | // Per Table 3-7, the beginning of a multibyte sequence must be a code unit in |
| | | 441 | | // the range [C2..F4]. If it's outside of that range, it's either a standalone |
| | | 442 | | // continuation byte, or it's an overlong two-byte sequence, or it's an out-of-range |
| | | 443 | | // four-byte sequence. |
| | | 444 | | |
| | | 445 | | // Try reading source[1]. |
| | | 446 | | |
| | 2579395 | 447 | | index = 1; |
| | 2579395 | 448 | | if (!UnicodeUtility.IsInRangeInclusive(tempValue, 0xC2, 0xF4)) |
| | | 449 | | { |
| | | 450 | | goto Invalid; |
| | | 451 | | } |
| | | 452 | | |
| | 1537997 | 453 | | tempValue = (tempValue - 0xC2) << 6; |
| | | 454 | | |
| | 1537997 | 455 | | if (source.Length <= 1) |
| | | 456 | | { |
| | | 457 | | goto NeedsMoreData; |
| | | 458 | | } |
| | | 459 | | |
| | | 460 | | // Continuation bytes are of the form [10xxxxxx], which means that their two's |
| | | 461 | | // complement representation is in the range [-65..-128]. This allows us to |
| | | 462 | | // perform a single comparison to see if a byte is a continuation byte. |
| | | 463 | | |
| | 1350540 | 464 | | int thisByteSignExtended = (sbyte)source[1]; |
| | 1350540 | 465 | | if (thisByteSignExtended >= -64) |
| | | 466 | | { |
| | | 467 | | goto Invalid; |
| | | 468 | | } |
| | | 469 | | |
| | 193792 | 470 | | tempValue += (uint)thisByteSignExtended; |
| | 193792 | 471 | | tempValue += 0x80; // remove the continuation byte marker |
| | 193792 | 472 | | tempValue += (0xC2 - 0xC0) << 6; // remove the leading byte marker |
| | | 473 | | |
| | 193792 | 474 | | if (tempValue < 0x0800) |
| | | 475 | | { |
| | 36737 | 476 | | Debug.Assert(UnicodeUtility.IsInRangeInclusive(tempValue, 0x0080, 0x07FF)); |
| | 36737 | 477 | | goto Finish; // this is a valid 2-byte sequence |
| | | 478 | | } |
| | | 479 | | |
| | | 480 | | // This appears to be a 3- or 4-byte sequence. Since per Table 3-7 we now have |
| | | 481 | | // enough information (from just two code units) to detect overlong or surrogate |
| | | 482 | | // sequences, we need to perform these checks now. |
| | | 483 | | |
| | 157055 | 484 | | if (!UnicodeUtility.IsInRangeInclusive(tempValue, ((0xE0 - 0xC0) << 6) + (0xA0 - 0x80), ((0xF4 - 0xC0) << 6) |
| | | 485 | | { |
| | | 486 | | // The first two bytes were not in the range [[E0 A0]..[F4 8F]]. |
| | | 487 | | // This is an overlong 3-byte sequence or an out-of-range 4-byte sequence. |
| | | 488 | | goto Invalid; |
| | | 489 | | } |
| | | 490 | | |
| | 130319 | 491 | | if (UnicodeUtility.IsInRangeInclusive(tempValue, ((0xED - 0xC0) << 6) + (0xA0 - 0x80), ((0xED - 0xC0) << 6) |
| | | 492 | | { |
| | | 493 | | // This is a UTF-16 surrogate code point, which is invalid in UTF-8. |
| | | 494 | | goto Invalid; |
| | | 495 | | } |
| | | 496 | | |
| | 115524 | 497 | | if (UnicodeUtility.IsInRangeInclusive(tempValue, ((0xF0 - 0xC0) << 6) + (0x80 - 0x80), ((0xF0 - 0xC0) << 6) |
| | | 498 | | { |
| | | 499 | | // This is an overlong 4-byte sequence. |
| | | 500 | | goto Invalid; |
| | | 501 | | } |
| | | 502 | | |
| | | 503 | | // The first two bytes were just fine. We don't need to perform any other checks |
| | | 504 | | // on the remaining bytes other than to see that they're valid continuation bytes. |
| | | 505 | | |
| | | 506 | | // Try reading source[2]. |
| | | 507 | | |
| | 115470 | 508 | | index = 2; |
| | 115470 | 509 | | if (source.Length <= 2) |
| | | 510 | | { |
| | | 511 | | goto NeedsMoreData; |
| | | 512 | | } |
| | | 513 | | |
| | 96288 | 514 | | thisByteSignExtended = (sbyte)source[2]; |
| | 96288 | 515 | | if (thisByteSignExtended >= -64) |
| | | 516 | | { |
| | | 517 | | goto Invalid; // this byte is not a UTF-8 continuation byte |
| | | 518 | | } |
| | | 519 | | |
| | 51417 | 520 | | tempValue <<= 6; |
| | 51417 | 521 | | tempValue += (uint)thisByteSignExtended; |
| | 51417 | 522 | | tempValue += 0x80; // remove the continuation byte marker |
| | 51417 | 523 | | tempValue -= (0xE0 - 0xC0) << 12; // remove the leading byte marker |
| | | 524 | | |
| | 51417 | 525 | | if (tempValue <= 0xFFFF) |
| | | 526 | | { |
| | 14921 | 527 | | Debug.Assert(UnicodeUtility.IsInRangeInclusive(tempValue, 0x0800, 0xFFFF)); |
| | 14921 | 528 | | goto Finish; // this is a valid 3-byte sequence |
| | | 529 | | } |
| | | 530 | | |
| | | 531 | | // Try reading source[3]. |
| | | 532 | | |
| | 36496 | 533 | | index = 3; |
| | 36496 | 534 | | if (source.Length <= 3) |
| | | 535 | | { |
| | | 536 | | goto NeedsMoreData; |
| | | 537 | | } |
| | | 538 | | |
| | 31139 | 539 | | thisByteSignExtended = (sbyte)source[3]; |
| | 31139 | 540 | | if (thisByteSignExtended >= -64) |
| | | 541 | | { |
| | | 542 | | goto Invalid; // this byte is not a UTF-8 continuation byte |
| | | 543 | | } |
| | | 544 | | |
| | 3190 | 545 | | tempValue <<= 6; |
| | 3190 | 546 | | tempValue += (uint)thisByteSignExtended; |
| | 3190 | 547 | | tempValue += 0x80; // remove the continuation byte marker |
| | 3190 | 548 | | tempValue -= (0xF0 - 0xE0) << 18; // remove the leading byte marker |
| | | 549 | | |
| | | 550 | | // Valid 4-byte sequence |
| | 3190 | 551 | | UnicodeDebug.AssertIsValidSupplementaryPlaneScalar(tempValue); |
| | | 552 | | |
| | | 553 | | Finish: |
| | | 554 | | |
| | 54848 | 555 | | bytesConsumed = index + 1; |
| | 54848 | 556 | | Debug.Assert(1 <= bytesConsumed && bytesConsumed <= 4); // Valid subsequences are always length [1..4] |
| | 54848 | 557 | | result = UnsafeCreate(tempValue); |
| | 54848 | 558 | | return OperationStatus.Done; |
| | | 559 | | |
| | | 560 | | NeedsMoreData: |
| | | 561 | | |
| | 211996 | 562 | | Debug.Assert(0 <= index && index <= 3); // Incomplete subsequences are always length 0..3 |
| | 211996 | 563 | | bytesConsumed = index; |
| | 211996 | 564 | | result = ReplacementChar; |
| | 211996 | 565 | | return OperationStatus.NeedMoreData; |
| | | 566 | | |
| | | 567 | | Invalid: |
| | | 568 | | |
| | 2312551 | 569 | | Debug.Assert(1 <= index && index <= 3); // Invalid subsequences are always length 1..3 |
| | 2312551 | 570 | | bytesConsumed = index; |
| | 2312551 | 571 | | result = ReplacementChar; |
| | 2312551 | 572 | | return OperationStatus.InvalidData; |
| | | 573 | | } |
| | | 574 | | |
| | | 575 | | /// <summary> |
| | | 576 | | /// Decodes the <see cref="Rune"/> at the end of the provided UTF-16 source buffer. |
| | | 577 | | /// </summary> |
| | | 578 | | /// <remarks> |
| | | 579 | | /// This method is very similar to <see cref="DecodeFromUtf16(ReadOnlySpan{char}, out Rune, out int)"/>, but it |
| | | 580 | | /// the caller to loop backward instead of forward. The typical calling convention is that on each iteration |
| | | 581 | | /// of the loop, the caller should slice off the final <paramref name="charsConsumed"/> elements of |
| | | 582 | | /// the <paramref name="source"/> buffer. |
| | | 583 | | /// </remarks> |
| | | 584 | | public static OperationStatus DecodeLastFromUtf16(ReadOnlySpan<char> source, out Rune result, out int charsConsu |
| | | 585 | | { |
| | 0 | 586 | | int index = source.Length - 1; |
| | 0 | 587 | | if ((uint)index < (uint)source.Length) |
| | | 588 | | { |
| | | 589 | | // First, check for the common case of a BMP scalar value. |
| | | 590 | | // If this is correct, return immediately. |
| | | 591 | | |
| | 0 | 592 | | char finalChar = source[index]; |
| | 0 | 593 | | if (TryCreate(finalChar, out result)) |
| | | 594 | | { |
| | 0 | 595 | | charsConsumed = 1; |
| | 0 | 596 | | return OperationStatus.Done; |
| | | 597 | | } |
| | | 598 | | |
| | 0 | 599 | | if (char.IsLowSurrogate(finalChar)) |
| | | 600 | | { |
| | | 601 | | // The final character was a UTF-16 low surrogate code point. |
| | | 602 | | // This must be preceded by a UTF-16 high surrogate code point, otherwise |
| | | 603 | | // we have a standalone low surrogate, which is always invalid. |
| | | 604 | | |
| | 0 | 605 | | index--; |
| | 0 | 606 | | if ((uint)index < (uint)source.Length) |
| | | 607 | | { |
| | 0 | 608 | | char penultimateChar = source[index]; |
| | 0 | 609 | | if (TryCreate(penultimateChar, finalChar, out result)) |
| | | 610 | | { |
| | | 611 | | // Success! Formed a supplementary scalar value. |
| | 0 | 612 | | charsConsumed = 2; |
| | 0 | 613 | | return OperationStatus.Done; |
| | | 614 | | } |
| | | 615 | | } |
| | | 616 | | |
| | | 617 | | // If we got to this point, we saw a standalone low surrogate |
| | | 618 | | // and must report an error. |
| | | 619 | | |
| | 0 | 620 | | charsConsumed = 1; // standalone surrogate |
| | 0 | 621 | | result = ReplacementChar; |
| | 0 | 622 | | return OperationStatus.InvalidData; |
| | | 623 | | } |
| | | 624 | | } |
| | | 625 | | |
| | | 626 | | // If we got this far, the source buffer was empty, or the source buffer ended |
| | | 627 | | // with a UTF-16 high surrogate code point. These aren't errors since they could |
| | | 628 | | // be valid given more input data. |
| | | 629 | | |
| | 0 | 630 | | charsConsumed = (int)((uint)(-source.Length) >> 31); // 0 -> 0, all other lengths -> 1 |
| | 0 | 631 | | result = ReplacementChar; |
| | 0 | 632 | | return OperationStatus.NeedMoreData; |
| | | 633 | | } |
| | | 634 | | |
| | | 635 | | /// <summary> |
| | | 636 | | /// Decodes the <see cref="Rune"/> at the end of the provided UTF-8 source buffer. |
| | | 637 | | /// </summary> |
| | | 638 | | /// <remarks> |
| | | 639 | | /// This method is very similar to <see cref="DecodeFromUtf8(ReadOnlySpan{byte}, out Rune, out int)"/>, but it a |
| | | 640 | | /// the caller to loop backward instead of forward. The typical calling convention is that on each iteration |
| | | 641 | | /// of the loop, the caller should slice off the final <paramref name="bytesConsumed"/> elements of |
| | | 642 | | /// the <paramref name="source"/> buffer. |
| | | 643 | | /// </remarks> |
| | | 644 | | public static OperationStatus DecodeLastFromUtf8(ReadOnlySpan<byte> source, out Rune value, out int bytesConsume |
| | | 645 | | { |
| | 0 | 646 | | int index = source.Length - 1; |
| | 0 | 647 | | if ((uint)index < (uint)source.Length) |
| | | 648 | | { |
| | | 649 | | // The buffer contains at least one byte. Let's check the fast case where the |
| | | 650 | | // buffer ends with an ASCII byte. |
| | | 651 | | |
| | 0 | 652 | | uint tempValue = source[index]; |
| | 0 | 653 | | if (UnicodeUtility.IsAsciiCodePoint(tempValue)) |
| | | 654 | | { |
| | 0 | 655 | | bytesConsumed = 1; |
| | 0 | 656 | | value = UnsafeCreate(tempValue); |
| | 0 | 657 | | return OperationStatus.Done; |
| | | 658 | | } |
| | | 659 | | |
| | | 660 | | // If the final byte is not an ASCII byte, we may be beginning or in the middle of |
| | | 661 | | // a UTF-8 multi-code unit sequence. We need to back up until we see the start of |
| | | 662 | | // the multi-code unit sequence; we can detect the leading byte because all multi-byte |
| | | 663 | | // sequences begin with a byte whose 0x40 bit is set. Since all multi-byte sequences |
| | | 664 | | // are no greater than 4 code units in length, we only need to search back a maximum |
| | | 665 | | // of four bytes. |
| | | 666 | | |
| | 0 | 667 | | if (((byte)tempValue & 0x40) != 0) |
| | | 668 | | { |
| | | 669 | | // This is a UTF-8 leading byte. We'll do a forward read from here. |
| | | 670 | | // It'll return invalid (if given C0, F5, etc.) or incomplete. Both are fine. |
| | | 671 | | |
| | 0 | 672 | | return DecodeFromUtf8(source.Slice(index), out value, out bytesConsumed); |
| | | 673 | | } |
| | | 674 | | |
| | | 675 | | // If we got to this point, the final byte was a UTF-8 continuation byte. |
| | | 676 | | // Let's check the three bytes immediately preceding this, looking for the starting byte. |
| | | 677 | | |
| | 0 | 678 | | for (int i = 3; i > 0; i--) |
| | | 679 | | { |
| | 0 | 680 | | index--; |
| | 0 | 681 | | if ((uint)index >= (uint)source.Length) |
| | | 682 | | { |
| | | 683 | | goto Invalid; // out of data |
| | | 684 | | } |
| | | 685 | | |
| | | 686 | | // The check below will get hit for ASCII (values 00..7F) and for UTF-8 starting bytes |
| | | 687 | | // (bits 0xC0 set, values C0..FF). In two's complement this is the range [-64..127]. |
| | | 688 | | // It's just a fast way for us to terminate the search. |
| | | 689 | | |
| | 0 | 690 | | if ((sbyte)source[index] >= -64) |
| | | 691 | | { |
| | | 692 | | goto ForwardDecode; |
| | | 693 | | } |
| | | 694 | | } |
| | | 695 | | |
| | | 696 | | Invalid: |
| | | 697 | | |
| | | 698 | | // If we got to this point, either: |
| | | 699 | | // - the last 4 bytes of the input buffer are continuation bytes; |
| | | 700 | | // - the entire input buffer (if fewer than 4 bytes) consists only of continuation bytes; or |
| | | 701 | | // - there's no UTF-8 leading byte between the final continuation byte of the buffer and |
| | | 702 | | // the previous well-formed subsequence or maximal invalid subsequence. |
| | | 703 | | // |
| | | 704 | | // In all of these cases, the final byte must be a maximal invalid subsequence of length 1. |
| | | 705 | | // See comment near the end of this method for more information. |
| | | 706 | | |
| | 0 | 707 | | value = ReplacementChar; |
| | 0 | 708 | | bytesConsumed = 1; |
| | 0 | 709 | | return OperationStatus.InvalidData; |
| | | 710 | | |
| | | 711 | | ForwardDecode: |
| | | 712 | | |
| | | 713 | | // If we got to this point, we found an ASCII byte or a UTF-8 starting byte at position source[index]. |
| | | 714 | | // Technically this could also mean we found an invalid byte like C0 or F5 at this position, but that's |
| | | 715 | | // fine since it'll be handled by the forward read. From this position, we'll perform a forward read |
| | | 716 | | // and see if we consumed the entirety of the buffer. |
| | | 717 | | |
| | 0 | 718 | | source = source.Slice(index); |
| | 0 | 719 | | Debug.Assert(!source.IsEmpty, "Shouldn't reach this for empty inputs."); |
| | | 720 | | |
| | 0 | 721 | | OperationStatus operationStatus = DecodeFromUtf8(source, out Rune tempRune, out int tempBytesConsumed); |
| | 0 | 722 | | if (tempBytesConsumed == source.Length) |
| | | 723 | | { |
| | | 724 | | // If this forward read consumed the entirety of the end of the input buffer, we can return it |
| | | 725 | | // as the result of this function. It could be well-formed, incomplete, or invalid. If it's |
| | | 726 | | // invalid and we consumed the remainder of the buffer, we know we've found the maximal invalid |
| | | 727 | | // subsequence, which is what we wanted anyway. |
| | | 728 | | |
| | 0 | 729 | | bytesConsumed = tempBytesConsumed; |
| | 0 | 730 | | value = tempRune; |
| | 0 | 731 | | return operationStatus; |
| | | 732 | | } |
| | | 733 | | |
| | | 734 | | // If we got to this point, we know that the final continuation byte wasn't consumed by the forward |
| | | 735 | | // read that we just performed above. This means that the continuation byte has to be part of an |
| | | 736 | | // invalid subsequence since there's no UTF-8 leading byte between what we just consumed and the |
| | | 737 | | // continuation byte at the end of the input. Furthermore, since any maximal invalid subsequence |
| | | 738 | | // of length > 1 must have a UTF-8 leading byte as its first code unit, this implies that the |
| | | 739 | | // continuation byte at the end of the buffer is itself a maximal invalid subsequence of length 1. |
| | | 740 | | |
| | | 741 | | goto Invalid; |
| | | 742 | | } |
| | | 743 | | else |
| | | 744 | | { |
| | | 745 | | // Source buffer was empty. |
| | 0 | 746 | | value = ReplacementChar; |
| | 0 | 747 | | bytesConsumed = 0; |
| | 0 | 748 | | return OperationStatus.NeedMoreData; |
| | | 749 | | } |
| | | 750 | | } |
| | | 751 | | |
| | | 752 | | /// <summary> |
| | | 753 | | /// Encodes this <see cref="Rune"/> to a UTF-16 destination buffer. |
| | | 754 | | /// </summary> |
| | | 755 | | /// <param name="destination">The buffer to which to write this value as UTF-16.</param> |
| | | 756 | | /// <returns>The number of <see cref="char"/>s written to <paramref name="destination"/>.</returns> |
| | | 757 | | /// <exception cref="ArgumentException"> |
| | | 758 | | /// If <paramref name="destination"/> is not large enough to hold the output. |
| | | 759 | | /// </exception> |
| | | 760 | | public int EncodeToUtf16(Span<char> destination) |
| | | 761 | | { |
| | 0 | 762 | | if (!TryEncodeToUtf16(destination, out int charsWritten)) |
| | | 763 | | { |
| | 0 | 764 | | ThrowHelper.ThrowArgumentException_DestinationTooShort(); |
| | | 765 | | } |
| | | 766 | | |
| | 0 | 767 | | return charsWritten; |
| | | 768 | | } |
| | | 769 | | |
| | | 770 | | /// <summary> |
| | | 771 | | /// Encodes this <see cref="Rune"/> to a UTF-8 destination buffer. |
| | | 772 | | /// </summary> |
| | | 773 | | /// <param name="destination">The buffer to which to write this value as UTF-8.</param> |
| | | 774 | | /// <returns>The number of <see cref="byte"/>s written to <paramref name="destination"/>.</returns> |
| | | 775 | | /// <exception cref="ArgumentException"> |
| | | 776 | | /// If <paramref name="destination"/> is not large enough to hold the output. |
| | | 777 | | /// </exception> |
| | | 778 | | public int EncodeToUtf8(Span<byte> destination) |
| | | 779 | | { |
| | 0 | 780 | | if (!TryEncodeToUtf8(destination, out int bytesWritten)) |
| | | 781 | | { |
| | 0 | 782 | | ThrowHelper.ThrowArgumentException_DestinationTooShort(); |
| | | 783 | | } |
| | | 784 | | |
| | 0 | 785 | | return bytesWritten; |
| | | 786 | | } |
| | | 787 | | |
| | 0 | 788 | | public override bool Equals([NotNullWhen(true)] object? obj) => (obj is Rune other) && Equals(other); |
| | | 789 | | |
| | 0 | 790 | | public bool Equals(Rune other) => this == other; |
| | | 791 | | |
| | | 792 | | /// <summary> |
| | | 793 | | /// Returns a value that indicates whether the current instance and a specified rune are equal using the specifi |
| | | 794 | | /// </summary> |
| | | 795 | | /// <param name="other">The rune to compare with the current instance.</param> |
| | | 796 | | /// <param name="comparisonType">One of the enumeration values that specifies the rules to use in the comparison |
| | | 797 | | /// <returns><see langword="true"/> if the current instance and <paramref name="other"/> are equal; otherwise, < |
| | | 798 | | public unsafe bool Equals(Rune other, StringComparison comparisonType) |
| | | 799 | | { |
| | 0 | 800 | | if (comparisonType is StringComparison.Ordinal) |
| | | 801 | | { |
| | 0 | 802 | | return this == other; |
| | | 803 | | } |
| | | 804 | | |
| | | 805 | | // Convert this to span |
| | 0 | 806 | | ReadOnlySpan<char> thisChars = AsSpan(stackalloc char[MaxUtf16CharsPerRune]); |
| | | 807 | | |
| | | 808 | | // Convert other to span |
| | 0 | 809 | | ReadOnlySpan<char> otherChars = other.AsSpan(stackalloc char[MaxUtf16CharsPerRune]); |
| | | 810 | | |
| | | 811 | | // Compare span equality |
| | 0 | 812 | | return thisChars.Equals(otherChars, comparisonType); |
| | | 813 | | } |
| | | 814 | | |
| | 0 | 815 | | public override int GetHashCode() => Value; |
| | | 816 | | |
| | | 817 | | #if SYSTEM_PRIVATE_CORELIB |
| | | 818 | | /// <summary> |
| | | 819 | | /// Gets the <see cref="Rune"/> which begins at index <paramref name="index"/> in |
| | | 820 | | /// string <paramref name="input"/>. |
| | | 821 | | /// </summary> |
| | | 822 | | /// <remarks> |
| | | 823 | | /// Throws if <paramref name="input"/> is null, if <paramref name="index"/> is out of range, or |
| | | 824 | | /// if <paramref name="index"/> does not reference the start of a valid scalar value within <paramref name="inpu |
| | | 825 | | /// </remarks> |
| | | 826 | | public static Rune GetRuneAt(string input, int index) |
| | | 827 | | { |
| | 0 | 828 | | int runeValue = ReadRuneFromString(input, index); |
| | 0 | 829 | | if (runeValue < 0) |
| | | 830 | | { |
| | 0 | 831 | | ThrowHelper.ThrowArgumentException_CannotExtractScalar(ExceptionArgument.index); |
| | | 832 | | } |
| | | 833 | | |
| | 0 | 834 | | return UnsafeCreate((uint)runeValue); |
| | | 835 | | } |
| | | 836 | | #endif |
| | | 837 | | |
| | | 838 | | /// <summary> |
| | | 839 | | /// Returns <see langword="true"/> iff <paramref name="value"/> is a valid Unicode scalar |
| | | 840 | | /// value, i.e., is in [ U+0000..U+D7FF ], inclusive; or [ U+E000..U+10FFFF ], inclusive. |
| | | 841 | | /// </summary> |
| | 0 | 842 | | public static bool IsValid(int value) => IsValid((uint)value); |
| | | 843 | | |
| | | 844 | | /// <summary> |
| | | 845 | | /// Returns <see langword="true"/> iff <paramref name="value"/> is a valid Unicode scalar |
| | | 846 | | /// value, i.e., is in [ U+0000..U+D7FF ], inclusive; or [ U+E000..U+10FFFF ], inclusive. |
| | | 847 | | /// </summary> |
| | | 848 | | [CLSCompliant(false)] |
| | 0 | 849 | | public static bool IsValid(uint value) => UnicodeUtility.IsValidUnicodeScalar(value); |
| | | 850 | | |
| | | 851 | | // returns a negative number on failure |
| | | 852 | | internal static int ReadFirstRuneFromUtf16Buffer(ReadOnlySpan<char> input) |
| | | 853 | | { |
| | 0 | 854 | | if (input.IsEmpty) |
| | | 855 | | { |
| | 0 | 856 | | return -1; |
| | | 857 | | } |
| | | 858 | | |
| | | 859 | | // Optimistically assume input is within BMP. |
| | | 860 | | |
| | 0 | 861 | | uint returnValue = input[0]; |
| | 0 | 862 | | if (UnicodeUtility.IsSurrogateCodePoint(returnValue)) |
| | | 863 | | { |
| | 0 | 864 | | if (!UnicodeUtility.IsHighSurrogateCodePoint(returnValue)) |
| | | 865 | | { |
| | 0 | 866 | | return -1; |
| | | 867 | | } |
| | | 868 | | |
| | | 869 | | // Treat 'returnValue' as the high surrogate. |
| | | 870 | | |
| | 0 | 871 | | if (input.Length <= 1) |
| | | 872 | | { |
| | 0 | 873 | | return -1; // not an argument exception - just a "bad data" failure |
| | | 874 | | } |
| | | 875 | | |
| | 0 | 876 | | uint potentialLowSurrogate = input[1]; |
| | 0 | 877 | | if (!UnicodeUtility.IsLowSurrogateCodePoint(potentialLowSurrogate)) |
| | | 878 | | { |
| | 0 | 879 | | return -1; |
| | | 880 | | } |
| | | 881 | | |
| | 0 | 882 | | returnValue = UnicodeUtility.GetScalarFromUtf16SurrogatePair(returnValue, potentialLowSurrogate); |
| | | 883 | | } |
| | | 884 | | |
| | 0 | 885 | | return (int)returnValue; |
| | | 886 | | } |
| | | 887 | | |
| | | 888 | | #if SYSTEM_PRIVATE_CORELIB |
| | | 889 | | // returns a negative number on failure |
| | | 890 | | private static int ReadRuneFromString(string input, int index) |
| | | 891 | | { |
| | 0 | 892 | | if (input is null) |
| | | 893 | | { |
| | 0 | 894 | | ThrowHelper.ThrowArgumentNullException(ExceptionArgument.input); |
| | | 895 | | } |
| | | 896 | | |
| | 0 | 897 | | if ((uint)index >= (uint)input.Length) |
| | | 898 | | { |
| | 0 | 899 | | ThrowHelper.ThrowArgumentOutOfRange_IndexMustBeLessException(); |
| | | 900 | | } |
| | | 901 | | |
| | | 902 | | // Optimistically assume input is within BMP. |
| | | 903 | | |
| | 0 | 904 | | uint returnValue = input[index]; |
| | 0 | 905 | | if (UnicodeUtility.IsSurrogateCodePoint(returnValue)) |
| | | 906 | | { |
| | 0 | 907 | | if (!UnicodeUtility.IsHighSurrogateCodePoint(returnValue)) |
| | | 908 | | { |
| | 0 | 909 | | return -1; |
| | | 910 | | } |
| | | 911 | | |
| | | 912 | | // Treat 'returnValue' as the high surrogate. |
| | | 913 | | // |
| | | 914 | | // If this becomes a hot code path, we can skip the below bounds check by reading |
| | | 915 | | // off the end of the string using unsafe code. Since strings are null-terminated, |
| | | 916 | | // we're guaranteed not to read a valid low surrogate, so we'll fail correctly if |
| | | 917 | | // the string terminates unexpectedly. |
| | | 918 | | |
| | 0 | 919 | | index++; |
| | 0 | 920 | | if ((uint)index >= (uint)input.Length) |
| | | 921 | | { |
| | 0 | 922 | | return -1; // not an argument exception - just a "bad data" failure |
| | | 923 | | } |
| | | 924 | | |
| | 0 | 925 | | uint potentialLowSurrogate = input[index]; |
| | 0 | 926 | | if (!UnicodeUtility.IsLowSurrogateCodePoint(potentialLowSurrogate)) |
| | | 927 | | { |
| | 0 | 928 | | return -1; |
| | | 929 | | } |
| | | 930 | | |
| | 0 | 931 | | returnValue = UnicodeUtility.GetScalarFromUtf16SurrogatePair(returnValue, potentialLowSurrogate); |
| | | 932 | | } |
| | | 933 | | |
| | 0 | 934 | | return (int)returnValue; |
| | | 935 | | } |
| | | 936 | | #endif |
| | | 937 | | |
| | | 938 | | /// <summary> |
| | | 939 | | /// Returns a <see cref="string"/> representation of this <see cref="Rune"/> instance. |
| | | 940 | | /// </summary> |
| | | 941 | | public override unsafe string ToString() |
| | | 942 | | { |
| | | 943 | | #if SYSTEM_PRIVATE_CORELIB |
| | 0 | 944 | | if (IsBmp) |
| | | 945 | | { |
| | 0 | 946 | | return string.CreateFromChar((char)_value); |
| | | 947 | | } |
| | | 948 | | else |
| | | 949 | | { |
| | 0 | 950 | | UnicodeUtility.GetUtf16SurrogatesFromSupplementaryPlaneScalar(_value, out char high, out char low); |
| | 0 | 951 | | return string.CreateFromChar(high, low); |
| | | 952 | | } |
| | | 953 | | #else |
| | | 954 | | if (IsBmp) |
| | | 955 | | { |
| | | 956 | | return ((char)_value).ToString(); |
| | | 957 | | } |
| | | 958 | | else |
| | | 959 | | { |
| | | 960 | | Span<char> buffer = stackalloc char[MaxUtf16CharsPerRune]; |
| | | 961 | | UnicodeUtility.GetUtf16SurrogatesFromSupplementaryPlaneScalar(_value, out buffer[0], out buffer[1]); |
| | | 962 | | return buffer.ToString(); |
| | | 963 | | } |
| | | 964 | | #endif |
| | | 965 | | } |
| | | 966 | | |
| | | 967 | | #if SYSTEM_PRIVATE_CORELIB |
| | | 968 | | bool ISpanFormattable.TryFormat(Span<char> destination, out int charsWritten, ReadOnlySpan<char> format, IFormat |
| | 0 | 969 | | TryEncodeToUtf16(destination, out charsWritten); |
| | | 970 | | |
| | | 971 | | bool IUtf8SpanFormattable.TryFormat(Span<byte> utf8Destination, out int bytesWritten, ReadOnlySpan<char> format, |
| | 0 | 972 | | TryEncodeToUtf8(utf8Destination, out bytesWritten); |
| | | 973 | | |
| | | 974 | | /// <inheritdoc cref="IUtf8SpanParsable{TSelf}.TryParse(ReadOnlySpan{byte}, IFormatProvider?, out TSelf)" /> |
| | | 975 | | static bool IUtf8SpanParsable<Rune>.TryParse(ReadOnlySpan<byte> utf8Text, IFormatProvider? provider, out Rune re |
| | | 976 | | { |
| | 0 | 977 | | if (DecodeFromUtf8(utf8Text, out result, out int bytesConsumed) == OperationStatus.Done) |
| | | 978 | | { |
| | 0 | 979 | | if (bytesConsumed == utf8Text.Length) |
| | | 980 | | { |
| | 0 | 981 | | return true; |
| | | 982 | | } |
| | | 983 | | |
| | 0 | 984 | | result = ReplacementChar; |
| | | 985 | | } |
| | | 986 | | |
| | 0 | 987 | | return false; |
| | | 988 | | } |
| | | 989 | | |
| | | 990 | | /// <inheritdoc cref="IUtf8SpanParsable{TSelf}.Parse(ReadOnlySpan{byte}, IFormatProvider?)" /> |
| | | 991 | | static Rune IUtf8SpanParsable<Rune>.Parse(ReadOnlySpan<byte> utf8Text, System.IFormatProvider? provider) |
| | | 992 | | { |
| | 0 | 993 | | if (DecodeFromUtf8(utf8Text, out Rune result, out int bytesConsumed) != OperationStatus.Done || bytesConsume |
| | | 994 | | { |
| | 0 | 995 | | ThrowHelper.ThrowFormatInvalidString(); |
| | | 996 | | } |
| | | 997 | | |
| | 0 | 998 | | return result; |
| | | 999 | | } |
| | | 1000 | | |
| | | 1001 | | /// <inheritdoc cref="IParsable{TSelf}.Parse(string, IFormatProvider?)" /> |
| | | 1002 | | static Rune IParsable<Rune>.Parse(string s, IFormatProvider? provider) |
| | | 1003 | | { |
| | 0 | 1004 | | ArgumentNullException.ThrowIfNull(s); |
| | | 1005 | | |
| | 0 | 1006 | | if (DecodeFromUtf16(s, out Rune result, out int charsConsumed) != OperationStatus.Done || charsConsumed != s |
| | | 1007 | | { |
| | 0 | 1008 | | ThrowHelper.ThrowFormatInvalidString(); |
| | | 1009 | | } |
| | | 1010 | | |
| | 0 | 1011 | | return result; |
| | | 1012 | | } |
| | | 1013 | | |
| | | 1014 | | /// <inheritdoc cref="IParsable{TSelf}.TryParse(string?, IFormatProvider?, out TSelf)" /> |
| | | 1015 | | static bool IParsable<Rune>.TryParse([NotNullWhen(true)] string? s, IFormatProvider? provider, out Rune result) |
| | | 1016 | | { |
| | 0 | 1017 | | if (DecodeFromUtf16(s, out result, out int charsConsumed) != OperationStatus.Done || charsConsumed != s!.Len |
| | | 1018 | | { |
| | 0 | 1019 | | result = ReplacementChar; |
| | 0 | 1020 | | return false; |
| | | 1021 | | } |
| | | 1022 | | |
| | 0 | 1023 | | return true; |
| | | 1024 | | } |
| | | 1025 | | |
| | | 1026 | | /// <inheritdoc cref="ISpanParsable{TSelf}.Parse(ReadOnlySpan{char}, IFormatProvider?)" /> |
| | | 1027 | | static Rune ISpanParsable<Rune>.Parse(ReadOnlySpan<char> s, IFormatProvider? provider) |
| | | 1028 | | { |
| | 0 | 1029 | | if (DecodeFromUtf16(s, out Rune result, out int charsConsumed) != OperationStatus.Done || charsConsumed != s |
| | | 1030 | | { |
| | 0 | 1031 | | ThrowHelper.ThrowFormatInvalidString(); |
| | | 1032 | | } |
| | | 1033 | | |
| | 0 | 1034 | | return result; |
| | | 1035 | | } |
| | | 1036 | | |
| | | 1037 | | /// <inheritdoc cref="ISpanParsable{TSelf}.TryParse(ReadOnlySpan{char}, IFormatProvider?, out TSelf)" /> |
| | | 1038 | | static bool ISpanParsable<Rune>.TryParse(ReadOnlySpan<char> s, IFormatProvider? provider, out Rune result) |
| | | 1039 | | { |
| | 0 | 1040 | | if (DecodeFromUtf16(s, out result, out int charsConsumed) != OperationStatus.Done || charsConsumed != s.Leng |
| | | 1041 | | { |
| | 0 | 1042 | | result = ReplacementChar; |
| | 0 | 1043 | | return false; |
| | | 1044 | | } |
| | | 1045 | | |
| | 0 | 1046 | | return true; |
| | | 1047 | | } |
| | | 1048 | | |
| | 0 | 1049 | | string IFormattable.ToString(string? format, IFormatProvider? formatProvider) => ToString(); |
| | | 1050 | | #endif |
| | | 1051 | | |
| | | 1052 | | /// <summary> |
| | | 1053 | | /// Attempts to create a <see cref="Rune"/> from the provided input value. |
| | | 1054 | | /// </summary> |
| | | 1055 | | public static bool TryCreate(char ch, out Rune result) |
| | | 1056 | | { |
| | 1865796 | 1057 | | uint extendedValue = ch; |
| | 1865796 | 1058 | | if (!UnicodeUtility.IsSurrogateCodePoint(extendedValue)) |
| | | 1059 | | { |
| | 1865796 | 1060 | | result = UnsafeCreate(extendedValue); |
| | 1865796 | 1061 | | return true; |
| | | 1062 | | } |
| | | 1063 | | else |
| | | 1064 | | { |
| | 0 | 1065 | | result = default; |
| | 0 | 1066 | | return false; |
| | | 1067 | | } |
| | | 1068 | | } |
| | | 1069 | | |
| | | 1070 | | /// <summary> |
| | | 1071 | | /// Attempts to create a <see cref="Rune"/> from the provided UTF-16 surrogate pair. |
| | | 1072 | | /// Returns <see langword="false"/> if the input values don't represent a well-formed UTF-16surrogate pair. |
| | | 1073 | | /// </summary> |
| | | 1074 | | public static bool TryCreate(char highSurrogate, char lowSurrogate, out Rune result) |
| | | 1075 | | { |
| | | 1076 | | // First, extend both to 32 bits, then calculate the offset of |
| | | 1077 | | // each candidate surrogate char from the start of its range. |
| | | 1078 | | |
| | 0 | 1079 | | uint highSurrogateOffset = (uint)highSurrogate - HighSurrogateStart; |
| | 0 | 1080 | | uint lowSurrogateOffset = (uint)lowSurrogate - LowSurrogateStart; |
| | | 1081 | | |
| | | 1082 | | // This is a single comparison which allows us to check both for validity at once since |
| | | 1083 | | // both the high surrogate range and the low surrogate range are the same length. |
| | | 1084 | | // If the comparison fails, we call to a helper method to throw the correct exception message. |
| | | 1085 | | |
| | 0 | 1086 | | if ((highSurrogateOffset | lowSurrogateOffset) <= HighSurrogateRange) |
| | | 1087 | | { |
| | | 1088 | | // The 0x40u << 10 below is to account for uuuuu = wwww + 1 in the surrogate encoding. |
| | 0 | 1089 | | result = UnsafeCreate((highSurrogateOffset << 10) + ((uint)lowSurrogate - LowSurrogateStart) + (0x40u << |
| | 0 | 1090 | | return true; |
| | | 1091 | | } |
| | | 1092 | | else |
| | | 1093 | | { |
| | | 1094 | | // Didn't have a high surrogate followed by a low surrogate. |
| | 0 | 1095 | | result = default; |
| | 0 | 1096 | | return false; |
| | | 1097 | | } |
| | | 1098 | | } |
| | | 1099 | | |
| | | 1100 | | /// <summary> |
| | | 1101 | | /// Attempts to create a <see cref="Rune"/> from the provided input value. |
| | | 1102 | | /// </summary> |
| | 0 | 1103 | | public static bool TryCreate(int value, out Rune result) => TryCreate((uint)value, out result); |
| | | 1104 | | |
| | | 1105 | | /// <summary> |
| | | 1106 | | /// Attempts to create a <see cref="Rune"/> from the provided input value. |
| | | 1107 | | /// </summary> |
| | | 1108 | | [CLSCompliant(false)] |
| | | 1109 | | public static bool TryCreate(uint value, out Rune result) |
| | | 1110 | | { |
| | 0 | 1111 | | if (UnicodeUtility.IsValidUnicodeScalar(value)) |
| | | 1112 | | { |
| | 0 | 1113 | | result = UnsafeCreate(value); |
| | 0 | 1114 | | return true; |
| | | 1115 | | } |
| | | 1116 | | else |
| | | 1117 | | { |
| | 0 | 1118 | | result = default; |
| | 0 | 1119 | | return false; |
| | | 1120 | | } |
| | | 1121 | | } |
| | | 1122 | | |
| | | 1123 | | /// <summary> |
| | | 1124 | | /// Encodes this <see cref="Rune"/> to a UTF-16 destination buffer. |
| | | 1125 | | /// </summary> |
| | | 1126 | | /// <param name="destination">The buffer to which to write this value as UTF-16.</param> |
| | | 1127 | | /// <param name="charsWritten"> |
| | | 1128 | | /// The number of <see cref="char"/>s written to <paramref name="destination"/>, |
| | | 1129 | | /// or 0 if the destination buffer is not large enough to contain the output.</param> |
| | | 1130 | | /// <returns>True if the value was written to the buffer; otherwise, false.</returns> |
| | | 1131 | | /// <remarks> |
| | | 1132 | | /// The <see cref="Utf16SequenceLength"/> property can be queried ahead of time to determine |
| | | 1133 | | /// the required size of the <paramref name="destination"/> buffer. |
| | | 1134 | | /// </remarks> |
| | | 1135 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1136 | | public bool TryEncodeToUtf16(Span<char> destination, out int charsWritten) |
| | | 1137 | | { |
| | | 1138 | | // The Rune type fits cleanly into a register, so pass byval rather than byref |
| | | 1139 | | // to avoid stack-spilling the 'this' parameter. |
| | 175880 | 1140 | | return TryEncodeToUtf16(this, destination, out charsWritten); |
| | | 1141 | | } |
| | | 1142 | | |
| | | 1143 | | private static bool TryEncodeToUtf16(Rune value, Span<char> destination, out int charsWritten) |
| | | 1144 | | { |
| | 175880 | 1145 | | if (!destination.IsEmpty) |
| | | 1146 | | { |
| | 175880 | 1147 | | if (value.IsBmp) |
| | | 1148 | | { |
| | 172690 | 1149 | | destination[0] = (char)value._value; |
| | 172690 | 1150 | | charsWritten = 1; |
| | 172690 | 1151 | | return true; |
| | | 1152 | | } |
| | 3190 | 1153 | | else if (destination.Length > 1) |
| | | 1154 | | { |
| | 3190 | 1155 | | UnicodeUtility.GetUtf16SurrogatesFromSupplementaryPlaneScalar((uint)value._value, out destination[0] |
| | 3190 | 1156 | | charsWritten = 2; |
| | 3190 | 1157 | | return true; |
| | | 1158 | | } |
| | | 1159 | | } |
| | | 1160 | | |
| | | 1161 | | // Destination buffer not large enough |
| | | 1162 | | |
| | 0 | 1163 | | charsWritten = default; |
| | 0 | 1164 | | return false; |
| | | 1165 | | } |
| | | 1166 | | |
| | | 1167 | | /// <summary> |
| | | 1168 | | /// Encodes this <see cref="Rune"/> to a destination buffer as UTF-8 bytes. |
| | | 1169 | | /// </summary> |
| | | 1170 | | /// <param name="destination">The buffer to which to write this value as UTF-8.</param> |
| | | 1171 | | /// <param name="bytesWritten"> |
| | | 1172 | | /// The number of <see cref="byte"/>s written to <paramref name="destination"/>, |
| | | 1173 | | /// or 0 if the destination buffer is not large enough to contain the output.</param> |
| | | 1174 | | /// <returns>True if the value was written to the buffer; otherwise, false.</returns> |
| | | 1175 | | /// <remarks> |
| | | 1176 | | /// The <see cref="Utf8SequenceLength"/> property can be queried ahead of time to determine |
| | | 1177 | | /// the required size of the <paramref name="destination"/> buffer. |
| | | 1178 | | /// </remarks> |
| | | 1179 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1180 | | public bool TryEncodeToUtf8(Span<byte> destination, out int bytesWritten) |
| | | 1181 | | { |
| | | 1182 | | // The Rune type fits cleanly into a register, so pass byval rather than byref |
| | | 1183 | | // to avoid stack-spilling the 'this' parameter. |
| | 0 | 1184 | | return TryEncodeToUtf8(this, destination, out bytesWritten); |
| | | 1185 | | } |
| | | 1186 | | |
| | | 1187 | | private static bool TryEncodeToUtf8(Rune value, Span<byte> destination, out int bytesWritten) |
| | | 1188 | | { |
| | | 1189 | | // The bit patterns below come from the Unicode Standard, Table 3-6. |
| | | 1190 | | |
| | 0 | 1191 | | if (!destination.IsEmpty) |
| | | 1192 | | { |
| | 0 | 1193 | | if (value.IsAscii) |
| | | 1194 | | { |
| | 0 | 1195 | | destination[0] = (byte)value._value; |
| | 0 | 1196 | | bytesWritten = 1; |
| | 0 | 1197 | | return true; |
| | | 1198 | | } |
| | | 1199 | | |
| | 0 | 1200 | | if (destination.Length > 1) |
| | | 1201 | | { |
| | 0 | 1202 | | if (value.Value <= 0x7FFu) |
| | | 1203 | | { |
| | | 1204 | | // Scalar 00000yyy yyxxxxxx -> bytes [ 110yyyyy 10xxxxxx ] |
| | 0 | 1205 | | destination[0] = (byte)((value._value + (0b110u << 11)) >> 6); |
| | 0 | 1206 | | destination[1] = (byte)((value._value & 0x3Fu) + 0x80u); |
| | 0 | 1207 | | bytesWritten = 2; |
| | 0 | 1208 | | return true; |
| | | 1209 | | } |
| | | 1210 | | |
| | 0 | 1211 | | if (destination.Length > 2) |
| | | 1212 | | { |
| | 0 | 1213 | | if (value.Value <= 0xFFFFu) |
| | | 1214 | | { |
| | | 1215 | | // Scalar zzzzyyyy yyxxxxxx -> bytes [ 1110zzzz 10yyyyyy 10xxxxxx ] |
| | 0 | 1216 | | destination[0] = (byte)((value._value + (0b1110 << 16)) >> 12); |
| | 0 | 1217 | | destination[1] = (byte)(((value._value & (0x3Fu << 6)) >> 6) + 0x80u); |
| | 0 | 1218 | | destination[2] = (byte)((value._value & 0x3Fu) + 0x80u); |
| | 0 | 1219 | | bytesWritten = 3; |
| | 0 | 1220 | | return true; |
| | | 1221 | | } |
| | | 1222 | | |
| | 0 | 1223 | | if (destination.Length > 3) |
| | | 1224 | | { |
| | | 1225 | | // Scalar 000uuuuu zzzzyyyy yyxxxxxx -> bytes [ 11110uuu 10uuzzzz 10yyyyyy 10xxxxxx ] |
| | 0 | 1226 | | destination[0] = (byte)((value._value + (0b11110 << 21)) >> 18); |
| | 0 | 1227 | | destination[1] = (byte)(((value._value & (0x3Fu << 12)) >> 12) + 0x80u); |
| | 0 | 1228 | | destination[2] = (byte)(((value._value & (0x3Fu << 6)) >> 6) + 0x80u); |
| | 0 | 1229 | | destination[3] = (byte)((value._value & 0x3Fu) + 0x80u); |
| | 0 | 1230 | | bytesWritten = 4; |
| | 0 | 1231 | | return true; |
| | | 1232 | | } |
| | | 1233 | | } |
| | | 1234 | | } |
| | | 1235 | | } |
| | | 1236 | | |
| | | 1237 | | // Destination buffer not large enough |
| | | 1238 | | |
| | 0 | 1239 | | bytesWritten = default; |
| | 0 | 1240 | | return false; |
| | | 1241 | | } |
| | | 1242 | | |
| | | 1243 | | #if SYSTEM_PRIVATE_CORELIB |
| | | 1244 | | /// <summary> |
| | | 1245 | | /// Attempts to get the <see cref="Rune"/> which begins at index <paramref name="index"/> in |
| | | 1246 | | /// string <paramref name="input"/>. |
| | | 1247 | | /// </summary> |
| | | 1248 | | /// <returns><see langword="true"/> if a scalar value was successfully extracted from the specified index, |
| | | 1249 | | /// <see langword="false"/> if a value could not be extracted due to invalid data.</returns> |
| | | 1250 | | /// <remarks> |
| | | 1251 | | /// Throws only if <paramref name="input"/> is null or <paramref name="index"/> is out of range. |
| | | 1252 | | /// </remarks> |
| | | 1253 | | public static bool TryGetRuneAt(string input, int index, out Rune value) |
| | | 1254 | | { |
| | 0 | 1255 | | int runeValue = ReadRuneFromString(input, index); |
| | 0 | 1256 | | if (runeValue >= 0) |
| | | 1257 | | { |
| | 0 | 1258 | | value = UnsafeCreate((uint)runeValue); |
| | 0 | 1259 | | return true; |
| | | 1260 | | } |
| | | 1261 | | else |
| | | 1262 | | { |
| | 0 | 1263 | | value = default; |
| | 0 | 1264 | | return false; |
| | | 1265 | | } |
| | | 1266 | | } |
| | | 1267 | | #endif |
| | | 1268 | | |
| | | 1269 | | // Allows constructing a Unicode scalar value from an arbitrary 32-bit integer without |
| | | 1270 | | // validation. It is the caller's responsibility to have performed manual validation |
| | | 1271 | | // before calling this method. If a Rune instance is forcibly constructed |
| | | 1272 | | // from invalid input, the APIs on this type have undefined behavior, potentially including |
| | | 1273 | | // introducing a security hole in the consuming application. |
| | | 1274 | | // |
| | | 1275 | | // An example of a security hole resulting from an invalid Rune value, which could result |
| | | 1276 | | // in a stack overflow. |
| | | 1277 | | // |
| | | 1278 | | // public int GetMarvin32HashCode(Rune r) { |
| | | 1279 | | // Span<char> buffer = stackalloc char[r.Utf16SequenceLength]; |
| | | 1280 | | // r.TryEncode(buffer, ...); |
| | | 1281 | | // return Marvin32.ComputeHash(buffer.AsBytes()); |
| | | 1282 | | // } |
| | | 1283 | | |
| | | 1284 | | /// <summary> |
| | | 1285 | | /// Creates a <see cref="Rune"/> without performing validation on the input. |
| | | 1286 | | /// </summary> |
| | | 1287 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 4445191 | 1288 | | internal static Rune UnsafeCreate(uint scalarValue) => new Rune(scalarValue, false); |
| | | 1289 | | |
| | | 1290 | | // These are analogs of APIs on System.Char |
| | | 1291 | | |
| | | 1292 | | public static double GetNumericValue(Rune value) |
| | | 1293 | | { |
| | 0 | 1294 | | if (value.IsAscii) |
| | | 1295 | | { |
| | 0 | 1296 | | uint baseNum = value._value - '0'; |
| | 0 | 1297 | | return (baseNum <= 9) ? (double)baseNum : -1; |
| | | 1298 | | } |
| | | 1299 | | else |
| | | 1300 | | { |
| | | 1301 | | // not an ASCII char; fall back to globalization table |
| | | 1302 | | #if SYSTEM_PRIVATE_CORELIB |
| | 0 | 1303 | | return CharUnicodeInfo.GetNumericValue(value.Value); |
| | | 1304 | | #else |
| | | 1305 | | if (value.IsBmp) |
| | | 1306 | | { |
| | | 1307 | | return CharUnicodeInfo.GetNumericValue((char)value._value); |
| | | 1308 | | } |
| | | 1309 | | return CharUnicodeInfo.GetNumericValue(value.ToString(), 0); |
| | | 1310 | | #endif |
| | | 1311 | | } |
| | | 1312 | | } |
| | | 1313 | | |
| | | 1314 | | public static UnicodeCategory GetUnicodeCategory(Rune value) |
| | | 1315 | | { |
| | 0 | 1316 | | if (value.IsAscii) |
| | | 1317 | | { |
| | 0 | 1318 | | return (UnicodeCategory)(AsciiCharInfo[value.Value] & UnicodeCategoryMask); |
| | | 1319 | | } |
| | | 1320 | | else |
| | | 1321 | | { |
| | 0 | 1322 | | return GetUnicodeCategoryNonAscii(value); |
| | | 1323 | | } |
| | | 1324 | | } |
| | | 1325 | | |
| | | 1326 | | private static UnicodeCategory GetUnicodeCategoryNonAscii(Rune value) |
| | | 1327 | | { |
| | 0 | 1328 | | Debug.Assert(!value.IsAscii, "Shouldn't use this non-optimized code path for ASCII characters."); |
| | | 1329 | | #if (!NETSTANDARD2_0 && !NETFRAMEWORK) |
| | 0 | 1330 | | return CharUnicodeInfo.GetUnicodeCategory(value.Value); |
| | | 1331 | | #else |
| | | 1332 | | if (value.IsBmp) |
| | | 1333 | | { |
| | | 1334 | | return CharUnicodeInfo.GetUnicodeCategory((char)value._value); |
| | | 1335 | | } |
| | | 1336 | | return CharUnicodeInfo.GetUnicodeCategory(value.ToString(), 0); |
| | | 1337 | | #endif |
| | | 1338 | | } |
| | | 1339 | | |
| | | 1340 | | // Returns true iff this Unicode category represents a letter |
| | | 1341 | | private static bool IsCategoryLetter(UnicodeCategory category) |
| | | 1342 | | { |
| | 0 | 1343 | | return UnicodeUtility.IsInRangeInclusive((uint)category, (uint)UnicodeCategory.UppercaseLetter, (uint)Unicod |
| | | 1344 | | } |
| | | 1345 | | |
| | | 1346 | | // Returns true iff this Unicode category represents a letter or a decimal digit |
| | | 1347 | | private static bool IsCategoryLetterOrDecimalDigit(UnicodeCategory category) |
| | | 1348 | | { |
| | 0 | 1349 | | return UnicodeUtility.IsInRangeInclusive((uint)category, (uint)UnicodeCategory.UppercaseLetter, (uint)Unicod |
| | 0 | 1350 | | || (category == UnicodeCategory.DecimalDigitNumber); |
| | | 1351 | | } |
| | | 1352 | | |
| | | 1353 | | // Returns true iff this Unicode category represents a number |
| | | 1354 | | private static bool IsCategoryNumber(UnicodeCategory category) |
| | | 1355 | | { |
| | 0 | 1356 | | return UnicodeUtility.IsInRangeInclusive((uint)category, (uint)UnicodeCategory.DecimalDigitNumber, (uint)Uni |
| | | 1357 | | } |
| | | 1358 | | |
| | | 1359 | | // Returns true iff this Unicode category represents a punctuation mark |
| | | 1360 | | private static bool IsCategoryPunctuation(UnicodeCategory category) |
| | | 1361 | | { |
| | 0 | 1362 | | return UnicodeUtility.IsInRangeInclusive((uint)category, (uint)UnicodeCategory.ConnectorPunctuation, (uint)U |
| | | 1363 | | } |
| | | 1364 | | |
| | | 1365 | | // Returns true iff this Unicode category represents a separator |
| | | 1366 | | private static bool IsCategorySeparator(UnicodeCategory category) |
| | | 1367 | | { |
| | 0 | 1368 | | return UnicodeUtility.IsInRangeInclusive((uint)category, (uint)UnicodeCategory.SpaceSeparator, (uint)Unicode |
| | | 1369 | | } |
| | | 1370 | | |
| | | 1371 | | // Returns true iff this Unicode category represents a symbol |
| | | 1372 | | private static bool IsCategorySymbol(UnicodeCategory category) |
| | | 1373 | | { |
| | 0 | 1374 | | return UnicodeUtility.IsInRangeInclusive((uint)category, (uint)UnicodeCategory.MathSymbol, (uint)UnicodeCate |
| | | 1375 | | } |
| | | 1376 | | |
| | | 1377 | | public static bool IsControl(Rune value) |
| | | 1378 | | { |
| | | 1379 | | // Per the Unicode stability policy, the set of control characters |
| | | 1380 | | // is forever fixed at [ U+0000..U+001F ], [ U+007F..U+009F ]. No |
| | | 1381 | | // characters will ever be added to or removed from the "control characters" |
| | | 1382 | | // group. See https://www.unicode.org/policies/stability_policy.html. |
| | | 1383 | | |
| | | 1384 | | // Logic below depends on Rune.Value never being -1 (since Rune is a validating type) |
| | | 1385 | | // 00..1F (+1) => 01..20 (&~80) => 01..20 |
| | | 1386 | | // 7F..9F (+1) => 80..A0 (&~80) => 00..20 |
| | | 1387 | | |
| | 0 | 1388 | | return ((value._value + 1) & ~0x80u) <= 0x20u; |
| | | 1389 | | } |
| | | 1390 | | |
| | | 1391 | | public static bool IsDigit(Rune value) |
| | | 1392 | | { |
| | 0 | 1393 | | if (value.IsAscii) |
| | | 1394 | | { |
| | 0 | 1395 | | return UnicodeUtility.IsInRangeInclusive(value._value, '0', '9'); |
| | | 1396 | | } |
| | | 1397 | | else |
| | | 1398 | | { |
| | 0 | 1399 | | return GetUnicodeCategoryNonAscii(value) == UnicodeCategory.DecimalDigitNumber; |
| | | 1400 | | } |
| | | 1401 | | } |
| | | 1402 | | |
| | | 1403 | | public static bool IsLetter(Rune value) |
| | | 1404 | | { |
| | 0 | 1405 | | if (value.IsAscii) |
| | | 1406 | | { |
| | 0 | 1407 | | return ((value._value - 'A') & ~0x20u) <= (uint)('Z' - 'A'); // [A-Za-z] |
| | | 1408 | | } |
| | | 1409 | | else |
| | | 1410 | | { |
| | 0 | 1411 | | return IsCategoryLetter(GetUnicodeCategoryNonAscii(value)); |
| | | 1412 | | } |
| | | 1413 | | } |
| | | 1414 | | |
| | | 1415 | | public static bool IsLetterOrDigit(Rune value) |
| | | 1416 | | { |
| | 0 | 1417 | | if (value.IsAscii) |
| | | 1418 | | { |
| | 0 | 1419 | | return (AsciiCharInfo[value.Value] & IsLetterOrDigitFlag) != 0; |
| | | 1420 | | } |
| | | 1421 | | else |
| | | 1422 | | { |
| | 0 | 1423 | | return IsCategoryLetterOrDecimalDigit(GetUnicodeCategoryNonAscii(value)); |
| | | 1424 | | } |
| | | 1425 | | } |
| | | 1426 | | |
| | | 1427 | | public static bool IsLower(Rune value) |
| | | 1428 | | { |
| | 0 | 1429 | | if (value.IsAscii) |
| | | 1430 | | { |
| | 0 | 1431 | | return UnicodeUtility.IsInRangeInclusive(value._value, 'a', 'z'); |
| | | 1432 | | } |
| | | 1433 | | else |
| | | 1434 | | { |
| | 0 | 1435 | | return GetUnicodeCategoryNonAscii(value) == UnicodeCategory.LowercaseLetter; |
| | | 1436 | | } |
| | | 1437 | | } |
| | | 1438 | | |
| | | 1439 | | public static bool IsNumber(Rune value) |
| | | 1440 | | { |
| | 0 | 1441 | | if (value.IsAscii) |
| | | 1442 | | { |
| | 0 | 1443 | | return UnicodeUtility.IsInRangeInclusive(value._value, '0', '9'); |
| | | 1444 | | } |
| | | 1445 | | else |
| | | 1446 | | { |
| | 0 | 1447 | | return IsCategoryNumber(GetUnicodeCategoryNonAscii(value)); |
| | | 1448 | | } |
| | | 1449 | | } |
| | | 1450 | | |
| | | 1451 | | public static bool IsPunctuation(Rune value) |
| | | 1452 | | { |
| | 0 | 1453 | | return IsCategoryPunctuation(GetUnicodeCategory(value)); |
| | | 1454 | | } |
| | | 1455 | | |
| | | 1456 | | public static bool IsSeparator(Rune value) |
| | | 1457 | | { |
| | 0 | 1458 | | return IsCategorySeparator(GetUnicodeCategory(value)); |
| | | 1459 | | } |
| | | 1460 | | |
| | | 1461 | | public static bool IsSymbol(Rune value) |
| | | 1462 | | { |
| | 0 | 1463 | | return IsCategorySymbol(GetUnicodeCategory(value)); |
| | | 1464 | | } |
| | | 1465 | | |
| | | 1466 | | public static bool IsUpper(Rune value) |
| | | 1467 | | { |
| | 0 | 1468 | | if (value.IsAscii) |
| | | 1469 | | { |
| | 0 | 1470 | | return UnicodeUtility.IsInRangeInclusive(value._value, 'A', 'Z'); |
| | | 1471 | | } |
| | | 1472 | | else |
| | | 1473 | | { |
| | 0 | 1474 | | return GetUnicodeCategoryNonAscii(value) == UnicodeCategory.UppercaseLetter; |
| | | 1475 | | } |
| | | 1476 | | } |
| | | 1477 | | |
| | | 1478 | | public static bool IsWhiteSpace(Rune value) |
| | | 1479 | | { |
| | 0 | 1480 | | if (value.IsAscii) |
| | | 1481 | | { |
| | 0 | 1482 | | return (AsciiCharInfo[value.Value] & IsWhiteSpaceFlag) != 0; |
| | | 1483 | | } |
| | | 1484 | | |
| | | 1485 | | // Only BMP code points can be white space, so only call into CharUnicodeInfo |
| | | 1486 | | // if the incoming value is within the BMP. |
| | | 1487 | | |
| | 0 | 1488 | | return value.IsBmp && |
| | 0 | 1489 | | #if SYSTEM_PRIVATE_CORELIB |
| | 0 | 1490 | | CharUnicodeInfo.GetIsWhiteSpace((char)value._value); |
| | | 1491 | | #else |
| | | 1492 | | char.IsWhiteSpace((char)value._value); |
| | | 1493 | | #endif |
| | | 1494 | | } |
| | | 1495 | | |
| | | 1496 | | public static Rune ToLower(Rune value, CultureInfo culture) |
| | | 1497 | | { |
| | 0 | 1498 | | if (culture is null) |
| | | 1499 | | { |
| | 0 | 1500 | | ThrowHelper.ThrowArgumentNullException(ExceptionArgument.culture); |
| | | 1501 | | } |
| | | 1502 | | |
| | | 1503 | | // We don't want to special-case ASCII here since the specified culture might handle |
| | | 1504 | | // ASCII characters differently than the invariant culture (e.g., Turkish I). Instead |
| | | 1505 | | // we'll just jump straight to the globalization tables if they're available. |
| | | 1506 | | |
| | | 1507 | | #if SYSTEM_PRIVATE_CORELIB |
| | 0 | 1508 | | if (GlobalizationMode.Invariant) |
| | | 1509 | | { |
| | 0 | 1510 | | return ToLowerInvariant(value); |
| | | 1511 | | } |
| | | 1512 | | |
| | 0 | 1513 | | return ChangeCaseCultureAware(value, culture.TextInfo, toUpper: false); |
| | | 1514 | | #else |
| | | 1515 | | return ChangeCaseCultureAware(value, culture, toUpper: false); |
| | | 1516 | | #endif |
| | | 1517 | | } |
| | | 1518 | | |
| | | 1519 | | public static Rune ToLowerInvariant(Rune value) |
| | | 1520 | | { |
| | | 1521 | | // Handle the most common case (ASCII data) first. Within the common case, we expect |
| | | 1522 | | // that there'll be a mix of lowercase & uppercase chars, so make the conversion branchless. |
| | | 1523 | | |
| | 0 | 1524 | | if (value.IsAscii) |
| | | 1525 | | { |
| | | 1526 | | // It's ok for us to use the UTF-16 conversion utility for this since the high |
| | | 1527 | | // 16 bits of the value will never be set so will be left unchanged. |
| | 0 | 1528 | | return UnsafeCreate(Utf16Utility.ConvertAllAsciiCharsInUInt32ToLowercase(value._value)); |
| | | 1529 | | } |
| | | 1530 | | |
| | | 1531 | | #if SYSTEM_PRIVATE_CORELIB |
| | 0 | 1532 | | if (GlobalizationMode.Invariant) |
| | | 1533 | | { |
| | 0 | 1534 | | return UnsafeCreate(CharUnicodeInfo.ToLower(value._value)); |
| | | 1535 | | } |
| | | 1536 | | |
| | | 1537 | | // Non-ASCII data requires going through the case folding tables. |
| | | 1538 | | |
| | 0 | 1539 | | return ChangeCaseCultureAware(value, TextInfo.Invariant, toUpper: false); |
| | | 1540 | | #else |
| | | 1541 | | return ChangeCaseCultureAware(value, CultureInfo.InvariantCulture, toUpper: false); |
| | | 1542 | | #endif |
| | | 1543 | | } |
| | | 1544 | | |
| | | 1545 | | public static Rune ToUpper(Rune value, CultureInfo culture) |
| | | 1546 | | { |
| | 0 | 1547 | | if (culture is null) |
| | | 1548 | | { |
| | 0 | 1549 | | ThrowHelper.ThrowArgumentNullException(ExceptionArgument.culture); |
| | | 1550 | | } |
| | | 1551 | | |
| | | 1552 | | // We don't want to special-case ASCII here since the specified culture might handle |
| | | 1553 | | // ASCII characters differently than the invariant culture (e.g., Turkish I). Instead |
| | | 1554 | | // we'll just jump straight to the globalization tables if they're available. |
| | | 1555 | | |
| | | 1556 | | #if SYSTEM_PRIVATE_CORELIB |
| | 0 | 1557 | | if (GlobalizationMode.Invariant) |
| | | 1558 | | { |
| | 0 | 1559 | | return ToUpperInvariant(value); |
| | | 1560 | | } |
| | | 1561 | | |
| | 0 | 1562 | | return ChangeCaseCultureAware(value, culture.TextInfo, toUpper: true); |
| | | 1563 | | #else |
| | | 1564 | | return ChangeCaseCultureAware(value, culture, toUpper: true); |
| | | 1565 | | #endif |
| | | 1566 | | } |
| | | 1567 | | |
| | | 1568 | | public static Rune ToUpperInvariant(Rune value) |
| | | 1569 | | { |
| | | 1570 | | // Handle the most common case (ASCII data) first. Within the common case, we expect |
| | | 1571 | | // that there'll be a mix of lowercase & uppercase chars, so make the conversion branchless. |
| | | 1572 | | |
| | 0 | 1573 | | if (value.IsAscii) |
| | | 1574 | | { |
| | | 1575 | | // It's ok for us to use the UTF-16 conversion utility for this since the high |
| | | 1576 | | // 16 bits of the value will never be set so will be left unchanged. |
| | 0 | 1577 | | return UnsafeCreate(Utf16Utility.ConvertAllAsciiCharsInUInt32ToUppercase(value._value)); |
| | | 1578 | | } |
| | | 1579 | | |
| | | 1580 | | #if SYSTEM_PRIVATE_CORELIB |
| | 0 | 1581 | | if (GlobalizationMode.Invariant) |
| | | 1582 | | { |
| | 0 | 1583 | | return UnsafeCreate(CharUnicodeInfo.ToUpper(value._value)); |
| | | 1584 | | } |
| | | 1585 | | |
| | | 1586 | | // Non-ASCII data requires going through the case folding tables. |
| | | 1587 | | |
| | 0 | 1588 | | return ChangeCaseCultureAware(value, TextInfo.Invariant, toUpper: true); |
| | | 1589 | | #else |
| | | 1590 | | return ChangeCaseCultureAware(value, CultureInfo.InvariantCulture, toUpper: true); |
| | | 1591 | | #endif |
| | | 1592 | | } |
| | | 1593 | | |
| | | 1594 | | #if SYSTEM_PRIVATE_CORELIB |
| | | 1595 | | /// <summary> |
| | | 1596 | | /// Returns a copy of <paramref name="value"/> converted to uppercase using the casing rules used by |
| | | 1597 | | /// <see cref="StringComparison.OrdinalIgnoreCase"/> comparisons. |
| | | 1598 | | /// </summary> |
| | | 1599 | | /// <param name="value">The character to convert.</param> |
| | | 1600 | | /// <returns>The uppercase equivalent of <paramref name="value"/>.</returns> |
| | | 1601 | | public static Rune ToUpperOrdinal(Rune value) |
| | | 1602 | | { |
| | 0 | 1603 | | if (value.IsAscii) |
| | | 1604 | | { |
| | 0 | 1605 | | return UnsafeCreate(Utf16Utility.ConvertAllAsciiCharsInUInt32ToUppercase(value._value)); |
| | | 1606 | | } |
| | | 1607 | | |
| | 0 | 1608 | | if (value.IsBmp) |
| | | 1609 | | { |
| | 0 | 1610 | | return UnsafeCreate(TextInfo.ToUpperOrdinal((char)value._value)); |
| | | 1611 | | } |
| | | 1612 | | |
| | | 1613 | | // Supplementary characters use the same simple scalar mapping as OrdinalIgnoreCase comparisons. |
| | 0 | 1614 | | uint upper = CharUnicodeInfo.ToUpper(value._value); |
| | 0 | 1615 | | return UnsafeCreate(GlobalizationMode.UseNls |
| | 0 | 1616 | | ? Ordinal.PreserveNlsOrdinalCasingClass(value._value, upper) |
| | 0 | 1617 | | : upper); |
| | | 1618 | | } |
| | | 1619 | | |
| | | 1620 | | /// <summary> |
| | | 1621 | | /// Returns a copy of <paramref name="value"/> converted to lowercase using ordinal (simple, one-to-one) casing |
| | | 1622 | | /// </summary> |
| | | 1623 | | /// <param name="value">The character to convert.</param> |
| | | 1624 | | /// <returns>The lowercase equivalent of <paramref name="value"/>.</returns> |
| | | 1625 | | public static Rune ToLowerOrdinal(Rune value) |
| | | 1626 | | { |
| | 0 | 1627 | | if (value.IsAscii) |
| | | 1628 | | { |
| | 0 | 1629 | | return UnsafeCreate(Utf16Utility.ConvertAllAsciiCharsInUInt32ToLowercase(value._value)); |
| | | 1630 | | } |
| | | 1631 | | |
| | 0 | 1632 | | if (value.IsBmp) |
| | | 1633 | | { |
| | 0 | 1634 | | return UnsafeCreate(TextInfo.ToLowerOrdinal((char)value._value)); |
| | | 1635 | | } |
| | | 1636 | | |
| | 0 | 1637 | | uint lower = CharUnicodeInfo.ToLower(value._value); |
| | 0 | 1638 | | return UnsafeCreate(GlobalizationMode.UseNls |
| | 0 | 1639 | | ? Ordinal.PreserveNlsOrdinalCasingClass(value._value, lower) |
| | 0 | 1640 | | : lower); |
| | | 1641 | | } |
| | | 1642 | | #endif |
| | | 1643 | | |
| | | 1644 | | /// <inheritdoc cref="IComparable.CompareTo" /> |
| | | 1645 | | int IComparable.CompareTo(object? obj) |
| | | 1646 | | { |
| | 0 | 1647 | | if (obj is null) |
| | | 1648 | | { |
| | 0 | 1649 | | return 1; // non-null ("this") always sorts after null |
| | | 1650 | | } |
| | | 1651 | | |
| | 0 | 1652 | | if (obj is Rune other) |
| | | 1653 | | { |
| | 0 | 1654 | | return this.CompareTo(other); |
| | | 1655 | | } |
| | | 1656 | | |
| | | 1657 | | #if SYSTEM_PRIVATE_CORELIB |
| | 0 | 1658 | | throw new ArgumentException(SR.Arg_MustBeRune); |
| | | 1659 | | #else |
| | | 1660 | | throw new ArgumentException(); |
| | | 1661 | | #endif |
| | | 1662 | | } |
| | | 1663 | | } |
| | | 1664 | | } |
| | | 1665 | | |