| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | using System.Runtime.CompilerServices; |
| | | 5 | | |
| | | 6 | | namespace System.Text |
| | | 7 | | { |
| | | 8 | | internal static class UnicodeUtility |
| | | 9 | | { |
| | | 10 | | /// <summary> |
| | | 11 | | /// The Unicode replacement character U+FFFD. |
| | | 12 | | /// </summary> |
| | | 13 | | public const uint ReplacementChar = 0xFFFD; |
| | | 14 | | |
| | | 15 | | /// <summary> |
| | | 16 | | /// Returns the Unicode plane (0 through 16, inclusive) which contains this code point. |
| | | 17 | | /// </summary> |
| | | 18 | | public static int GetPlane(uint codePoint) |
| | | 19 | | { |
| | 0 | 20 | | UnicodeDebug.AssertIsValidCodePoint(codePoint); |
| | | 21 | | |
| | 0 | 22 | | return (int)(codePoint >> 16); |
| | | 23 | | } |
| | | 24 | | |
| | | 25 | | /// <summary> |
| | | 26 | | /// Returns a Unicode scalar value from two code points representing a UTF-16 surrogate pair. |
| | | 27 | | /// </summary> |
| | | 28 | | public static uint GetScalarFromUtf16SurrogatePair(uint highSurrogateCodePoint, uint lowSurrogateCodePoint) |
| | | 29 | | { |
| | 0 | 30 | | UnicodeDebug.AssertIsHighSurrogateCodePoint(highSurrogateCodePoint); |
| | 0 | 31 | | UnicodeDebug.AssertIsLowSurrogateCodePoint(lowSurrogateCodePoint); |
| | | 32 | | |
| | | 33 | | // This calculation comes from the Unicode specification, Table 3-5. |
| | | 34 | | // Need to remove the D800 marker from the high surrogate and the DC00 marker from the low surrogate, |
| | | 35 | | // then fix up the "wwww = uuuuu - 1" section of the bit distribution. The code is written as below |
| | | 36 | | // to become just two instructions: shl, lea. |
| | | 37 | | |
| | 0 | 38 | | return (highSurrogateCodePoint << 10) + lowSurrogateCodePoint - ((0xD800U << 10) + 0xDC00U - (1 << 16)); |
| | | 39 | | } |
| | | 40 | | |
| | | 41 | | /// <summary> |
| | | 42 | | /// Given a Unicode scalar value, gets the number of UTF-16 code units required to represent this value. |
| | | 43 | | /// </summary> |
| | | 44 | | public static int GetUtf16SequenceLength(uint value) |
| | | 45 | | { |
| | 811866 | 46 | | UnicodeDebug.AssertIsValidScalar(value); |
| | | 47 | | |
| | 811866 | 48 | | value -= 0x10000; // if value < 0x10000, high byte = 0xFF; else high byte = 0x00 |
| | 811866 | 49 | | value += (2 << 24); // if value < 0x10000, high byte = 0x01; else high byte = 0x02 |
| | 811866 | 50 | | value >>= 24; // shift high byte down |
| | 811866 | 51 | | return (int)value; // and return it |
| | | 52 | | } |
| | | 53 | | |
| | | 54 | | /// <summary> |
| | | 55 | | /// Decomposes an astral Unicode scalar into UTF-16 high and low surrogate code units. |
| | | 56 | | /// </summary> |
| | | 57 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 58 | | public static void GetUtf16SurrogatesFromSupplementaryPlaneScalar(uint value, out char highSurrogateCodePoint, o |
| | | 59 | | { |
| | 3190 | 60 | | UnicodeDebug.AssertIsValidSupplementaryPlaneScalar(value); |
| | | 61 | | |
| | | 62 | | // This calculation comes from the Unicode specification, Table 3-5. |
| | | 63 | | |
| | 3190 | 64 | | highSurrogateCodePoint = (char)((value + ((0xD800u - 0x40u) << 10)) >> 10); |
| | 3190 | 65 | | lowSurrogateCodePoint = (char)((value & 0x3FFu) + 0xDC00u); |
| | 3190 | 66 | | } |
| | | 67 | | |
| | | 68 | | /// <summary> |
| | | 69 | | /// Given a Unicode scalar value, gets the number of UTF-8 code units required to represent this value. |
| | | 70 | | /// </summary> |
| | | 71 | | public static int GetUtf8SequenceLength(uint value) |
| | | 72 | | { |
| | 0 | 73 | | UnicodeDebug.AssertIsValidScalar(value); |
| | | 74 | | |
| | | 75 | | // The logic below can handle all valid scalar values branchlessly. |
| | | 76 | | // It gives generally good performance across all inputs, and on x86 |
| | | 77 | | // it's only six instructions: lea, sar, xor, add, shr, lea. |
| | | 78 | | |
| | | 79 | | // 'a' will be -1 if input is < 0x800; else 'a' will be 0 |
| | | 80 | | // => 'a' will be -1 if input is 1 or 2 UTF-8 code units; else 'a' will be 0 |
| | | 81 | | |
| | 0 | 82 | | int a = ((int)value - 0x0800) >> 31; |
| | | 83 | | |
| | | 84 | | // The number of UTF-8 code units for a given scalar is as follows: |
| | | 85 | | // - U+0000..U+007F => 1 code unit |
| | | 86 | | // - U+0080..U+07FF => 2 code units |
| | | 87 | | // - U+0800..U+FFFF => 3 code units |
| | | 88 | | // - U+10000+ => 4 code units |
| | | 89 | | // |
| | | 90 | | // If we XOR the incoming scalar with 0xF800, the chart mutates: |
| | | 91 | | // - U+0000..U+F7FF => 3 code units |
| | | 92 | | // - U+F800..U+F87F => 1 code unit |
| | | 93 | | // - U+F880..U+FFFF => 2 code units |
| | | 94 | | // - U+10000+ => 4 code units |
| | | 95 | | // |
| | | 96 | | // Since the 1- and 3-code unit cases are now clustered, they can |
| | | 97 | | // both be checked together very cheaply. |
| | | 98 | | |
| | 0 | 99 | | value ^= 0xF800u; |
| | 0 | 100 | | value -= 0xF880u; // if scalar is 1 or 3 code units, high byte = 0xFF; else high byte = 0x00 |
| | 0 | 101 | | value += (4 << 24); // if scalar is 1 or 3 code units, high byte = 0x03; else high byte = 0x04 |
| | 0 | 102 | | value >>= 24; // shift high byte down |
| | | 103 | | |
| | | 104 | | // Final return value: |
| | | 105 | | // - U+0000..U+007F => 3 + (-1) * 2 = 1 |
| | | 106 | | // - U+0080..U+07FF => 4 + (-1) * 2 = 2 |
| | | 107 | | // - U+0800..U+FFFF => 3 + ( 0) * 2 = 3 |
| | | 108 | | // - U+10000+ => 4 + ( 0) * 2 = 4 |
| | 0 | 109 | | return (int)value + (a * 2); |
| | | 110 | | } |
| | | 111 | | |
| | | 112 | | /// <summary> |
| | | 113 | | /// Returns <see langword="true"/> iff <paramref name="value"/> is an ASCII |
| | | 114 | | /// character ([ U+0000..U+007F ]). |
| | | 115 | | /// </summary> |
| | | 116 | | /// <remarks> |
| | | 117 | | /// Per http://www.unicode.org/glossary/#ASCII, ASCII is only U+0000..U+007F. |
| | | 118 | | /// </remarks> |
| | | 119 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 2579395 | 120 | | public static bool IsAsciiCodePoint(uint value) => value <= 0x7Fu; |
| | | 121 | | |
| | | 122 | | /// <summary> |
| | | 123 | | /// Returns <see langword="true"/> iff <paramref name="value"/> is in the |
| | | 124 | | /// Basic Multilingual Plane (BMP). |
| | | 125 | | /// </summary> |
| | | 126 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 182260 | 127 | | public static bool IsBmpCodePoint(uint value) => value <= 0xFFFFu; |
| | | 128 | | |
| | | 129 | | /// <summary> |
| | | 130 | | /// Returns <see langword="true"/> iff <paramref name="value"/> is a UTF-16 high surrogate code point, |
| | | 131 | | /// i.e., is in [ U+D800..U+DBFF ], inclusive. |
| | | 132 | | /// </summary> |
| | | 133 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 0 | 134 | | public static bool IsHighSurrogateCodePoint(uint value) => IsInRangeInclusive(value, 0xD800U, 0xDBFFU); |
| | | 135 | | |
| | | 136 | | /// <summary> |
| | | 137 | | /// Returns <see langword="true"/> iff <paramref name="value"/> is between |
| | | 138 | | /// <paramref name="lowerBound"/> and <paramref name="upperBound"/>, inclusive. |
| | | 139 | | /// </summary> |
| | | 140 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 5361695 | 141 | | public static bool IsInRangeInclusive(uint value, uint lowerBound, uint upperBound) => (value - lowerBound) <= ( |
| | | 142 | | |
| | | 143 | | /// <summary> |
| | | 144 | | /// Returns <see langword="true"/> iff <paramref name="value"/> is a UTF-16 low surrogate code point, |
| | | 145 | | /// i.e., is in [ U+DC00..U+DFFF ], inclusive. |
| | | 146 | | /// </summary> |
| | | 147 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 0 | 148 | | public static bool IsLowSurrogateCodePoint(uint value) => IsInRangeInclusive(value, 0xDC00U, 0xDFFFU); |
| | | 149 | | |
| | | 150 | | /// <summary> |
| | | 151 | | /// Returns <see langword="true"/> iff <paramref name="value"/> is a UTF-16 surrogate code point, |
| | | 152 | | /// i.e., is in [ U+D800..U+DFFF ], inclusive. |
| | | 153 | | /// </summary> |
| | | 154 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 1866928 | 155 | | public static bool IsSurrogateCodePoint(uint value) => IsInRangeInclusive(value, 0xD800U, 0xDFFFU); |
| | | 156 | | |
| | | 157 | | /// <summary> |
| | | 158 | | /// Returns <see langword="true"/> iff <paramref name="codePoint"/> is a valid Unicode code |
| | | 159 | | /// point, i.e., is in [ U+0000..U+10FFFF ], inclusive. |
| | | 160 | | /// </summary> |
| | | 161 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | 0 | 162 | | public static bool IsValidCodePoint(uint codePoint) => codePoint <= 0x10FFFFU; |
| | | 163 | | |
| | | 164 | | /// <summary> |
| | | 165 | | /// Returns <see langword="true"/> iff <paramref name="value"/> is a valid Unicode scalar |
| | | 166 | | /// value, i.e., is in [ U+0000..U+D7FF ], inclusive; or [ U+E000..U+10FFFF ], inclusive. |
| | | 167 | | /// </summary> |
| | | 168 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 169 | | public static bool IsValidUnicodeScalar(uint value) |
| | | 170 | | { |
| | | 171 | | // This is an optimized check that on x86 is just three instructions: lea, xor, cmp. |
| | | 172 | | // |
| | | 173 | | // After the subtraction operation, the input value is modified as such: |
| | | 174 | | // [ 00000000..0010FFFF ] -> [ FFEF0000..FFFFFFFF ] |
| | | 175 | | // |
| | | 176 | | // We now want to _exclude_ the range [ FFEFD800..FFEFDFFF ] (surrogates) from being valid. |
| | | 177 | | // After the xor, this particular exclusion range becomes [ FFEF0000..FFEF07FF ]. |
| | | 178 | | // |
| | | 179 | | // So now the range [ FFEF0800..FFFFFFFF ] contains all valid code points, |
| | | 180 | | // excluding surrogates. This allows us to perform a single comparison. |
| | | 181 | | |
| | 5263437 | 182 | | return ((value - 0x110000u) ^ 0xD800u) >= 0xFFEF0800u; |
| | | 183 | | } |
| | | 184 | | } |
| | | 185 | | } |
| | | 186 | | |