< Summary

Line coverage
34%
Covered lines: 10
Uncovered lines: 19
Coverable lines: 29
Total lines: 186
Line coverage: 34.4%
Branch coverage
N/A
Covered branches: 0
Total branches: 0
Branch coverage: N/A
Method coverage

Feature is only available for sponsors

Upgrade to PRO version

Metrics

File(s)

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.CoreLib/src/System/Text/UnicodeUtility.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Runtime.CompilerServices;
 5
 6namespace System.Text
 7{
 8    internal static class UnicodeUtility
 9    {
 10        /// <summary>
 11        /// The Unicode replacement character U+FFFD.
 12        /// </summary>
 13        public const uint ReplacementChar = 0xFFFD;
 14
 15        /// <summary>
 16        /// Returns the Unicode plane (0 through 16, inclusive) which contains this code point.
 17        /// </summary>
 18        public static int GetPlane(uint codePoint)
 19        {
 020            UnicodeDebug.AssertIsValidCodePoint(codePoint);
 21
 022            return (int)(codePoint >> 16);
 23        }
 24
 25        /// <summary>
 26        /// Returns a Unicode scalar value from two code points representing a UTF-16 surrogate pair.
 27        /// </summary>
 28        public static uint GetScalarFromUtf16SurrogatePair(uint highSurrogateCodePoint, uint lowSurrogateCodePoint)
 29        {
 030            UnicodeDebug.AssertIsHighSurrogateCodePoint(highSurrogateCodePoint);
 031            UnicodeDebug.AssertIsLowSurrogateCodePoint(lowSurrogateCodePoint);
 32
 33            // This calculation comes from the Unicode specification, Table 3-5.
 34            // Need to remove the D800 marker from the high surrogate and the DC00 marker from the low surrogate,
 35            // then fix up the "wwww = uuuuu - 1" section of the bit distribution. The code is written as below
 36            // to become just two instructions: shl, lea.
 37
 038            return (highSurrogateCodePoint << 10) + lowSurrogateCodePoint - ((0xD800U << 10) + 0xDC00U - (1 << 16));
 39        }
 40
 41        /// <summary>
 42        /// Given a Unicode scalar value, gets the number of UTF-16 code units required to represent this value.
 43        /// </summary>
 44        public static int GetUtf16SequenceLength(uint value)
 45        {
 300363046            UnicodeDebug.AssertIsValidScalar(value);
 47
 300363048            value -= 0x10000;   // if value < 0x10000, high byte = 0xFF; else high byte = 0x00
 300363049            value += (2 << 24); // if value < 0x10000, high byte = 0x01; else high byte = 0x02
 300363050            value >>= 24;       // shift high byte down
 300363051            return (int)value;  // and return it
 52        }
 53
 54        /// <summary>
 55        /// Decomposes an astral Unicode scalar into UTF-16 high and low surrogate code units.
 56        /// </summary>
 57        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 58        public static void GetUtf16SurrogatesFromSupplementaryPlaneScalar(uint value, out char highSurrogateCodePoint, o
 59        {
 060            UnicodeDebug.AssertIsValidSupplementaryPlaneScalar(value);
 61
 62            // This calculation comes from the Unicode specification, Table 3-5.
 63
 064            highSurrogateCodePoint = (char)((value + ((0xD800u - 0x40u) << 10)) >> 10);
 065            lowSurrogateCodePoint = (char)((value & 0x3FFu) + 0xDC00u);
 066        }
 67
 68        /// <summary>
 69        /// Given a Unicode scalar value, gets the number of UTF-8 code units required to represent this value.
 70        /// </summary>
 71        public static int GetUtf8SequenceLength(uint value)
 72        {
 073            UnicodeDebug.AssertIsValidScalar(value);
 74
 75            // The logic below can handle all valid scalar values branchlessly.
 76            // It gives generally good performance across all inputs, and on x86
 77            // it's only six instructions: lea, sar, xor, add, shr, lea.
 78
 79            // 'a' will be -1 if input is < 0x800; else 'a' will be 0
 80            // => 'a' will be -1 if input is 1 or 2 UTF-8 code units; else 'a' will be 0
 81
 082            int a = ((int)value - 0x0800) >> 31;
 83
 84            // The number of UTF-8 code units for a given scalar is as follows:
 85            // - U+0000..U+007F => 1 code unit
 86            // - U+0080..U+07FF => 2 code units
 87            // - U+0800..U+FFFF => 3 code units
 88            // - U+10000+       => 4 code units
 89            //
 90            // If we XOR the incoming scalar with 0xF800, the chart mutates:
 91            // - U+0000..U+F7FF => 3 code units
 92            // - U+F800..U+F87F => 1 code unit
 93            // - U+F880..U+FFFF => 2 code units
 94            // - U+10000+       => 4 code units
 95            //
 96            // Since the 1- and 3-code unit cases are now clustered, they can
 97            // both be checked together very cheaply.
 98
 099            value ^= 0xF800u;
 0100            value -= 0xF880u;   // if scalar is 1 or 3 code units, high byte = 0xFF; else high byte = 0x00
 0101            value += (4 << 24); // if scalar is 1 or 3 code units, high byte = 0x03; else high byte = 0x04
 0102            value >>= 24;       // shift high byte down
 103
 104            // Final return value:
 105            // - U+0000..U+007F => 3 + (-1) * 2 = 1
 106            // - U+0080..U+07FF => 4 + (-1) * 2 = 2
 107            // - U+0800..U+FFFF => 3 + ( 0) * 2 = 3
 108            // - U+10000+       => 4 + ( 0) * 2 = 4
 0109            return (int)value + (a * 2);
 110        }
 111
 112        /// <summary>
 113        /// Returns <see langword="true"/> iff <paramref name="value"/> is an ASCII
 114        /// character ([ U+0000..U+007F ]).
 115        /// </summary>
 116        /// <remarks>
 117        /// Per http://www.unicode.org/glossary/#ASCII, ASCII is only U+0000..U+007F.
 118        /// </remarks>
 119        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 5341840120        public static bool IsAsciiCodePoint(uint value) => value <= 0x7Fu;
 121
 122        /// <summary>
 123        /// Returns <see langword="true"/> iff <paramref name="value"/> is in the
 124        /// Basic Multilingual Plane (BMP).
 125        /// </summary>
 126        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 1708595127        public static bool IsBmpCodePoint(uint value) => value <= 0xFFFFu;
 128
 129        /// <summary>
 130        /// Returns <see langword="true"/> iff <paramref name="value"/> is a UTF-16 high surrogate code point,
 131        /// i.e., is in [ U+D800..U+DBFF ], inclusive.
 132        /// </summary>
 133        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 0134        public static bool IsHighSurrogateCodePoint(uint value) => IsInRangeInclusive(value, 0xD800U, 0xDBFFU);
 135
 136        /// <summary>
 137        /// Returns <see langword="true"/> iff <paramref name="value"/> is between
 138        /// <paramref name="lowerBound"/> and <paramref name="upperBound"/>, inclusive.
 139        /// </summary>
 140        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 13387492141        public static bool IsInRangeInclusive(uint value, uint lowerBound, uint upperBound) => (value - lowerBound) <= (
 142
 143        /// <summary>
 144        /// Returns <see langword="true"/> iff <paramref name="value"/> is a UTF-16 low surrogate code point,
 145        /// i.e., is in [ U+DC00..U+DFFF ], inclusive.
 146        /// </summary>
 147        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 0148        public static bool IsLowSurrogateCodePoint(uint value) => IsInRangeInclusive(value, 0xDC00U, 0xDFFFU);
 149
 150        /// <summary>
 151        /// Returns <see langword="true"/> iff <paramref name="value"/> is a UTF-16 surrogate code point,
 152        /// i.e., is in [ U+D800..U+DFFF ], inclusive.
 153        /// </summary>
 154        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 6340810155        public static bool IsSurrogateCodePoint(uint value) => IsInRangeInclusive(value, 0xD800U, 0xDFFFU);
 156
 157        /// <summary>
 158        /// Returns <see langword="true"/> iff <paramref name="codePoint"/> is a valid Unicode code
 159        /// point, i.e., is in [ U+0000..U+10FFFF ], inclusive.
 160        /// </summary>
 161        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 0162        public static bool IsValidCodePoint(uint codePoint) => codePoint <= 0x10FFFFU;
 163
 164        /// <summary>
 165        /// Returns <see langword="true"/> iff <paramref name="value"/> is a valid Unicode scalar
 166        /// value, i.e., is in [ U+0000..U+D7FF ], inclusive; or [ U+E000..U+10FFFF ], inclusive.
 167        /// </summary>
 168        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 169        public static bool IsValidUnicodeScalar(uint value)
 170        {
 171            // This is an optimized check that on x86 is just three instructions: lea, xor, cmp.
 172            //
 173            // After the subtraction operation, the input value is modified as such:
 174            // [ 00000000..0010FFFF ] -> [ FFEF0000..FFFFFFFF ]
 175            //
 176            // We now want to _exclude_ the range [ FFEFD800..FFEFDFFF ] (surrogates) from being valid.
 177            // After the xor, this particular exclusion range becomes [ FFEF0000..FFEF07FF ].
 178            //
 179            // So now the range [ FFEF0800..FFFFFFFF ] contains all valid code points,
 180            // excluding surrogates. This allows us to perform a single comparison.
 181
 14727125182            return ((value - 0x110000u) ^ 0xD800u) >= 0xFFEF0800u;
 183        }
 184    }
 185}
 186