< Summary

Line coverage
20%
Covered lines: 25
Uncovered lines: 95
Coverable lines: 120
Total lines: 625
Line coverage: 20.8%
Branch coverage
7%
Covered branches: 3
Total branches: 40
Branch coverage: 7.5%
Method coverage

Feature is only available for sponsors

Upgrade to PRO version

Metrics

File(s)

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.CoreLib/src/System/Text/Unicode/Utf16Utility.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Diagnostics;
 5using System.Runtime.CompilerServices;
 6using System.Runtime.InteropServices;
 7
 8#if NET
 9using System.Runtime.Intrinsics;
 10#endif
 11
 12namespace System.Text.Unicode
 13{
 14    internal static partial class Utf16Utility
 15    {
 16        /// <summary>
 17        /// Returns true iff the UInt32 represents two ASCII UTF-16 characters in machine endianness.
 18        /// </summary>
 19        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 20        internal static bool AllCharsInUInt32AreAscii(uint value)
 21        {
 822            return (value & ~0x007F_007Fu) == 0;
 23        }
 24
 25        /// <summary>
 26        /// Returns true iff the UInt64 represents four ASCII UTF-16 characters in machine endianness.
 27        /// </summary>
 28        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 29        internal static bool AllCharsInUInt64AreAscii(ulong value)
 30        {
 431            return (value & ~0x007F_007F_007F_007Ful) == 0;
 32        }
 33
 34        /// <summary>
 35        /// Given a UInt32 that represents two ASCII UTF-16 characters, returns the invariant
 36        /// lowercase representation of those characters. Requires the input value to contain
 37        /// two ASCII UTF-16 characters in machine endianness.
 38        /// </summary>
 39        /// <remarks>
 40        /// This is a branchless implementation.
 41        /// </remarks>
 42        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 43        internal static uint ConvertAllAsciiCharsInUInt32ToLowercase(uint value)
 44        {
 45            // ASSUMPTION: Caller has validated that input value is ASCII.
 046            Debug.Assert(AllCharsInUInt32AreAscii(value));
 47
 48            // the 0x80 bit of each word of 'lowerIndicator' will be set iff the word has value >= 'A'
 049            uint lowerIndicator = value + 0x0080_0080u - 0x0041_0041u;
 50
 51            // the 0x80 bit of each word of 'upperIndicator' will be set iff the word has value > 'Z'
 052            uint upperIndicator = value + 0x0080_0080u - 0x005B_005Bu;
 53
 54            // the 0x80 bit of each word of 'combinedIndicator' will be set iff the word has value >= 'A' and <= 'Z'
 055            uint combinedIndicator = (lowerIndicator ^ upperIndicator);
 56
 57            // the 0x20 bit of each word of 'mask' will be set iff the word has value >= 'A' and <= 'Z'
 058            uint mask = (combinedIndicator & 0x0080_0080u) >> 2;
 59
 060            return value ^ mask; // bit flip uppercase letters [A-Z] => [a-z]
 61        }
 62
 63        /// <summary>
 64        /// Given a UInt32 that represents two ASCII UTF-16 characters, returns the invariant
 65        /// uppercase representation of those characters. Requires the input value to contain
 66        /// two ASCII UTF-16 characters in machine endianness.
 67        /// </summary>
 68        /// <remarks>
 69        /// This is a branchless implementation.
 70        /// </remarks>
 71        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 72        internal static uint ConvertAllAsciiCharsInUInt32ToUppercase(uint value)
 73        {
 74            // Intrinsified in mono interpreter
 75            // ASSUMPTION: Caller has validated that input value is ASCII.
 076            Debug.Assert(AllCharsInUInt32AreAscii(value));
 77
 78            // the 0x80 bit of each word of 'lowerIndicator' will be set iff the word has value >= 'a'
 079            uint lowerIndicator = value + 0x0080_0080u - 0x0061_0061u;
 80
 81            // the 0x80 bit of each word of 'upperIndicator' will be set iff the word has value > 'z'
 082            uint upperIndicator = value + 0x0080_0080u - 0x007B_007Bu;
 83
 84            // the 0x80 bit of each word of 'combinedIndicator' will be set iff the word has value >= 'a' and <= 'z'
 085            uint combinedIndicator = (lowerIndicator ^ upperIndicator);
 86
 87            // the 0x20 bit of each word of 'mask' will be set iff the word has value >= 'a' and <= 'z'
 088            uint mask = (combinedIndicator & 0x0080_0080u) >> 2;
 89
 090            return value ^ mask; // bit flip lowercase letters [a-z] => [A-Z]
 91        }
 92
 93        /// <summary>
 94        /// Given a UInt64 that represents four ASCII UTF-16 characters, returns the invariant
 95        /// uppercase representation of those characters. Requires the input value to contain
 96        /// four ASCII UTF-16 characters in machine endianness.
 97        /// </summary>
 98        /// <remarks>
 99        /// This is a branchless implementation.
 100        /// </remarks>
 101        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 102        internal static ulong ConvertAllAsciiCharsInUInt64ToUppercase(ulong value)
 103        {
 104            // ASSUMPTION: Caller has validated that input value is ASCII.
 0105            Debug.Assert(AllCharsInUInt64AreAscii(value));
 106
 107            // the 0x80 bit of each word of 'lowerIndicator' will be set iff the word has value >= 'a'
 0108            ulong lowerIndicator = value + 0x0080_0080_0080_0080ul - 0x0061_0061_0061_0061ul;
 109
 110            // the 0x80 bit of each word of 'upperIndicator' will be set iff the word has value > 'z'
 0111            ulong upperIndicator = value + 0x0080_0080_0080_0080ul - 0x007B_007B_007B_007Bul;
 112
 113            // the 0x80 bit of each word of 'combinedIndicator' will be set iff the word has value >= 'a' and <= 'z'
 0114            ulong combinedIndicator = (lowerIndicator ^ upperIndicator);
 115
 116            // the 0x20 bit of each word of 'mask' will be set iff the word has value >= 'a' and <= 'z'
 0117            ulong mask = (combinedIndicator & 0x0080_0080_0080_0080ul) >> 2;
 118
 0119            return value ^ mask; // bit flip lowercase letters [a-z] => [A-Z]
 120        }
 121
 122        /// <summary>
 123        /// Given a UInt64 that represents four ASCII UTF-16 characters, returns the invariant
 124        /// lowercase representation of those characters. Requires the input value to contain
 125        /// four ASCII UTF-16 characters in machine endianness.
 126        /// </summary>
 127        /// <remarks>
 128        /// This is a branchless implementation.
 129        /// </remarks>
 130        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 131        internal static ulong ConvertAllAsciiCharsInUInt64ToLowercase(ulong value)
 132        {
 133            // ASSUMPTION: Caller has validated that input value is ASCII.
 0134            Debug.Assert(AllCharsInUInt64AreAscii(value));
 135
 136            // the 0x80 bit of each word of 'lowerIndicator' will be set iff the word has value >= 'A'
 0137            ulong lowerIndicator = value + 0x0080_0080_0080_0080ul - 0x0041_0041_0041_0041ul;
 138
 139            // the 0x80 bit of each word of 'upperIndicator' will be set iff the word has value > 'Z'
 0140            ulong upperIndicator = value + 0x0080_0080_0080_0080ul - 0x005B_005B_005B_005Bul;
 141
 142            // the 0x80 bit of each word of 'combinedIndicator' will be set iff the word has value >= 'a' and <= 'z'
 0143            ulong combinedIndicator = (lowerIndicator ^ upperIndicator);
 144
 145            // the 0x20 bit of each word of 'mask' will be set iff the word has value >= 'a' and <= 'z'
 0146            ulong mask = (combinedIndicator & 0x0080_0080_0080_0080ul) >> 2;
 147
 0148            return value ^ mask; // bit flip uppercase letters [A-Z] => [a-z]
 149        }
 150
 151        /// <summary>
 152        /// Given a UInt32 that represents two ASCII UTF-16 characters, returns true iff
 153        /// the input contains one or more lowercase ASCII characters.
 154        /// </summary>
 155        /// <remarks>
 156        /// This is a branchless implementation.
 157        /// </remarks>
 158        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 159        internal static bool UInt32ContainsAnyLowercaseAsciiChar(uint value)
 160        {
 161            // ASSUMPTION: Caller has validated that input value is ASCII.
 0162            Debug.Assert(AllCharsInUInt32AreAscii(value));
 163
 164            // the 0x80 bit of each word of 'lowerIndicator' will be set iff the word has value >= 'a'
 0165            uint lowerIndicator = value + 0x0080_0080u - 0x0061_0061u;
 166
 167            // the 0x80 bit of each word of 'upperIndicator' will be set iff the word has value > 'z'
 0168            uint upperIndicator = value + 0x0080_0080u - 0x007B_007Bu;
 169
 170            // the 0x80 bit of each word of 'combinedIndicator' will be set iff the word has value >= 'a' and <= 'z'
 0171            uint combinedIndicator = (lowerIndicator ^ upperIndicator);
 172
 0173            return (combinedIndicator & 0x0080_0080u) != 0;
 174        }
 175
 176        /// <summary>
 177        /// Given a UInt32 that represents two ASCII UTF-16 characters, returns true iff
 178        /// the input contains one or more uppercase ASCII characters.
 179        /// </summary>
 180        /// <remarks>
 181        /// This is a branchless implementation.
 182        /// </remarks>
 183        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 184        internal static bool UInt32ContainsAnyUppercaseAsciiChar(uint value)
 185        {
 186            // ASSUMPTION: Caller has validated that input value is ASCII.
 2187            Debug.Assert(AllCharsInUInt32AreAscii(value));
 188
 189            // the 0x80 bit of each word of 'lowerIndicator' will be set iff the word has value >= 'A'
 2190            uint lowerIndicator = value + 0x0080_0080u - 0x0041_0041u;
 191
 192            // the 0x80 bit of each word of 'upperIndicator' will be set iff the word has value > 'Z'
 2193            uint upperIndicator = value + 0x0080_0080u - 0x005B_005Bu;
 194
 195            // the 0x80 bit of each word of 'combinedIndicator' will be set iff the word has value >= 'A' and <= 'Z'
 2196            uint combinedIndicator = (lowerIndicator ^ upperIndicator);
 197
 2198            return (combinedIndicator & 0x0080_0080u) != 0;
 199        }
 200
 201        /// <summary>
 202        /// Given two UInt32s that represent two ASCII UTF-16 characters each, returns true iff
 203        /// the two inputs are equal using an ordinal case-insensitive comparison.
 204        /// </summary>
 205        /// <remarks>
 206        /// This is a branchless implementation.
 207        /// </remarks>
 208        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 209        internal static bool UInt32OrdinalIgnoreCaseAscii(uint valueA, uint valueB)
 210        {
 211            // Intrinsified in mono interpreter
 212            // ASSUMPTION: Caller has validated that input values are ASCII.
 0213            Debug.Assert(AllCharsInUInt32AreAscii(valueA));
 0214            Debug.Assert(AllCharsInUInt32AreAscii(valueB));
 215
 216            // Generate a mask of all bits which are different between A and B. Since [A-Z]
 217            // and [a-z] differ by the 0x20 bit, we'll left-shift this by 2 now so that
 218            // this is moved over to the 0x80 bit, which nicely aligns with the calculation
 219            // we're going to do on the indicator flag later.
 220            //
 221            // n.b. All of the logic below assumes we have at least 2 "known zero" bits leading
 222            // each of the 7-bit ASCII values. This assumption won't hold if this method is
 223            // ever adapted to deal with packed bytes instead of packed chars.
 224
 0225            uint differentBits = (valueA ^ valueB) << 2;
 226
 227            // Now, we want to generate a mask where for each word in the input, the mask contains
 228            // 0xFF7F if the word is [A-Za-z], 0xFFFF if the word is not [A-Za-z]. We know each
 229            // input word is ASCII (only low 7 bit set), so we can use a combination of addition
 230            // and logical operators as follows.
 231            //
 232            // original input   +05         |A0         +1A
 233            // ====================================================
 234            //         00 .. 3F -> 05 .. 44 -> A5 .. E4 -> BF .. FE
 235            //               40 ->       45 ->       E5 ->       FF
 236            // ([A-Z]) 41 .. 5A -> 46 .. 5F -> E6 .. FF -> 00 .. 19
 237            //         5B .. 5F -> 60 .. 64 -> E0 .. E4 -> FA .. FE
 238            //               60 ->       65 ->       E5 ->       FF
 239            // ([a-z]) 61 .. 7A -> 66 .. 7F -> E6 .. FF -> 00 .. 19
 240            //         7B .. 7F -> 80 .. 84 -> A0 .. A4 -> BA .. BE
 241            //
 242            // This combination of operations results in the 0x80 bit of each word being set
 243            // iff the original word value was *not* [A-Za-z].
 244
 0245            uint indicator = valueA + 0x0005_0005u;
 0246            indicator |= 0x00A0_00A0u;
 0247            indicator += 0x001A_001Au;
 0248            indicator |= 0xFF7F_FF7Fu; // normalize each word to 0xFF7F or 0xFFFF
 249
 250            // At this point, 'indicator' contains the mask of bits which are *not* allowed to
 251            // differ between the inputs, and 'differentBits' contains the mask of bits which
 252            // actually differ between the inputs. If these masks have any bits in common, then
 253            // the two values are *not* equal under an OrdinalIgnoreCase comparer.
 254
 0255            return (differentBits & indicator) == 0;
 256        }
 257
 258        /// <summary>
 259        /// Given two UInt64s that represent four ASCII UTF-16 characters each, returns true iff
 260        /// the two inputs are equal using an ordinal case-insensitive comparison.
 261        /// </summary>
 262        /// <remarks>
 263        /// This is a branchless implementation.
 264        /// </remarks>
 265        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 266        internal static bool UInt64OrdinalIgnoreCaseAscii(ulong valueA, ulong valueB)
 267        {
 268            // Intrinsified in mono interpreter
 269            // ASSUMPTION: Caller has validated that input values are ASCII.
 2270            Debug.Assert(AllCharsInUInt64AreAscii(valueA));
 2271            Debug.Assert(AllCharsInUInt64AreAscii(valueB));
 272
 273            // Duplicate of logic in UInt32OrdinalIgnoreCaseAscii, but using 64-bit consts.
 274            // See comments in that method for more info.
 275
 2276            ulong differentBits = (valueA ^ valueB) << 2;
 2277            ulong indicator = valueA + 0x0005_0005_0005_0005ul;
 2278            indicator |= 0x00A0_00A0_00A0_00A0ul;
 2279            indicator += 0x001A_001A_001A_001Aul;
 2280            indicator |= 0xFF7F_FF7F_FF7F_FF7Ful;
 2281            return (differentBits & indicator) == 0;
 282        }
 283
 284#if SYSTEM_PRIVATE_CORELIB
 285        /// <summary>
 286        /// Returns true iff the TVector represents ASCII UTF-16 characters in machine endianness.
 287        /// </summary>
 288        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 289        internal static bool AllCharsInVectorAreAscii<TVector>(TVector vec)
 290            where TVector : struct, ISimdVector<TVector, ushort>
 291        {
 2292            return (vec & TVector.Create(unchecked((ushort)~0x007F))) == TVector.Zero;
 293        }
 294#endif
 295
 296#if NET
 297        /// <summary>
 298        /// Returns the char index in <paramref name="utf16Data"/> where the first invalid UTF-16 sequence begins,
 299        /// or -1 if the buffer contains no invalid sequences.
 300        /// </summary>
 301        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 302        public static unsafe int GetIndexOfFirstInvalidUtf16Sequence(ReadOnlySpan<char> utf16Data)
 0303        {
 0304            fixed (char* pValue = &MemoryMarshal.GetReference(utf16Data))
 305            {
 0306                char* pFirstInvalidChar = GetPointerToFirstInvalidChar(pValue, utf16Data.Length, out _, out _);
 0307                int index = (int)(pFirstInvalidChar - pValue);
 308
 0309                return (index < utf16Data.Length) ? index : -1;
 310            }
 311        }
 312#endif
 313    }
 314}
 315

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.CoreLib/src/System/Text/Unicode/Utf16Utility.Validation.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Diagnostics;
 5using System.Diagnostics.CodeAnalysis;
 6using System.Numerics;
 7using System.Runtime.CompilerServices;
 8using System.Runtime.Intrinsics;
 9using System.Runtime.Intrinsics.Arm;
 10
 11namespace System.Text.Unicode
 12{
 13    internal static unsafe partial class Utf16Utility
 14    {
 15
 16        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 17        private static nuint GetSurrogateMask(Vector128<ushort> cmp)
 18        {
 19            // Convert the comparison result to a scalar surrogate mask.
 20            // The elements in 'cmp' should be either all bits set or zero.
 21
 22            if (AdvSimd.Arm64.IsSupported)
 23            {
 24                // Since ExtractMostSignificantBits is very slow on AdvSimd,
 25                // we use a 64-bit value to encode the mask, where each byte represents one element:
 26                //   0x01 for all bits set, 0x00 for zero.
 27                ulong mask = AdvSimd.Arm64.UnzipOdd(cmp.AsByte(), cmp.AsByte()).AsUInt64().ToScalar();
 28                return (nuint)(mask & 0x0101010101010101u);
 29            }
 30
 31            // Otherwise, encode the mask with 8-bits (one byte), where each bit represents one element.
 032            return cmp.ExtractMostSignificantBits();
 33        }
 34
 35        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 36        private static bool IsSurrogatesMatch(nuint maskHigh, nuint maskLow)
 37        {
 38            // Make sure that each high surrogate is followed by a low surrogate character,
 39            // and each low surrogate follows a high surrogate character.
 40            // The last character is discarded as it will be checked by 'IsLastCharHighSurrogate'.
 41            // The first character must not be a low surrogate. This is checked by matching
 42            // 'maskLow' aganist the zeros inserted after shifting 'maskHigh' to the left.
 43
 44            if (AdvSimd.Arm64.IsSupported)
 45            {
 46                // Each surrogate character is 8 bits apart.
 47                return (maskHigh << 8) == maskLow;
 48            }
 49            // Each surrogate character is 1 bit apart.
 050            return (byte)(maskHigh << 1) == (byte)maskLow;
 51        }
 52
 53        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 54        private static bool IsLastCharHighSurrogate(nuint maskHigh)
 55        {
 56            if (AdvSimd.Arm64.IsSupported)
 57            {
 58                // Check if the top byte is not zero.
 59                return (maskHigh >>> 56) != 0;
 60            }
 61            // Check if the top bit (of a byte) is not zero.
 062            return ((byte)maskHigh >>> 7) != 0;
 63        }
 64
 65        // Returns &inputBuffer[inputLength] if the input buffer is valid.
 66        /// <summary>
 67        /// Given an input buffer <paramref name="pInputBuffer"/> of char length <paramref name="inputLength"/>,
 68        /// returns a pointer to where the first invalid data appears in <paramref name="pInputBuffer"/>.
 69        /// </summary>
 70        /// <remarks>
 71        /// Returns a pointer to the end of <paramref name="pInputBuffer"/> if the buffer is well-formed.
 72        /// </remarks>
 73        public static char* GetPointerToFirstInvalidChar(char* pInputBuffer, int inputLength, out long utf8CodeUnitCount
 74        {
 75            Debug.Assert(inputLength >= 0, "Input length must not be negative.");
 1676            Debug.Assert(pInputBuffer != null || inputLength == 0, "Input length must be zero if input buffer pointer is
 77
 78            // First, we'll handle the common case of all-ASCII. If this is able to
 79            // consume the entire buffer, we'll skip the remainder of this method's logic.
 80
 1681            int numAsciiCharsConsumedJustNow = (int)Ascii.GetIndexOfFirstNonAsciiChar(pInputBuffer, (uint)inputLength);
 1682            Debug.Assert(0 <= numAsciiCharsConsumedJustNow && numAsciiCharsConsumedJustNow <= inputLength);
 83
 1684            pInputBuffer += (uint)numAsciiCharsConsumedJustNow;
 1685            inputLength -= numAsciiCharsConsumedJustNow;
 86
 1687            if (inputLength == 0)
 88            {
 1689                utf8CodeUnitCountAdjustment = 0;
 1690                scalarCountAdjustment = 0;
 1691                return pInputBuffer;
 92            }
 93
 94            // If we got here, it means we saw some non-ASCII data, so within our
 95            // vectorized code paths below we'll handle all non-surrogate UTF-16
 96            // code points branchlessly. We'll only branch if we see surrogates.
 97            //
 98            // We still optimistically assume the data is mostly ASCII. This means that the
 99            // number of UTF-8 code units and the number of scalars almost matches the number
 100            // of UTF-16 code units. As we go through the input and find non-ASCII
 101            // characters, we'll keep track of these "adjustment" fixups. To get the
 102            // total number of UTF-8 code units required to encode the input data, add
 103            // the UTF-8 code unit count adjustment to the number of UTF-16 code units
 104            // seen.  To get the total number of scalars present in the input data,
 105            // add the scalar count adjustment to the number of UTF-16 code units seen.
 106
 0107            long tempUtf8CodeUnitCountAdjustment = 0;
 0108            int tempScalarCountAdjustment = 0;
 0109            char* pEndOfInputBuffer = pInputBuffer + (uint)inputLength;
 110
 0111            if (Vector128.IsHardwareAccelerated)
 112            {
 0113                if (inputLength >= Vector128<ushort>.Count)
 114                {
 0115                    Vector128<ushort> vector0080 = Vector128.Create<ushort>(0x0080);
 0116                    Vector128<ushort> vector0400 = Vector128.Create<ushort>(0x0400);
 0117                    Vector128<ushort> vector0800 = Vector128.Create<ushort>(0x0800);
 0118                    Vector128<ushort> vectorD800 = Vector128.Create<ushort>(0xD800);
 119
 0120                    char* pHighestAddressWhereCanReadOneVector = pEndOfInputBuffer - Vector128<ushort>.Count;
 0121                    Debug.Assert(pHighestAddressWhereCanReadOneVector >= pInputBuffer);
 122
 123                    do
 124                    {
 0125                        Vector128<ushort> utf16Data = Vector128.Load((ushort*)pInputBuffer);
 126
 127                        // Calculate the popcnt for UTF-8 adjustments, which is the number of *additional*
 128                        // UTF-8 bytes that each UTF-16 code unit requires as it expands.
 129                        // This results in the wrong count for UTF-16 surrogate code units (we just counted
 130                        // that each individual code unit expands to 3 bytes, but in reality a well-formed
 131                        // UTF-16 surrogate pair expands to 4 bytes). We'll handle this in just a moment.
 132                        //
 133                        // For now, compute the popcnt but squirrel it away. We'll fold it in to the
 134                        // cumulative UTF-8 adjustment factor once we determine that there are no
 135                        // unpaired surrogates in our data. (Unpaired surrogates would invalidate
 136                        // our computed result and we'd have to throw it away.)
 137
 138                        uint popcnt;
 139
 140                        // On AdvSimd ExtractMostSignificantBits is very slow, so a different algorithm is used to avoid
 141                        // the poor performance.
 142
 143                        if (AdvSimd.Arm64.IsSupported)
 144                        {
 145                            // The 'twoOrMoreUtf8Bytes' and 'threeOrMoreUtf8Bytes' vectors will contain
 146                            // elements whose values are 0xFFFF (-1 as signed word) iff the corresponding
 147                            // UTF-16 code unit was >= 0x0080 and >= 0x0800, respectively. By summing these
 148                            // vectors, each element of the sum will contain one of three values:
 149                            //
 150                            // 0x0000 ( 0) = original char was 0000..007F
 151                            // 0xFFFF (-1) = original char was 0080..07FF
 152                            // 0xFFFE (-2) = original char was 0800..FFFF
 153                            //
 154                            // We'll negate them to produce a value 0..2 for each element, then sum all the
 155                            // elements together to produce the number of *additional* UTF-8 code units
 156                            // required to represent this UTF-16 data.
 157
 158                            Vector128<ushort> twoOrMoreUtf8Bytes = Vector128.GreaterThanOrEqual(utf16Data, vector0080);
 159                            Vector128<ushort> threeOrMoreUtf8Bytes = Vector128.GreaterThanOrEqual(utf16Data, vector0800)
 160                            Vector128<ushort> sumVector = Vector128<ushort>.Zero - twoOrMoreUtf8Bytes - threeOrMoreUtf8B
 161                            popcnt = Vector128.Sum(sumVector);
 162                        }
 163                        else
 164                        {
 0165                            Vector128<ushort> vector7800 = Vector128.Create<ushort>(0x7800);
 166
 167                            // Sets the 0x0080 bit of each element in 'charIsNonAscii' if the corresponding
 168                            // input was 0x0080 <= [value]. (i.e., [value] is non-ASCII.)
 169
 0170                            Vector128<ushort> charIsNonAscii = Vector128.Min(utf16Data, vector0080);
 171
 172#if DEBUG
 173                            // Quick check to ensure we didn't accidentally set the 0x8000 bit of any element.
 0174                            uint debugMask = charIsNonAscii.AsByte().ExtractMostSignificantBits();
 0175                            Debug.Assert((debugMask & 0b_1010_1010_1010_1010) == 0, "Shouldn't have set the 0x8000 bit o
 176#endif // DEBUG
 177
 178                            // Sets the 0x8080 bits of each element in 'charIsNonAscii' if the corresponding
 179                            // input was 0x0800 <= [value]. This also handles the missing range a few lines above.
 180                            // Since 3-byte elements have a value >= 0x0800, we'll perform a saturating add of 0x7800 in
 181                            // get all 3-byte elements to have their 0x8000 bits set. A saturating add will not set the 
 182                            // bit for 1-byte or 2-byte elements. The 0x0080 bit will already have been set for non-ASCI
 183                            // and 3-byte) elements.
 184
 0185                            Vector128<ushort> charIsThreeByteUtf8Encoded = Vector128.AddSaturate(utf16Data, vector7800);
 186
 187                            // Each even bit of mask will be 1 only if the char was >= 0x0080,
 188                            // and each odd bit of mask will be 1 only if the char was >= 0x0800.
 189                            //
 190                            // Example for UTF-16 input "[ 0123 ] [ 1234 ] ...":
 191                            //
 192                            //            ,-- set if char[1] is >= 0x0800
 193                            //            |   ,-- set if char[0] is >= 0x0800
 194                            //            v   v
 195                            // mask = ... 1 1 0 1
 196                            //              ^   ^-- set if char[0] is non-ASCII
 197                            //              `-- set if char[1] is non-ASCII
 198
 0199                            uint mask = (charIsNonAscii | charIsThreeByteUtf8Encoded).AsByte().ExtractMostSignificantBit
 0200                            popcnt = (uint)BitOperations.PopCount(mask); // on x64, perform zero-extension for free
 201                        }
 202
 203                        // Now check for surrogates.
 204
 0205                        utf16Data -= vectorD800;
 0206                        nuint maskSurr = GetSurrogateMask(Vector128.LessThan(utf16Data, vector0800));
 0207                        if (maskSurr != 0)
 208                        {
 209                            // Get the surrogate masks for high and low surrogates.
 210                            // A high surrogate will be less than 0x0400 after subtracting by 0xD800.
 211                            // A low surrogate is a surrogate that is not a high surrogate.
 212
 0213                            nuint maskHigh = GetSurrogateMask(Vector128.LessThan(utf16Data, vector0400));
 0214                            nuint maskLow  = ~maskHigh & maskSurr;
 215
 0216                            if (!IsSurrogatesMatch(maskHigh, maskLow))
 217                            {
 218                                break; // error: mismatched surrogate pair; break out of vectorized logic
 219                            }
 220
 0221                            if (IsLastCharHighSurrogate(maskHigh))
 222                            {
 223                                // There was a standalone high surrogate at the end of the vector.
 224                                // We'll adjust our counters so that we don't consider this char consumed.
 225
 0226                                pInputBuffer--;
 0227                                popcnt -= 2;
 228                            }
 229
 230                            // If all the surrogate pairs are valid, then the number of surrogate pairs
 231                            // is equal to the number of low surrogates.
 232
 0233                            nint surrogatePairsCountNint = (nint)BitOperations.PopCount(maskLow);
 234
 235                            // 2 UTF-16 chars become 1 Unicode scalar
 236
 0237                            tempScalarCountAdjustment -= (int)surrogatePairsCountNint;
 238
 239                            // Since each surrogate code unit was >= 0x0800, we eagerly assumed
 240                            // it'd be encoded as 3 UTF-8 code units. Each surrogate half is only
 241                            // encoded as 2 UTF-8 code units (for 4 UTF-8 code units total),
 242                            // so we'll adjust this now.
 243
 0244                            tempUtf8CodeUnitCountAdjustment -= surrogatePairsCountNint;
 0245                            tempUtf8CodeUnitCountAdjustment -= surrogatePairsCountNint;
 246                        }
 247
 0248                        tempUtf8CodeUnitCountAdjustment += popcnt;
 0249                        pInputBuffer += Vector128<ushort>.Count;
 0250                    } while (pInputBuffer <= pHighestAddressWhereCanReadOneVector);
 251                }
 252            }
 253
 254            // Vectorization isn't supported on our current platform, or the input was too small to benefit
 255            // from vectorization, or we saw invalid UTF-16 data in the vectorized code paths and need to
 256            // drain remaining valid chars before we report failure.
 257
 0258            for (; pInputBuffer < pEndOfInputBuffer; pInputBuffer++)
 259            {
 0260                uint thisChar = pInputBuffer[0];
 0261                if (thisChar <= 0x7F)
 262                {
 263                    continue;
 264                }
 265
 266                // Bump adjustment by +1 for U+0080..U+07FF; by +2 for U+0800..U+FFFF.
 267                // This optimistically assumes no surrogates, which we'll handle shortly.
 268
 0269                tempUtf8CodeUnitCountAdjustment += (thisChar + 0x0001_F800u) >> 16;
 270
 0271                if (!UnicodeUtility.IsSurrogateCodePoint(thisChar))
 272                {
 273                    continue;
 274                }
 275
 276                // Found a surrogate char. Back out the adjustment we made above, then
 277                // try to consume the entire surrogate pair all at once. We won't bother
 278                // trying to interpret the surrogate pair as a scalar value; we'll only
 279                // validate that its bit pattern matches what's expected for a surrogate pair.
 280
 0281                tempUtf8CodeUnitCountAdjustment -= 2;
 282
 0283                if ((nuint)pEndOfInputBuffer - (nuint)pInputBuffer < sizeof(uint))
 284                {
 285                    goto Error; // input buffer too small to read a surrogate pair
 286                }
 287
 0288                thisChar = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 0289                if (((thisChar - (BitConverter.IsLittleEndian ? 0xDC00_D800u : 0xD800_DC00u)) & 0xFC00_FC00u) != 0)
 290                {
 291                    goto Error; // not a well-formed surrogate pair
 292                }
 293
 0294                tempScalarCountAdjustment--; // 2 UTF-16 code units -> 1 scalar
 0295                tempUtf8CodeUnitCountAdjustment += 2; // 2 UTF-16 code units -> 4 UTF-8 code units
 296
 0297                pInputBuffer++; // consumed one extra char
 298            }
 299
 300        Error:
 301
 302            // Also used for normal return.
 303
 0304            utf8CodeUnitCountAdjustment = tempUtf8CodeUnitCountAdjustment;
 0305            scalarCountAdjustment = tempScalarCountAdjustment;
 0306            return pInputBuffer;
 307        }
 308    }
 309}
 310