< Summary

Line coverage
59%
Covered lines: 411
Uncovered lines: 281
Coverable lines: 692
Total lines: 3321
Line coverage: 59.3%
Branch coverage
61%
Covered branches: 277
Total branches: 450
Branch coverage: 61.5%
Method coverage

Feature is only available for sponsors

Upgrade to PRO version

Metrics

MethodBranch coverage Cyclomatic complexity NPath complexity Sequence coverage
File 1: GetIndexOfFirstInvalidUtf8Sequence(...)0%220%
File 1: AllBytesInUInt32AreAscii(...)100%110%
File 1: AllBytesInUInt64AreAscii(...)100%110%
File 1: ConvertAllAsciiBytesInUInt32ToLowercase(...)100%110%
File 1: ConvertAllAsciiBytesInUInt32ToUppercase(...)100%110%
File 1: ConvertAllAsciiBytesInUInt64ToUppercase(...)100%110%
File 1: ConvertAllAsciiBytesInUInt64ToLowercase(...)100%110%
File 1: UInt64OrdinalIgnoreCaseAscii(...)100%110%
File 1: AllBytesInVector128AreAscii(...)100%110%
File 2: ExtractCharFromFirstThreeByteSequence(...)100%11100%
File 2: ExtractCharFromFirstTwoByteSequence(...)50%22100%
File 2: ExtractCharsFromFourByteSequence(...)100%11100%
File 2: ExtractFourUtf8BytesFromSurrogatePair(...)100%110%
File 2: ExtractTwoCharsPackedFromTwoAdjacentTwoByteSequences(...)100%11100%
File 2: ExtractTwoUtf8TwoByteSequencesFromTwoPackedUtf16Chars(...)0%220%
File 2: ExtractUtf8TwoByteSequenceFromFirstUtf16Char(...)100%110%
File 2: IsLowByteUtf8ContinuationByte(...)100%11100%
File 2: IsUtf8ContinuationByte(...)100%11100%
File 2: ToLittleEndian(...)100%11100%
File 2: UInt32BeginsWithOverlongUtf8TwoByteSequence(...)100%44100%
File 2: UInt32BeginsWithValidUtf8TwoByteSequenceLittleEndian(...)100%44100%
File 2: UInt32EndsWithValidUtf8TwoByteSequenceLittleEndian(...)100%44100%
File 2: WriteTwoUtf16CharsAsTwoUtf8ThreeByteSequences(...)0%440%
File 2: WriteFirstUtf16CharAsUtf8ThreeByteSequence(...)0%220%
File 3: TranscodeToUtf16(...)86.36%15415495.35%
File 3: TranscodeToUtf8(...)2.23%1341344.8%
File 4: GetPointerToFirstInvalidByte(...)92.75%13813895.48%

File(s)

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.CoreLib/src/System/Text/Unicode/Utf8Utility.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Diagnostics;
 5using System.Runtime.CompilerServices;
 6using System.Runtime.InteropServices;
 7#if NET
 8using System.Runtime.Intrinsics;
 9#endif
 10
 11namespace System.Text.Unicode
 12{
 13    internal static partial class Utf8Utility
 14    {
 15        /// <summary>
 16        /// The maximum number of bytes that can result from UTF-8 transcoding
 17        /// any Unicode scalar value.
 18        /// </summary>
 19        internal const int MaxBytesPerScalar = 4;
 20
 21        /// <summary>
 22        /// Returns the byte index in <paramref name="utf8Data"/> where the first invalid UTF-8 sequence begins,
 23        /// or -1 if the buffer contains no invalid sequences. Also outs the <paramref name="isAscii"/> parameter
 24        /// stating whether all data observed (up to the first invalid sequence or the end of the buffer, whichever
 25        /// comes first) is ASCII.
 26        /// </summary>
 27        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 28        public static unsafe int GetIndexOfFirstInvalidUtf8Sequence(ReadOnlySpan<byte> utf8Data, out bool isAscii)
 029        {
 030            fixed (byte* pUtf8Data = &MemoryMarshal.GetReference(utf8Data))
 31            {
 032                byte* pFirstInvalidByte = GetPointerToFirstInvalidByte(pUtf8Data, utf8Data.Length, out int utf16CodeUnit
 033                int index = (int)(void*)Unsafe.ByteOffset(ref *pUtf8Data, ref *pFirstInvalidByte);
 34
 035                isAscii = (utf16CodeUnitCountAdjustment == 0); // If UTF-16 char count == UTF-8 byte count, it's ASCII.
 036                return (index < utf8Data.Length) ? index : -1;
 37            }
 38        }
 39
 40        /// <summary>
 41        /// Returns true iff the UInt32 represents four ASCII UTF-8 characters in machine endianness.
 42        /// </summary>
 43        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 044        internal static bool AllBytesInUInt32AreAscii(uint value) => (value & ~0x7F7F_7F7Fu) == 0;
 45
 46        /// <summary>
 47        /// Returns true iff the UInt64 represents eighty ASCII UTF-8 characters in machine endianness.
 48        /// </summary>
 49        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 050        internal static bool AllBytesInUInt64AreAscii(ulong value) => (value & ~0x7F7F_7F7F_7F7F_7F7Ful) == 0;
 51
 52        /// <summary>
 53        /// Given a UInt32 that represents four ASCII UTF-8 characters, returns the invariant
 54        /// lowercase representation of those characters. Requires the input value to contain
 55        /// four ASCII UTF-8 characters in machine endianness.
 56        /// </summary>
 57        /// <remarks>
 58        /// This is a branchless implementation.
 59        /// </remarks>
 60        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 61        internal static uint ConvertAllAsciiBytesInUInt32ToLowercase(uint value)
 62        {
 63            // ASSUMPTION: Caller has validated that input value is ASCII.
 064            Debug.Assert(AllBytesInUInt32AreAscii(value));
 65
 66            // the 0x80 bit of each byte of 'lowerIndicator' will be set iff the word has value >= 'A'
 067            uint lowerIndicator = value + 0x8080_8080u - 0x4141_4141u;
 68
 69            // the 0x80 bit of each byte of 'upperIndicator' will be set iff the word has value > 'Z'
 070            uint upperIndicator = value + 0x8080_8080u - 0x5B5B_5B5Bu;
 71
 72            // the 0x80 bit of each byte of 'combinedIndicator' will be set iff the word has value >= 'A' and <= 'Z'
 073            uint combinedIndicator = (lowerIndicator ^ upperIndicator);
 74
 75            // the 0x20 bit of each byte of 'mask' will be set iff the word has value >= 'A' and <= 'Z'
 076            uint mask = (combinedIndicator & 0x8080_8080u) >> 2;
 77
 078            return value ^ mask; // bit flip uppercase letters [A-Z] => [a-z]
 79        }
 80
 81        /// <summary>
 82        /// Given a UInt32 that represents four ASCII UTF-8 characters, returns the invariant
 83        /// uppercase representation of those characters. Requires the input value to contain
 84        /// four ASCII UTF-8 characters in machine endianness.
 85        /// </summary>
 86        /// <remarks>
 87        /// This is a branchless implementation.
 88        /// </remarks>
 89        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 90        internal static uint ConvertAllAsciiBytesInUInt32ToUppercase(uint value)
 91        {
 92            // Intrinsified in mono interpreter
 93            // ASSUMPTION: Caller has validated that input value is ASCII.
 094            Debug.Assert(AllBytesInUInt32AreAscii(value));
 95
 96            // the 0x80 bit of each byte of 'lowerIndicator' will be set iff the word has value >= 'a'
 097            uint lowerIndicator = value + 0x8080_8080u - 0x6161_6161u;
 98
 99            // the 0x80 bit of each byte of 'upperIndicator' will be set iff the word has value > 'z'
 0100            uint upperIndicator = value + 0x8080_8080u - 0x7B7B_7B7Bu;
 101
 102            // the 0x80 bit of each byte of 'combinedIndicator' will be set iff the word has value >= 'a' and <= 'z'
 0103            uint combinedIndicator = (lowerIndicator ^ upperIndicator);
 104
 105            // the 0x20 bit of each byte of 'mask' will be set iff the word has value >= 'a' and <= 'z'
 0106            uint mask = (combinedIndicator & 0x8080_8080u) >> 2;
 107
 0108            return value ^ mask; // bit flip lowercase letters [a-z] => [A-Z]
 109        }
 110
 111        /// <summary>
 112        /// Given a UInt64 that represents eight ASCII UTF-8 characters, returns the invariant
 113        /// uppercase representation of those characters. Requires the input value to contain
 114        /// eight ASCII UTF-8 characters in machine endianness.
 115        /// </summary>
 116        /// <remarks>
 117        /// This is a branchless implementation.
 118        /// </remarks>
 119        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 120        internal static ulong ConvertAllAsciiBytesInUInt64ToUppercase(ulong value)
 121        {
 122            // ASSUMPTION: Caller has validated that input value is ASCII.
 0123            Debug.Assert(AllBytesInUInt64AreAscii(value));
 124
 125            // the 0x80 bit of each byte of 'lowerIndicator' will be set iff the word has value >= 'a'
 0126            ulong lowerIndicator = value + 0x8080_8080_8080_8080ul - 0x6161_6161_6161_6161ul;
 127
 128            // the 0x80 bit of each byte of 'upperIndicator' will be set iff the word has value > 'z'
 0129            ulong upperIndicator = value + 0x8080_8080_8080_8080ul - 0x7B7B_7B7B_7B7B_7B7Bul;
 130
 131            // the 0x80 bit of each byte of 'combinedIndicator' will be set iff the word has value >= 'a' and <= 'z'
 0132            ulong combinedIndicator = (lowerIndicator ^ upperIndicator);
 133
 134            // the 0x20 bit of each byte of 'mask' will be set iff the word has value >= 'a' and <= 'z'
 0135            ulong mask = (combinedIndicator & 0x8080_8080_8080_8080ul) >> 2;
 136
 0137            return value ^ mask; // bit flip lowercase letters [a-z] => [A-Z]
 138        }
 139
 140        /// <summary>
 141        /// Given a UInt64 that represents eight ASCII UTF-8 characters, returns the invariant
 142        /// uppercase representation of those characters. Requires the input value to contain
 143        /// eight ASCII UTF-8 characters in machine endianness.
 144        /// </summary>
 145        /// <remarks>
 146        /// This is a branchless implementation.
 147        /// </remarks>
 148        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 149        internal static ulong ConvertAllAsciiBytesInUInt64ToLowercase(ulong value)
 150        {
 151            // ASSUMPTION: Caller has validated that input value is ASCII.
 0152            Debug.Assert(AllBytesInUInt64AreAscii(value));
 153
 154            // the 0x80 bit of each byte of 'lowerIndicator' will be set iff the word has value >= 'A'
 0155            ulong lowerIndicator = value + 0x8080_8080_8080_8080ul - 0x4141_4141_4141_4141ul;
 156
 157            // the 0x80 bit of each byte of 'upperIndicator' will be set iff the word has value > 'Z'
 0158            ulong upperIndicator = value + 0x8080_8080_8080_8080ul - 0x5B5B_5B5B_5B5B_5B5Bul;
 159
 160            // the 0x80 bit of each byte of 'combinedIndicator' will be set iff the word has value >= 'a' and <= 'z'
 0161            ulong combinedIndicator = (lowerIndicator ^ upperIndicator);
 162
 163            // the 0x20 bit of each byte of 'mask' will be set iff the word has value >= 'a' and <= 'z'
 0164            ulong mask = (combinedIndicator & 0x8080_8080_8080_8080ul) >> 2;
 165
 0166            return value ^ mask; // bit flip uppercase letters [A-Z] => [a-z]
 167        }
 168
 169        /// <summary>
 170        /// Given two UInt64s that represent eight ASCII UTF-8 characters each, returns true iff
 171        /// the two inputs are equal using an ordinal case-insensitive comparison.
 172        /// </summary>
 173        /// <remarks>
 174        /// This is a branchless implementation.
 175        /// </remarks>
 176        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 177        internal static bool UInt64OrdinalIgnoreCaseAscii(ulong valueA, ulong valueB)
 178        {
 179            // ASSUMPTION: Caller has validated that input values are ASCII.
 0180            Debug.Assert(AllBytesInUInt64AreAscii(valueA));
 0181            Debug.Assert(AllBytesInUInt64AreAscii(valueB));
 182
 183            // The 0x80 bit of each byte is set iff 'A' <= byte <= 'Z'; shifting it right by 2
 184            // gives 0x20, which lowercases those bytes before comparing.
 0185            ulong letterMaskA = (((valueA + 0x3F3F3F3F3F3F3F3F) ^ (valueA + 0x2525252525252525)) & 0x8080808080808080) >
 0186            ulong letterMaskB = (((valueB + 0x3F3F3F3F3F3F3F3F) ^ (valueB + 0x2525252525252525)) & 0x8080808080808080) >
 187
 0188            return (valueA | letterMaskA) == (valueB | letterMaskB);
 189        }
 190
 191#if NET
 192        /// <summary>
 193        /// Returns true iff the Vector128 represents 16 ASCII UTF-8 characters in machine endianness.
 194        /// </summary>
 195        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 196        internal static bool AllBytesInVector128AreAscii(Vector128<byte> vec)
 197        {
 0198            return (vec & Vector128.Create(unchecked((byte)(~0x7F)))) == Vector128<byte>.Zero;
 199        }
 200#endif
 201    }
 202}
 203

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.CoreLib/src/System/Text/Unicode/Utf8Utility.Helpers.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Buffers.Binary;
 5using System.Diagnostics;
 6using System.Numerics;
 7using System.Runtime.CompilerServices;
 8
 9namespace System.Text.Unicode
 10{
 11    internal static partial class Utf8Utility
 12    {
 13        /// <summary>
 14        /// Given a machine-endian DWORD which four bytes of UTF-8 data, interprets the
 15        /// first three bytes as a three-byte UTF-8 subsequence and returns the UTF-16 representation.
 16        /// </summary>
 17        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 18        private static uint ExtractCharFromFirstThreeByteSequence(uint value)
 19        {
 20            Debug.Assert(UInt32BeginsWithUtf8ThreeByteMask(value));
 21
 56570422            if (BitConverter.IsLittleEndian)
 23            {
 24                // value = [ ######## | 10xxxxxx 10yyyyyy 1110zzzz ]
 56570425                return ((value & 0x003F0_000u) >> 16)
 56570426                    | ((value & 0x0000_3F00u) >> 2)
 56570427                    | ((value & 0x0000_000Fu) << 12);
 28            }
 29            else
 30            {
 31                // value = [ 1110zzzz 10yyyyyy 10xxxxxx | ######## ]
 32                return ((value & 0x0F00_0000u) >> 12)
 33                    | ((value & 0x003F_0000u) >> 10)
 34                    | ((value & 0x0000_3F00u) >> 8);
 35            }
 36        }
 37
 38        /// <summary>
 39        /// Given a machine-endian DWORD which four bytes of UTF-8 data, interprets the
 40        /// first two bytes as a two-byte UTF-8 subsequence and returns the UTF-16 representation.
 41        /// </summary>
 42        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 43        private static uint ExtractCharFromFirstTwoByteSequence(uint value)
 44        {
 45            Debug.Assert(UInt32BeginsWithUtf8TwoByteMask(value) && !UInt32BeginsWithOverlongUtf8TwoByteSequence(value));
 46
 21972647            if (BitConverter.IsLittleEndian)
 48            {
 49                // value = [ ######## ######## | 10xxxxxx 110yyyyy ]
 21972650                uint leadingByte = (uint)(byte)value << 6;
 21972651                return (uint)(byte)(value >> 8) + leadingByte - (0xC0u << 6) - 0x80u; // remove header bits
 52            }
 53            else
 54            {
 55                // value = [ 110yyyyy 10xxxxxx | ######## ######## ]
 56                return (char)(((value & 0x1F00_0000u) >> 18) | ((value & 0x003F_0000u) >> 16));
 57            }
 58        }
 59
 60        /// <summary>
 61        /// Given a machine-endian DWORD which represents four bytes of UTF-8 data, interprets the input as a
 62        /// four-byte UTF-8 sequence and returns the machine-endian DWORD of the UTF-16 representation.
 63        /// </summary>
 64        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 65        private static uint ExtractCharsFromFourByteSequence(uint value)
 66        {
 67            if (BitConverter.IsLittleEndian)
 68            {
 69                // input is UTF8 [ 10xxxxxx 10yyyyyy 10uuzzzz 11110uuu ] = scalar 000uuuuu zzzzyyyy yyxxxxxx
 70                // want to return UTF16 scalar 000uuuuuzzzzyyyyyyxxxxxx = [ 110111yy yyxxxxxx 110110ww wwzzzzyy ]
 71                // where wwww = uuuuu - 1
 8169072                uint retVal = (uint)(byte)value << 8; // retVal = [ 00000000 00000000 11110uuu 00000000 ]
 8169073                retVal |= (value & 0x0000_3F00u) >> 6; // retVal = [ 00000000 00000000 11110uuu uuzzzz00 ]
 8169074                retVal |= (value & 0x0030_0000u) >> 20; // retVal = [ 00000000 00000000 11110uuu uuzzzzyy ]
 8169075                retVal |= (value & 0x3F00_0000u) >> 8; // retVal = [ 00000000 00xxxxxx 11110uuu uuzzzzyy ]
 8169076                retVal |= (value & 0x000F_0000u) << 6; // retVal = [ 000000yy yyxxxxxx 11110uuu uuzzzzyy ]
 8169077                retVal -= 0x0000_0040u; // retVal = [ 000000yy yyxxxxxx 111100ww wwzzzzyy ]
 8169078                retVal -= 0x0000_2000u; // retVal = [ 000000yy yyxxxxxx 110100ww wwzzzzyy ]
 8169079                retVal += 0x0000_0800u; // retVal = [ 000000yy yyxxxxxx 110110ww wwzzzzyy ]
 8169080                retVal += 0xDC00_0000u; // retVal = [ 110111yy yyxxxxxx 110110ww wwzzzzyy ]
 8169081                return retVal;
 82            }
 83            else
 84            {
 85                // input is UTF8 [ 11110uuu 10uuzzzz 10yyyyyy 10xxxxxx ] = scalar 000uuuuu zzzzyyyy yyxxxxxx
 86                // want to return UTF16 scalar 000uuuuuxxxxxxxxxxxxxxxx = [ 110110wwwwxxxxxx 110111xxxxxxxxx ]
 87                // where wwww = uuuuu - 1
 88                uint retVal = value & 0xFF00_0000u; // retVal = [ 11110uuu 00000000 00000000 00000000 ]
 89                retVal |= (value & 0x003F_0000u) << 2; // retVal = [ 11110uuu uuzzzz00 00000000 00000000 ]
 90                retVal |= (value & 0x0000_3000u) << 4; // retVal = [ 11110uuu uuzzzzyy 00000000 00000000 ]
 91                retVal |= (value & 0x0000_0F00u) >> 2; // retVal = [ 11110uuu uuzzzzyy 000000yy yy000000 ]
 92                retVal |= (value & 0x0000_003Fu); // retVal = [ 11110uuu uuzzzzyy 000000yy yyxxxxxx ]
 93                retVal -= 0x2000_0000u; // retVal = [ 11010uuu uuzzzzyy 000000yy yyxxxxxx ]
 94                retVal -= 0x0040_0000u; // retVal = [ 110100ww wwzzzzyy 000000yy yyxxxxxx ]
 95                retVal += 0x0000_DC00u; // retVal = [ 110100ww wwzzzzyy 110111yy yyxxxxxx ]
 96                retVal += 0x0800_0000u; // retVal = [ 110110ww wwzzzzyy 110111yy yyxxxxxx ]
 97                return retVal;
 98            }
 99        }
 100
 101        /// <summary>
 102        /// Given a 32-bit integer that represents a valid packed UTF-16 surrogate pair, all in machine-endian order,
 103        /// returns the packed 4-byte UTF-8 representation of this scalar value, also in machine-endian order.
 104        /// </summary>
 105        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 106        private static uint ExtractFourUtf8BytesFromSurrogatePair(uint value)
 107        {
 108            Debug.Assert(IsWellFormedUtf16SurrogatePair(value));
 109
 0110            if (BitConverter.IsLittleEndian)
 111            {
 112                // input = [ 110111yyyyxxxxxx 110110wwwwzzzzyy ] = scalar (000uuuuu zzzzyyyy yyxxxxxx)
 113                // must return [ 10xxxxxx 10yyyyyy 10uuzzzz 11110uuu ], where wwww = uuuuu - 1
 114
 0115                value += 0x0000_0040u; // = [ 110111yyyyxxxxxx 11011uuuuuzzzzyy ]
 116
 0117                uint tempA = BinaryPrimitives.ReverseEndianness(value & 0x003F_0700u); // = [ 00000000 00000uuu 00xxxxxx
 0118                tempA = BitOperations.RotateLeft(tempA, 16); // = [ 00xxxxxx 00000000 00000000 00000uuu ]
 119
 0120                uint tempB = (value & 0x00FCu) << 6; // = [ 00000000 00000000 00uuzzzz 00000000 ]
 0121                uint tempC = (value >> 6) & 0x000F_0000u; // = [ 00000000 0000yyyy 00000000 00000000 ]
 0122                tempC |= tempB;
 123
 0124                uint tempD = (value & 0x03u) << 20; // = [ 00000000 00yy0000 00000000 00000000 ]
 0125                tempD |= 0x8080_80F0u;
 126
 0127                return tempD | tempA | tempC; // = [ 10xxxxxx 10yyyyyy 10uuzzzz 11110uuu ]
 128            }
 129            else
 130            {
 131                // input = [ 110110wwwwzzzzyy 110111yyyyxxxxxx ], where wwww = uuuuu - 1
 132                // must return [ 11110uuu 10uuzzzz 10yyyyyy 10xxxxxx ], where wwww = uuuuu - 1
 133
 134                value -= 0xD800_DC00u; // = [ 000000wwwwzzzzyy 000000yyyyxxxxxx ]
 135                value += 0x0040_0000u; // = [ 00000uuuuuzzzzyy 000000yyyyxxxxxx ]
 136
 137                uint tempA = value & 0x0700_0000u; // = [ 00000uuu 00000000 00000000 00000000 ]
 138                uint tempB = (value >> 2) & 0x003F_0000u; // = [ 00000000 00uuzzzz 00000000 00000000 ]
 139                tempB |= tempA;
 140
 141                uint tempC = (value << 2) & 0x0000_0F00u; // = [ 00000000 00000000 0000yyyy 00000000 ]
 142                uint tempD = (value >> 4) & 0x0000_3000u; // = [ 00000000 00000000 00yy0000 00000000 ]
 143                tempD |= tempC;
 144
 145                uint tempE = (value & 0x3Fu) + 0xF080_8080u; // = [ 11110000 10000000 10000000 10xxxxxx ]
 146                return tempE | tempB | tempD; // = [ 11110uuu 10uuzzzz 10yyyyyy 10xxxxxx ]
 147            }
 148        }
 149
 150        /// <summary>
 151        /// Given a machine-endian DWORD which represents two adjacent UTF-8 two-byte sequences,
 152        /// returns the machine-endian DWORD representation of that same data as two adjacent
 153        /// UTF-16 byte sequences.
 154        /// </summary>
 155        /// <param name="value"></param>
 156        /// <returns></returns>
 157        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 158        private static uint ExtractTwoCharsPackedFromTwoAdjacentTwoByteSequences(uint value)
 159        {
 160            // We don't want to swap the position of the high and low WORDs,
 161            // as the buffer was read in machine order and will be written in
 162            // machine order.
 163
 164            if (BitConverter.IsLittleEndian)
 165            {
 166                // value = [ 10xxxxxx 110yyyyy | 10xxxxxx 110yyyyy ]
 35112167                return ((value & 0x3F003F00u) >> 8) | ((value & 0x001F001Fu) << 6);
 168            }
 169            else
 170            {
 171                // value = [ 110yyyyy 10xxxxxx | 110yyyyy 10xxxxxx ]
 172                return ((value & 0x1F001F00u) >> 2) | (value & 0x003F003Fu);
 173            }
 174        }
 175
 176        /// <summary>
 177        /// Given a machine-endian DWORD which represents two adjacent UTF-16 sequences,
 178        /// returns the machine-endian DWORD representation of that same data as two
 179        /// adjacent UTF-8 two-byte sequences.
 180        /// </summary>
 181        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 182        private static uint ExtractTwoUtf8TwoByteSequencesFromTwoPackedUtf16Chars(uint value)
 183        {
 184            // stays in machine endian
 185
 186            Debug.Assert(IsFirstCharTwoUtf8Bytes(value) && IsSecondCharTwoUtf8Bytes(value));
 187
 0188            if (BitConverter.IsLittleEndian)
 189            {
 190                // value = [ 00000YYY YYXXXXXX 00000yyy yyxxxxxx ]
 191                // want to return [ 10XXXXXX 110YYYYY 10xxxxxx 110yyyyy ]
 192
 0193                return ((value >> 6) & 0x001F_001Fu) + ((value << 8) & 0x3F00_3F00u) + 0x80C0_80C0u;
 194            }
 195            else
 196            {
 197                // value = [ 00000YYY YYXXXXXX 00000yyy yyxxxxxx ]
 198                // want to return [ 110YYYYY 10XXXXXX 110yyyyy 10xxxxxx ]
 199
 200                return ((value << 2) & 0x1F00_1F00u) + (value & 0x003F_003Fu) + 0xC080_C080u;
 201            }
 202        }
 203
 204        /// <summary>
 205        /// Given a machine-endian DWORD which represents two adjacent UTF-16 sequences,
 206        /// returns the machine-endian DWORD representation of the first UTF-16 char
 207        /// as a UTF-8 two-byte sequence packed into a WORD and zero-extended to DWORD.
 208        /// </summary>
 209        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 210        private static uint ExtractUtf8TwoByteSequenceFromFirstUtf16Char(uint value)
 211        {
 212            // stays in machine endian
 213
 214            Debug.Assert(IsFirstCharTwoUtf8Bytes(value));
 215
 0216            if (BitConverter.IsLittleEndian)
 217            {
 218                // value = [ ######## ######## 00000yyy yyxxxxxx ]
 219                // want to return [ ######## ######## 10xxxxxx 110yyyyy ]
 220
 0221                uint temp = (value << 2) & 0x1F00u; // [ 00000000 00000000 000yyyyy 00000000 ]
 0222                value &= 0x3Fu; // [ 00000000 00000000 00000000 00xxxxxx ]
 0223                return BinaryPrimitives.ReverseEndianness((ushort)(temp + value + 0xC080u)); // [ 00000000 00000000 10xx
 224            }
 225            else
 226            {
 227                // value = [ 00000yyy yyxxxxxx ######## ######## ]
 228                // want to return [ ######## ######## 110yyyyy 10xxxxxx ]
 229
 230                uint temp = (value >> 16) & 0x3Fu; // [ 00000000 00000000 00000000 00xxxxxx ]
 231                value = (value >> 14) & 0x1F00u; // [ 00000000 00000000 000yyyyy 0000000 ]
 232                return value + temp + 0xC080u;
 233            }
 234        }
 235
 236        /// <summary>
 237        /// Given a 32-bit integer that represents two packed UTF-16 characters, all in machine-endian order,
 238        /// returns true iff the first UTF-16 character is ASCII.
 239        /// </summary>
 240        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 241        private static bool IsFirstCharAscii(uint value)
 242        {
 243            // Little-endian: Given [ #### AAAA ], return whether AAAA is in range [ 0000..007F ].
 244            // Big-endian: Given [ AAAA #### ], return whether AAAA is in range [ 0000..007F ].
 245
 246            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 247
 248            return (BitConverter.IsLittleEndian && (value & 0xFF80u) == 0)
 249                || (!BitConverter.IsLittleEndian && value < 0x0080_0000u);
 250        }
 251
 252        /// <summary>
 253        /// Given a 32-bit integer that represents two packed UTF-16 characters, all in machine-endian order,
 254        /// returns true iff the first UTF-16 character requires *at least* 3 bytes to encode in UTF-8.
 255        /// This also returns true if the first UTF-16 character is a surrogate character (well-formedness is not valida
 256        /// </summary>
 257        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 258        private static bool IsFirstCharAtLeastThreeUtf8Bytes(uint value)
 259        {
 260            // Little-endian: Given [ #### AAAA ], return whether AAAA is in range [ 0800..FFFF ].
 261            // Big-endian: Given [ AAAA #### ], return whether AAAA is in range [ 0800..FFFF ].
 262
 263            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 264
 265            return (BitConverter.IsLittleEndian && (value & 0xF800u) != 0)
 266                || (!BitConverter.IsLittleEndian && value >= 0x0800_0000u);
 267        }
 268
 269        /// <summary>
 270        /// Given a 32-bit integer that represents two packed UTF-16 characters, all in machine-endian order,
 271        /// returns true iff the first UTF-16 character is a surrogate character (either high or low).
 272        /// </summary>
 273        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 274        private static bool IsFirstCharSurrogate(uint value)
 275        {
 276            // Little-endian: Given [ #### AAAA ], return whether AAAA is in range [ D800..DFFF ].
 277            // Big-endian: Given [ AAAA #### ], return whether AAAA is in range [ D800..DFFF ].
 278
 279            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 280
 281            return (BitConverter.IsLittleEndian && ((value - 0xD800u) & 0xF800u) == 0)
 282                || (!BitConverter.IsLittleEndian && (value - 0xD800_0000u) < 0x0800_0000u);
 283        }
 284
 285        /// <summary>
 286        /// Given a 32-bit integer that represents two packed UTF-16 characters, all in machine-endian order,
 287        /// returns true iff the first UTF-16 character would be encoded as exactly 2 bytes in UTF-8.
 288        /// </summary>
 289        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 290        private static bool IsFirstCharTwoUtf8Bytes(uint value)
 291        {
 292            // Little-endian: Given [ #### AAAA ], return whether AAAA is in range [ 0080..07FF ].
 293            // Big-endian: Given [ AAAA #### ], return whether AAAA is in range [ 0080..07FF ].
 294
 295            // TODO: I'd like to be able to write "(ushort)(value - 0x0080u) < 0x0780u" for the little-endian
 296            // case, but the JIT doesn't currently emit 16-bit comparisons efficiently.
 297            // Tracked as https://github.com/dotnet/runtime/issues/10337.
 298
 299            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 300
 301            return (BitConverter.IsLittleEndian && ((value - 0x0080u) & 0xFFFFu) < 0x0780u)
 302                || (!BitConverter.IsLittleEndian && UnicodeUtility.IsInRangeInclusive(value, 0x0080_0000u, 0x07FF_FFFFu)
 303        }
 304
 305        /// <summary>
 306        /// Returns <see langword="true"/> iff the low byte of <paramref name="value"/>
 307        /// is a UTF-8 continuation byte.
 308        /// </summary>
 309        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 310        private static bool IsLowByteUtf8ContinuationByte(uint value)
 311        {
 312            // The JIT won't emit a single 8-bit signed cmp instruction (see IsUtf8ContinuationByte),
 313            // so the best we can do for now is the lea / cmp pair.
 314            // Tracked as https://github.com/dotnet/runtime/issues/10337.
 315
 19222316            return (byte)(value - 0x80u) <= 0x3Fu;
 317        }
 318
 319        /// <summary>
 320        /// Given a 32-bit integer that represents two packed UTF-16 characters, all in machine-endian order,
 321        /// returns true iff the second UTF-16 character is ASCII.
 322        /// </summary>
 323        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 324        private static bool IsSecondCharAscii(uint value)
 325        {
 326            // Little-endian: Given [ BBBB #### ], return whether BBBB is in range [ 0000..007F ].
 327            // Big-endian: Given [ #### BBBB ], return whether BBBB is in range [ 0000..007F ].
 328
 329            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 330
 331            return (BitConverter.IsLittleEndian && value < 0x0080_0000u)
 332                || (!BitConverter.IsLittleEndian && (value & 0xFF80u) == 0);
 333        }
 334
 335        /// <summary>
 336        /// Given a 32-bit integer that represents two packed UTF-16 characters, all in machine-endian order,
 337        /// returns true iff the second UTF-16 character requires *at least* 3 bytes to encode in UTF-8.
 338        /// This also returns true if the second UTF-16 character is a surrogate character (well-formedness is not valid
 339        /// </summary>
 340        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 341        private static bool IsSecondCharAtLeastThreeUtf8Bytes(uint value)
 342        {
 343            // Little-endian: Given [ BBBB #### ], return whether BBBB is in range [ 0800..FFFF ].
 344            // Big-endian: Given [ #### BBBB ], return whether ABBBBAAA is in range [ 0800..FFFF ].
 345
 346            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 347
 348            return (BitConverter.IsLittleEndian && (value & 0xF800_0000u) != 0)
 349                || (!BitConverter.IsLittleEndian && (value & 0xF800u) != 0);
 350        }
 351
 352        /// <summary>
 353        /// Given a 32-bit integer that represents two packed UTF-16 characters, all in machine-endian order,
 354        /// returns true iff the second UTF-16 character is a surrogate character (either high or low).
 355        /// </summary>
 356        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 357        private static bool IsSecondCharSurrogate(uint value)
 358        {
 359            // Little-endian: Given [ BBBB #### ], return whether BBBB is in range [ D800..DFFF ].
 360            // Big-endian: Given [ #### BBBB ], return whether BBBB is in range [ D800..DFFF ].
 361
 362            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 363
 364            return (BitConverter.IsLittleEndian && (value - 0xD800_0000u) < 0x0800_0000u)
 365                || (!BitConverter.IsLittleEndian && ((value - 0xD800u) & 0xF800u) == 0);
 366        }
 367
 368        /// <summary>
 369        /// Given a 32-bit integer that represents two packed UTF-16 characters, all in machine-endian order,
 370        /// returns true iff the second UTF-16 character would be encoded as exactly 2 bytes in UTF-8.
 371        /// </summary>
 372        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 373        private static bool IsSecondCharTwoUtf8Bytes(uint value)
 374        {
 375            // Little-endian: Given [ BBBB #### ], return whether BBBB is in range [ 0080..07FF ].
 376            // Big-endian: Given [ #### BBBB ], return whether BBBB is in range [ 0080..07FF ].
 377
 378            // TODO: I'd like to be able to write "(ushort)(value - 0x0080u) < 0x0780u" for the big-endian
 379            // case, but the JIT doesn't currently emit 16-bit comparisons efficiently.
 380            // Tracked as https://github.com/dotnet/runtime/issues/10337.
 381
 382            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 383
 384            return (BitConverter.IsLittleEndian && UnicodeUtility.IsInRangeInclusive(value, 0x0080_0000u, 0x07FF_FFFFu))
 385                || (!BitConverter.IsLittleEndian && ((value - 0x0080u) & 0xFFFFu) < 0x0780u);
 386        }
 387
 388        /// <summary>
 389        /// Returns <see langword="true"/> iff <paramref name="value"/> is a UTF-8 continuation byte;
 390        /// i.e., has binary representation 10xxxxxx, where x is any bit.
 391        /// </summary>
 392        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 393        internal static bool IsUtf8ContinuationByte(in byte value)
 394        {
 395            // This API takes its input as a readonly ref so that the JIT can emit "cmp ModRM" statements
 396            // directly rather than bounce a temporary through a register. That is, we want the JIT to be
 397            // able to emit a single "cmp byte ptr [data], C0h" statement if we're querying a memory location
 398            // to see if it's a continuation byte. Data that's already enregistered will go through the
 399            // normal "cmp reg, C0h" code paths, perhaps with some extra unnecessary "movzx" instructions.
 400            //
 401            // The below check takes advantage of the two's complement representation of negative numbers.
 402            // [ 0b1000_0000, 0b1011_1111 ] is [ -127 (sbyte.MinValue), -65 ]
 403
 55570404            return (sbyte)value < -64;
 405        }
 406
 407        /// <summary>
 408        /// Given a 32-bit integer that represents two packed UTF-16 characters, all in machine-endian order,
 409        /// returns true iff the two characters represent a well-formed UTF-16 surrogate pair.
 410        /// </summary>
 411        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 412        private static bool IsWellFormedUtf16SurrogatePair(uint value)
 413        {
 414            // Little-endian: Given [ LLLL HHHH ], validate that LLLL in [ DC00..DFFF ] and HHHH in [ D800..DBFF ].
 415            // Big-endian: Given [ HHHH LLLL ], validate that HHHH in [ D800..DBFF ] and LLLL in [ DC00..DFFF ].
 416            //
 417            // We're essentially performing a range check on each component of the input in parallel. The allowed range
 418            // ends up being "< 0x0400" after the beginning of the allowed range is subtracted from each element. We
 419            // can't perform the equivalent of two CMPs in parallel, but we can take advantage of the fact that 0x0400
 420            // is a whole power of 2, which means that a CMP is really just a glorified TEST operation. Two TESTs *can*
 421            // be performed in parallel. The logic below then becomes 3 operations: "add/lea; test; jcc".
 422
 423            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 424
 425            return (BitConverter.IsLittleEndian && ((value - 0xDC00_D800u) & 0xFC00_FC00u) == 0)
 426                || (!BitConverter.IsLittleEndian && ((value - 0xD800_DC00u) & 0xFC00_FC00u) == 0);
 427        }
 428
 429        /// <summary>
 430        /// Converts a DWORD from machine-endian to little-endian.
 431        /// </summary>
 432        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 433        private static uint ToLittleEndian(uint value)
 434        {
 435            if (BitConverter.IsLittleEndian)
 436            {
 73572437                return value;
 438            }
 439            else
 440            {
 441                return BinaryPrimitives.ReverseEndianness(value);
 442            }
 443        }
 444
 445        /// <summary>
 446        /// Given a UTF-8 buffer which has been read into a DWORD in machine endianness,
 447        /// returns <see langword="true"/> iff the first two bytes of the buffer are
 448        /// an overlong representation of a sequence that should be represented as one byte.
 449        /// This method *does not* validate that the sequence matches the appropriate
 450        /// 2-byte sequence mask (see <see cref="UInt32BeginsWithUtf8TwoByteMask"/>).
 451        /// </summary>
 452        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 453        private static bool UInt32BeginsWithOverlongUtf8TwoByteSequence(uint value)
 454        {
 455            // ASSUMPTION: Caller has already checked the '110yyyyy 10xxxxxx' mask of the input.
 456            Debug.Assert(UInt32BeginsWithUtf8TwoByteMask(value));
 457
 458            // Per Table 3-7, first byte of two-byte sequence must be within range C2 .. DF.
 459            // Since we already validated it's 80 <= ?? <= DF (per mask check earlier), now only need
 460            // to check that it's < C2.
 461
 462            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 463
 500197464            return (BitConverter.IsLittleEndian && ((byte)value < 0xC2u))
 500197465                || (!BitConverter.IsLittleEndian && (value < 0xC200_0000u));
 466        }
 467
 468        /// <summary>
 469        /// Given a UTF-8 buffer which has been read into a DWORD in machine endianness,
 470        /// returns <see langword="true"/> iff the first four bytes of the buffer match
 471        /// the UTF-8 4-byte sequence mask [ 11110www 10zzzzzz 10yyyyyy 10xxxxxx ]. This
 472        /// method *does not* validate that the sequence is well-formed; the caller must
 473        /// still perform overlong form or out-of-range checking.
 474        /// </summary>
 475        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 476        private static bool UInt32BeginsWithUtf8FourByteMask(uint value)
 477        {
 478            // The code in this method is equivalent to the code
 479            // below but is slightly more optimized.
 480            //
 481            // if (BitConverter.IsLittleEndian)
 482            // {
 483            //     const uint mask = 0xC0C0C0F8U;
 484            //     const uint comparand = 0x808080F0U;
 485            //     return ((value & mask) == comparand);
 486            // }
 487            // else
 488            // {
 489            //     const uint mask = 0xF8C0C0C0U;
 490            //     const uint comparand = 0xF0808000U;
 491            //     return ((value & mask) == comparand);
 492            // }
 493
 494            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 495
 496            return (BitConverter.IsLittleEndian && (((value - 0x8080_80F0u) & 0xC0C0_C0F8u) == 0))
 497                || (!BitConverter.IsLittleEndian && (((value - 0xF080_8080u) & 0xF8C0_C0C0u) == 0));
 498        }
 499
 500        /// <summary>
 501        /// Given a UTF-8 buffer which has been read into a DWORD in machine endianness,
 502        /// returns <see langword="true"/> iff the first three bytes of the buffer match
 503        /// the UTF-8 3-byte sequence mask [ 1110zzzz 10yyyyyy 10xxxxxx ]. This method *does not*
 504        /// validate that the sequence is well-formed; the caller must still perform
 505        /// overlong form or surrogate checking.
 506        /// </summary>
 507        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 508        private static bool UInt32BeginsWithUtf8ThreeByteMask(uint value)
 509        {
 510            // The code in this method is equivalent to the code
 511            // below but is slightly more optimized.
 512            //
 513            // if (BitConverter.IsLittleEndian)
 514            // {
 515            //     const uint mask = 0x00C0C0F0U;
 516            //     const uint comparand = 0x008080E0U;
 517            //     return ((value & mask) == comparand);
 518            // }
 519            // else
 520            // {
 521            //     const uint mask = 0xF0C0C000U;
 522            //     const uint comparand = 0xE0808000U;
 523            //     return ((value & mask) == comparand);
 524            // }
 525
 526            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 527
 528            return (BitConverter.IsLittleEndian && (((value - 0x0080_80E0u) & 0x00C0_C0F0u) == 0))
 529                || (!BitConverter.IsLittleEndian && (((value - 0xE080_8000u) & 0xF0C0_C000u) == 0));
 530        }
 531
 532        /// <summary>
 533        /// Given a UTF-8 buffer which has been read into a DWORD in machine endianness,
 534        /// returns <see langword="true"/> iff the first two bytes of the buffer match
 535        /// the UTF-8 2-byte sequence mask [ 110yyyyy 10xxxxxx ]. This method *does not*
 536        /// validate that the sequence is well-formed; the caller must still perform
 537        /// overlong form checking.
 538        /// </summary>
 539        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 540        private static bool UInt32BeginsWithUtf8TwoByteMask(uint value)
 541        {
 542            // The code in this method is equivalent to the code
 543            // below but is slightly more optimized.
 544            //
 545            // if (BitConverter.IsLittleEndian)
 546            // {
 547            //     const uint mask = 0x0000C0E0U;
 548            //     const uint comparand = 0x000080C0U;
 549            //     return ((value & mask) == comparand);
 550            // }
 551            // else
 552            // {
 553            //     const uint mask = 0xE0C00000U;
 554            //     const uint comparand = 0xC0800000U;
 555            //     return ((value & mask) == comparand);
 556            // }
 557
 558            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 559
 560            return (BitConverter.IsLittleEndian && (((value - 0x0000_80C0u) & 0x0000_C0E0u) == 0))
 561                || (!BitConverter.IsLittleEndian && (((value - 0xC080_0000u) & 0xE0C0_0000u) == 0));
 562        }
 563
 564        /// <summary>
 565        /// Given a UTF-8 buffer which has been read into a DWORD in machine endianness,
 566        /// returns <see langword="true"/> iff the first two bytes of the buffer are
 567        /// an overlong representation of a sequence that should be represented as one byte.
 568        /// This method *does not* validate that the sequence matches the appropriate
 569        /// 2-byte sequence mask (see <see cref="UInt32BeginsWithUtf8TwoByteMask"/>).
 570        /// </summary>
 571        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 572        private static bool UInt32EndsWithOverlongUtf8TwoByteSequence(uint value)
 573        {
 574            // ASSUMPTION: Caller has already checked the '110yyyyy 10xxxxxx' mask of the input.
 575            Debug.Assert(UInt32EndsWithUtf8TwoByteMask(value));
 576
 577            // Per Table 3-7, first byte of two-byte sequence must be within range C2 .. DF.
 578            // We already validated that it's 80 .. DF (per mask check earlier).
 579            // C2 = 1100 0010
 580            // DF = 1101 1111
 581            // This means that we can AND the leading byte with the mask 0001 1110 (1E),
 582            // and if the result is zero the sequence is overlong.
 583
 584            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 585
 586            return (BitConverter.IsLittleEndian && ((value & 0x001E_0000u) == 0))
 587                || (!BitConverter.IsLittleEndian && ((value & 0x1E00u) == 0));
 588        }
 589
 590        /// <summary>
 591        /// Given a UTF-8 buffer which has been read into a DWORD in machine endianness,
 592        /// returns <see langword="true"/> iff the last two bytes of the buffer match
 593        /// the UTF-8 2-byte sequence mask [ 110yyyyy 10xxxxxx ]. This method *does not*
 594        /// validate that the sequence is well-formed; the caller must still perform
 595        /// overlong form checking.
 596        /// </summary>
 597        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 598        private static bool UInt32EndsWithUtf8TwoByteMask(uint value)
 599        {
 600            // The code in this method is equivalent to the code
 601            // below but is slightly more optimized.
 602            //
 603            // if (BitConverter.IsLittleEndian)
 604            // {
 605            //     const uint mask = 0xC0E00000U;
 606            //     const uint comparand = 0x80C00000U;
 607            //     return ((value & mask) == comparand);
 608            // }
 609            // else
 610            // {
 611            //     const uint mask = 0x0000E0C0U;
 612            //     const uint comparand = 0x0000C080U;
 613            //     return ((value & mask) == comparand);
 614            // }
 615
 616            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 617
 618            return (BitConverter.IsLittleEndian && (((value - 0x80C0_0000u) & 0xC0E0_0000u) == 0))
 619                || (!BitConverter.IsLittleEndian && (((value - 0x0000_C080u) & 0x0000_E0C0u) == 0));
 620        }
 621
 622        /// <summary>
 623        /// Given a UTF-8 buffer which has been read into a DWORD on a little-endian machine,
 624        /// returns <see langword="true"/> iff the first two bytes of the buffer are a well-formed
 625        /// UTF-8 two-byte sequence. This wraps the mask check and the overlong check into a
 626        /// single operation. Returns <see langword="false"/> if running on a big-endian machine.
 627        /// </summary>
 628        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 629        private static bool UInt32BeginsWithValidUtf8TwoByteSequenceLittleEndian(uint value)
 630        {
 631            // Per Table 3-7, valid 2-byte sequences are [ C2..DF ] [ 80..BF ].
 632            // In little-endian, that would be represented as:
 633            // [ ######## ######## 10xxxxxx 110yyyyy ].
 634            // Due to the little-endian representation we can perform a trick by ANDing the low
 635            // WORD with the bitmask [ 11000000 11111111 ] and checking that the value is within
 636            // the range [ 10000000_11000010, 10000000_11011111 ]. This performs both the
 637            // 2-byte-sequence bitmask check and overlong form validation with one comparison.
 638
 639            Debug.Assert(BitConverter.IsLittleEndian);
 640
 641            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 642
 46400643            return (BitConverter.IsLittleEndian && UnicodeUtility.IsInRangeInclusive(value & 0xC0FFu, 0x80C2u, 0x80DFu))
 46400644                || (!BitConverter.IsLittleEndian && false);
 645        }
 646
 647        /// <summary>
 648        /// Given a UTF-8 buffer which has been read into a DWORD on a little-endian machine,
 649        /// returns <see langword="true"/> iff the last two bytes of the buffer are a well-formed
 650        /// UTF-8 two-byte sequence. This wraps the mask check and the overlong check into a
 651        /// single operation. Returns <see langword="false"/> if running on a big-endian machine.
 652        /// </summary>
 653        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 654        private static bool UInt32EndsWithValidUtf8TwoByteSequenceLittleEndian(uint value)
 655        {
 656            // See comments in UInt32BeginsWithValidUtf8TwoByteSequenceLittleEndian.
 657
 658            Debug.Assert(BitConverter.IsLittleEndian);
 659
 660            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 661
 339784662            return (BitConverter.IsLittleEndian && UnicodeUtility.IsInRangeInclusive(value & 0xC0FF_0000u, 0x80C2_0000u,
 339784663                || (!BitConverter.IsLittleEndian && false);
 664        }
 665
 666        /// <summary>
 667        /// Given a UTF-8 buffer which has been read into a DWORD in machine endianness,
 668        /// returns <see langword="true"/> iff the first byte of the buffer is ASCII.
 669        /// </summary>
 670        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 671        private static bool UInt32FirstByteIsAscii(uint value)
 672        {
 673            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 674
 675            return (BitConverter.IsLittleEndian && ((value & 0x80u) == 0))
 676                || (!BitConverter.IsLittleEndian && ((int)value >= 0));
 677        }
 678
 679        /// <summary>
 680        /// Given a UTF-8 buffer which has been read into a DWORD in machine endianness,
 681        /// returns <see langword="true"/> iff the fourth byte of the buffer is ASCII.
 682        /// </summary>
 683        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 684        private static bool UInt32FourthByteIsAscii(uint value)
 685        {
 686            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 687
 688            return (BitConverter.IsLittleEndian && ((int)value >= 0))
 689                || (!BitConverter.IsLittleEndian && ((value & 0x80u) == 0));
 690        }
 691
 692        /// <summary>
 693        /// Given a UTF-8 buffer which has been read into a DWORD in machine endianness,
 694        /// returns <see langword="true"/> iff the second byte of the buffer is ASCII.
 695        /// </summary>
 696        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 697        private static bool UInt32SecondByteIsAscii(uint value)
 698        {
 699            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 700
 701            return (BitConverter.IsLittleEndian && ((value & 0x8000u) == 0))
 702                || (!BitConverter.IsLittleEndian && ((value & 0x0080_0000u) == 0));
 703        }
 704
 705        /// <summary>
 706        /// Given a UTF-8 buffer which has been read into a DWORD in machine endianness,
 707        /// returns <see langword="true"/> iff the third byte of the buffer is ASCII.
 708        /// </summary>
 709        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 710        private static bool UInt32ThirdByteIsAscii(uint value)
 711        {
 712            // Return statement is written this way to work around https://github.com/dotnet/runtime/issues/4207.
 713
 714            return (BitConverter.IsLittleEndian && ((value & 0x0080_0000u) == 0))
 715                || (!BitConverter.IsLittleEndian && ((value & 0x8000u) == 0));
 716        }
 717
 718        /// <summary>
 719        /// Given a DWORD which represents a buffer of 2 packed UTF-16 values in machine endianness,
 720        /// converts those scalar values to their 3-byte UTF-8 representation and writes the
 721        /// resulting 6 bytes to the destination buffer.
 722        /// </summary>
 723        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 724        private static void WriteTwoUtf16CharsAsTwoUtf8ThreeByteSequences(ref byte outputBuffer, uint value)
 725        {
 726            Debug.Assert(IsFirstCharAtLeastThreeUtf8Bytes(value) && !IsFirstCharSurrogate(value), "First half of value s
 0727            Debug.Assert(IsSecondCharAtLeastThreeUtf8Bytes(value) && !IsSecondCharSurrogate(value), "Second half of valu
 728
 0729            if (BitConverter.IsLittleEndian)
 730            {
 731                // value = [ ZZZZYYYY YYXXXXXX zzzzyyyy yyxxxxxx ]
 732                // want to write [ 1110ZZZZ 10xxxxxx 10yyyyyy 1110zzzz ] [ 10XXXXXX 10YYYYYY ]
 733
 0734                uint tempA = ((value << 2) & 0x3F00u) | ((value & 0x3Fu) << 16); // = [ 00000000 00xxxxxx 00yyyyyy 00000
 0735                uint tempB = ((value >> 4) & 0x0F00_0000u) | ((value >> 12) & 0x0Fu); // = [ 0000ZZZZ 00000000 00000000 
 0736                Unsafe.WriteUnaligned(ref outputBuffer, tempA + tempB + 0xE080_80E0u); // = [ 1110ZZZZ 10xxxxxx 10yyyyyy
 0737                Unsafe.WriteUnaligned(ref Unsafe.Add(ref outputBuffer, 4), (ushort)(((value >> 22) & 0x3Fu) + ((value >>
 738            }
 739            else
 740            {
 741                // value = [ zzzzyyyy yyxxxxxx ZZZZYYYY YYXXXXXX ]
 742                // want to write [ 1110zzzz ] [ 10yyyyyy ] [ 10xxxxxx ] [ 1110ZZZZ ] [ 10YYYYYY ] [ 10XXXXXX ]
 743
 744                Unsafe.Add(ref outputBuffer, 5) = (byte)((value & 0x3Fu) | 0x80u);
 745                Unsafe.Add(ref outputBuffer, 4) = (byte)(((value >>= 6) & 0x3Fu) | 0x80u);
 746                Unsafe.Add(ref outputBuffer, 3) = (byte)(((value >>= 6) & 0x0Fu) | 0xE0u);
 747                Unsafe.Add(ref outputBuffer, 2) = (byte)(((value >>= 4) & 0x3Fu) | 0x80u);
 748                Unsafe.Add(ref outputBuffer, 1) = (byte)(((value >>= 6) & 0x3Fu) | 0x80u);
 749                outputBuffer = (byte)((value >>= 6) | 0xE0u);
 750            }
 751        }
 752
 753        /// <summary>
 754        /// Given a DWORD which represents a buffer of 2 packed UTF-16 values in machine endianness,
 755        /// converts the first UTF-16 value to its 3-byte UTF-8 representation and writes the
 756        /// resulting 3 bytes to the destination buffer.
 757        /// </summary>
 758        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 759        private static void WriteFirstUtf16CharAsUtf8ThreeByteSequence(ref byte outputBuffer, uint value)
 760        {
 761            Debug.Assert(IsFirstCharAtLeastThreeUtf8Bytes(value) && !IsFirstCharSurrogate(value), "First half of value s
 762
 0763            if (BitConverter.IsLittleEndian)
 764            {
 765                // value = [ ######## ######## zzzzyyyy yyxxxxxx ]
 766                // want to write [ 10yyyyyy 1110zzzz ] [ 10xxxxxx ]
 767
 0768                uint tempA = (value << 2) & 0x3F00u; // [ 00yyyyyy 00000000 ]
 0769                uint tempB = ((uint)(ushort)value >> 12); // [ 00000000 0000zzzz ]
 0770                Unsafe.WriteUnaligned(ref outputBuffer, (ushort)(tempA + tempB + 0x80E0u)); // [ 10yyyyyy 1110zzzz ]
 0771                Unsafe.Add(ref outputBuffer, 2) = (byte)((value & 0x3Fu) | ~0x7Fu); // [ 10xxxxxx ]
 772            }
 773            else
 774            {
 775                // value = [ zzzzyyyy yyxxxxxx ######## ######## ]
 776                // want to write [ 1110zzzz ] [ 10yyyyyy ] [ 10xxxxxx ]
 777
 778                Unsafe.Add(ref outputBuffer, 2) = (byte)(((value >>= 16) & 0x3Fu) | 0x80u);
 779                Unsafe.Add(ref outputBuffer, 1) = (byte)(((value >>= 6) & 0x3Fu) | 0x80u);
 780                outputBuffer = (byte)((value >>= 6) | 0xE0u);
 781            }
 782        }
 783    }
 784}
 785

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.CoreLib/src/System/Text/Unicode/Utf8Utility.Transcoding.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Buffers;
 5using System.Buffers.Text;
 6using System.Diagnostics;
 7using System.Diagnostics.CodeAnalysis;
 8using System.Numerics;
 9using System.Runtime.CompilerServices;
 10#if NET
 11using System.Runtime.Intrinsics;
 12using System.Runtime.Intrinsics.Arm;
 13using System.Runtime.Intrinsics.Wasm;
 14using System.Runtime.Intrinsics.X86;
 15#endif
 16
 17namespace System.Text.Unicode
 18{
 19    internal static unsafe partial class Utf8Utility
 20    {
 21        // On method return, pInputBufferRemaining and pOutputBufferRemaining will both point to where
 22        // the next byte would have been consumed from / the next char would have been written to.
 23        // inputLength in bytes, outputCharsRemaining in chars.
 24        public static OperationStatus TranscodeToUtf16(byte* pInputBuffer, int inputLength, char* pOutputBuffer, int out
 25        {
 26            Debug.Assert(inputLength >= 0, "Input length must not be negative.");
 202136827            Debug.Assert(pInputBuffer != null || inputLength == 0, "Input length must be zero if input buffer pointer is
 28
 202136829            Debug.Assert(outputCharsRemaining >= 0, "Destination length must not be negative.");
 202136830            Debug.Assert(pOutputBuffer != null || outputCharsRemaining == 0, "Destination length must be zero if destina
 31
 32            // First, try vectorized conversion.
 33            {
 202136834                nuint numElementsConverted = Ascii.WidenAsciiToUtf16(pInputBuffer, pOutputBuffer, (uint)Math.Min(inputLe
 35
 202136836                pInputBuffer += numElementsConverted;
 202136837                pOutputBuffer += numElementsConverted;
 38
 39                // Quick check - did we just end up consuming the entire input buffer?
 40                // If so, short-circuit the remainder of the method.
 41
 202136842                if ((int)numElementsConverted == inputLength)
 43                {
 941144                    pInputBufferRemaining = pInputBuffer;
 941145                    pOutputBufferRemaining = pOutputBuffer;
 941146                    return OperationStatus.Done;
 47                }
 48
 201195749                inputLength -= (int)numElementsConverted;
 201195750                outputCharsRemaining -= (int)numElementsConverted;
 51            }
 52
 201195753            if (inputLength < sizeof(uint))
 54            {
 55                goto ProcessInputOfLessThanDWordSize;
 56            }
 57
 197318558            byte* pFinalPosWhereCanReadDWordFromInputBuffer = pInputBuffer + (uint)inputLength - 4;
 59
 60            // Begin the main loop.
 61
 62#if DEBUG
 197318563            byte* pLastBufferPosProcessed = null; // used for invariant checking in debug builds
 64#endif
 65
 197318566            Debug.Assert(pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer);
 67            do
 68            {
 69                // Read 32 bits at a time. This is enough to hold any possible UTF8-encoded scalar.
 70
 218290471                uint thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 72
 73            AfterReadDWord:
 74
 75#if DEBUG
 236717676                Debug.Assert(pLastBufferPosProcessed < pInputBuffer, "Algorithm should've made forward progress since la
 236717677                pLastBufferPosProcessed = pInputBuffer;
 78#endif
 79                // First, check for the common case of all-ASCII bytes.
 80
 236717681                if (Ascii.AllBytesInUInt32AreAscii(thisDWord))
 82                {
 83                    // We read an all-ASCII sequence.
 84
 8448085                    if (outputCharsRemaining < sizeof(uint))
 86                    {
 87                        goto ProcessRemainingBytesSlow; // running out of space, but may be able to write some data
 88                    }
 89
 8448090                    Ascii.WidenFourAsciiBytesToUtf16AndWriteToBuffer(ref *pOutputBuffer, thisDWord);
 8448091                    pInputBuffer += 4;
 8448092                    pOutputBuffer += 4;
 8448093                    outputCharsRemaining -= 4;
 94
 95                    // If we saw a sequence of all ASCII, there's a good chance a significant amount of following data i
 96                    // Below is basically unrolled loops with poor man's vectorization.
 97
 8448098                    uint remainingInputBytes = (uint)(void*)Unsafe.ByteOffset(ref *pInputBuffer, ref *pFinalPosWhereCanR
 8448099                    uint maxIters = Math.Min(remainingInputBytes, (uint)outputCharsRemaining) / (2 * sizeof(uint));
 100                    uint secondDWord;
 101                    int i;
 210648102                    for (i = 0; (uint)i < maxIters; i++)
 103                    {
 104                        // Reading two DWORDs in parallel benchmarked faster than reading a single QWORD.
 105
 103415106                        thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 103415107                        secondDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer + sizeof(uint));
 108
 103415109                        if (!Ascii.AllBytesInUInt32AreAscii(thisDWord | secondDWord))
 110                        {
 111                            goto LoopTerminatedEarlyDueToNonAsciiData;
 112                        }
 113
 20844114                        pInputBuffer += 8;
 115
 20844116                        Ascii.WidenFourAsciiBytesToUtf16AndWriteToBuffer(ref pOutputBuffer[0], thisDWord);
 20844117                        Ascii.WidenFourAsciiBytesToUtf16AndWriteToBuffer(ref pOutputBuffer[4], secondDWord);
 118
 20844119                        pOutputBuffer += 8;
 120                    }
 121
 1909122                    outputCharsRemaining -= 8 * i;
 123
 1909124                    continue; // need to perform a bounds check because we might be running out of data
 125
 126                LoopTerminatedEarlyDueToNonAsciiData:
 127
 82571128                    if (Ascii.AllBytesInUInt32AreAscii(thisDWord))
 129                    {
 130                        // The first DWORD contained all-ASCII bytes, so expand it.
 131
 41742132                        Ascii.WidenFourAsciiBytesToUtf16AndWriteToBuffer(ref *pOutputBuffer, thisDWord);
 133
 134                        // continue the outer loop from the second DWORD
 135
 41742136                        Debug.Assert(!Ascii.AllBytesInUInt32AreAscii(secondDWord));
 41742137                        thisDWord = secondDWord;
 138
 41742139                        pInputBuffer += 4;
 41742140                        pOutputBuffer += 4;
 41742141                        outputCharsRemaining -= 4;
 142                    }
 143
 82571144                    outputCharsRemaining -= 8 * i;
 145
 146                    // We know that there's *at least* one DWORD of data remaining in the buffer.
 147                    // We also know that it's not all-ASCII. We can skip the logic at the beginning of the main loop.
 148
 149                    goto AfterReadDWordSkipAllBytesAsciiCheck;
 150                }
 151
 152            AfterReadDWordSkipAllBytesAsciiCheck:
 153
 2365267154                Debug.Assert(!Ascii.AllBytesInUInt32AreAscii(thisDWord)); // this should have been handled earlier
 155
 156                // Next, try stripping off ASCII bytes one at a time.
 157                // We only handle up to three ASCII bytes here since we handled the four ASCII byte case above.
 158
 2365267159                if (UInt32FirstByteIsAscii(thisDWord))
 160                {
 73572161                    if (outputCharsRemaining >= 3)
 162                    {
 163                        // Fast-track: we don't need to check the destination length for subsequent
 164                        // ASCII bytes since we know we can write them all now.
 165
 73462166                        uint thisDWordLittleEndian = ToLittleEndian(thisDWord);
 167
 73462168                        nuint adjustment = 1;
 73462169                        pOutputBuffer[0] = (char)(byte)thisDWordLittleEndian;
 170
 73462171                        if (UInt32SecondByteIsAscii(thisDWord))
 172                        {
 43806173                            adjustment++;
 43806174                            thisDWordLittleEndian >>= 8;
 43806175                            pOutputBuffer[1] = (char)(byte)thisDWordLittleEndian;
 176
 43806177                            if (UInt32ThirdByteIsAscii(thisDWord))
 178                            {
 21096179                                adjustment++;
 21096180                                thisDWordLittleEndian >>= 8;
 21096181                                pOutputBuffer[2] = (char)(byte)thisDWordLittleEndian;
 182                            }
 183                        }
 184
 73462185                        pInputBuffer += adjustment;
 73462186                        pOutputBuffer += adjustment;
 73462187                        outputCharsRemaining -= (int)adjustment;
 188                    }
 189                    else
 190                    {
 191                        // Slow-track: we need to make sure each individual write has enough
 192                        // of a buffer so that we don't overrun the destination.
 193
 110194                        if (outputCharsRemaining == 0)
 195                        {
 196                            goto OutputBufferTooSmall;
 197                        }
 198
 110199                        uint thisDWordLittleEndian = ToLittleEndian(thisDWord);
 200
 110201                        pInputBuffer++;
 110202                        *pOutputBuffer++ = (char)(byte)thisDWordLittleEndian;
 110203                        outputCharsRemaining--;
 204
 110205                        if (UInt32SecondByteIsAscii(thisDWord))
 206                        {
 0207                            if (outputCharsRemaining == 0)
 208                            {
 209                                goto OutputBufferTooSmall;
 210                            }
 211
 0212                            pInputBuffer++;
 0213                            thisDWordLittleEndian >>= 8;
 0214                            *pOutputBuffer++ = (char)(byte)thisDWordLittleEndian;
 215
 216                            // We can perform a small optimization here. We know at this point that
 217                            // the output buffer is fully consumed (we read two ASCII bytes and wrote
 218                            // two ASCII chars, and we checked earlier that the destination buffer
 219                            // can't store a third byte). If the next byte is ASCII, we can jump straight
 220                            // to the return statement since the end-of-method logic only relies on the
 221                            // destination buffer pointer -- NOT the output chars remaining count -- being
 222                            // correct. If the next byte is not ASCII, we'll need to continue with the
 223                            // rest of the main loop, but we can set the buffer length directly to zero
 224                            // rather than decrementing it from 1 to 0.
 225
 0226                            Debug.Assert(outputCharsRemaining == 1);
 227
 0228                            if (UInt32ThirdByteIsAscii(thisDWord))
 229                            {
 230                                goto OutputBufferTooSmall;
 231                            }
 232                            else
 233                            {
 0234                                outputCharsRemaining = 0;
 235                            }
 236                        }
 237                    }
 238
 73572239                    if (pInputBuffer > pFinalPosWhereCanReadDWordFromInputBuffer)
 240                    {
 241                        goto ProcessRemainingBytesSlow; // input buffer doesn't contain enough data to read a DWORD
 242                    }
 243                    else
 244                    {
 245                        // The input buffer at the current offset contains a non-ASCII byte.
 246                        // Read an entire DWORD and fall through to multi-byte consumption logic.
 71946247                        thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 248                    }
 249                }
 250
 251            BeforeProcessTwoByteSequence:
 252
 253                // At this point, we know we're working with a multi-byte code unit,
 254                // but we haven't yet validated it.
 255
 256                // The masks and comparands are derived from the Unicode Standard, Table 3-6.
 257                // Additionally, we need to check for valid byte sequences per Table 3-7.
 258
 259                // Check the 2-byte case.
 260
 2420215261                if (UInt32BeginsWithUtf8TwoByteMask(thisDWord))
 262                {
 263                    // Per Table 3-7, valid sequences are:
 264                    // [ C2..DF ] [ 80..BF ]
 265
 280471266                    if (UInt32BeginsWithOverlongUtf8TwoByteSequence(thisDWord))
 267                    {
 268                        goto Error;
 269                    }
 270
 271                ProcessTwoByteSequenceSkipOverlongFormCheck:
 272
 273                    // Optimization: If this is a two-byte-per-character language like Cyrillic or Hebrew,
 274                    // there's a good chance that if we see one two-byte run then there's another two-byte
 275                    // run immediately after. Let's check that now.
 276
 277                    // On little-endian platforms, we can check for the two-byte UTF8 mask *and* validate that
 278                    // the value isn't overlong using a single comparison. On big-endian platforms, we'll need
 279                    // to validate the mask and validate that the sequence isn't overlong as two separate comparisons.
 280
 254838281                    if ((BitConverter.IsLittleEndian && UInt32EndsWithValidUtf8TwoByteSequenceLittleEndian(thisDWord))
 254838282                        || (!BitConverter.IsLittleEndian && (UInt32EndsWithUtf8TwoByteMask(thisDWord) && !UInt32EndsWith
 283                    {
 284                        // We have two runs of two bytes each.
 285
 35112286                        if (outputCharsRemaining < 2)
 287                        {
 288                            goto ProcessRemainingBytesSlow; // running out of output buffer
 289                        }
 290
 35112291                        Unsafe.WriteUnaligned(pOutputBuffer, ExtractTwoCharsPackedFromTwoAdjacentTwoByteSequences(thisDW
 292
 35112293                        pInputBuffer += 4;
 35112294                        pOutputBuffer += 2;
 35112295                        outputCharsRemaining -= 2;
 296
 35112297                        if (pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer)
 298                        {
 299                            // Optimization: If we read a long run of two-byte sequences, the next sequence is probably
 300                            // also two bytes. Check for that first before going back to the beginning of the loop.
 301
 34800302                            thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 303
 34800304                            if (BitConverter.IsLittleEndian)
 305                            {
 34800306                                if (UInt32BeginsWithValidUtf8TwoByteSequenceLittleEndian(thisDWord))
 307                                {
 308                                    // The next sequence is a valid two-byte sequence.
 20430309                                    goto ProcessTwoByteSequenceSkipOverlongFormCheck;
 310                                }
 311                            }
 312                            else
 313                            {
 314                                if (UInt32BeginsWithUtf8TwoByteMask(thisDWord))
 315                                {
 316                                    if (UInt32BeginsWithOverlongUtf8TwoByteSequence(thisDWord))
 317                                    {
 318                                        goto Error; // The next sequence purports to be a 2-byte sequence but is overlon
 319                                    }
 320
 321                                    goto ProcessTwoByteSequenceSkipOverlongFormCheck;
 322                                }
 323                            }
 324
 325                            // If we reached this point, the next sequence is something other than a valid
 326                            // two-byte sequence, so go back to the beginning of the loop.
 327                            goto AfterReadDWord;
 328                        }
 329                        else
 330                        {
 331                            goto ProcessRemainingBytesSlow; // Running out of data - go down slow path
 332                        }
 333                    }
 334
 335                    // The buffer contains a 2-byte sequence followed by 2 bytes that aren't a 2-byte sequence.
 336                    // Unlikely that a 3-byte sequence would follow a 2-byte sequence, so perhaps remaining
 337                    // bytes are ASCII?
 338
 219726339                    uint charToWrite = ExtractCharFromFirstTwoByteSequence(thisDWord); // optimistically compute this no
 340
 219726341                    if (UInt32ThirdByteIsAscii(thisDWord))
 342                    {
 185688343                        if (UInt32FourthByteIsAscii(thisDWord))
 344                        {
 128004345                            if (outputCharsRemaining < 3)
 346                            {
 347                                goto ProcessRemainingBytesSlow; // running out of output buffer
 348                            }
 349
 128004350                            pOutputBuffer[0] = (char)charToWrite;
 128004351                            if (BitConverter.IsLittleEndian)
 352                            {
 128004353                                thisDWord >>= 16;
 128004354                                pOutputBuffer[1] = (char)(byte)thisDWord;
 128004355                                thisDWord >>= 8;
 128004356                                pOutputBuffer[2] = (char)thisDWord;
 357                            }
 358                            else
 359                            {
 360                                pOutputBuffer[2] = (char)(byte)thisDWord;
 361                                pOutputBuffer[1] = (char)(byte)(thisDWord >> 8);
 362                            }
 128004363                            pInputBuffer += 4;
 128004364                            pOutputBuffer += 3;
 128004365                            outputCharsRemaining -= 3;
 366
 128004367                            continue; // go back to original bounds check and check for ASCII
 368                        }
 369                        else
 370                        {
 57684371                            if (outputCharsRemaining < 2)
 372                            {
 373                                goto ProcessRemainingBytesSlow; // running out of output buffer
 374                            }
 375
 57684376                            pOutputBuffer[0] = (char)charToWrite;
 57684377                            pOutputBuffer[1] = (char)(byte)(thisDWord >> (BitConverter.IsLittleEndian ? 16 : 8));
 57684378                            pInputBuffer += 3;
 57684379                            pOutputBuffer += 2;
 57684380                            outputCharsRemaining -= 2;
 381
 382                            // A two-byte sequence followed by an ASCII byte followed by a non-ASCII byte.
 383                            // Read in the next DWORD and jump directly to the start of the multi-byte processing block.
 384
 57684385                            if (pFinalPosWhereCanReadDWordFromInputBuffer < pInputBuffer)
 386                            {
 387                                goto ProcessRemainingBytesSlow; // Running out of data - go down slow path
 388                            }
 389                            else
 390                            {
 56574391                                thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 56574392                                goto BeforeProcessTwoByteSequence;
 393                            }
 394                        }
 395                    }
 396                    else
 397                    {
 34038398                        if (outputCharsRemaining == 0)
 399                        {
 400                            goto ProcessRemainingBytesSlow; // running out of output buffer
 401                        }
 402
 34038403                        pOutputBuffer[0] = (char)charToWrite;
 34038404                        pInputBuffer += 2;
 34038405                        pOutputBuffer++;
 34038406                        outputCharsRemaining--;
 407
 34038408                        if (pFinalPosWhereCanReadDWordFromInputBuffer < pInputBuffer)
 409                        {
 410                            goto ProcessRemainingBytesSlow; // Running out of data - go down slow path
 411                        }
 412                        else
 413                        {
 33360414                            thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 415                            goto BeforeProcessThreeByteSequence; // we know the next byte isn't ASCII, and it's not the 
 416                        }
 417                    }
 418                }
 419
 420            // Check the 3-byte case.
 421
 422            BeforeProcessThreeByteSequence:
 423
 2173104424                if (UInt32BeginsWithUtf8ThreeByteMask(thisDWord))
 425                {
 426                ProcessThreeByteSequenceWithCheck:
 427
 428                    // We need to check for overlong or surrogate three-byte sequences.
 429                    //
 430                    // Per Table 3-7, valid sequences are:
 431                    // [   E0   ] [ A0..BF ] [ 80..BF ]
 432                    // [ E1..EC ] [ 80..BF ] [ 80..BF ]
 433                    // [   ED   ] [ 80..9F ] [ 80..BF ]
 434                    // [ EE..EF ] [ 80..BF ] [ 80..BF ]
 435                    //
 436                    // Big-endian examples of using the above validation table:
 437                    // E0A0 = 1110 0000 1010 0000 => invalid (overlong ) patterns are 1110 0000 100# ####
 438                    // ED9F = 1110 1101 1001 1111 => invalid (surrogate) patterns are 1110 1101 101# ####
 439                    // If using the bitmask ......................................... 0000 1111 0010 0000 (=0F20),
 440                    // Then invalid (overlong) patterns match the comparand ......... 0000 0000 0000 0000 (=0000),
 441                    // And invalid (surrogate) patterns match the comparand ......... 0000 1101 0010 0000 (=0D20).
 442
 623462443                    if (BitConverter.IsLittleEndian)
 444                    {
 445                        // The "overlong or surrogate" check can be implemented using a single jump, but there's
 446                        // some overhead to moving the bits into the correct locations in order to perform the
 447                        // correct comparison, and in practice the processor's branch prediction capability is
 448                        // good enough that we shouldn't bother. So we'll use two jumps instead.
 449
 450                        // Can't extract this check into its own helper method because JITter produces suboptimal
 451                        // assembly, even with aggressive inlining.
 452
 453                        // Code below becomes 5 instructions: test, jz, lea, test, jz
 454
 623462455                        if (((thisDWord & 0x0000_200Fu) == 0) || (((thisDWord - 0x0000_200Du) & 0x0000_200Fu) == 0))
 456                        {
 152911457                            goto Error; // overlong or surrogate
 458                        }
 459                    }
 460                    else
 461                    {
 462                        if (((thisDWord & 0x0F20_0000u) == 0) || (((thisDWord - 0x0D20_0000u) & 0x0F20_0000u) == 0))
 463                        {
 464                            goto Error; // overlong or surrogate
 465                        }
 466                    }
 467
 468                    // At this point, we know the incoming scalar is well-formed.
 469
 413334470                    if (outputCharsRemaining == 0)
 471                    {
 472                        goto OutputBufferTooSmall; // not enough space in the destination buffer to write
 473                    }
 474
 475                    // As an optimization, on compatible platforms check if a second three-byte sequence immediately
 476                    // follows the one we just read, and if so extract them together.
 477
 413334478                    if (BitConverter.IsLittleEndian)
 479                    {
 480                        // First, check that the leftover byte from the original DWORD is in the range [ E0..EF ], which
 481                        // would indicate the potential start of a second three-byte sequence.
 482
 413334483                        if (((thisDWord - 0xE000_0000u) & 0xF000_0000u) == 0)
 484                        {
 485                            // The const '3' below is correct because pFinalPosWhereCanReadDWordFromInputBuffer represen
 486                            // the final place where we can safely perform a DWORD read, and we want to probe whether it
 487                            // safe to read a DWORD beginning at address &pInputBuffer[3].
 488
 311112489                            if (outputCharsRemaining > 1 && (nint)(void*)Unsafe.ByteOffset(ref *pInputBuffer, ref *pFina
 490                            {
 491                                // We're going to attempt to read a second 3-byte sequence and write them both out one a
 492                                // We need to check the continuation bit mask on the remaining two bytes (and we may as 
 493                                // byte mask again since it's free), then perform overlong + surrogate checks. If the ov
 494                                // checks fail, we'll fall through to the remainder of the logic which will transcode th
 495                                // 3-byte UTF-8 sequence we read; and on the next iteration of the loop the validation r
 496                                // fail, and redirect control flow to the error handling logic at the very end of this m
 497
 310284498                                uint secondDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer + 3);
 499
 310284500                                if (UInt32BeginsWithUtf8ThreeByteMask(secondDWord)
 310284501                                    && ((secondDWord & 0x0000_200Fu) != 0)
 310284502                                    && (((secondDWord - 0x0000_200Du) & 0x0000_200Fu) != 0))
 503                                {
 152370504                                    pOutputBuffer[0] = (char)ExtractCharFromFirstThreeByteSequence(thisDWord);
 152370505                                    pOutputBuffer[1] = (char)ExtractCharFromFirstThreeByteSequence(secondDWord);
 152370506                                    pInputBuffer += 6;
 152370507                                    pOutputBuffer += 2;
 152370508                                    outputCharsRemaining -= 2;
 509
 510                                    // Drain any ASCII data following the second three-byte sequence.
 511
 152370512                                    goto CheckForAsciiByteAfterThreeByteSequence;
 513                                }
 514                            }
 515                        }
 516                    }
 517
 518                    // Couldn't extract 2x three-byte sequences together, just do this one by itself.
 519
 260964520                    *pOutputBuffer = (char)ExtractCharFromFirstThreeByteSequence(thisDWord);
 260964521                    pInputBuffer += 3;
 260964522                    pOutputBuffer++;
 260964523                    outputCharsRemaining--;
 524
 525                CheckForAsciiByteAfterThreeByteSequence:
 526
 527                    // Occasionally one-off ASCII characters like spaces, periods, or newlines will make their way
 528                    // in to the text. If this happens strip it off now before seeing if the next character
 529                    // consists of three code units.
 530
 413334531                    if (UInt32FourthByteIsAscii(thisDWord))
 532                    {
 40740533                        if (outputCharsRemaining == 0)
 534                        {
 535                            goto OutputBufferTooSmall;
 536                        }
 537
 40740538                        if (BitConverter.IsLittleEndian)
 539                        {
 40740540                            *pOutputBuffer = (char)(thisDWord >> 24);
 541                        }
 542                        else
 543                        {
 544                            *pOutputBuffer = (char)(byte)thisDWord;
 545                        }
 546
 40740547                        pInputBuffer++;
 40740548                        pOutputBuffer++;
 40740549                        outputCharsRemaining--;
 550                    }
 551
 413334552                    if (pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer)
 553                    {
 409392554                        thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 555
 556                        // Optimization: A three-byte character could indicate CJK text, which makes it likely
 557                        // that the character following this one is also CJK. We'll check for a three-byte sequence
 558                        // marker now and jump directly to three-byte sequence processing if we see one, skipping
 559                        // all of the logic at the beginning of the loop.
 560
 409392561                        if (UInt32BeginsWithUtf8ThreeByteMask(thisDWord))
 562                        {
 239490563                            goto ProcessThreeByteSequenceWithCheck; // found a three-byte sequence marker; validate and 
 564                        }
 565                        else
 566                        {
 567                            goto AfterReadDWord; // probably ASCII punctuation or whitespace
 568                        }
 569                    }
 570                    else
 571                    {
 572                        goto ProcessRemainingBytesSlow; // Running out of data - go down slow path
 573                    }
 574                }
 575
 576                // Assume the 4-byte case, but we need to validate.
 577
 578                {
 579                    // We need to check for overlong or invalid (over U+10FFFF) four-byte sequences.
 580                    //
 581                    // Per Table 3-7, valid sequences are:
 582                    // [   F0   ] [ 90..BF ] [ 80..BF ] [ 80..BF ]
 583                    // [ F1..F3 ] [ 80..BF ] [ 80..BF ] [ 80..BF ]
 584                    // [   F4   ] [ 80..8F ] [ 80..BF ] [ 80..BF ]
 585
 1789132586                    if (!UInt32BeginsWithUtf8FourByteMask(thisDWord))
 587                    {
 588                        goto Error;
 589                    }
 590
 591                    // Now check for overlong / out-of-range sequences.
 592
 83558593                    if (BitConverter.IsLittleEndian)
 594                    {
 595                        // The DWORD we read is [ 10xxxxxx 10yyyyyy 10zzzzzz 11110www ].
 596                        // We want to get the 'w' byte in front of the 'z' byte so that we can perform
 597                        // a single range comparison. We'll take advantage of the fact that the JITter
 598                        // can detect a ROR / ROL operation, then we'll just zero out the bytes that
 599                        // aren't involved in the range check.
 600
 83558601                        uint toCheck = thisDWord & 0x0000_FFFFu;
 602
 603                        // At this point, toCheck = [ 00000000 00000000 10zzzzzz 11110www ].
 604
 83558605                        toCheck = BitOperations.RotateRight(toCheck, 8);
 606
 607                        // At this point, toCheck = [ 11110www 00000000 00000000 10zzzzzz ].
 608
 83558609                        if (!UnicodeUtility.IsInRangeInclusive(toCheck, 0xF000_0090u, 0xF400_008Fu))
 610                        {
 1868611                            goto Error;
 612                        }
 613                    }
 614                    else
 615                    {
 616                        if (!UnicodeUtility.IsInRangeInclusive(thisDWord, 0xF090_0000u, 0xF48F_FFFFu))
 617                        {
 618                            goto Error;
 619                        }
 620                    }
 621
 622                    // Validation complete.
 623
 81690624                    if (outputCharsRemaining < 2)
 625                    {
 626                        // There's no point to falling back to the "drain the input buffer" logic, since we know
 627                        // we can't write anything to the destination. So we'll just exit immediately.
 628                        goto OutputBufferTooSmall;
 629                    }
 630
 81690631                    Unsafe.WriteUnaligned(pOutputBuffer, ExtractCharsFromFourByteSequence(thisDWord));
 632
 81690633                    pInputBuffer += 4;
 81690634                    pOutputBuffer += 2;
 81690635                    outputCharsRemaining -= 2;
 636
 637                    continue; // go back to beginning of loop for processing
 638                }
 211603639            } while (pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer);
 640
 641        ProcessRemainingBytesSlow:
 9552642            inputLength = (int)(void*)Unsafe.ByteOffset(ref *pInputBuffer, ref *pFinalPosWhereCanReadDWordFromInputBuffe
 643
 644        ProcessInputOfLessThanDWordSize:
 55584645            while (inputLength > 0)
 646            {
 49746647                uint firstByte = pInputBuffer[0];
 49746648                if (firstByte <= 0x7Fu)
 649                {
 2784650                    if (outputCharsRemaining == 0)
 651                    {
 652                        goto OutputBufferTooSmall; // we have no hope of writing anything to the output
 653                    }
 654
 655                    // 1-byte (ASCII) case
 2784656                    *pOutputBuffer = (char)firstByte;
 657
 2784658                    pInputBuffer++;
 2784659                    pOutputBuffer++;
 2784660                    inputLength--;
 2784661                    outputCharsRemaining--;
 2784662                    continue;
 663                }
 664
 665                // Potentially the start of a multi-byte sequence?
 666
 46962667                firstByte -= 0xC2u;
 46962668                if ((byte)firstByte <= (0xDFu - 0xC2u))
 669                {
 670                    // Potentially a 2-byte sequence?
 6298671                    if (inputLength < 2)
 672                    {
 673                        goto InputBufferTooSmall; // out of data
 674                    }
 675
 5397676                    uint secondByte = pInputBuffer[1];
 5397677                    if (!IsLowByteUtf8ContinuationByte(secondByte))
 678                    {
 679                        goto Error; // 2-byte marker not followed by continuation byte
 680                    }
 681
 3246682                    if (outputCharsRemaining == 0)
 683                    {
 684                        goto OutputBufferTooSmall; // we have no hope of writing anything to the output
 685                    }
 686
 3246687                    uint asChar = (firstByte << 6) + secondByte + ((0xC2u - 0xC0u) << 6) - 0x80u; // remove UTF-8 marker
 3246688                    *pOutputBuffer = (char)asChar;
 689
 3246690                    pInputBuffer += 2;
 3246691                    pOutputBuffer++;
 3246692                    inputLength -= 2;
 3246693                    outputCharsRemaining--;
 3246694                    continue;
 695                }
 40664696                else if ((byte)firstByte <= (0xEFu - 0xC2u))
 697                {
 698                    // Potentially a 3-byte sequence?
 8616699                    if (inputLength >= 3)
 700                    {
 3534701                        uint secondByte = pInputBuffer[1];
 3534702                        uint thirdByte = pInputBuffer[2];
 3534703                        if (!IsLowByteUtf8ContinuationByte(secondByte) || !IsLowByteUtf8ContinuationByte(thirdByte))
 704                        {
 705                            goto Error; // 3-byte marker not followed by 2 continuation bytes
 706                        }
 707
 708                        // To speed up the validation logic below, we're not going to remove the UTF-8 markers from the 
 709                        // We account for this in the comparisons below.
 710
 1536711                        uint partialChar = (firstByte << 12) + (secondByte << 6);
 1536712                        if (partialChar < ((0xE0u - 0xC2u) << 12) + (0xA0u << 6))
 713                        {
 714                            goto Error; // this is an overlong encoding; fail
 715                        }
 716
 1317717                        partialChar -= ((0xEDu - 0xC2u) << 12) + (0xA0u << 6); // if partialChar = 0, we're at beginning
 1317718                        if (partialChar < 0x0800u /* number of code points in UTF-16 surrogate code point range */)
 719                        {
 720                            goto Error; // attempted to encode a UTF-16 surrogate code point; fail
 721                        }
 722
 1230723                        if (outputCharsRemaining == 0)
 724                        {
 725                            goto OutputBufferTooSmall; // we have no hope of writing anything to the output
 726                        }
 727
 728                        // Now restore the full scalar value.
 729
 1230730                        partialChar += thirdByte;
 1230731                        partialChar += 0xD800; // undo "move to beginning of UTF-16 surrogate code point range" from ear
 1230732                        partialChar -= 0x80u; // remove third byte continuation marker
 733
 1230734                        *pOutputBuffer = (char)partialChar;
 735
 1230736                        pInputBuffer += 3;
 1230737                        pOutputBuffer++;
 1230738                        inputLength -= 3;
 1230739                        outputCharsRemaining--;
 1230740                        continue;
 741                    }
 5082742                    else if (inputLength >= 2)
 743                    {
 2649744                        uint secondByte = pInputBuffer[1];
 2649745                        if (!IsLowByteUtf8ContinuationByte(secondByte))
 746                        {
 747                            goto Error; // 3-byte marker not followed by continuation byte
 748                        }
 749
 750                        // We can't build up the entire scalar value now, but we can check for overlong / surrogate repr
 751                        // from just the first two bytes.
 752
 1881753                        uint partialChar = (firstByte << 6) + secondByte; // don't worry about fixing up the UTF-8 marke
 1881754                        if (partialChar < ((0xE0u - 0xC2u) << 6) + 0xA0u)
 755                        {
 756                            goto Error; // failed overlong check
 757                        }
 918758                        if (UnicodeUtility.IsInRangeInclusive(partialChar, ((0xEDu - 0xC2u) << 6) + 0xA0u, ((0xEEu - 0xC
 759                        {
 321760                            goto Error; // failed surrogate check
 761                        }
 762                    }
 763
 764                    goto InputBufferTooSmall; // out of data
 765                }
 32048766                else if ((byte)firstByte <= (0xF4u - 0xC2u))
 767                {
 768                    // Potentially a 4-byte sequence?
 769
 2602770                    if (inputLength < 2)
 771                    {
 772                        goto InputBufferTooSmall; // ran out of data
 773                    }
 774
 1748775                    uint nextByte = pInputBuffer[1];
 1748776                    if (!IsLowByteUtf8ContinuationByte(nextByte))
 777                    {
 778                        goto Error; // 4-byte marker not followed by a continuation byte
 779                    }
 780
 1285781                    uint asPartialChar = (firstByte << 6) + nextByte; // don't worry about fixing up the UTF-8 markers; 
 1285782                    if (!UnicodeUtility.IsInRangeInclusive(asPartialChar, ((0xF0u - 0xC2u) << 6) + 0x90u, ((0xF4u - 0xC2
 783                    {
 784                        goto Error; // failed overlong / out-of-range check
 785                    }
 786
 1239787                    if (inputLength < 3)
 788                    {
 789                        goto InputBufferTooSmall; // ran out of data
 790                    }
 791
 1147792                    if (!IsLowByteUtf8ContinuationByte(pInputBuffer[2]))
 793                    {
 794                        goto Error; // third byte in 4-byte sequence not a continuation byte
 795                    }
 796
 1106797                    if (inputLength < 4)
 798                    {
 799                        goto InputBufferTooSmall; // ran out of data
 800                    }
 801
 0802                    if (!IsLowByteUtf8ContinuationByte(pInputBuffer[3]))
 803                    {
 0804                        goto Error; // fourth byte in 4-byte sequence not a continuation byte
 805                    }
 806
 807                    // If we read a valid astral scalar value, the only way we could've fallen down this code path
 808                    // is that we didn't have enough output buffer to write the result.
 809
 810                    goto OutputBufferTooSmall;
 811                }
 812                else
 813                {
 814                    goto Error; // didn't begin with [ C2 .. F4 ], so invalid multi-byte sequence header byte
 815                }
 816            }
 817
 5838818            OperationStatus retVal = OperationStatus.Done;
 5838819            goto ReturnCommon;
 820
 821        InputBufferTooSmall:
 5983822            retVal = OperationStatus.NeedMoreData;
 5983823            goto ReturnCommon;
 824
 825        OutputBufferTooSmall:
 0826            retVal = OperationStatus.DestinationTooSmall;
 0827            goto ReturnCommon;
 828
 829        Error:
 2000136830            retVal = OperationStatus.InvalidData;
 831            goto ReturnCommon;
 832
 833        ReturnCommon:
 2011957834            pInputBufferRemaining = pInputBuffer;
 2011957835            pOutputBufferRemaining = pOutputBuffer;
 2011957836            return retVal;
 837        }
 838
 839        // On method return, pInputBufferRemaining and pOutputBufferRemaining will both point to where
 840        // the next char would have been consumed from / the next byte would have been written to.
 841        // inputLength in chars, outputBytesRemaining in bytes.
 842        public static OperationStatus TranscodeToUtf8(char* pInputBuffer, int inputLength, byte* pOutputBuffer, int outp
 843        {
 844            const int CharsPerDWord = sizeof(uint) / sizeof(char);
 845
 846            Debug.Assert(inputLength >= 0, "Input length must not be negative.");
 358847            Debug.Assert(pInputBuffer != null || inputLength == 0, "Input length must be zero if input buffer pointer is
 848
 358849            Debug.Assert(outputBytesRemaining >= 0, "Destination length must not be negative.");
 358850            Debug.Assert(pOutputBuffer != null || outputBytesRemaining == 0, "Destination length must be zero if destina
 851
 852            // First, try vectorized conversion.
 853
 854            {
 358855                nuint numElementsConverted = Ascii.NarrowUtf16ToAscii(pInputBuffer, pOutputBuffer, (uint)Math.Min(inputL
 856
 358857                pInputBuffer += numElementsConverted;
 358858                pOutputBuffer += numElementsConverted;
 859
 860                // Quick check - did we just end up consuming the entire input buffer?
 861                // If so, short-circuit the remainder of the method.
 862
 358863                if ((int)numElementsConverted == inputLength)
 864                {
 358865                    pInputBufferRemaining = pInputBuffer;
 358866                    pOutputBufferRemaining = pOutputBuffer;
 358867                    return OperationStatus.Done;
 868                }
 869
 0870                inputLength -= (int)numElementsConverted;
 0871                outputBytesRemaining -= (int)numElementsConverted;
 872            }
 873
 0874            if (inputLength < CharsPerDWord)
 875            {
 876                goto ProcessInputOfLessThanDWordSize;
 877            }
 878
 0879            char* pFinalPosWhereCanReadDWordFromInputBuffer = pInputBuffer + (uint)inputLength - CharsPerDWord;
 880
 881            // We have paths for SSE4.1 vectorization inside the inner loop. Since the below
 882            // vector is only used in those code paths, we leave it uninitialized if SSE4.1
 883            // is not enabled.
 884
 885#if NET
 886            Vector128<short> nonAsciiUtf16DataMask;
 887
 0888            if (Sse41.X64.IsSupported || (AdvSimd.Arm64.IsSupported && BitConverter.IsLittleEndian) || PackedSimd.IsSupp
 889            {
 0890                nonAsciiUtf16DataMask = Vector128.Create(unchecked((short)0xFF80)); // mask of non-ASCII bits in a UTF-1
 891            }
 892#endif
 893
 894            // Begin the main loop.
 895
 896#if DEBUG
 0897            char* pLastBufferPosProcessed = null; // used for invariant checking in debug builds
 898#endif
 899
 900            uint thisDWord;
 901
 0902            Debug.Assert(pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer);
 903            do
 904            {
 905                // Read 32 bits at a time. This is enough to hold any possible UTF16-encoded scalar.
 906
 0907                thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 908
 909            AfterReadDWord:
 910
 911#if DEBUG
 0912                Debug.Assert(pLastBufferPosProcessed < pInputBuffer, "Algorithm should've made forward progress since la
 0913                pLastBufferPosProcessed = pInputBuffer;
 914#endif
 915
 916                // First, check for the common case of all-ASCII chars.
 917
 0918                if (Utf16Utility.AllCharsInUInt32AreAscii(thisDWord))
 919                {
 920                    // We read an all-ASCII sequence (2 chars).
 921
 0922                    if (outputBytesRemaining < 2)
 923                    {
 924                        goto ProcessOneCharFromCurrentDWordAndFinish; // running out of space, but may be able to write 
 925                    }
 926
 927                    // The high WORD of the local declared below might be populated with garbage
 928                    // as a result of our shifts below, but that's ok since we're only going to
 929                    // write the low WORD.
 930                    //
 931                    // [ 00000000 0bbbbbbb | 00000000 0aaaaaaa ] -> [ 00000000 0bbbbbbb | 0bbbbbbb 0aaaaaaa ]
 932                    // (Same logic works regardless of endianness.)
 0933                    uint valueToWrite = thisDWord | (thisDWord >> 8);
 934
 0935                    Unsafe.WriteUnaligned(pOutputBuffer, (ushort)valueToWrite);
 936
 0937                    pInputBuffer += 2;
 0938                    pOutputBuffer += 2;
 0939                    outputBytesRemaining -= 2;
 940
 941                    // If we saw a sequence of all ASCII, there's a good chance a significant amount of following data i
 942                    // Below is basically unrolled loops with poor man's vectorization.
 943
 0944                    uint inputCharsRemaining = (uint)(pFinalPosWhereCanReadDWordFromInputBuffer - pInputBuffer) + 2;
 0945                    uint minElementsRemaining = (uint)Math.Min(inputCharsRemaining, outputBytesRemaining);
 946
 947#if NET
 0948                    if (Sse41.X64.IsSupported || (AdvSimd.Arm64.IsSupported && BitConverter.IsLittleEndian) || PackedSim
 949                    {
 950                        // Try reading and writing 8 elements per iteration.
 0951                        uint maxIters = minElementsRemaining / 8;
 952                        ulong possibleNonAsciiQWord;
 953                        int i;
 954                        Vector128<short> utf16Data;
 0955                        for (i = 0; (uint)i < maxIters; i++)
 956                        {
 957                            // The trimmer won't trim out nonAsciiUtf16DataMask unless this is in the loop.
 958                            // Luckily, this is a nop and will be elided by the JIT
 0959                            Unsafe.SkipInit(out nonAsciiUtf16DataMask);
 960
 0961                            utf16Data = Unsafe.ReadUnaligned<Vector128<short>>(pInputBuffer);
 962
 963                            if (AdvSimd.Arm64.IsSupported)
 964                            {
 965                                Vector128<short> isUtf16DataNonAscii = AdvSimd.CompareTest(utf16Data, nonAsciiUtf16DataM
 966                                bool hasNonAsciiDataInVector = AdvSimd.Arm64.MinPairwise(isUtf16DataNonAscii, isUtf16Dat
 967
 968                                if (hasNonAsciiDataInVector)
 969                                {
 970                                    goto LoopTerminatedDueToNonAsciiDataInVectorLocal;
 971                                }
 972
 973                                Vector64<byte> lower = AdvSimd.ExtractNarrowingSaturateUnsignedLower(utf16Data);
 974                                AdvSimd.Store(pOutputBuffer, lower);
 975                            }
 0976                            else if (Sse41.IsSupported)
 977                            {
 0978                                if ((utf16Data & nonAsciiUtf16DataMask) != Vector128<short>.Zero)
 979                                {
 980                                    goto LoopTerminatedDueToNonAsciiDataInVectorLocal;
 981                                }
 982
 983                                // narrow and write
 0984                                Sse2.StoreScalar((ulong*)pOutputBuffer /* unaligned */, Sse2.PackUnsignedSaturate(utf16D
 985                            }
 986                            else if (PackedSimd.IsSupported)
 987                            {
 988                                if ((utf16Data & nonAsciiUtf16DataMask) != Vector128<short>.Zero)
 989                                {
 990                                    goto LoopTerminatedDueToNonAsciiDataInVectorLocal;
 991                                }
 992
 993                                // narrow and write low 8 bytes
 994                                Vector128<byte> narrowed = PackedSimd.ConvertNarrowingSaturateUnsigned(utf16Data, utf16D
 995                                Unsafe.WriteUnaligned<ulong>(pOutputBuffer, narrowed.AsUInt64().ToScalar());
 996                            }
 997                            else
 998                            {
 999                                // We explicitly recheck each IsSupported query to ensure that the trimmer can see which
 01000                                ThrowHelper.ThrowUnreachableException();
 1001                            }
 1002
 01003                            pInputBuffer += 8;
 01004                            pOutputBuffer += 8;
 1005                        }
 1006
 01007                        outputBytesRemaining -= 8 * i;
 1008
 1009                        // Can we perform one more iteration, but reading & writing 4 elements instead of 8?
 1010
 01011                        if ((minElementsRemaining & 4) != 0)
 1012                        {
 01013                            possibleNonAsciiQWord = Unsafe.ReadUnaligned<ulong>(pInputBuffer);
 01014                            if (!Utf16Utility.AllCharsInUInt64AreAscii(possibleNonAsciiQWord))
 1015                            {
 1016                                goto LoopTerminatedDueToNonAsciiDataInPossibleNonAsciiQWordLocal;
 1017                            }
 1018
 01019                            utf16Data = Vector128.CreateScalarUnsafe(possibleNonAsciiQWord).AsInt16();
 1020
 1021                            if (AdvSimd.IsSupported)
 1022                            {
 1023                                Vector64<byte> lower = AdvSimd.ExtractNarrowingSaturateUnsignedLower(utf16Data);
 1024                                AdvSimd.StoreSelectedScalar((uint*)pOutputBuffer, lower.AsUInt32(), 0);
 1025                            }
 01026                            else if (Sse2.IsSupported)
 1027                            {
 01028                                Unsafe.WriteUnaligned(pOutputBuffer, Sse2.ConvertToUInt32(Sse2.PackUnsignedSaturate(utf1
 1029                            }
 1030                            else if (PackedSimd.IsSupported)
 1031                            {
 1032                                Vector128<byte> narrowed = PackedSimd.ConvertNarrowingSaturateUnsigned(utf16Data, utf16D
 1033                                Unsafe.WriteUnaligned<uint>(pOutputBuffer, narrowed.AsUInt32().ToScalar());
 1034                            }
 1035                            else
 1036                            {
 1037                                // We explicitly recheck each IsSupported query to ensure that the trimmer can see which
 01038                                ThrowHelper.ThrowUnreachableException();
 1039                            }
 1040
 01041                            pInputBuffer += 4;
 01042                            pOutputBuffer += 4;
 01043                            outputBytesRemaining -= 4;
 1044                        }
 1045
 01046                        continue; // Go back to beginning of main loop, read data, check for ASCII
 1047
 1048                    LoopTerminatedDueToNonAsciiDataInVectorLocal:
 1049
 01050                        outputBytesRemaining -= 8 * i;
 1051
 01052                        if (Sse2.X64.IsSupported)
 1053                        {
 01054                            possibleNonAsciiQWord = Sse2.X64.ConvertToUInt64(utf16Data.AsUInt64());
 1055                        }
 1056                        else
 1057                        {
 01058                            possibleNonAsciiQWord = utf16Data.AsUInt64().ToScalar();
 1059                        }
 1060
 1061                        // Temporarily set 'possibleNonAsciiQWord' to be the low 64 bits of the vector,
 1062                        // then check whether it's all-ASCII. If so, narrow and write to the destination
 1063                        // buffer. Since we know that either the high 64 bits or the low 64 bits of the
 1064                        // vector contains non-ASCII data, by the end of the following block the
 1065                        // 'possibleNonAsciiQWord' local is guaranteed to contain the non-ASCII segment.
 1066
 01067                        if (Utf16Utility.AllCharsInUInt64AreAscii(possibleNonAsciiQWord)) // all chars in first QWORD ar
 1068                        {
 1069                            if (AdvSimd.IsSupported)
 1070                            {
 1071                                Vector64<byte> lower = AdvSimd.ExtractNarrowingSaturateUnsignedLower(utf16Data);
 1072                                AdvSimd.StoreSelectedScalar((uint*)pOutputBuffer, lower.AsUInt32(), 0);
 1073                            }
 01074                            else if (Sse2.IsSupported)
 1075                            {
 01076                                Unsafe.WriteUnaligned(pOutputBuffer, Sse2.ConvertToUInt32(Sse2.PackUnsignedSaturate(utf1
 1077                            }
 1078                            else if (PackedSimd.IsSupported)
 1079                            {
 1080                                Vector128<byte> narrowed = PackedSimd.ConvertNarrowingSaturateUnsigned(utf16Data, utf16D
 1081                                Unsafe.WriteUnaligned<uint>(pOutputBuffer, narrowed.AsUInt32().ToScalar());
 1082                            }
 1083                            else
 1084                            {
 1085                                // We explicitly recheck each IsSupported query to ensure that the trimmer can see which
 01086                                ThrowHelper.ThrowUnreachableException();
 1087                            }
 01088                            pInputBuffer += 4;
 01089                            pOutputBuffer += 4;
 01090                            outputBytesRemaining -= 4;
 01091                            possibleNonAsciiQWord = utf16Data.AsUInt64().GetElement(1);
 1092                        }
 1093
 1094                    LoopTerminatedDueToNonAsciiDataInPossibleNonAsciiQWordLocal:
 1095
 01096                        Debug.Assert(!Utf16Utility.AllCharsInUInt64AreAscii(possibleNonAsciiQWord)); // this condition s
 1097
 01098                        thisDWord = (uint)possibleNonAsciiQWord;
 01099                        if (Utf16Utility.AllCharsInUInt32AreAscii(thisDWord))
 1100                        {
 1101                            // [ 00000000 0bbbbbbb | 00000000 0aaaaaaa ] -> [ 00000000 0bbbbbbb | 0bbbbbbb 0aaaaaaa ]
 01102                            Unsafe.WriteUnaligned(pOutputBuffer, (ushort)(thisDWord | (thisDWord >> 8)));
 01103                            pInputBuffer += 2;
 01104                            pOutputBuffer += 2;
 01105                            outputBytesRemaining -= 2;
 01106                            thisDWord = (uint)(possibleNonAsciiQWord >> 32);
 1107                        }
 1108
 01109                        goto AfterReadDWordSkipAllCharsAsciiCheck;
 1110                    }
 1111                    else
 1112#endif
 1113                    {
 1114                        // Can't use SSE41 x64, so we'll only read and write 4 elements per iteration.
 01115                        uint maxIters = minElementsRemaining / 4;
 1116                        uint secondDWord;
 1117                        int i;
 01118                        for (i = 0; (uint)i < maxIters; i++)
 1119                        {
 01120                            thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 01121                            secondDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer + 2);
 1122
 01123                            if (!Utf16Utility.AllCharsInUInt32AreAscii(thisDWord | secondDWord))
 1124                            {
 1125                                goto LoopTerminatedDueToNonAsciiData;
 1126                            }
 1127
 1128                            // [ 00000000 0bbbbbbb | 00000000 0aaaaaaa ] -> [ 00000000 0bbbbbbb | 0bbbbbbb 0aaaaaaa ]
 1129                            // (Same logic works regardless of endianness.)
 01130                            Unsafe.WriteUnaligned(pOutputBuffer, (ushort)(thisDWord | (thisDWord >> 8)));
 01131                            Unsafe.WriteUnaligned(pOutputBuffer + 2, (ushort)(secondDWord | (secondDWord >> 8)));
 1132
 01133                            pInputBuffer += 4;
 01134                            pOutputBuffer += 4;
 1135                        }
 1136
 01137                        outputBytesRemaining -= 4 * i;
 1138
 01139                        continue; // Go back to beginning of main loop, read data, check for ASCII
 1140
 1141                    LoopTerminatedDueToNonAsciiData:
 1142
 01143                        outputBytesRemaining -= 4 * i;
 1144
 1145                        // First, see if we can drain any ASCII data from the first DWORD.
 1146
 01147                        if (Utf16Utility.AllCharsInUInt32AreAscii(thisDWord))
 1148                        {
 1149                            // [ 00000000 0bbbbbbb | 00000000 0aaaaaaa ] -> [ 00000000 0bbbbbbb | 0bbbbbbb 0aaaaaaa ]
 1150                            // (Same logic works regardless of endianness.)
 01151                            Unsafe.WriteUnaligned(pOutputBuffer, (ushort)(thisDWord | (thisDWord >> 8)));
 01152                            pInputBuffer += 2;
 01153                            pOutputBuffer += 2;
 01154                            outputBytesRemaining -= 2;
 01155                            thisDWord = secondDWord;
 1156                        }
 1157
 1158                        goto AfterReadDWordSkipAllCharsAsciiCheck;
 1159                    }
 1160                }
 1161
 1162            AfterReadDWordSkipAllCharsAsciiCheck:
 1163
 01164                Debug.Assert(!Utf16Utility.AllCharsInUInt32AreAscii(thisDWord)); // this should have been handled earlie
 1165
 1166                // Next, try stripping off the first ASCII char if it exists.
 1167                // We don't check for a second ASCII char since that should have been handled above.
 1168
 01169                if (IsFirstCharAscii(thisDWord))
 1170                {
 01171                    if (outputBytesRemaining == 0)
 1172                    {
 1173                        goto OutputBufferTooSmall;
 1174                    }
 1175
 01176                    if (BitConverter.IsLittleEndian)
 1177                    {
 01178                        pOutputBuffer[0] = (byte)thisDWord; // extract [ ## ## 00 AA ]
 1179                    }
 1180                    else
 1181                    {
 1182                        pOutputBuffer[0] = (byte)(thisDWord >> 16); // extract [ 00 AA ## ## ]
 1183                    }
 1184
 01185                    pInputBuffer++;
 01186                    pOutputBuffer++;
 01187                    outputBytesRemaining--;
 1188
 01189                    if (pInputBuffer > pFinalPosWhereCanReadDWordFromInputBuffer)
 1190                    {
 1191                        goto ProcessNextCharAndFinish; // input buffer doesn't contain enough data to read a DWORD
 1192                    }
 1193                    else
 1194                    {
 1195                        // The input buffer at the current offset contains a non-ASCII char.
 1196                        // Read an entire DWORD and fall through to non-ASCII consumption logic.
 01197                        thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 1198                    }
 1199                }
 1200
 1201                // At this point, we know the first char in the buffer is non-ASCII, but we haven't yet validated it.
 1202
 01203                if (!IsFirstCharAtLeastThreeUtf8Bytes(thisDWord))
 1204                {
 1205                TryConsumeMultipleTwoByteSequences:
 1206
 1207                    // For certain text (Greek, Cyrillic, ...), 2-byte sequences tend to be clustered. We'll try transco
 1208                    // a tight loop without falling back to the main loop.
 1209
 01210                    if (IsSecondCharTwoUtf8Bytes(thisDWord))
 1211                    {
 1212                        // We have two runs of two bytes each.
 1213
 01214                        if (outputBytesRemaining < 4)
 1215                        {
 1216                            goto ProcessOneCharFromCurrentDWordAndFinish; // running out of output buffer
 1217                        }
 1218
 01219                        Unsafe.WriteUnaligned(pOutputBuffer, ExtractTwoUtf8TwoByteSequencesFromTwoPackedUtf16Chars(thisD
 1220
 01221                        pInputBuffer += 2;
 01222                        pOutputBuffer += 4;
 01223                        outputBytesRemaining -= 4;
 1224
 01225                        if (pInputBuffer > pFinalPosWhereCanReadDWordFromInputBuffer)
 1226                        {
 1227                            goto ProcessNextCharAndFinish; // Running out of data - go down slow path
 1228                        }
 1229                        else
 1230                        {
 1231                            // Optimization: If we read a long run of two-byte sequences, the next sequence is probably
 1232                            // also two bytes. Check for that first before going back to the beginning of the loop.
 1233
 01234                            thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 1235
 01236                            if (IsFirstCharTwoUtf8Bytes(thisDWord))
 1237                            {
 1238                                // Validated we have a two-byte sequence coming up
 01239                                goto TryConsumeMultipleTwoByteSequences;
 1240                            }
 1241
 1242                            // If we reached this point, the next sequence is something other than a valid
 1243                            // two-byte sequence, so go back to the beginning of the loop.
 1244                            goto AfterReadDWord;
 1245                        }
 1246                    }
 1247
 01248                    if (outputBytesRemaining < 2)
 1249                    {
 1250                        goto OutputBufferTooSmall;
 1251                    }
 1252
 01253                    Unsafe.WriteUnaligned(pOutputBuffer, (ushort)ExtractUtf8TwoByteSequenceFromFirstUtf16Char(thisDWord)
 1254
 1255                    // The buffer contains a 2-byte sequence followed by 2 bytes that aren't a 2-byte sequence.
 1256                    // Unlikely that a 3-byte sequence would follow a 2-byte sequence, so perhaps remaining
 1257                    // char is ASCII?
 1258
 01259                    if (IsSecondCharAscii(thisDWord))
 1260                    {
 01261                        if (outputBytesRemaining >= 3)
 1262                        {
 01263                            if (BitConverter.IsLittleEndian)
 1264                            {
 01265                                thisDWord >>= 16;
 1266                            }
 01267                            pOutputBuffer[2] = (byte)thisDWord;
 1268
 01269                            pInputBuffer += 2;
 01270                            pOutputBuffer += 3;
 01271                            outputBytesRemaining -= 3;
 1272
 01273                            continue; // go back to original bounds check and check for ASCII
 1274                        }
 1275                        else
 1276                        {
 01277                            pInputBuffer++;
 01278                            pOutputBuffer += 2;
 01279                            goto OutputBufferTooSmall;
 1280                        }
 1281                    }
 1282                    else
 1283                    {
 01284                        pInputBuffer++;
 01285                        pOutputBuffer += 2;
 01286                        outputBytesRemaining -= 2;
 1287
 01288                        if (pInputBuffer > pFinalPosWhereCanReadDWordFromInputBuffer)
 1289                        {
 1290                            goto ProcessNextCharAndFinish; // Running out of data - go down slow path
 1291                        }
 1292                        else
 1293                        {
 01294                            thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 1295                            goto BeforeProcessThreeByteSequence; // we know the next byte isn't ASCII, and it's not the 
 1296                        }
 1297                    }
 1298                }
 1299
 1300            // Check the 3-byte case.
 1301
 1302            BeforeProcessThreeByteSequence:
 1303
 01304                if (!IsFirstCharSurrogate(thisDWord))
 1305                {
 1306                    // Optimization: A three-byte character could indicate CJK text, which makes it likely
 1307                    // that the character following this one is also CJK. We'll perform the check now
 1308                    // rather than jumping to the beginning of the main loop.
 1309
 01310                    if (IsSecondCharAtLeastThreeUtf8Bytes(thisDWord))
 1311                    {
 01312                        if (!IsSecondCharSurrogate(thisDWord))
 1313                        {
 01314                            if (outputBytesRemaining < 6)
 1315                            {
 1316                                goto ConsumeSingleThreeByteRun; // not enough space - try consuming as much as we can
 1317                            }
 1318
 01319                            WriteTwoUtf16CharsAsTwoUtf8ThreeByteSequences(ref *pOutputBuffer, thisDWord);
 1320
 01321                            pInputBuffer += 2;
 01322                            pOutputBuffer += 6;
 01323                            outputBytesRemaining -= 6;
 1324
 1325                            // Try to remain in the 3-byte processing loop if at all possible.
 1326
 01327                            if (pInputBuffer > pFinalPosWhereCanReadDWordFromInputBuffer)
 1328                            {
 1329                                goto ProcessNextCharAndFinish; // Running out of data - go down slow path
 1330                            }
 1331                            else
 1332                            {
 01333                                thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 1334
 01335                                if (IsFirstCharAtLeastThreeUtf8Bytes(thisDWord))
 1336                                {
 01337                                    goto BeforeProcessThreeByteSequence;
 1338                                }
 1339                                else
 1340                                {
 1341                                    // Fall back to standard processing loop since we don't know how to optimize this.
 1342                                    goto AfterReadDWord;
 1343                                }
 1344                            }
 1345                        }
 1346                    }
 1347
 1348                ConsumeSingleThreeByteRun:
 1349
 01350                    if (outputBytesRemaining < 3)
 1351                    {
 1352                        goto OutputBufferTooSmall;
 1353                    }
 1354
 01355                    WriteFirstUtf16CharAsUtf8ThreeByteSequence(ref *pOutputBuffer, thisDWord);
 1356
 01357                    pInputBuffer++;
 01358                    pOutputBuffer += 3;
 01359                    outputBytesRemaining -= 3;
 1360
 1361                    // Occasionally one-off ASCII characters like spaces, periods, or newlines will make their way
 1362                    // in to the text. If this happens strip it off now before seeing if the next character
 1363                    // consists of three code units.
 1364
 01365                    if (IsSecondCharAscii(thisDWord))
 1366                    {
 01367                        if (outputBytesRemaining == 0)
 1368                        {
 1369                            goto OutputBufferTooSmall;
 1370                        }
 1371
 01372                        if (BitConverter.IsLittleEndian)
 1373                        {
 01374                            *pOutputBuffer = (byte)(thisDWord >> 16);
 1375                        }
 1376                        else
 1377                        {
 1378                            *pOutputBuffer = (byte)(thisDWord);
 1379                        }
 1380
 01381                        pInputBuffer++;
 01382                        pOutputBuffer++;
 01383                        outputBytesRemaining--;
 1384
 01385                        if (pInputBuffer > pFinalPosWhereCanReadDWordFromInputBuffer)
 1386                        {
 1387                            goto ProcessNextCharAndFinish; // Running out of data - go down slow path
 1388                        }
 1389                        else
 1390                        {
 01391                            thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 1392
 01393                            if (IsFirstCharAtLeastThreeUtf8Bytes(thisDWord))
 1394                            {
 01395                                goto BeforeProcessThreeByteSequence;
 1396                            }
 1397                            else
 1398                            {
 1399                                // Fall back to standard processing loop since we don't know how to optimize this.
 1400                                goto AfterReadDWord;
 1401                            }
 1402                        }
 1403                    }
 1404
 01405                    if (pInputBuffer > pFinalPosWhereCanReadDWordFromInputBuffer)
 1406                    {
 1407                        goto ProcessNextCharAndFinish; // Running out of data - go down slow path
 1408                    }
 1409                    else
 1410                    {
 01411                        thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 01412                        goto AfterReadDWordSkipAllCharsAsciiCheck; // we just checked above that this value isn't ASCII
 1413                    }
 1414                }
 1415
 1416                // Four byte sequence processing
 1417
 01418                if (IsWellFormedUtf16SurrogatePair(thisDWord))
 1419                {
 01420                    if (outputBytesRemaining < 4)
 1421                    {
 1422                        goto OutputBufferTooSmall;
 1423                    }
 1424
 01425                    Unsafe.WriteUnaligned(pOutputBuffer, ExtractFourUtf8BytesFromSurrogatePair(thisDWord));
 1426
 01427                    pInputBuffer += 2;
 01428                    pOutputBuffer += 4;
 01429                    outputBytesRemaining -= 4;
 1430
 1431                    continue; // go back to beginning of loop for processing
 1432                }
 1433
 1434                goto Error; // an ill-formed surrogate sequence: high not followed by low, or low not preceded by high
 01435            } while (pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer);
 1436
 1437        ProcessNextCharAndFinish:
 01438            inputLength = (int)(pFinalPosWhereCanReadDWordFromInputBuffer - pInputBuffer) + CharsPerDWord;
 1439
 1440        ProcessInputOfLessThanDWordSize:
 01441            Debug.Assert(inputLength < CharsPerDWord);
 1442
 01443            if (inputLength == 0)
 1444            {
 1445                goto InputBufferFullyConsumed;
 1446            }
 1447
 01448            uint thisChar = *pInputBuffer;
 01449            goto ProcessFinalChar;
 1450
 1451        ProcessOneCharFromCurrentDWordAndFinish:
 01452            if (BitConverter.IsLittleEndian)
 1453            {
 01454                thisChar = thisDWord & 0xFFFFu; // preserve only the first char
 1455            }
 1456            else
 1457            {
 1458                thisChar = thisDWord >> 16; // preserve only the first char
 1459            }
 1460
 1461        ProcessFinalChar:
 1462            {
 01463                if (thisChar <= 0x7Fu)
 1464                {
 01465                    if (outputBytesRemaining == 0)
 1466                    {
 1467                        goto OutputBufferTooSmall; // we have no hope of writing anything to the output
 1468                    }
 1469
 1470                    // 1-byte (ASCII) case
 01471                    *pOutputBuffer = (byte)thisChar;
 1472
 01473                    pInputBuffer++;
 01474                    pOutputBuffer++;
 1475                }
 01476                else if (thisChar < 0x0800u)
 1477                {
 01478                    if (outputBytesRemaining < 2)
 1479                    {
 1480                        goto OutputBufferTooSmall; // we have no hope of writing anything to the output
 1481                    }
 1482
 1483                    // 2-byte case
 01484                    pOutputBuffer[1] = (byte)((thisChar & 0x3Fu) | unchecked((uint)(sbyte)0x80)); // [ 10xxxxxx ]
 01485                    pOutputBuffer[0] = (byte)((thisChar >> 6) | unchecked((uint)(sbyte)0xC0)); // [ 110yyyyy ]
 1486
 01487                    pInputBuffer++;
 01488                    pOutputBuffer += 2;
 1489                }
 01490                else if (!UnicodeUtility.IsSurrogateCodePoint(thisChar))
 1491                {
 01492                    if (outputBytesRemaining < 3)
 1493                    {
 1494                        goto OutputBufferTooSmall; // we have no hope of writing anything to the output
 1495                    }
 1496
 1497                    // 3-byte case
 01498                    pOutputBuffer[2] = (byte)((thisChar & 0x3Fu) | unchecked((uint)(sbyte)0x80)); // [ 10xxxxxx ]
 01499                    pOutputBuffer[1] = (byte)(((thisChar >> 6) & 0x3Fu) | unchecked((uint)(sbyte)0x80)); // [ 10yyyyyy ]
 01500                    pOutputBuffer[0] = (byte)((thisChar >> 12) | unchecked((uint)(sbyte)0xE0)); // [ 1110zzzz ]
 1501
 01502                    pInputBuffer++;
 01503                    pOutputBuffer += 3;
 1504                }
 01505                else if (thisChar <= 0xDBFFu)
 1506                {
 1507                    // UTF-16 high surrogate code point with no trailing data, report incomplete input buffer
 01508                    goto InputBufferTooSmall;
 1509                }
 1510                else
 1511                {
 1512                    // UTF-16 low surrogate code point with no leading data, report error
 1513                    goto Error;
 1514                }
 1515            }
 1516
 1517            // There are two ways we can end up here. Either we were running low on input data,
 1518            // or we were running low on space in the destination buffer. If we're running low on
 1519            // input data (label targets ProcessInputOfLessThanDWordSize and ProcessNextCharAndFinish),
 1520            // then the inputLength value is guaranteed to be between 0 and 1, and we should return Done.
 1521            // If we're running low on destination buffer space (label target ProcessOneCharFromCurrentDWordAndFinish),
 1522            // then we didn't modify inputLength since entering the main loop, which means it should
 1523            // still have a value of >= 2. So checking the value of inputLength is all we need to do to determine
 1524            // which of the two scenarios we're in.
 1525
 01526            if (inputLength > 1)
 1527            {
 1528                goto OutputBufferTooSmall;
 1529            }
 1530
 1531        InputBufferFullyConsumed:
 01532            OperationStatus retVal = OperationStatus.Done;
 01533            goto ReturnCommon;
 1534
 1535        InputBufferTooSmall:
 01536            retVal = OperationStatus.NeedMoreData;
 01537            goto ReturnCommon;
 1538
 1539        OutputBufferTooSmall:
 01540            retVal = OperationStatus.DestinationTooSmall;
 01541            goto ReturnCommon;
 1542
 1543        Error:
 01544            retVal = OperationStatus.InvalidData;
 1545            goto ReturnCommon;
 1546
 1547        ReturnCommon:
 01548            pInputBufferRemaining = pInputBuffer;
 01549            pOutputBufferRemaining = pOutputBuffer;
 01550            return retVal;
 1551        }
 1552    }
 1553}
 1554

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.CoreLib/src/System/Text/Unicode/Utf8Utility.Validation.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Buffers.Text;
 5using System.Diagnostics;
 6using System.Diagnostics.CodeAnalysis;
 7using System.Numerics;
 8using System.Runtime.CompilerServices;
 9#if NET
 10using System.Runtime.Intrinsics;
 11using System.Runtime.Intrinsics.Arm;
 12using System.Runtime.Intrinsics.Wasm;
 13using System.Runtime.Intrinsics.X86;
 14#endif
 15
 16namespace System.Text.Unicode
 17{
 18    internal static unsafe partial class Utf8Utility
 19    {
 20        // Returns &inputBuffer[inputLength] if the input buffer is valid.
 21        /// <summary>
 22        /// Given an input buffer <paramref name="pInputBuffer"/> of byte length <paramref name="inputLength"/>,
 23        /// returns a pointer to where the first invalid data appears in <paramref name="pInputBuffer"/>.
 24        /// </summary>
 25        /// <remarks>
 26        /// Returns a pointer to the end of <paramref name="pInputBuffer"/> if the buffer is well-formed.
 27        /// </remarks>
 28        public static byte* GetPointerToFirstInvalidByte(byte* pInputBuffer, int inputLength, out int utf16CodeUnitCount
 29        {
 30            Debug.Assert(inputLength >= 0, "Input length must not be negative.");
 67303431            Debug.Assert(pInputBuffer != null || inputLength == 0, "Input length must be zero if input buffer pointer is
 32
 33            // First, try to drain off as many ASCII bytes as we can from the beginning.
 67303434            nuint numAsciiBytesCounted = Ascii.GetIndexOfFirstNonAsciiByte(pInputBuffer, (uint)inputLength);
 67303435            pInputBuffer += numAsciiBytesCounted;
 36
 37            // Quick check - did we just end up consuming the entire input buffer?
 38            // If so, short-circuit the remainder of the method.
 39
 67303440            inputLength -= (int)numAsciiBytesCounted;
 67303441            if (inputLength == 0)
 42            {
 398843                utf16CodeUnitCountAdjustment = 0;
 398844                scalarCountAdjustment = 0;
 398845                return pInputBuffer;
 46            }
 47
 48#if DEBUG
 49            // Keep these around for final validation at the end of the method.
 66904650            byte* pOriginalInputBuffer = pInputBuffer;
 66904651            int originalInputLength = inputLength;
 52#endif
 53
 54            // Enregistered locals that we'll eventually out to our caller.
 55
 66904656            int tempUtf16CodeUnitCountAdjustment = 0;
 66904657            int tempScalarCountAdjustment = 0;
 58
 66904659            if (inputLength < sizeof(uint))
 60            {
 61                goto ProcessInputOfLessThanDWordSize;
 62            }
 63
 65621264            byte* pFinalPosWhereCanReadDWordFromInputBuffer = pInputBuffer + (uint)inputLength - sizeof(uint);
 65
 66            // Begin the main loop.
 67
 68#if DEBUG
 65621269            byte* pLastBufferPosProcessed = null; // used for invariant checking in debug builds
 70#endif
 71
 74546272            while (pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer)
 73            {
 74                // Read 32 bits at a time. This is enough to hold any possible UTF8-encoded scalar.
 75
 74421676                uint thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 77
 78            AfterReadDWord:
 79
 80#if DEBUG
 79950281                Debug.Assert(pLastBufferPosProcessed < pInputBuffer, "Algorithm should've made forward progress since la
 79950282                pLastBufferPosProcessed = pInputBuffer;
 83#endif
 84
 85                // First, check for the common case of all-ASCII bytes.
 86
 79950287                if (Ascii.AllBytesInUInt32AreAscii(thisDWord))
 88                {
 89                    // We read an all-ASCII sequence.
 90
 2863491                    pInputBuffer += sizeof(uint);
 92
 93                    // If we saw a sequence of all ASCII, there's a good chance a significant amount of following data i
 94                    // Below is basically unrolled loops with poor man's vectorization.
 95
 96                    // Below check is "can I read at least five DWORDs from the input stream?"
 97                    // n.b. Since we incremented pInputBuffer above the below subtraction may result in a negative value
 98                    // hence using nint instead of nuint.
 99
 28634100                    if ((nint)(void*)Unsafe.ByteOffset(ref *pInputBuffer, ref *pFinalPosWhereCanReadDWordFromInputBuffer
 101                    {
 102                        // We want reads in the inner loop to be aligned. So let's perform a quick
 103                        // ASCII check of the next 32 bits (4 bytes) now, and if that succeeds bump
 104                        // the read pointer up to the next aligned address.
 105
 27094106                        thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 27094107                        if (!Ascii.AllBytesInUInt32AreAscii(thisDWord))
 108                        {
 109                            goto AfterReadDWordSkipAllBytesAsciiCheck;
 110                        }
 111
 14376112                        pInputBuffer = (byte*)((nuint)(pInputBuffer + 4) & ~(nuint)3);
 113
 114                        // At this point, the input buffer offset points to an aligned DWORD. We also know that there's
 115                        // enough room to read at least four DWORDs from the buffer. (Heed the comment a few lines above
 116                        // the original 'if' check confirmed that there were 5 DWORDs before the alignment check, and
 117                        // the alignment check consumes at most a single DWORD.)
 118
 14376119                        byte* pInputBufferFinalPosAtWhichCanSafelyLoop = pFinalPosWhereCanReadDWordFromInputBuffer - 3 *
 120
 121                        // pInputBuffer is 32-bit aligned but not necessary 128-bit aligned, so we're
 122                        // going to perform an unaligned load. We don't necessarily care about aligning
 123                        // this because we pessimistically assume we'll encounter non-ASCII data at some
 124                        // point in the not-too-distant future (otherwise we would've stayed entirely
 125                        // within the all-ASCII vectorized code at the entry to this method).
 126#if NET
 127                        nuint trailingZeroCount;
 128                        if (AdvSimd.Arm64.IsSupported && BitConverter.IsLittleEndian)
 129                        {
 130                            // declare bitMask128 inside of the AdvSimd.Arm64.IsSupported check
 131                            // so it gets removed on non-Arm64 builds.
 132                            Vector128<byte> bitMask128 = BitConverter.IsLittleEndian ?
 133                                Vector128.Create((ushort)0x1001).AsByte() :
 134                                Vector128.Create((ushort)0x0110).AsByte();
 135                            do
 136                            {
 137                                ulong mask = GetNonAsciiBytes(AdvSimd.LoadVector128(pInputBuffer), bitMask128);
 138                                if (mask != 0)
 139                                {
 140                                    trailingZeroCount = (nuint)BitOperations.TrailingZeroCount(mask) >> 2;
 141                                    goto LoopTerminatedEarlyDueToNonAsciiData;
 142                                }
 143
 144                                pInputBuffer += 4 * sizeof(uint); // consumed 4 DWORDs
 145                            } while (pInputBuffer <= pInputBufferFinalPosAtWhichCanSafelyLoop);
 146                        }
 147                        else
 148#endif
 149                        {
 150                            do
 151                            {
 152#if NET
 17110153                                if (Sse2.IsSupported)
 154                                {
 17110155                                    uint mask = (uint)Sse2.MoveMask(Sse2.LoadVector128(pInputBuffer));
 17110156                                    if (mask != 0)
 157                                    {
 14280158                                        trailingZeroCount = (nuint)BitOperations.TrailingZeroCount(mask);
 14280159                                        goto LoopTerminatedEarlyDueToNonAsciiData;
 160                                    }
 161                                }
 162                                else if (PackedSimd.IsSupported)
 163                                {
 164                                    uint mask = Vector128.LoadUnsafe(ref *pInputBuffer).ExtractMostSignificantBits();
 165                                    if (mask != 0)
 166                                    {
 167                                        trailingZeroCount = (nuint)BitOperations.TrailingZeroCount(mask);
 168                                        goto LoopTerminatedEarlyDueToNonAsciiData;
 169                                    }
 170                                }
 171                                else
 172#endif
 173                                {
 0174                                    if (!Ascii.AllBytesInUInt32AreAscii(((uint*)pInputBuffer)[0] | ((uint*)pInputBuffer)
 175                                    {
 176                                        goto LoopTerminatedEarlyDueToNonAsciiDataInFirstPair;
 177                                    }
 178
 0179                                    if (!Ascii.AllBytesInUInt32AreAscii(((uint*)pInputBuffer)[2] | ((uint*)pInputBuffer)
 180                                    {
 181                                        goto LoopTerminatedEarlyDueToNonAsciiDataInSecondPair;
 182                                    }
 183                                }
 184
 2830185                                pInputBuffer += 4 * sizeof(uint); // consumed 4 DWORDs
 2830186                            } while (pInputBuffer <= pInputBufferFinalPosAtWhichCanSafelyLoop);
 187                        }
 188
 96189                        continue; // need to perform a bounds check because we might be running out of data
 190
 191#if NET
 192                    LoopTerminatedEarlyDueToNonAsciiData:
 193                        // x86 and Wasm can only be little endian, while ARM can be big or little endian,
 194                        // so if we reached this label we need to check the LE-restricted combinations as well.
 195                        Debug.Assert((AdvSimd.Arm64.IsSupported && BitConverter.IsLittleEndian) || Sse2.IsSupported || P
 196
 197
 198                        // The 'mask' value will have a 0 bit for each ASCII byte we saw and a 1 bit
 199                        // for each non-ASCII byte we saw. trailingZeroCount will count the number of ASCII bytes,
 200                        // bump our input counter by that amount, and resume processing from the
 201                        // "the first byte is no longer ASCII" portion of the main loop.
 202                        // We should not expect a total number of zeroes equal or larger than 16.
 14280203                        Debug.Assert(trailingZeroCount < 16);
 204
 14280205                        pInputBuffer += trailingZeroCount;
 14280206                        if (pInputBuffer > pFinalPosWhereCanReadDWordFromInputBuffer)
 207                        {
 208                            goto ProcessRemainingBytesSlow;
 209                        }
 210
 14270211                        thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); // no longer guaranteed to be aligned
 14270212                        goto BeforeProcessTwoByteSequence;
 213#endif
 214
 215                    LoopTerminatedEarlyDueToNonAsciiDataInSecondPair:
 216
 0217                        pInputBuffer += 2 * sizeof(uint); // consumed 2 DWORDs
 218
 219                    LoopTerminatedEarlyDueToNonAsciiDataInFirstPair:
 220
 221                        // We know that there's *at least* two DWORDs of data remaining in the buffer.
 222                        // We also know that one of them (or both of them) contains non-ASCII data somewhere.
 223                        // Let's perform a quick check here to bypass the logic at the beginning of the main loop.
 224
 0225                        thisDWord = *(uint*)pInputBuffer; // still aligned here
 0226                        if (Ascii.AllBytesInUInt32AreAscii(thisDWord))
 227                        {
 0228                            pInputBuffer += sizeof(uint); // consumed 1 more DWORD
 0229                            thisDWord = *(uint*)pInputBuffer; // still aligned here
 230                        }
 231
 232                        goto AfterReadDWordSkipAllBytesAsciiCheck;
 233                    }
 234
 235                    continue; // not enough data remaining to unroll loop - go back to beginning with bounds checks
 236                }
 237
 238            AfterReadDWordSkipAllBytesAsciiCheck:
 239
 783586240                Debug.Assert(!Ascii.AllBytesInUInt32AreAscii(thisDWord)); // this should have been handled earlier
 241
 242                // Next, try stripping off ASCII bytes one at a time.
 243                // We only handle up to three ASCII bytes here since we handled the four ASCII byte case above.
 244
 245                {
 783586246                    uint numLeadingAsciiBytes = Ascii.CountNumberOfLeadingAsciiBytesFromUInt32WithSomeNonAsciiData(thisD
 783586247                    pInputBuffer += numLeadingAsciiBytes;
 248
 783586249                    if (pFinalPosWhereCanReadDWordFromInputBuffer < pInputBuffer)
 250                    {
 251                        goto ProcessRemainingBytesSlow; // Input buffer doesn't contain enough data to read a DWORD
 252                    }
 253                    else
 254                    {
 255                        // The input buffer at the current offset contains a non-ASCII byte.
 256                        // Read an entire DWORD and fall through to multi-byte consumption logic.
 783074257                        thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 258                    }
 259                }
 260
 261            BeforeProcessTwoByteSequence:
 262
 263                // At this point, we suspect we're working with a multi-byte code unit sequence,
 264                // but we haven't yet validated it for well-formedness.
 265
 266                // The masks and comparands are derived from the Unicode Standard, Table 3-6.
 267                // Additionally, we need to check for valid byte sequences per Table 3-7.
 268
 269                // Check the 2-byte case.
 270
 816202271                thisDWord -= (BitConverter.IsLittleEndian) ? 0x0000_80C0u : 0xC080_0000u;
 816202272                if ((thisDWord & (BitConverter.IsLittleEndian ? 0x0000_C0E0u : 0xE0C0_0000u)) == 0)
 273                {
 274                    // Per Table 3-7, valid sequences are:
 275                    // [ C2..DF ] [ 80..BF ]
 276                    //
 277                    // Due to our modification of 'thisDWord' above, this becomes:
 278                    // [ 02..1F ] [ 00..3F ]
 279                    //
 280                    // We've already checked that the leading byte was originally in the range [ C0..DF ]
 281                    // and that the trailing byte was originally in the range [ 80..BF ], so now we only need
 282                    // to check that the modified leading byte is >= [ 02 ].
 283
 93612284                    if ((BitConverter.IsLittleEndian && (byte)thisDWord < 0x02u)
 93612285                        || (!BitConverter.IsLittleEndian && thisDWord < 0x0200_0000u))
 286                    {
 287                        goto Error; // overlong form - leading byte was [ C0 ] or [ C1 ]
 288                    }
 289
 290                ProcessTwoByteSequenceSkipOverlongFormCheck:
 291
 292                    // Optimization: If this is a two-byte-per-character language like Cyrillic or Hebrew,
 293                    // there's a good chance that if we see one two-byte run then there's another two-byte
 294                    // run immediately after. Let's check that now.
 295
 296                    // On little-endian platforms, we can check for the two-byte UTF8 mask *and* validate that
 297                    // the value isn't overlong using a single comparison. On big-endian platforms, we'll need
 298                    // to validate the mask and validate that the sequence isn't overlong as two separate comparisons.
 299
 84946300                    if ((BitConverter.IsLittleEndian && UInt32EndsWithValidUtf8TwoByteSequenceLittleEndian(thisDWord))
 84946301                        || (!BitConverter.IsLittleEndian && (UInt32EndsWithUtf8TwoByteMask(thisDWord) && !UInt32EndsWith
 302                    {
 303                        // We have two runs of two bytes each.
 11704304                        pInputBuffer += 4;
 11704305                        tempUtf16CodeUnitCountAdjustment -= 2; // 4 UTF-8 code units -> 2 UTF-16 code units (and 2 scala
 306
 11704307                        if (pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer)
 308                        {
 309                            // Optimization: If we read a long run of two-byte sequences, the next sequence is probably
 310                            // also two bytes. Check for that first before going back to the beginning of the loop.
 311
 11600312                            thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 313
 11600314                            if (BitConverter.IsLittleEndian)
 315                            {
 11600316                                if (UInt32BeginsWithValidUtf8TwoByteSequenceLittleEndian(thisDWord))
 317                                {
 318                                    // The next sequence is a valid two-byte sequence.
 6810319                                    goto ProcessTwoByteSequenceSkipOverlongFormCheck;
 320                                }
 321                            }
 322                            else
 323                            {
 324                                if (UInt32BeginsWithUtf8TwoByteMask(thisDWord))
 325                                {
 326                                    if (UInt32BeginsWithOverlongUtf8TwoByteSequence(thisDWord))
 327                                    {
 328                                        goto Error; // The next sequence purports to be a 2-byte sequence but is overlon
 329                                    }
 330
 331                                    goto ProcessTwoByteSequenceSkipOverlongFormCheck;
 332                                }
 333                            }
 334
 335                            // If we reached this point, the next sequence is something other than a valid
 336                            // two-byte sequence, so go back to the beginning of the loop.
 337                            goto AfterReadDWord;
 338                        }
 339                        else
 340                        {
 341                            goto ProcessRemainingBytesSlow; // Running out of data - go down slow path
 342                        }
 343                    }
 344
 345                    // The buffer contains a 2-byte sequence followed by 2 bytes that aren't a 2-byte sequence.
 346                    // Unlikely that a 3-byte sequence would follow a 2-byte sequence, so perhaps remaining
 347                    // bytes are ASCII?
 348
 73242349                    tempUtf16CodeUnitCountAdjustment--; // 2-byte sequence + (some number of ASCII bytes) -> 1 UTF-16 co
 350
 73242351                    if (UInt32ThirdByteIsAscii(thisDWord))
 352                    {
 61896353                        if (UInt32FourthByteIsAscii(thisDWord))
 354                        {
 42668355                            pInputBuffer += 4;
 356                        }
 357                        else
 358                        {
 19228359                            pInputBuffer += 3;
 360
 361                            // A two-byte sequence followed by an ASCII byte followed by a non-ASCII byte.
 362                            // Read in the next DWORD and jump directly to the start of the multi-byte processing block.
 363
 19228364                            if (pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer)
 365                            {
 18858366                                thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 18858367                                goto BeforeProcessTwoByteSequence;
 368                            }
 369                        }
 370                    }
 371                    else
 372                    {
 11346373                        pInputBuffer += 2;
 374                    }
 375
 11346376                    continue;
 377                }
 378
 379                // Check the 3-byte case.
 380                // We need to restore the C0 leading byte we stripped out earlier, then we can strip out the expected E0
 381
 722590382                thisDWord -= (BitConverter.IsLittleEndian) ? (0x0080_00E0u - 0x0000_00C0u) : (0xE000_8000u - 0xC000_0000
 722590383                if ((thisDWord & (BitConverter.IsLittleEndian ? 0x00C0_C0F0u : 0xF0C0_C000u)) == 0)
 384                {
 385                ProcessThreeByteSequenceWithCheck:
 386
 387                    // We assume the caller has confirmed that the bit pattern is representative of a three-byte
 388                    // sequence, but it may still be overlong or surrogate. We need to check for these possibilities.
 389                    //
 390                    // Per Table 3-7, valid sequences are:
 391                    // [   E0   ] [ A0..BF ] [ 80..BF ]
 392                    // [ E1..EC ] [ 80..BF ] [ 80..BF ]
 393                    // [   ED   ] [ 80..9F ] [ 80..BF ]
 394                    // [ EE..EF ] [ 80..BF ] [ 80..BF ]
 395                    //
 396                    // Big-endian examples of using the above validation table:
 397                    // E0A0 = 1110 0000 1010 0000 => invalid (overlong ) patterns are 1110 0000 100# ####
 398                    // ED9F = 1110 1101 1001 1111 => invalid (surrogate) patterns are 1110 1101 101# ####
 399                    // If using the bitmask ......................................... 0000 1111 0010 0000 (=0F20),
 400                    // Then invalid (overlong) patterns match the comparand ......... 0000 0000 0000 0000 (=0000),
 401                    // And invalid (surrogate) patterns match the comparand ......... 0000 1101 0010 0000 (=0D20).
 402                    //
 403                    // It's ok if the caller has manipulated 'thisDWord' (e.g., by subtracting 0xE0 or 0x80)
 404                    // as long as they haven't touched the bits we're about to use in our mask checking below.
 405
 159944406                    if (BitConverter.IsLittleEndian)
 407                    {
 408                        // The "overlong or surrogate" check can be implemented using a single jump, but there's
 409                        // some overhead to moving the bits into the correct locations in order to perform the
 410                        // correct comparison, and in practice the processor's branch prediction capability is
 411                        // good enough that we shouldn't bother. So we'll use two jumps instead.
 412
 413                        // Can't extract this check into its own helper method because JITter produces suboptimal
 414                        // assembly, even with aggressive inlining.
 415
 416                        // Code below becomes 5 instructions: test, jz, lea, test, jz
 417
 159944418                        if (((thisDWord & 0x0000_200Fu) == 0) || (((thisDWord - 0x0000_200Du) & 0x0000_200Fu) == 0))
 419                        {
 25412420                            goto Error; // overlong or surrogate
 421                        }
 422                    }
 423                    else
 424                    {
 425                        if (((thisDWord & 0x0F20_0000u) == 0) || (((thisDWord - 0x0D20_0000u) & 0x0F20_0000u) == 0))
 426                        {
 427                            goto Error; // overlong or surrogate
 428                        }
 429                    }
 430
 431                ProcessSingleThreeByteSequenceSkipOverlongAndSurrogateChecks:
 432
 433                    // Occasionally one-off ASCII characters like spaces, periods, or newlines will make their way
 434                    // in to the text. If this happens strip it off now before seeing if the next character
 435                    // consists of three code units.
 436
 437                    // Branchless: consume a 3-byte UTF-8 sequence and optionally an extra ASCII byte from the end.
 438
 439                    nint asciiAdjustment;
 149664440                    if (BitConverter.IsLittleEndian)
 441                    {
 149664442                        asciiAdjustment = (int)thisDWord >> 31; // smear most significant bit across entire value
 443                    }
 444                    else
 445                    {
 446                        asciiAdjustment = (nint)(sbyte)thisDWord >> 7; // smear most significant bit of least significan
 447                    }
 448
 449                    // asciiAdjustment = 0 if fourth byte is ASCII; -1 otherwise
 450
 451                    // Please *DO NOT* reorder the below two lines. It provides extra defense in depth in case this meth
 452                    // is ever changed such that pInputBuffer becomes a 'ref byte' instead of a simple 'byte*'. It's val
 453                    // to add 4 before backing up since we already checked previously that the input buffer contains at
 454                    // least a DWORD's worth of data, so we're not going to run past the end of the buffer where the GC 
 455                    // no longer track the reference. However, we can't back up before adding 4, since we might back up 
 456                    // before the start of the buffer, and the GC isn't guaranteed to be able to track this.
 457
 149664458                    pInputBuffer += 4; // optimistically, assume consumed a 3-byte UTF-8 sequence plus an extra ASCII by
 149664459                    pInputBuffer += asciiAdjustment; // back up if we didn't actually consume an ASCII byte
 460
 149664461                    tempUtf16CodeUnitCountAdjustment -= 2; // 3 (or 4) UTF-8 bytes -> 1 (or 2) UTF-16 code unit (and 1 [
 462
 463                SuccessfullyProcessedThreeByteSequence:
 464
 465                    if (IntPtr.Size >= 8 && BitConverter.IsLittleEndian)
 466                    {
 467                        // x64 little-endian optimization: A three-byte character could indicate CJK text,
 468                        // which makes it likely that the character following this one is also CJK.
 469                        // We'll try to process several three-byte sequences at a time.
 470
 471                        // The check below is really "can we read 9 bytes from the input buffer?" since 'pFinalPos...' i
 472                        // n.b. The subtraction below could result in a negative value (since we advanced pInputBuffer a
 473                        // use nint instead of nuint.
 474
 158634475                        if ((nint)(pFinalPosWhereCanReadDWordFromInputBuffer - pInputBuffer) >= 5)
 476                        {
 154858477                            ulong thisQWord = Unsafe.ReadUnaligned<ulong>(pInputBuffer);
 478
 479                            // Stage the next 32 bits into 'thisDWord' so that it's ready for us in case we need to jump
 480                            // to a previous location in the loop. This offers defense against reading main memory again
 481                            // have been modified and could lead to a race condition).
 482
 154858483                            thisDWord = (uint)thisQWord;
 484
 485                            // Is this three 3-byte sequences in a row?
 486                            // thisQWord = [ 10yyyyyy 1110zzzz | 10xxxxxx 10yyyyyy 1110zzzz | 10xxxxxx 10yyyyyy 1110zzzz
 487                            //               ---- CHAR 3  ----   --------- CHAR 2 ---------   --------- CHAR 1 ---------
 154858488                            if ((thisQWord & 0xC0F0_C0C0_F0C0_C0F0ul) == 0x80E0_8080_E080_80E0ul && IsUtf8ContinuationBy
 489                            {
 490                                // Saw a proper bitmask for three incoming 3-byte sequences, perform the
 491                                // overlong and surrogate sequence checking now.
 492
 493                                // Check the first character.
 494                                // If the first character is overlong or a surrogate, fail immediately.
 495
 46100496                                if ((((uint)thisQWord & 0x200Fu) == 0) || ((((uint)thisQWord - 0x200Du) & 0x200Fu) == 0)
 497                                {
 498                                    goto Error;
 499                                }
 500
 501                                // Check the second character.
 502                                // At this point, we now know the first three bytes represent a well-formed sequence.
 503                                // If there's an error beyond here, we'll jump back to the "process three known good byt
 504                                // logic.
 505
 29146506                                thisQWord >>= 24;
 29146507                                if ((((uint)thisQWord & 0x200Fu) == 0) || ((((uint)thisQWord - 0x200Du) & 0x200Fu) == 0)
 508                                {
 509                                    goto ProcessSingleThreeByteSequenceSkipOverlongAndSurrogateChecks;
 510                                }
 511
 512                                // Check the third character (we already checked that it's followed by a continuation by
 513
 15394514                                thisQWord >>= 24;
 15394515                                if ((((uint)thisQWord & 0x200Fu) == 0) || ((((uint)thisQWord - 0x200Du) & 0x200Fu) == 0)
 516                                {
 517                                    goto ProcessSingleThreeByteSequenceSkipOverlongAndSurrogateChecks;
 518                                }
 519
 8970520                                pInputBuffer += 9;
 8970521                                tempUtf16CodeUnitCountAdjustment -= 6; // 9 UTF-8 bytes -> 3 UTF-16 code units (and 3 sc
 522
 8970523                                goto SuccessfullyProcessedThreeByteSequence;
 524                            }
 525
 526                            // Is this two 3-byte sequences in a row?
 527                            // thisQWord = [ ######## ######## | 10xxxxxx 10yyyyyy 1110zzzz | 10xxxxxx 10yyyyyy 1110zzzz
 528                            //                                   --------- CHAR 2 ---------   --------- CHAR 1 ---------
 108758529                            if ((thisQWord & 0xC0C0_F0C0_C0F0ul) == 0x8080_E080_80E0ul)
 530                            {
 531                                // Saw a proper bitmask for two incoming 3-byte sequences, perform the
 532                                // overlong and surrogate sequence checking now.
 533
 534                                // Check the first character.
 535                                // If the first character is overlong or a surrogate, fail immediately.
 536
 28524537                                if ((((uint)thisQWord & 0x200Fu) == 0) || ((((uint)thisQWord - 0x200Du) & 0x200Fu) == 0)
 538                                {
 539                                    goto Error;
 540                                }
 541
 542                                // Check the second character.
 543                                // At this point, we now know the first three bytes represent a well-formed sequence.
 544                                // If there's an error beyond here, we'll jump back to the "process three known good byt
 545                                // logic.
 546
 14342547                                thisQWord >>= 24;
 14342548                                if ((((uint)thisQWord & 0x200Fu) == 0) || ((((uint)thisQWord - 0x200Du) & 0x200Fu) == 0)
 549                                {
 550                                    goto ProcessSingleThreeByteSequenceSkipOverlongAndSurrogateChecks;
 551                                }
 552
 6000553                                pInputBuffer += 6;
 6000554                                tempUtf16CodeUnitCountAdjustment -= 4; // 6 UTF-8 bytes -> 2 UTF-16 code units (and 2 sc
 555
 556                                // The next byte in the sequence didn't have a 3-byte marker, so it's probably
 557                                // an ASCII character. Jump back to the beginning of loop processing.
 558
 6000559                                continue;
 560                            }
 561
 80234562                            if (UInt32BeginsWithUtf8ThreeByteMask(thisDWord))
 563                            {
 564                                // A single three-byte sequence.
 31210565                                goto ProcessThreeByteSequenceWithCheck;
 566                            }
 567                            else
 568                            {
 569                                // Not a three-byte sequence; perhaps ASCII?
 570                                goto AfterReadDWord;
 571                            }
 572                        }
 573                    }
 574
 3776575                    if (pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer)
 576                    {
 2464577                        thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer);
 578
 579                        // Optimization: A three-byte character could indicate CJK text, which makes it likely
 580                        // that the character following this one is also CJK. We'll check for a three-byte sequence
 581                        // marker now and jump directly to three-byte sequence processing if we see one, skipping
 582                        // all of the logic at the beginning of the loop.
 583
 2464584                        if (UInt32BeginsWithUtf8ThreeByteMask(thisDWord))
 585                        {
 992586                            goto ProcessThreeByteSequenceWithCheck; // Found another [not yet validated] three-byte sequ
 587                        }
 588                        else
 589                        {
 590                            goto AfterReadDWord; // Probably ASCII punctuation or whitespace; go back to start of loop
 591                        }
 592                    }
 593                    else
 594                    {
 595                        goto ProcessRemainingBytesSlow; // Running out of data
 596                    }
 597                }
 598
 599                // Assume the 4-byte case, but we need to validate.
 600
 594848601                if (BitConverter.IsLittleEndian)
 602                {
 594848603                    thisDWord &= 0xC0C0_FFFFu;
 604
 605                    // After the above modifications earlier in this method, we expect 'thisDWord'
 606                    // to have the structure [ 10000000 00000000 00uuzzzz 00010uuu ]. We'll now
 607                    // perform two checks to confirm this. The first will verify the
 608                    // [ 10000000 00000000 00###### ######## ] structure by taking advantage of two's
 609                    // complement representation to perform a single *signed* integer check.
 610
 594848611                    if ((int)thisDWord > unchecked((int)0x8000_3FFF))
 612                    {
 613                        goto Error; // didn't have three trailing bytes
 614                    }
 615
 616                    // Now we want to confirm that 0x01 <= uuuuu (otherwise this is an overlong encoding)
 617                    // and that uuuuu <= 0x10 (otherwise this is an out-of-range encoding).
 618
 49878619                    thisDWord = BitOperations.RotateRight(thisDWord, 8);
 620
 621                    // Now, thisDWord = [ 00010uuu 10000000 00000000 00uuzzzz ].
 622                    // The check is now a simple add / cmp / jcc combo.
 623
 49878624                    if (!UnicodeUtility.IsInRangeInclusive(thisDWord, 0x1080_0010u, 0x1480_000Fu))
 625                    {
 22648626                        goto Error; // overlong or out-of-range
 627                    }
 628                }
 629                else
 630                {
 631                    thisDWord -= 0x80u;
 632
 633                    // After the above modifications earlier in this method, we expect 'thisDWord'
 634                    // to have the structure [ 00010uuu 00uuzzzz 00yyyyyy 00xxxxxx ]. We'll now
 635                    // perform two checks to confirm this. The first will verify the
 636                    // [ ######## 00###### 00###### 00###### ] structure.
 637
 638                    if ((thisDWord & 0x00C0_C0C0u) != 0)
 639                    {
 640                        goto Error; // didn't have three trailing bytes
 641                    }
 642
 643                    // Now we want to confirm that 0x01 <= uuuuu (otherwise this is an overlong encoding)
 644                    // and that uuuuu <= 0x10 (otherwise this is an out-of-range encoding).
 645                    // This is a simple range check. (We don't care about the low two bytes.)
 646
 647                    if (!UnicodeUtility.IsInRangeInclusive(thisDWord, 0x1010_0000u, 0x140F_FFFFu))
 648                    {
 649                        goto Error; // overlong or out-of-range
 650                    }
 651                }
 652
 653                // Validation of 4-byte case complete.
 654
 27230655                pInputBuffer += 4;
 27230656                tempUtf16CodeUnitCountAdjustment -= 2; // 4 UTF-8 bytes -> 2 UTF-16 code units
 27230657                tempScalarCountAdjustment--; // 2 UTF-16 code units -> 1 scalar
 658
 659                continue; // go back to beginning of loop for processing
 660            }
 661
 1246662            goto ProcessRemainingBytesSlow;
 663
 664        ProcessInputOfLessThanDWordSize:
 665
 12834666            Debug.Assert(inputLength < 4);
 12834667            nuint inputBufferRemainingBytes = (uint)inputLength;
 12834668            goto ProcessSmallBufferCommon;
 669
 670        ProcessRemainingBytesSlow:
 671
 3184672            inputBufferRemainingBytes = (nuint)(void*)Unsafe.ByteOffset(ref *pInputBuffer, ref *pFinalPosWhereCanReadDWo
 673
 674        ProcessSmallBufferCommon:
 675
 16018676            Debug.Assert(inputBufferRemainingBytes < 4);
 18364677            while (inputBufferRemainingBytes > 0)
 678            {
 16418679                uint firstByte = pInputBuffer[0];
 680
 16418681                if ((byte)firstByte < 0x80u)
 682                {
 683                    // 1-byte (ASCII) case
 860684                    pInputBuffer++;
 860685                    inputBufferRemainingBytes--;
 860686                    continue;
 687                }
 15558688                else if (inputBufferRemainingBytes >= 2)
 689                {
 10536690                    uint secondByte = pInputBuffer[1]; // typed as 32-bit since we perform arithmetic (not just comparis
 10536691                    if ((byte)firstByte < 0xE0u)
 692                    {
 693                        // 2-byte case
 6470694                        if ((byte)firstByte >= 0xC2u && IsLowByteUtf8ContinuationByte(secondByte))
 695                        {
 1082696                            pInputBuffer += 2;
 1082697                            tempUtf16CodeUnitCountAdjustment--; // 2 UTF-8 bytes -> 1 UTF-16 code unit (and 1 scalar)
 1082698                            inputBufferRemainingBytes -= 2;
 1082699                            continue;
 700                        }
 701                    }
 4066702                    else if (inputBufferRemainingBytes >= 3)
 703                    {
 2304704                        if ((byte)firstByte < 0xF0u)
 705                        {
 1160706                            if ((byte)firstByte == 0xE0u)
 707                            {
 100708                                if (!UnicodeUtility.IsInRangeInclusive(secondByte, 0xA0u, 0xBFu))
 709                                {
 96710                                    goto Error; // overlong encoding
 711                                }
 712                            }
 1060713                            else if ((byte)firstByte == 0xEDu)
 714                            {
 168715                                if (!UnicodeUtility.IsInRangeInclusive(secondByte, 0x80u, 0x9Fu))
 716                                {
 156717                                    goto Error; // would be a UTF-16 surrogate code point
 718                                }
 719                            }
 720                            else
 721                            {
 892722                                if (!IsLowByteUtf8ContinuationByte(secondByte))
 723                                {
 724                                    goto Error; // first trailing byte doesn't have proper continuation marker
 725                                }
 726                            }
 727
 550728                            if (IsUtf8ContinuationByte(in pInputBuffer[2]))
 729                            {
 404730                                pInputBuffer += 3;
 404731                                tempUtf16CodeUnitCountAdjustment -= 2; // 3 UTF-8 bytes -> 2 UTF-16 code units (and 2 sc
 404732                                inputBufferRemainingBytes -= 3;
 733                                continue;
 734                            }
 735                        }
 736                    }
 737                }
 738
 739                // Error - no match.
 740
 741                goto Error;
 742            }
 743
 744            // If we reached this point, we're out of data, and we saw no bad UTF8 sequence.
 745
 746#if DEBUG
 747            // Quick check that for the success case we're going to fulfill our contract of returning &inputBuffer[input
 1946748            Debug.Assert(pOriginalInputBuffer + originalInputLength == pInputBuffer, "About to return an unexpected valu
 749#endif
 750
 751        Error:
 752
 753            // Report back to our caller how far we got before seeing invalid data.
 754            // (Also used for normal termination when falling out of the loop above.)
 755
 669046756            utf16CodeUnitCountAdjustment = tempUtf16CodeUnitCountAdjustment;
 669046757            scalarCountAdjustment = tempScalarCountAdjustment;
 669046758            return pInputBuffer;
 759        }
 760
 761#if NET
 762        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 763        [CompExactlyDependsOn(typeof(AdvSimd.Arm64))]
 764        private static ulong GetNonAsciiBytes(Vector128<byte> value, Vector128<byte> bitMask128)
 765        {
 766            if (!AdvSimd.Arm64.IsSupported || !BitConverter.IsLittleEndian)
 767            {
 768                throw new PlatformNotSupportedException();
 769            }
 770
 771            Vector128<byte> mostSignificantBitIsSet = (value.AsSByte() >> 7).AsByte();
 772            Vector128<byte> extractedBits = mostSignificantBitIsSet & bitMask128;
 773            extractedBits = AdvSimd.Arm64.AddPairwise(extractedBits, extractedBits);
 774            return extractedBits.AsUInt64().ToScalar();
 775        }
 776#endif
 777    }
 778}
 779

Methods/Properties

GetIndexOfFirstInvalidUtf8Sequence(System.ReadOnlySpan`1<System.Byte>,System.Boolean&)
AllBytesInUInt32AreAscii(System.UInt32)
AllBytesInUInt64AreAscii(System.UInt64)
ConvertAllAsciiBytesInUInt32ToLowercase(System.UInt32)
ConvertAllAsciiBytesInUInt32ToUppercase(System.UInt32)
ConvertAllAsciiBytesInUInt64ToUppercase(System.UInt64)
ConvertAllAsciiBytesInUInt64ToLowercase(System.UInt64)
UInt64OrdinalIgnoreCaseAscii(System.UInt64,System.UInt64)
AllBytesInVector128AreAscii(System.Runtime.Intrinsics.Vector128`1<System.Byte>)
ExtractCharFromFirstThreeByteSequence(System.UInt32)
ExtractCharFromFirstTwoByteSequence(System.UInt32)
ExtractCharsFromFourByteSequence(System.UInt32)
ExtractFourUtf8BytesFromSurrogatePair(System.UInt32)
ExtractTwoCharsPackedFromTwoAdjacentTwoByteSequences(System.UInt32)
ExtractTwoUtf8TwoByteSequencesFromTwoPackedUtf16Chars(System.UInt32)
ExtractUtf8TwoByteSequenceFromFirstUtf16Char(System.UInt32)
IsLowByteUtf8ContinuationByte(System.UInt32)
IsUtf8ContinuationByte(System.Byte&)
ToLittleEndian(System.UInt32)
UInt32BeginsWithOverlongUtf8TwoByteSequence(System.UInt32)
UInt32BeginsWithValidUtf8TwoByteSequenceLittleEndian(System.UInt32)
UInt32EndsWithValidUtf8TwoByteSequenceLittleEndian(System.UInt32)
WriteTwoUtf16CharsAsTwoUtf8ThreeByteSequences(System.Byte&,System.UInt32)
WriteFirstUtf16CharAsUtf8ThreeByteSequence(System.Byte&,System.UInt32)
TranscodeToUtf16(System.Byte*,System.Int32,System.Char*,System.Int32,System.Byte*&,System.Char*&)
TranscodeToUtf8(System.Char*,System.Int32,System.Byte*,System.Int32,System.Char*&,System.Byte*&)
GetPointerToFirstInvalidByte(System.Byte*,System.Int32,System.Int32&,System.Int32&)