< Summary

Line coverage
0%
Covered lines: 0
Uncovered lines: 321
Coverable lines: 321
Total lines: 1222
Line coverage: 0%
Branch coverage
0%
Covered branches: 0
Total branches: 186
Branch coverage: 0%
Method coverage

Feature is only available for sponsors

Upgrade to PRO version

Metrics

File(s)

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.CoreLib/src/System/Text/Latin1Utility.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Diagnostics;
 5using System.Diagnostics.CodeAnalysis;
 6using System.Numerics;
 7using System.Runtime.CompilerServices;
 8using System.Runtime.Intrinsics;
 9using System.Runtime.Intrinsics.X86;
 10
 11namespace System.Text
 12{
 13    internal static partial class Latin1Utility
 14    {
 15        /// <summary>
 16        /// Returns the index in <paramref name="pBuffer"/> where the first non-Latin1 char is found.
 17        /// Returns <paramref name="bufferLength"/> if the buffer is empty or all-Latin1.
 18        /// </summary>
 19        /// <returns>A Latin-1 char is defined as 0x0000 - 0x00FF, inclusive.</returns>
 20        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 21        public static unsafe nuint GetIndexOfFirstNonLatin1Char(char* pBuffer, nuint bufferLength /* in chars */)
 22        {
 23            // If SSE2 is supported, use those specific intrinsics instead of the generic vectorized
 24            // code below. This has two benefits: (a) we can take advantage of specific instructions like
 25            // pmovmskb which we know are optimized, and (b) we can avoid downclocking the processor while
 26            // this method is running.
 27
 028            return (Sse2.IsSupported)
 029                ? GetIndexOfFirstNonLatin1Char_Sse2(pBuffer, bufferLength)
 030                : GetIndexOfFirstNonLatin1Char_Default(pBuffer, bufferLength);
 31        }
 32
 33        private static unsafe nuint GetIndexOfFirstNonLatin1Char_Default(char* pBuffer, nuint bufferLength /* in chars *
 34        {
 35            // Squirrel away the original buffer reference.This method works by determining the exact
 36            // char reference where non-Latin1 data begins, so we need this base value to perform the
 37            // final subtraction at the end of the method to get the index into the original buffer.
 38
 039            char* pOriginalBuffer = pBuffer;
 40
 041            Debug.Assert(bufferLength <= nuint.MaxValue / sizeof(char));
 42
 43            // Before we drain off char-by-char, try a generic vectorized loop.
 44            // Only run the loop if we have at least two vectors we can pull out.
 45
 046            if (Vector.IsHardwareAccelerated && bufferLength >= 2 * (uint)Vector<ushort>.Count)
 47            {
 048                uint SizeOfVectorInChars = (uint)Vector<ushort>.Count; // JIT will make this a const
 049                uint SizeOfVectorInBytes = (uint)Vector<byte>.Count; // JIT will make this a const
 50
 051                Vector<ushort> maxLatin1 = new Vector<ushort>(0x00FF);
 52
 053                if (Vector.LessThanOrEqualAll(Unsafe.ReadUnaligned<Vector<ushort>>(pBuffer), maxLatin1))
 54                {
 55                    // The first several elements of the input buffer were Latin-1. Bump up the pointer to the
 56                    // next aligned boundary, then perform aligned reads from here on out until we find non-Latin-1
 57                    // data or we approach the end of the buffer. It's possible we'll reread data; this is ok.
 58
 059                    char* pFinalVectorReadPos = pBuffer + bufferLength - SizeOfVectorInChars;
 060                    pBuffer = (char*)(((nuint)pBuffer + SizeOfVectorInBytes) & ~(nuint)(SizeOfVectorInBytes - 1));
 61
 62#if DEBUG
 063                    long numCharsRead = pBuffer - pOriginalBuffer;
 064                    Debug.Assert(0 < numCharsRead && numCharsRead <= SizeOfVectorInChars, "We should've made forward pro
 065                    Debug.Assert((nuint)numCharsRead <= bufferLength, "We shouldn't have read past the end of the input 
 66#endif
 67
 068                    Debug.Assert(pBuffer <= pFinalVectorReadPos, "Should be able to read at least one vector.");
 69
 70                    do
 71                    {
 072                        Debug.Assert((nuint)pBuffer % SizeOfVectorInChars == 0, "Vector read should be aligned.");
 073                        if (Vector.GreaterThanAny(Unsafe.Read<Vector<ushort>>(pBuffer), maxLatin1))
 74                        {
 75                            break; // found non-Latin-1 data
 76                        }
 077                        pBuffer += SizeOfVectorInChars;
 078                    } while (pBuffer <= pFinalVectorReadPos);
 79
 80                    // Adjust the remaining buffer length for the number of elements we just consumed.
 81
 082                    bufferLength -= ((nuint)pBuffer - (nuint)pOriginalBuffer) / sizeof(char);
 83                }
 84            }
 85
 86            // At this point, the buffer length wasn't enough to perform a vectorized search, or we did perform
 87            // a vectorized search and encountered non-Latin-1 data. In either case go down a non-vectorized code
 88            // path to drain any remaining Latin-1 chars.
 89            //
 90            // We're going to perform unaligned reads, so prefer 32-bit reads instead of 64-bit reads.
 91            // This also allows us to perform more optimized bit twiddling tricks to count the number of Latin-1 chars.
 92
 93            uint currentUInt32;
 94
 95            // Try reading 64 bits at a time in a loop.
 96
 097            for (; bufferLength >= 4; bufferLength -= 4) // 64 bits = 4 * 16-bit chars
 98            {
 099                currentUInt32 = Unsafe.ReadUnaligned<uint>(pBuffer);
 0100                uint nextUInt32 = Unsafe.ReadUnaligned<uint>(pBuffer + 4 / sizeof(char));
 101
 0102                if (!AllCharsInUInt32AreLatin1(currentUInt32 | nextUInt32))
 103                {
 104                    // One of these two values contains non-Latin-1 chars.
 105                    // Figure out which one it is, then put it in 'current' so that we can drain the Latin-1 chars.
 106
 0107                    if (AllCharsInUInt32AreLatin1(currentUInt32))
 108                    {
 0109                        currentUInt32 = nextUInt32;
 0110                        pBuffer += 2;
 111                    }
 112
 0113                    goto FoundNonLatin1Data;
 114                }
 115
 0116                pBuffer += 4; // consumed 4 Latin-1 chars
 117            }
 118
 119            // From this point forward we don't need to keep track of the remaining buffer length.
 120            // Try reading 32 bits.
 121
 0122            if ((bufferLength & 2) != 0) // 32 bits = 2 * 16-bit chars
 123            {
 0124                currentUInt32 = Unsafe.ReadUnaligned<uint>(pBuffer);
 0125                if (!AllCharsInUInt32AreLatin1(currentUInt32))
 126                {
 127                    goto FoundNonLatin1Data;
 128                }
 129
 0130                pBuffer += 2;
 131            }
 132
 133            // Try reading 16 bits.
 134            // No need to try an 8-bit read after this since we're working with chars.
 135
 0136            if ((bufferLength & 1) != 0)
 137            {
 138                // If the buffer contains non-Latin-1 data, the comparison below will fail, and
 139                // we'll end up not incrementing the buffer reference.
 140
 0141                if (*pBuffer <= byte.MaxValue)
 142                {
 0143                    pBuffer++;
 144                }
 145            }
 146
 147        Finish:
 148
 0149            nuint totalNumBytesRead = (nuint)pBuffer - (nuint)pOriginalBuffer;
 0150            Debug.Assert(totalNumBytesRead % sizeof(char) == 0, "Total number of bytes read should be even since we're w
 0151            return totalNumBytesRead / sizeof(char); // convert byte count -> char count before returning
 152
 153        FoundNonLatin1Data:
 154
 0155            Debug.Assert(!AllCharsInUInt32AreLatin1(currentUInt32), "Shouldn't have reached this point if we have an all
 156
 157            // We don't bother looking at the second char - only the first char.
 158
 0159            if (FirstCharInUInt32IsLatin1(currentUInt32))
 160            {
 0161                pBuffer++;
 162            }
 163
 0164            goto Finish;
 165        }
 166
 167        [CompExactlyDependsOn(typeof(Sse2))]
 168        private static unsafe nuint GetIndexOfFirstNonLatin1Char_Sse2(char* pBuffer, nuint bufferLength /* in chars */)
 169        {
 170            // This method contains logic optimized for both SSE2 and SSE41. Much of the logic in this method
 171            // will be elided by JIT once we determine which specific ISAs we support.
 172
 173            // Quick check for empty inputs.
 174
 0175            if (bufferLength == 0)
 176            {
 0177                return 0;
 178            }
 179
 180            // JIT turns the below into constants
 181
 0182            uint SizeOfVector128InBytes = (uint)sizeof(Vector128<byte>);
 0183            uint SizeOfVector128InChars = SizeOfVector128InBytes / sizeof(char);
 184
 0185            Debug.Assert(Sse2.IsSupported, "Should've been checked by caller.");
 0186            Debug.Assert(BitConverter.IsLittleEndian, "SSE2 assumes little-endian.");
 187
 188            Vector128<ushort> firstVector, secondVector;
 189            uint currentMask;
 0190            char* pOriginalBuffer = pBuffer;
 191
 0192            if (bufferLength < SizeOfVector128InChars)
 193            {
 194                goto InputBufferLessThanOneVectorInLength; // can't vectorize; drain primitives instead
 195            }
 196
 197            // This method is written such that control generally flows top-to-bottom, avoiding
 198            // jumps as much as possible in the optimistic case of "all Latin-1". If we see non-Latin-1
 199            // data, we jump out of the hot paths to targets at the end of the method.
 200
 0201            Vector128<ushort> latin1MaskForTestZ = Vector128.Create((ushort)0xFF00); // used for PTEST on supported hard
 0202            Vector128<ushort> latin1MaskForAddSaturate = Vector128.Create((ushort)0x7F00); // used for PADDUSW
 203            const uint NonLatin1DataSeenMask = 0b_1010_1010_1010_1010; // used for determining whether 'currentMask' con
 204
 0205            Debug.Assert(bufferLength <= nuint.MaxValue / sizeof(char));
 206
 207            // Read the first vector unaligned.
 208
 0209            firstVector = Sse2.LoadVector128((ushort*)pBuffer); // unaligned load
 210
 211            // The operation below forces the 0x8000 bit of each WORD to be set iff the WORD element
 212            // has value >= 0x0100 (non-Latin-1). Then we'll treat the vector as a BYTE vector in order
 213            // to extract the mask. Reminder: the 0x0080 bit of each WORD should be ignored.
 214
 0215            currentMask = (uint)Sse2.MoveMask(Sse2.AddSaturate(firstVector, latin1MaskForAddSaturate).AsByte());
 216
 0217            if ((currentMask & NonLatin1DataSeenMask) != 0)
 218            {
 219                goto FoundNonLatin1DataInCurrentMask;
 220            }
 221
 222            // If we have less than 32 bytes to process, just go straight to the final unaligned
 223            // read. There's no need to mess with the loop logic in the middle of this method.
 224
 225            // Adjust the remaining length to account for what we just read.
 226            // For the remainder of this code path, bufferLength will be in bytes, not chars.
 227
 0228            bufferLength <<= 1; // chars to bytes
 229
 0230            if (bufferLength < 2 * SizeOfVector128InBytes)
 231            {
 232                goto IncrementCurrentOffsetBeforeFinalUnalignedVectorRead;
 233            }
 234
 235            // Now adjust the read pointer so that future reads are aligned.
 236
 0237            pBuffer = (char*)(((nuint)pBuffer + SizeOfVector128InBytes) & ~(nuint)(SizeOfVector128InBytes - 1));
 238
 239#if DEBUG
 0240            long numCharsRead = pBuffer - pOriginalBuffer;
 0241            Debug.Assert(0 < numCharsRead && numCharsRead <= SizeOfVector128InChars, "We should've made forward progress
 0242            Debug.Assert((nuint)numCharsRead <= bufferLength, "We shouldn't have read past the end of the input buffer."
 243#endif
 244
 245            // Adjust remaining buffer length.
 246
 0247            bufferLength += (nuint)pOriginalBuffer;
 0248            bufferLength -= (nuint)pBuffer;
 249
 250            // The buffer is now properly aligned.
 251            // Read 2 vectors at a time if possible.
 252
 0253            if (bufferLength >= 2 * SizeOfVector128InBytes)
 254            {
 0255                char* pFinalVectorReadPos = (char*)((nuint)pBuffer + bufferLength - 2 * SizeOfVector128InBytes);
 256
 257                // After this point, we no longer need to update the bufferLength value.
 258
 259                do
 260                {
 0261                    firstVector = Sse2.LoadAlignedVector128((ushort*)pBuffer);
 0262                    secondVector = Sse2.LoadAlignedVector128((ushort*)pBuffer + SizeOfVector128InChars);
 0263                    Vector128<ushort> combinedVector = firstVector | secondVector;
 264
 265#pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau
 0266                    if (Sse41.IsSupported)
 267#pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough
 268                    {
 269                        // If a non-Latin-1 bit is set in any WORD of the combined vector, we have seen non-Latin-1 data
 270                        // Jump to the non-Latin-1 handler to figure out which particular vector contained non-Latin-1 d
 0271                        if ((combinedVector & latin1MaskForTestZ) != Vector128<ushort>.Zero)
 272                        {
 0273                            goto FoundNonLatin1DataInFirstOrSecondVector;
 274                        }
 275                    }
 276                    else
 277                    {
 278                        // See comment earlier in the method for an explanation of how the below logic works.
 0279                        currentMask = (uint)Sse2.MoveMask(Sse2.AddSaturate(combinedVector, latin1MaskForAddSaturate).AsB
 0280                        if ((currentMask & NonLatin1DataSeenMask) != 0)
 281                        {
 282                            goto FoundNonLatin1DataInFirstOrSecondVector;
 283                        }
 284                    }
 285
 0286                    pBuffer += 2 * SizeOfVector128InChars;
 0287                } while (pBuffer <= pFinalVectorReadPos);
 288            }
 289
 290            // We have somewhere between 0 and (2 * vector length) - 1 bytes remaining to read from.
 291            // Since the above loop doesn't update bufferLength, we can't rely on its absolute value.
 292            // But we _can_ rely on it to tell us how much remaining data must be drained by looking
 293            // at what bits of it are set. This works because had we updated it within the loop above,
 294            // we would've been adding 2 * SizeOfVector128 on each iteration, but we only care about
 295            // bits which are less significant than those that the addition would've acted on.
 296
 297            // If there is fewer than one vector length remaining, skip the next aligned read.
 298            // Remember, at this point bufferLength is measured in bytes, not chars.
 299
 0300            if ((bufferLength & SizeOfVector128InBytes) == 0)
 301            {
 302                goto DoFinalUnalignedVectorRead;
 303            }
 304
 305            // At least one full vector's worth of data remains, so we can safely read it.
 306            // Remember, at this point pBuffer is still aligned.
 307
 0308            firstVector = Sse2.LoadAlignedVector128((ushort*)pBuffer);
 309
 310#pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau
 0311            if (Sse41.IsSupported)
 312#pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough
 313            {
 314                // If a non-Latin-1 bit is set in any WORD of the combined vector, we have seen non-Latin-1 data.
 315                // Jump to the non-Latin-1 handler to figure out which particular vector contained non-Latin-1 data.
 0316                if ((firstVector & latin1MaskForTestZ) != Vector128<ushort>.Zero)
 317                {
 0318                    goto FoundNonLatin1DataInFirstVector;
 319                }
 320            }
 321            else
 322            {
 323                // See comment earlier in the method for an explanation of how the below logic works.
 0324                currentMask = (uint)Sse2.MoveMask(Sse2.AddSaturate(firstVector, latin1MaskForAddSaturate).AsByte());
 0325                if ((currentMask & NonLatin1DataSeenMask) != 0)
 326                {
 327                    goto FoundNonLatin1DataInCurrentMask;
 328                }
 329            }
 330
 331        IncrementCurrentOffsetBeforeFinalUnalignedVectorRead:
 332
 0333            pBuffer += SizeOfVector128InChars;
 334
 335        DoFinalUnalignedVectorRead:
 336
 0337            if (((byte)bufferLength & (SizeOfVector128InBytes - 1)) != 0)
 338            {
 339                // Perform an unaligned read of the last vector.
 340                // We need to adjust the pointer because we're re-reading data.
 341
 0342                pBuffer = (char*)((byte*)pBuffer + (bufferLength & (SizeOfVector128InBytes - 1)) - SizeOfVector128InByte
 0343                firstVector = Sse2.LoadVector128((ushort*)pBuffer); // unaligned load
 344
 345#pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau
 0346                if (Sse41.IsSupported)
 347#pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough
 348                {
 349                    // If a non-Latin-1 bit is set in any WORD of the combined vector, we have seen non-Latin-1 data.
 350                    // Jump to the non-Latin-1 handler to figure out which particular vector contained non-Latin-1 data.
 0351                    if ((firstVector & latin1MaskForTestZ) != Vector128<ushort>.Zero)
 352                    {
 0353                        goto FoundNonLatin1DataInFirstVector;
 354                    }
 355                }
 356                else
 357                {
 358                    // See comment earlier in the method for an explanation of how the below logic works.
 0359                    currentMask = (uint)Sse2.MoveMask(Sse2.AddSaturate(firstVector, latin1MaskForAddSaturate).AsByte());
 0360                    if ((currentMask & NonLatin1DataSeenMask) != 0)
 361                    {
 362                        goto FoundNonLatin1DataInCurrentMask;
 363                    }
 364                }
 365
 0366                pBuffer += SizeOfVector128InChars;
 367            }
 368
 369        Finish:
 370
 0371            Debug.Assert(((nuint)pBuffer - (nuint)pOriginalBuffer) % 2 == 0, "Shouldn't have incremented any pointer by 
 0372            return ((nuint)pBuffer - (nuint)pOriginalBuffer) / sizeof(char); // and we're done! (remember to adjust for 
 373
 374        FoundNonLatin1DataInFirstOrSecondVector:
 375
 376            // We don't know if the first or the second vector contains non-Latin-1 data. Check the first
 377            // vector, and if that's all-Latin-1 then the second vector must be the culprit. Either way
 378            // we'll make sure the first vector local is the one that contains the non-Latin-1 data.
 379
 380            // See comment earlier in the method for an explanation of how the below logic works.
 381#pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau
 0382            if (Sse41.IsSupported)
 383#pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough
 384            {
 0385                if ((firstVector & latin1MaskForTestZ) != Vector128<ushort>.Zero)
 386                {
 0387                    goto FoundNonLatin1DataInFirstVector;
 388                }
 389            }
 390            else
 391            {
 0392                currentMask = (uint)Sse2.MoveMask(Sse2.AddSaturate(firstVector, latin1MaskForAddSaturate).AsByte());
 0393                if ((currentMask & NonLatin1DataSeenMask) != 0)
 394                {
 395                    goto FoundNonLatin1DataInCurrentMask;
 396                }
 397            }
 398
 399            // Wasn't the first vector; must be the second.
 400
 0401            pBuffer += SizeOfVector128InChars;
 0402            firstVector = secondVector;
 403
 404        FoundNonLatin1DataInFirstVector:
 405
 406            // See comment earlier in the method for an explanation of how the below logic works.
 0407            currentMask = (uint)Sse2.MoveMask(Sse2.AddSaturate(firstVector, latin1MaskForAddSaturate).AsByte());
 408
 409        FoundNonLatin1DataInCurrentMask:
 410
 411            // See comment earlier in the method accounting for the 0x8000 and 0x0080 bits set after the WORD-sized oper
 412
 0413            currentMask &= NonLatin1DataSeenMask;
 414
 415            // Now, the mask contains - from the LSB - a 0b00 pair for each Latin-1 char we saw, and a 0b10 pair for eac
 416            //
 417            // (Keep endianness in mind in the below examples.)
 418            // A non-Latin-1 char followed by two Latin-1 chars is 0b..._00_00_10. (tzcnt = 1)
 419            // A Latin-1 char followed by two non-Latin-1 chars is 0b..._10_10_00. (tzcnt = 3)
 420            // Two Latin-1 chars followed by a non-Latin-1 char is 0b..._10_00_00. (tzcnt = 5)
 421            //
 422            // This means tzcnt = 2 * numLeadingLatin1Chars + 1. We can conveniently take advantage of the fact
 423            // that the 2x multiplier already matches the char* stride length, then just subtract 1 at the end to
 424            // compute the correct final ending pointer value.
 425
 0426            Debug.Assert(currentMask != 0, "Shouldn't be here unless we see non-Latin-1 data.");
 0427            pBuffer = (char*)((byte*)pBuffer + (uint)BitOperations.TrailingZeroCount(currentMask) - 1);
 428
 0429            goto Finish;
 430
 431        FoundNonLatin1DataInCurrentDWord:
 432
 433            uint currentDWord;
 0434            Debug.Assert(!AllCharsInUInt32AreLatin1(currentDWord), "Shouldn't be here unless we see non-Latin-1 data.");
 435
 0436            if (FirstCharInUInt32IsLatin1(currentDWord))
 437            {
 0438                pBuffer++; // skip past the Latin-1 char
 439            }
 440
 0441            goto Finish;
 442
 443        InputBufferLessThanOneVectorInLength:
 444
 445            // These code paths get hit if the original input length was less than one vector in size.
 446            // We can't perform vectorized reads at this point, so we'll fall back to reading primitives
 447            // directly. Note that all of these reads are unaligned.
 448
 449            // Reminder: If this code path is hit, bufferLength is still a char count, not a byte count.
 450            // We skipped the code path that multiplied the count by sizeof(char).
 451
 0452            Debug.Assert(bufferLength < SizeOfVector128InChars);
 453
 454            // QWORD drain
 455
 0456            if ((bufferLength & 4) != 0)
 457            {
 458#pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau
 0459                if (Bmi1.X64.IsSupported)
 460#pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough
 461                {
 462                    // If we can use 64-bit tzcnt to count the number of leading Latin-1 chars, prefer it.
 463
 0464                    ulong candidateUInt64 = Unsafe.ReadUnaligned<ulong>(pBuffer);
 0465                    if (!AllCharsInUInt64AreLatin1(candidateUInt64))
 466                    {
 467                        // Clear the low 8 bits (the Latin-1 bits) of each char, then tzcnt.
 468                        // Remember the / 8 at the end to convert bit count to byte count,
 469                        // then the & ~1 at the end to treat a match in the high byte of
 470                        // any char the same as a match in the low byte of that same char.
 471
 0472                        candidateUInt64 &= 0xFF00FF00_FF00FF00ul;
 0473                        pBuffer = (char*)((byte*)pBuffer + ((nuint)(Bmi1.X64.TrailingZeroCount(candidateUInt64) / 8) & ~
 0474                        goto Finish;
 475                    }
 476                }
 477                else
 478                {
 479                    // If we can't use 64-bit tzcnt, no worries. We'll just do 2x 32-bit reads instead.
 480
 0481                    currentDWord = Unsafe.ReadUnaligned<uint>(pBuffer);
 0482                    uint nextDWord = Unsafe.ReadUnaligned<uint>(pBuffer + 4 / sizeof(char));
 483
 0484                    if (!AllCharsInUInt32AreLatin1(currentDWord | nextDWord))
 485                    {
 486                        // At least one of the values wasn't all-Latin-1.
 487                        // We need to figure out which one it was and stick it in the currentMask local.
 488
 0489                        if (AllCharsInUInt32AreLatin1(currentDWord))
 490                        {
 0491                            currentDWord = nextDWord; // this one is the culprit
 0492                            pBuffer += 4 / sizeof(char);
 493                        }
 494
 0495                        goto FoundNonLatin1DataInCurrentDWord;
 496                    }
 497                }
 498
 0499                pBuffer += 4; // successfully consumed 4 Latin-1 chars
 500            }
 501
 502            // DWORD drain
 503
 0504            if ((bufferLength & 2) != 0)
 505            {
 0506                currentDWord = Unsafe.ReadUnaligned<uint>(pBuffer);
 507
 0508                if (!AllCharsInUInt32AreLatin1(currentDWord))
 509                {
 510                    goto FoundNonLatin1DataInCurrentDWord;
 511                }
 512
 0513                pBuffer += 2; // successfully consumed 2 Latin-1 chars
 514            }
 515
 516            // WORD drain
 517            // This is the final drain; there's no need for a BYTE drain since our elemental type is 16-bit char.
 518
 0519            if ((bufferLength & 1) != 0)
 520            {
 0521                if (*pBuffer <= byte.MaxValue)
 522                {
 0523                    pBuffer++; // successfully consumed a single char
 524                }
 525            }
 526
 0527            goto Finish;
 528        }
 529
 530
 531        /// <summary>
 532        /// Copies as many Latin-1 characters (U+0000..U+00FF) as possible from <paramref name="pUtf16Buffer"/>
 533        /// to <paramref name="pLatin1Buffer"/>, stopping when the first non-Latin-1 character is encountered
 534        /// or once <paramref name="elementCount"/> elements have been converted. Returns the total number
 535        /// of elements that were able to be converted.
 536        /// </summary>
 537        public static unsafe nuint NarrowUtf16ToLatin1(char* pUtf16Buffer, byte* pLatin1Buffer, nuint elementCount)
 538        {
 539            nuint currentOffset = 0;
 540
 0541            uint utf16Data32BitsHigh = 0, utf16Data32BitsLow = 0;
 0542            ulong utf16Data64Bits = 0;
 543
 544            // If SSE2 is supported, use those specific intrinsics instead of the generic vectorized
 545            // code below. This has two benefits: (a) we can take advantage of specific instructions like
 546            // pmovmskb, ptest, vpminuw which we know are optimized, and (b) we can avoid downclocking the
 547            // processor while this method is running.
 548
 0549            if (Sse2.IsSupported)
 550            {
 0551                Debug.Assert(BitConverter.IsLittleEndian, "Assume little endian if SSE2 is supported.");
 552
 0553                if (elementCount >= 2 * (uint)sizeof(Vector128<byte>))
 554                {
 555                    // Since there's overhead to setting up the vectorized code path, we only want to
 556                    // call into it after a quick probe to ensure the next immediate characters really are Latin-1.
 557                    // If we see non-Latin-1 data, we'll jump immediately to the draining logic at the end of the method
 558
 559                    if (IntPtr.Size >= 8)
 560                    {
 0561                        utf16Data64Bits = Unsafe.ReadUnaligned<ulong>(pUtf16Buffer);
 0562                        if (!AllCharsInUInt64AreLatin1(utf16Data64Bits))
 563                        {
 0564                            goto FoundNonLatin1DataIn64BitRead;
 565                        }
 566                    }
 567                    else
 568                    {
 569                        utf16Data32BitsHigh = Unsafe.ReadUnaligned<uint>(pUtf16Buffer);
 570                        utf16Data32BitsLow = Unsafe.ReadUnaligned<uint>(pUtf16Buffer + 4 / sizeof(char));
 571                        if (!AllCharsInUInt32AreLatin1(utf16Data32BitsHigh | utf16Data32BitsLow))
 572                        {
 573                            goto FoundNonLatin1DataIn64BitRead;
 574                        }
 575                    }
 576
 0577                    currentOffset = NarrowUtf16ToLatin1_Sse2(pUtf16Buffer, pLatin1Buffer, elementCount);
 578                }
 579            }
 0580            else if (Vector.IsHardwareAccelerated)
 581            {
 0582                uint SizeOfVector = (uint)sizeof(Vector<byte>); // JIT will make this a const
 583
 584                // Only bother vectorizing if we have enough data to do so.
 0585                if (elementCount >= 2 * SizeOfVector)
 586                {
 587                    // Since there's overhead to setting up the vectorized code path, we only want to
 588                    // call into it after a quick probe to ensure the next immediate characters really are Latin-1.
 589                    // If we see non-Latin-1 data, we'll jump immediately to the draining logic at the end of the method
 590
 591                    if (IntPtr.Size >= 8)
 592                    {
 0593                        utf16Data64Bits = Unsafe.ReadUnaligned<ulong>(pUtf16Buffer);
 0594                        if (!AllCharsInUInt64AreLatin1(utf16Data64Bits))
 595                        {
 0596                            goto FoundNonLatin1DataIn64BitRead;
 597                        }
 598                    }
 599                    else
 600                    {
 601                        utf16Data32BitsHigh = Unsafe.ReadUnaligned<uint>(pUtf16Buffer);
 602                        utf16Data32BitsLow = Unsafe.ReadUnaligned<uint>(pUtf16Buffer + 4 / sizeof(char));
 603                        if (!AllCharsInUInt32AreLatin1(utf16Data32BitsHigh | utf16Data32BitsLow))
 604                        {
 605                            goto FoundNonLatin1DataIn64BitRead;
 606                        }
 607                    }
 608
 0609                    Vector<ushort> maxLatin1 = new Vector<ushort>(0x00FF);
 610
 0611                    nuint finalOffsetWhereCanLoop = elementCount - 2 * SizeOfVector;
 612                    do
 613                    {
 0614                        Vector<ushort> utf16VectorHigh = Unsafe.ReadUnaligned<Vector<ushort>>(pUtf16Buffer + currentOffs
 0615                        Vector<ushort> utf16VectorLow = Unsafe.ReadUnaligned<Vector<ushort>>(pUtf16Buffer + currentOffse
 616
 0617                        if (Vector.GreaterThanAny(Vector.BitwiseOr(utf16VectorHigh, utf16VectorLow), maxLatin1))
 618                        {
 619                            break; // found non-Latin-1 data
 620                        }
 621
 622                        // TODO: Is the below logic also valid for big-endian platforms?
 0623                        Vector<byte> latin1Vector = Vector.Narrow(utf16VectorHigh, utf16VectorLow);
 0624                        Unsafe.WriteUnaligned(pLatin1Buffer + currentOffset, latin1Vector);
 625
 0626                        currentOffset += SizeOfVector;
 0627                    } while (currentOffset <= finalOffsetWhereCanLoop);
 628                }
 629            }
 630
 0631            Debug.Assert(currentOffset <= elementCount);
 0632            nuint remainingElementCount = elementCount - currentOffset;
 633
 634            // Try to narrow 64 bits -> 32 bits at a time.
 635            // We needn't update remainingElementCount after this point.
 636
 0637            if (remainingElementCount >= 4)
 638            {
 0639                nuint finalOffsetWhereCanLoop = currentOffset + remainingElementCount - 4;
 640                do
 641                {
 642                    if (IntPtr.Size >= 8)
 643                    {
 644                        // Only perform QWORD reads on a 64-bit platform.
 0645                        utf16Data64Bits = Unsafe.ReadUnaligned<ulong>(pUtf16Buffer + currentOffset);
 0646                        if (!AllCharsInUInt64AreLatin1(utf16Data64Bits))
 647                        {
 648                            goto FoundNonLatin1DataIn64BitRead;
 649                        }
 650
 0651                        NarrowFourUtf16CharsToLatin1AndWriteToBuffer(ref pLatin1Buffer[currentOffset], utf16Data64Bits);
 652                    }
 653                    else
 654                    {
 655                        utf16Data32BitsHigh = Unsafe.ReadUnaligned<uint>(pUtf16Buffer + currentOffset);
 656                        utf16Data32BitsLow = Unsafe.ReadUnaligned<uint>(pUtf16Buffer + currentOffset + 4 / sizeof(char))
 657                        if (!AllCharsInUInt32AreLatin1(utf16Data32BitsHigh | utf16Data32BitsLow))
 658                        {
 659                            goto FoundNonLatin1DataIn64BitRead;
 660                        }
 661
 662                        NarrowTwoUtf16CharsToLatin1AndWriteToBuffer(ref pLatin1Buffer[currentOffset], utf16Data32BitsHig
 663                        NarrowTwoUtf16CharsToLatin1AndWriteToBuffer(ref pLatin1Buffer[currentOffset + 2], utf16Data32Bit
 664                    }
 665
 0666                    currentOffset += 4;
 0667                } while (currentOffset <= finalOffsetWhereCanLoop);
 668            }
 669
 670            // Try to narrow 32 bits -> 16 bits.
 671
 0672            if (((uint)remainingElementCount & 2) != 0)
 673            {
 0674                utf16Data32BitsHigh = Unsafe.ReadUnaligned<uint>(pUtf16Buffer + currentOffset);
 0675                if (!AllCharsInUInt32AreLatin1(utf16Data32BitsHigh))
 676                {
 677                    goto FoundNonLatin1DataInHigh32Bits;
 678                }
 679
 0680                NarrowTwoUtf16CharsToLatin1AndWriteToBuffer(ref pLatin1Buffer[currentOffset], utf16Data32BitsHigh);
 0681                currentOffset += 2;
 682            }
 683
 684            // Try to narrow 16 bits -> 8 bits.
 685
 0686            if (((uint)remainingElementCount & 1) != 0)
 687            {
 0688                utf16Data32BitsHigh = pUtf16Buffer[currentOffset];
 0689                if (utf16Data32BitsHigh <= byte.MaxValue)
 690                {
 0691                    pLatin1Buffer[currentOffset] = (byte)utf16Data32BitsHigh;
 0692                    currentOffset++;
 693                }
 694            }
 695
 696        Finish:
 697
 0698            return currentOffset;
 699
 700        FoundNonLatin1DataIn64BitRead:
 701
 702            if (IntPtr.Size >= 8)
 703            {
 704                // Try checking the first 32 bits of the buffer for non-Latin-1 data.
 705                // Regardless, we'll move the non-Latin-1 data into the utf16Data32BitsHigh local.
 706
 0707                if (BitConverter.IsLittleEndian)
 708                {
 0709                    utf16Data32BitsHigh = (uint)utf16Data64Bits;
 710                }
 711                else
 712                {
 713                    utf16Data32BitsHigh = (uint)(utf16Data64Bits >> 32);
 714                }
 715
 0716                if (AllCharsInUInt32AreLatin1(utf16Data32BitsHigh))
 717                {
 0718                    NarrowTwoUtf16CharsToLatin1AndWriteToBuffer(ref pLatin1Buffer[currentOffset], utf16Data32BitsHigh);
 719
 0720                    if (BitConverter.IsLittleEndian)
 721                    {
 0722                        utf16Data32BitsHigh = (uint)(utf16Data64Bits >> 32);
 723                    }
 724                    else
 725                    {
 726                        utf16Data32BitsHigh = (uint)utf16Data64Bits;
 727                    }
 728
 0729                    currentOffset += 2;
 730                }
 731            }
 732            else
 733            {
 734                // Need to determine if the high or the low 32-bit value contained non-Latin-1 data.
 735                // Regardless, we'll move the non-Latin-1 data into the utf16Data32BitsHigh local.
 736
 737                if (AllCharsInUInt32AreLatin1(utf16Data32BitsHigh))
 738                {
 739                    NarrowTwoUtf16CharsToLatin1AndWriteToBuffer(ref pLatin1Buffer[currentOffset], utf16Data32BitsHigh);
 740                    utf16Data32BitsHigh = utf16Data32BitsLow;
 741                    currentOffset += 2;
 742                }
 743            }
 744
 745        FoundNonLatin1DataInHigh32Bits:
 746
 0747            Debug.Assert(!AllCharsInUInt32AreLatin1(utf16Data32BitsHigh), "Shouldn't have reached this point if we have 
 748
 749            // There's at most one char that needs to be drained.
 750
 0751            if (FirstCharInUInt32IsLatin1(utf16Data32BitsHigh))
 752            {
 0753                if (!BitConverter.IsLittleEndian)
 754                {
 755                    utf16Data32BitsHigh >>= 16; // move high char down to low char
 756                }
 757
 0758                pLatin1Buffer[currentOffset] = (byte)utf16Data32BitsHigh;
 0759                currentOffset++;
 760            }
 761
 0762            goto Finish;
 763        }
 764
 765        [CompExactlyDependsOn(typeof(Sse2))]
 766        private static unsafe nuint NarrowUtf16ToLatin1_Sse2(char* pUtf16Buffer, byte* pLatin1Buffer, nuint elementCount
 767        {
 768            // This method contains logic optimized for both SSE2 and SSE41. Much of the logic in this method
 769            // will be elided by JIT once we determine which specific ISAs we support.
 770
 771            // JIT turns the below into constants
 772
 0773            uint SizeOfVector128 = (uint)sizeof(Vector128<byte>);
 0774            nuint MaskOfAllBitsInVector128 = SizeOfVector128 - 1;
 775
 776            // This method is written such that control generally flows top-to-bottom, avoiding
 777            // jumps as much as possible in the optimistic case of "all Latin-1". If we see non-Latin-1
 778            // data, we jump out of the hot paths to targets at the end of the method.
 779
 0780            Debug.Assert(Sse2.IsSupported);
 0781            Debug.Assert(BitConverter.IsLittleEndian);
 0782            Debug.Assert(elementCount >= 2 * SizeOfVector128);
 783
 0784            Vector128<short> latin1MaskForTestZ = Vector128.Create(unchecked((short)0xFF00)); // used for PTEST on suppo
 0785            Vector128<ushort> latin1MaskForAddSaturate = Vector128.Create((ushort)0x7F00); // used for PADDUSW
 786            const int NonLatin1DataSeenMask = 0b_1010_1010_1010_1010; // used for determining whether the pmovmskb opera
 787
 788            // First, perform an unaligned read of the first part of the input buffer.
 789
 0790            Vector128<short> utf16VectorFirst = Sse2.LoadVector128((short*)pUtf16Buffer); // unaligned load
 791
 792            // If there's non-Latin-1 data in the first 8 elements of the vector, there's nothing we can do.
 793            // See comments in GetIndexOfFirstNonLatin1Char_Sse2 for information about how this works.
 794
 795#pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau
 0796            if (Sse41.IsSupported)
 797#pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough
 798            {
 0799                if ((utf16VectorFirst & latin1MaskForTestZ) != Vector128<short>.Zero)
 800                {
 0801                    return 0;
 802                }
 803            }
 804            else
 805            {
 0806                if ((Sse2.MoveMask(Sse2.AddSaturate(utf16VectorFirst.AsUInt16(), latin1MaskForAddSaturate).AsByte()) & N
 807                {
 0808                    return 0;
 809                }
 810            }
 811
 812            // Turn the 8 Latin-1 chars we just read into 8 Latin-1 bytes, then copy it to the destination.
 813
 0814            Vector128<byte> latin1Vector = Sse2.PackUnsignedSaturate(utf16VectorFirst, utf16VectorFirst);
 0815            Sse2.StoreScalar((ulong*)pLatin1Buffer, latin1Vector.AsUInt64()); // ulong* calculated here is UNALIGNED
 816
 0817            nuint currentOffsetInElements = SizeOfVector128 / 2; // we processed 8 elements so far
 818
 819            // We're going to get the best performance when we have aligned writes, so we'll take the
 820            // hit of potentially unaligned reads in order to hit this sweet spot.
 821
 822            // pLatin1Buffer points to the start of the destination buffer, immediately before where we wrote
 823            // the 8 bytes previously. If the 0x08 bit is set at the pinned address, then the 8 bytes we wrote
 824            // previously mean that the 0x08 bit is *not* set at address &pLatin1Buffer[SizeOfVector128 / 2]. In
 825            // that case we can immediately back up to the previous aligned boundary and start the main loop.
 826            // If the 0x08 bit is *not* set at the pinned address, then it means the 0x08 bit *is* set at
 827            // address &pLatin1Buffer[SizeOfVector128 / 2], and we should perform one more 8-byte write to bump
 828            // just past the next aligned boundary address.
 829
 0830            if (((uint)pLatin1Buffer & (SizeOfVector128 / 2)) == 0)
 831            {
 832                // We need to perform one more partial vector write before we can get the alignment we want.
 833
 0834                utf16VectorFirst = Sse2.LoadVector128((short*)pUtf16Buffer + currentOffsetInElements); // unaligned load
 835
 836                // See comments earlier in this method for information about how this works.
 837#pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau
 0838                if (Sse41.IsSupported)
 839#pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough
 840                {
 0841                    if ((utf16VectorFirst & latin1MaskForTestZ) != Vector128<short>.Zero)
 842                    {
 0843                        goto Finish;
 844                    }
 845                }
 846                else
 847                {
 0848                    if ((Sse2.MoveMask(Sse2.AddSaturate(utf16VectorFirst.AsUInt16(), latin1MaskForAddSaturate).AsByte())
 849                    {
 850                        goto Finish;
 851                    }
 852                }
 853
 854                // Turn the 8 Latin-1 chars we just read into 8 Latin-1 bytes, then copy it to the destination.
 0855                latin1Vector = Sse2.PackUnsignedSaturate(utf16VectorFirst, utf16VectorFirst);
 0856                Sse2.StoreScalar((ulong*)(pLatin1Buffer + currentOffsetInElements), latin1Vector.AsUInt64()); // ulong* 
 857            }
 858
 859            // Calculate how many elements we wrote in order to get pLatin1Buffer to its next alignment
 860            // point, then use that as the base offset going forward.
 861
 0862            currentOffsetInElements = SizeOfVector128 - ((nuint)pLatin1Buffer & MaskOfAllBitsInVector128);
 0863            Debug.Assert(0 < currentOffsetInElements && currentOffsetInElements <= SizeOfVector128, "We wrote at least 1
 864
 0865            Debug.Assert(currentOffsetInElements <= elementCount, "Shouldn't have overrun the destination buffer.");
 0866            Debug.Assert(elementCount - currentOffsetInElements >= SizeOfVector128, "We should be able to run at least o
 867
 0868            nuint finalOffsetWhereCanRunLoop = elementCount - SizeOfVector128;
 869            do
 870            {
 871                // In a loop, perform two unaligned reads, narrow to a single vector, then aligned write one vector.
 872
 0873                utf16VectorFirst = Sse2.LoadVector128((short*)pUtf16Buffer + currentOffsetInElements); // unaligned load
 0874                Vector128<short> utf16VectorSecond = Sse2.LoadVector128((short*)pUtf16Buffer + currentOffsetInElements +
 0875                Vector128<short> combinedVector = utf16VectorFirst | utf16VectorSecond;
 876
 877                // See comments in GetIndexOfFirstNonLatin1Char_Sse2 for information about how this works.
 878#pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau
 0879                if (Sse41.IsSupported)
 880#pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough
 881                {
 0882                    if ((combinedVector & latin1MaskForTestZ) != Vector128<short>.Zero)
 883                    {
 0884                        goto FoundNonLatin1DataInLoop;
 885                    }
 886                }
 887                else
 888                {
 0889                    if ((Sse2.MoveMask(Sse2.AddSaturate(combinedVector.AsUInt16(), latin1MaskForAddSaturate).AsByte()) &
 890                    {
 891                        goto FoundNonLatin1DataInLoop;
 892                    }
 893                }
 894
 895                // Build up the Latin-1 vector and perform the store.
 896
 0897                latin1Vector = Sse2.PackUnsignedSaturate(utf16VectorFirst, utf16VectorSecond);
 898
 0899                Debug.Assert(((nuint)pLatin1Buffer + currentOffsetInElements) % SizeOfVector128 == 0, "Write should be a
 0900                Sse2.StoreAligned(pLatin1Buffer + currentOffsetInElements, latin1Vector); // aligned
 901
 0902                currentOffsetInElements += SizeOfVector128;
 0903            } while (currentOffsetInElements <= finalOffsetWhereCanRunLoop);
 904
 905        Finish:
 906
 907            // There might be some Latin-1 data left over. That's fine - we'll let our caller handle the final drain.
 0908            return currentOffsetInElements;
 909
 910        FoundNonLatin1DataInLoop:
 911
 912            // Can we at least narrow the high vector?
 913            // See comments in GetIndexOfFirstNonLatin1Char_Sse2 for information about how this works.
 914#pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau
 0915            if (Sse41.IsSupported)
 916#pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough
 917            {
 0918                if ((utf16VectorFirst & latin1MaskForTestZ) != Vector128<short>.Zero)
 919                {
 0920                    goto Finish; // found non-Latin-1 data
 921                }
 922            }
 923            else
 924            {
 0925                if ((Sse2.MoveMask(Sse2.AddSaturate(utf16VectorFirst.AsUInt16(), latin1MaskForAddSaturate).AsByte()) & N
 926                {
 927                    goto Finish; // found non-Latin-1 data
 928                }
 929            }
 930
 931            // First part was all Latin-1, narrow and aligned write. Note we're only filling in the low half of the vect
 0932            latin1Vector = Sse2.PackUnsignedSaturate(utf16VectorFirst, utf16VectorFirst);
 933
 0934            Debug.Assert(((nuint)pLatin1Buffer + currentOffsetInElements) % sizeof(ulong) == 0, "Destination should be u
 935
 0936            Sse2.StoreScalar((ulong*)(pLatin1Buffer + currentOffsetInElements), latin1Vector.AsUInt64()); // ulong* calc
 0937            currentOffsetInElements += SizeOfVector128 / 2;
 938
 0939            goto Finish;
 940        }
 941
 942        /// <summary>
 943        /// Copies Latin-1 (narrow character) data from <paramref name="pLatin1Buffer"/> to the UTF-16 (wide character)
 944        /// buffer <paramref name="pUtf16Buffer"/>, widening data while copying. <paramref name="elementCount"/>
 945        /// specifies the element count of both the source and destination buffers.
 946        /// </summary>
 947        public static unsafe void WidenLatin1ToUtf16(byte* pLatin1Buffer, char* pUtf16Buffer, nuint elementCount)
 948        {
 949            // If SSE2 is supported, use those specific intrinsics instead of the generic vectorized
 950            // code below. This has two benefits: (a) we can take advantage of specific instructions like
 951            // punpcklbw which we know are optimized, and (b) we can avoid downclocking the processor while
 952            // this method is running.
 953
 0954            if (Sse2.IsSupported)
 955            {
 0956                WidenLatin1ToUtf16_Sse2(pLatin1Buffer, pUtf16Buffer, elementCount);
 957            }
 958            else
 959            {
 0960                WidenLatin1ToUtf16_Fallback(pLatin1Buffer, pUtf16Buffer, elementCount);
 961            }
 0962        }
 963
 964        [CompExactlyDependsOn(typeof(Sse2))]
 965        private static unsafe void WidenLatin1ToUtf16_Sse2(byte* pLatin1Buffer, char* pUtf16Buffer, nuint elementCount)
 966        {
 967            // JIT turns the below into constants
 968
 0969            uint SizeOfVector128 = (uint)sizeof(Vector128<byte>);
 0970            nuint MaskOfAllBitsInVector128 = SizeOfVector128 - 1;
 971
 0972            Debug.Assert(Sse2.IsSupported);
 0973            Debug.Assert(BitConverter.IsLittleEndian);
 974
 0975            nuint currentOffset = 0;
 0976            Vector128<byte> zeroVector = Vector128<byte>.Zero;
 977            Vector128<byte> latin1Vector;
 978
 979            // We're going to get the best performance when we have aligned writes, so we'll take the
 980            // hit of potentially unaligned reads in order to hit this sweet spot. Our central loop
 981            // will perform 1x 128-bit reads followed by 2x 128-bit writes, so we want to make sure
 982            // we actually have 128 bits of input data before entering the loop.
 983
 0984            if (elementCount >= SizeOfVector128)
 985            {
 986                // First, perform an unaligned 1x 64-bit read from the input buffer and an unaligned
 987                // 1x 128-bit write to the destination buffer.
 988
 0989                latin1Vector = Sse2.LoadScalarVector128((ulong*)pLatin1Buffer).AsByte(); // unaligned load
 0990                Sse2.Store((byte*)pUtf16Buffer, Sse2.UnpackLow(latin1Vector, zeroVector)); // unaligned write
 991
 992                // Calculate how many elements we wrote in order to get pOutputBuffer to its next alignment
 993                // point, then use that as the base offset going forward. Remember the >> 1 to account for
 994                // that we wrote chars, not bytes. This means we may re-read data in the next iteration of
 995                // the loop, but this is ok.
 996
 0997                currentOffset = (SizeOfVector128 >> 1) - (((nuint)pUtf16Buffer >> 1) & (MaskOfAllBitsInVector128 >> 1));
 0998                Debug.Assert(0 < currentOffset && currentOffset <= SizeOfVector128 / sizeof(char));
 999
 1000                // Calculating the destination address outside the loop results in significant
 1001                // perf wins vs. relying on the JIT to fold memory addressing logic into the
 1002                // write instructions. See: https://github.com/dotnet/runtime/issues/33002
 1003
 01004                char* pCurrentWriteAddress = pUtf16Buffer + currentOffset;
 1005
 1006                // Now run the main 1x 128-bit read + 2x 128-bit write loop.
 1007
 01008                nuint finalOffsetWhereCanIterateLoop = elementCount - SizeOfVector128;
 01009                while (currentOffset <= finalOffsetWhereCanIterateLoop)
 1010                {
 01011                    latin1Vector = Sse2.LoadVector128(pLatin1Buffer + currentOffset); // unaligned load
 1012
 1013                    // Calculating the destination address in the below manner results in significant
 1014                    // performance wins vs. other patterns. See for more information:
 1015                    // https://github.com/dotnet/runtime/issues/33002
 1016
 01017                    Vector128<byte> low = Sse2.UnpackLow(latin1Vector, zeroVector);
 01018                    Sse2.StoreAligned((byte*)pCurrentWriteAddress, low);
 1019
 01020                    Vector128<byte> high = Sse2.UnpackHigh(latin1Vector, zeroVector);
 01021                    Sse2.StoreAligned((byte*)pCurrentWriteAddress + SizeOfVector128, high);
 1022
 01023                    currentOffset += SizeOfVector128;
 01024                    pCurrentWriteAddress += SizeOfVector128;
 1025                }
 1026            }
 1027
 01028            Debug.Assert(elementCount - currentOffset < SizeOfVector128, "Case where 2 vectors remained should've been i
 01029            uint remaining = (uint)elementCount - (uint)currentOffset;
 1030
 1031            // Now handle cases where we can't process two vectors at a time.
 1032
 01033            if ((remaining & 8) != 0)
 1034            {
 1035                // Read a single 64-bit vector; write a single 128-bit vector.
 1036
 01037                latin1Vector = Sse2.LoadScalarVector128((ulong*)(pLatin1Buffer + currentOffset)).AsByte(); // unaligned 
 01038                Sse2.Store((byte*)(pUtf16Buffer + currentOffset), Sse2.UnpackLow(latin1Vector, zeroVector)); // unaligne
 01039                currentOffset += 8;
 1040            }
 1041
 01042            if ((remaining & 4) != 0)
 1043            {
 1044                // Read a single 32-bit vector; write a single 64-bit vector.
 1045
 01046                latin1Vector = Sse2.LoadScalarVector128((uint*)(pLatin1Buffer + currentOffset)).AsByte(); // unaligned l
 01047                Sse2.StoreScalar((ulong*)(pUtf16Buffer + currentOffset), Sse2.UnpackLow(latin1Vector, zeroVector).AsUInt
 01048                currentOffset += 4;
 1049            }
 1050
 01051            if ((remaining & 3) != 0)
 1052            {
 1053                // 1, 2, or 3 bytes were left over
 01054                pUtf16Buffer[currentOffset] = (char)pLatin1Buffer[currentOffset];
 1055
 01056                if ((remaining & 2) != 0)
 1057                {
 1058                    // 2 or 3 bytes were left over
 01059                    pUtf16Buffer[currentOffset + 1] = (char)pLatin1Buffer[currentOffset + 1];
 1060
 01061                    if ((remaining & 1) != 0)
 1062                    {
 1063                        // 1 or 3 bytes were left over (and since '1' doesn't go down this branch, we know it was actual
 01064                        pUtf16Buffer[currentOffset + 2] = (char)pLatin1Buffer[currentOffset + 2];
 1065                    }
 1066                }
 1067            }
 01068        }
 1069
 1070        private static unsafe void WidenLatin1ToUtf16_Fallback(byte* pLatin1Buffer, char* pUtf16Buffer, nuint elementCou
 1071        {
 01072            Debug.Assert(!Sse2.IsSupported);
 1073
 01074            nuint currentOffset = 0;
 1075
 01076            if (Vector.IsHardwareAccelerated)
 1077            {
 1078                // In a loop, read 1x vector (unaligned) and write 2x vectors (unaligned).
 1079
 01080                uint SizeOfVector = (uint)Vector<byte>.Count; // JIT will make this a const
 1081
 1082                // Only bother vectorizing if we have enough data to do so.
 01083                if (elementCount >= SizeOfVector)
 1084                {
 01085                    nuint finalOffsetWhereCanIterate = elementCount - SizeOfVector;
 1086                    do
 1087                    {
 01088                        Vector<byte> latin1Vector = Unsafe.ReadUnaligned<Vector<byte>>(pLatin1Buffer + currentOffset);
 01089                        Vector.Widen(Vector.AsVectorByte(latin1Vector), out Vector<ushort> utf16LowVector, out Vector<us
 1090
 1091                        // TODO: Is the below logic also valid for big-endian platforms?
 01092                        Unsafe.WriteUnaligned(pUtf16Buffer + currentOffset, utf16LowVector);
 01093                        Unsafe.WriteUnaligned(pUtf16Buffer + currentOffset + Vector<ushort>.Count, utf16HighVector);
 1094
 01095                        currentOffset += SizeOfVector;
 01096                    } while (currentOffset <= finalOffsetWhereCanIterate);
 1097                }
 1098
 01099                Debug.Assert(elementCount - currentOffset < SizeOfVector, "Vectorized logic should result in less than a
 1100            }
 1101
 1102            // Flush any remaining data.
 1103
 01104            while (currentOffset < elementCount)
 1105            {
 01106                pUtf16Buffer[currentOffset] = (char)pLatin1Buffer[currentOffset];
 01107                currentOffset++;
 1108            }
 01109        }
 1110    }
 1111}
 1112

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.CoreLib/src/System/Text/Latin1Utility.Helpers.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Diagnostics;
 5using System.Runtime.CompilerServices;
 6using System.Runtime.Intrinsics;
 7using System.Runtime.Intrinsics.X86;
 8
 9namespace System.Text
 10{
 11    internal static partial class Latin1Utility
 12    {
 13        /// <summary>
 14        /// Returns <see langword="true"/> iff all chars in <paramref name="value"/> are Latin-1.
 15        /// </summary>
 16        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 17        private static bool AllCharsInUInt32AreLatin1(uint value)
 18        {
 019            return (value & ~0x00FF00FFu) == 0;
 20        }
 21
 22        /// <summary>
 23        /// Returns <see langword="true"/> iff all chars in <paramref name="value"/> are Latin-1.
 24        /// </summary>
 25        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 26        private static bool AllCharsInUInt64AreLatin1(ulong value)
 27        {
 028            return (value & ~0x00FF00FF_00FF00FFul) == 0;
 29        }
 30
 31        /// <summary>
 32        /// Given a DWORD which represents two packed chars in machine-endian order,
 33        /// <see langword="true"/> iff the first char (in machine-endian order) is Latin-1.
 34        /// </summary>
 35        /// <param name="value"></param>
 36        /// <returns></returns>
 37        private static bool FirstCharInUInt32IsLatin1(uint value)
 38        {
 39            return (BitConverter.IsLittleEndian && (value & 0xFF00u) == 0)
 40                || (!BitConverter.IsLittleEndian && (value & 0xFF000000u) == 0);
 41        }
 42
 43        /// <summary>
 44        /// Given a QWORD which represents a buffer of 4 Latin-1 chars in machine-endian order,
 45        /// narrows each WORD to a BYTE, then writes the 4-byte result to the output buffer
 46        /// also in machine-endian order.
 47        /// </summary>
 48        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 49        private static void NarrowFourUtf16CharsToLatin1AndWriteToBuffer(ref byte outputBuffer, ulong value)
 50        {
 51            Debug.Assert(AllCharsInUInt64AreLatin1(value));
 52
 053            if (Sse2.X64.IsSupported)
 54            {
 55                // Narrows a vector of words [ w0 w1 w2 w3 ] to a vector of bytes
 56                // [ b0 b1 b2 b3 b0 b1 b2 b3 ], then writes 4 bytes (32 bits) to the destination.
 57
 058                Vector128<short> vecWide = Sse2.X64.ConvertScalarToVector128UInt64(value).AsInt16();
 059                Vector128<uint> vecNarrow = Sse2.PackUnsignedSaturate(vecWide, vecWide).AsUInt32();
 060                Unsafe.WriteUnaligned(ref outputBuffer, Sse2.ConvertToUInt32(vecNarrow));
 61            }
 62            else
 63            {
 064                if (BitConverter.IsLittleEndian)
 65                {
 066                    outputBuffer = (byte)value;
 067                    value >>= 16;
 068                    Unsafe.Add(ref outputBuffer, 1) = (byte)value;
 069                    value >>= 16;
 070                    Unsafe.Add(ref outputBuffer, 2) = (byte)value;
 071                    value >>= 16;
 072                    Unsafe.Add(ref outputBuffer, 3) = (byte)value;
 73                }
 74                else
 75                {
 76                    Unsafe.Add(ref outputBuffer, 3) = (byte)value;
 77                    value >>= 16;
 78                    Unsafe.Add(ref outputBuffer, 2) = (byte)value;
 79                    value >>= 16;
 80                    Unsafe.Add(ref outputBuffer, 1) = (byte)value;
 81                    value >>= 16;
 82                    outputBuffer = (byte)value;
 83                }
 84            }
 85        }
 86
 87        /// <summary>
 88        /// Given a DWORD which represents a buffer of 2 Latin-1 chars in machine-endian order,
 89        /// narrows each WORD to a BYTE, then writes the 2-byte result to the output buffer also in
 90        /// machine-endian order.
 91        /// </summary>
 92        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 93        private static void NarrowTwoUtf16CharsToLatin1AndWriteToBuffer(ref byte outputBuffer, uint value)
 94        {
 95            Debug.Assert(AllCharsInUInt32AreLatin1(value));
 96
 097            if (BitConverter.IsLittleEndian)
 98            {
 099                outputBuffer = (byte)value;
 0100                Unsafe.Add(ref outputBuffer, 1) = (byte)(value >> 16);
 101            }
 102            else
 103            {
 104                Unsafe.Add(ref outputBuffer, 1) = (byte)value;
 105                outputBuffer = (byte)(value >> 16);
 106            }
 107        }
 108    }
 109}
 110