| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | using System.Diagnostics; |
| | | 5 | | using System.Diagnostics.CodeAnalysis; |
| | | 6 | | using System.Numerics; |
| | | 7 | | using System.Runtime.CompilerServices; |
| | | 8 | | using System.Runtime.Intrinsics; |
| | | 9 | | using System.Runtime.Intrinsics.X86; |
| | | 10 | | |
| | | 11 | | namespace System.Text |
| | | 12 | | { |
| | | 13 | | internal static partial class Latin1Utility |
| | | 14 | | { |
| | | 15 | | /// <summary> |
| | | 16 | | /// Returns the index in <paramref name="pBuffer"/> where the first non-Latin1 char is found. |
| | | 17 | | /// Returns <paramref name="bufferLength"/> if the buffer is empty or all-Latin1. |
| | | 18 | | /// </summary> |
| | | 19 | | /// <returns>A Latin-1 char is defined as 0x0000 - 0x00FF, inclusive.</returns> |
| | | 20 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 21 | | public static unsafe nuint GetIndexOfFirstNonLatin1Char(char* pBuffer, nuint bufferLength /* in chars */) |
| | | 22 | | { |
| | | 23 | | // If SSE2 is supported, use those specific intrinsics instead of the generic vectorized |
| | | 24 | | // code below. This has two benefits: (a) we can take advantage of specific instructions like |
| | | 25 | | // pmovmskb which we know are optimized, and (b) we can avoid downclocking the processor while |
| | | 26 | | // this method is running. |
| | | 27 | | |
| | 0 | 28 | | return (Sse2.IsSupported) |
| | 0 | 29 | | ? GetIndexOfFirstNonLatin1Char_Sse2(pBuffer, bufferLength) |
| | 0 | 30 | | : GetIndexOfFirstNonLatin1Char_Default(pBuffer, bufferLength); |
| | | 31 | | } |
| | | 32 | | |
| | | 33 | | private static unsafe nuint GetIndexOfFirstNonLatin1Char_Default(char* pBuffer, nuint bufferLength /* in chars * |
| | | 34 | | { |
| | | 35 | | // Squirrel away the original buffer reference.This method works by determining the exact |
| | | 36 | | // char reference where non-Latin1 data begins, so we need this base value to perform the |
| | | 37 | | // final subtraction at the end of the method to get the index into the original buffer. |
| | | 38 | | |
| | 0 | 39 | | char* pOriginalBuffer = pBuffer; |
| | | 40 | | |
| | 0 | 41 | | Debug.Assert(bufferLength <= nuint.MaxValue / sizeof(char)); |
| | | 42 | | |
| | | 43 | | // Before we drain off char-by-char, try a generic vectorized loop. |
| | | 44 | | // Only run the loop if we have at least two vectors we can pull out. |
| | | 45 | | |
| | 0 | 46 | | if (Vector.IsHardwareAccelerated && bufferLength >= 2 * (uint)Vector<ushort>.Count) |
| | | 47 | | { |
| | 0 | 48 | | uint SizeOfVectorInChars = (uint)Vector<ushort>.Count; // JIT will make this a const |
| | 0 | 49 | | uint SizeOfVectorInBytes = (uint)Vector<byte>.Count; // JIT will make this a const |
| | | 50 | | |
| | 0 | 51 | | Vector<ushort> maxLatin1 = new Vector<ushort>(0x00FF); |
| | | 52 | | |
| | 0 | 53 | | if (Vector.LessThanOrEqualAll(Unsafe.ReadUnaligned<Vector<ushort>>(pBuffer), maxLatin1)) |
| | | 54 | | { |
| | | 55 | | // The first several elements of the input buffer were Latin-1. Bump up the pointer to the |
| | | 56 | | // next aligned boundary, then perform aligned reads from here on out until we find non-Latin-1 |
| | | 57 | | // data or we approach the end of the buffer. It's possible we'll reread data; this is ok. |
| | | 58 | | |
| | 0 | 59 | | char* pFinalVectorReadPos = pBuffer + bufferLength - SizeOfVectorInChars; |
| | 0 | 60 | | pBuffer = (char*)(((nuint)pBuffer + SizeOfVectorInBytes) & ~(nuint)(SizeOfVectorInBytes - 1)); |
| | | 61 | | |
| | | 62 | | #if DEBUG |
| | 0 | 63 | | long numCharsRead = pBuffer - pOriginalBuffer; |
| | 0 | 64 | | Debug.Assert(0 < numCharsRead && numCharsRead <= SizeOfVectorInChars, "We should've made forward pro |
| | 0 | 65 | | Debug.Assert((nuint)numCharsRead <= bufferLength, "We shouldn't have read past the end of the input |
| | | 66 | | #endif |
| | | 67 | | |
| | 0 | 68 | | Debug.Assert(pBuffer <= pFinalVectorReadPos, "Should be able to read at least one vector."); |
| | | 69 | | |
| | | 70 | | do |
| | | 71 | | { |
| | 0 | 72 | | Debug.Assert((nuint)pBuffer % SizeOfVectorInChars == 0, "Vector read should be aligned."); |
| | 0 | 73 | | if (Vector.GreaterThanAny(Unsafe.Read<Vector<ushort>>(pBuffer), maxLatin1)) |
| | | 74 | | { |
| | | 75 | | break; // found non-Latin-1 data |
| | | 76 | | } |
| | 0 | 77 | | pBuffer += SizeOfVectorInChars; |
| | 0 | 78 | | } while (pBuffer <= pFinalVectorReadPos); |
| | | 79 | | |
| | | 80 | | // Adjust the remaining buffer length for the number of elements we just consumed. |
| | | 81 | | |
| | 0 | 82 | | bufferLength -= ((nuint)pBuffer - (nuint)pOriginalBuffer) / sizeof(char); |
| | | 83 | | } |
| | | 84 | | } |
| | | 85 | | |
| | | 86 | | // At this point, the buffer length wasn't enough to perform a vectorized search, or we did perform |
| | | 87 | | // a vectorized search and encountered non-Latin-1 data. In either case go down a non-vectorized code |
| | | 88 | | // path to drain any remaining Latin-1 chars. |
| | | 89 | | // |
| | | 90 | | // We're going to perform unaligned reads, so prefer 32-bit reads instead of 64-bit reads. |
| | | 91 | | // This also allows us to perform more optimized bit twiddling tricks to count the number of Latin-1 chars. |
| | | 92 | | |
| | | 93 | | uint currentUInt32; |
| | | 94 | | |
| | | 95 | | // Try reading 64 bits at a time in a loop. |
| | | 96 | | |
| | 0 | 97 | | for (; bufferLength >= 4; bufferLength -= 4) // 64 bits = 4 * 16-bit chars |
| | | 98 | | { |
| | 0 | 99 | | currentUInt32 = Unsafe.ReadUnaligned<uint>(pBuffer); |
| | 0 | 100 | | uint nextUInt32 = Unsafe.ReadUnaligned<uint>(pBuffer + 4 / sizeof(char)); |
| | | 101 | | |
| | 0 | 102 | | if (!AllCharsInUInt32AreLatin1(currentUInt32 | nextUInt32)) |
| | | 103 | | { |
| | | 104 | | // One of these two values contains non-Latin-1 chars. |
| | | 105 | | // Figure out which one it is, then put it in 'current' so that we can drain the Latin-1 chars. |
| | | 106 | | |
| | 0 | 107 | | if (AllCharsInUInt32AreLatin1(currentUInt32)) |
| | | 108 | | { |
| | 0 | 109 | | currentUInt32 = nextUInt32; |
| | 0 | 110 | | pBuffer += 2; |
| | | 111 | | } |
| | | 112 | | |
| | 0 | 113 | | goto FoundNonLatin1Data; |
| | | 114 | | } |
| | | 115 | | |
| | 0 | 116 | | pBuffer += 4; // consumed 4 Latin-1 chars |
| | | 117 | | } |
| | | 118 | | |
| | | 119 | | // From this point forward we don't need to keep track of the remaining buffer length. |
| | | 120 | | // Try reading 32 bits. |
| | | 121 | | |
| | 0 | 122 | | if ((bufferLength & 2) != 0) // 32 bits = 2 * 16-bit chars |
| | | 123 | | { |
| | 0 | 124 | | currentUInt32 = Unsafe.ReadUnaligned<uint>(pBuffer); |
| | 0 | 125 | | if (!AllCharsInUInt32AreLatin1(currentUInt32)) |
| | | 126 | | { |
| | | 127 | | goto FoundNonLatin1Data; |
| | | 128 | | } |
| | | 129 | | |
| | 0 | 130 | | pBuffer += 2; |
| | | 131 | | } |
| | | 132 | | |
| | | 133 | | // Try reading 16 bits. |
| | | 134 | | // No need to try an 8-bit read after this since we're working with chars. |
| | | 135 | | |
| | 0 | 136 | | if ((bufferLength & 1) != 0) |
| | | 137 | | { |
| | | 138 | | // If the buffer contains non-Latin-1 data, the comparison below will fail, and |
| | | 139 | | // we'll end up not incrementing the buffer reference. |
| | | 140 | | |
| | 0 | 141 | | if (*pBuffer <= byte.MaxValue) |
| | | 142 | | { |
| | 0 | 143 | | pBuffer++; |
| | | 144 | | } |
| | | 145 | | } |
| | | 146 | | |
| | | 147 | | Finish: |
| | | 148 | | |
| | 0 | 149 | | nuint totalNumBytesRead = (nuint)pBuffer - (nuint)pOriginalBuffer; |
| | 0 | 150 | | Debug.Assert(totalNumBytesRead % sizeof(char) == 0, "Total number of bytes read should be even since we're w |
| | 0 | 151 | | return totalNumBytesRead / sizeof(char); // convert byte count -> char count before returning |
| | | 152 | | |
| | | 153 | | FoundNonLatin1Data: |
| | | 154 | | |
| | 0 | 155 | | Debug.Assert(!AllCharsInUInt32AreLatin1(currentUInt32), "Shouldn't have reached this point if we have an all |
| | | 156 | | |
| | | 157 | | // We don't bother looking at the second char - only the first char. |
| | | 158 | | |
| | 0 | 159 | | if (FirstCharInUInt32IsLatin1(currentUInt32)) |
| | | 160 | | { |
| | 0 | 161 | | pBuffer++; |
| | | 162 | | } |
| | | 163 | | |
| | 0 | 164 | | goto Finish; |
| | | 165 | | } |
| | | 166 | | |
| | | 167 | | [CompExactlyDependsOn(typeof(Sse2))] |
| | | 168 | | private static unsafe nuint GetIndexOfFirstNonLatin1Char_Sse2(char* pBuffer, nuint bufferLength /* in chars */) |
| | | 169 | | { |
| | | 170 | | // This method contains logic optimized for both SSE2 and SSE41. Much of the logic in this method |
| | | 171 | | // will be elided by JIT once we determine which specific ISAs we support. |
| | | 172 | | |
| | | 173 | | // Quick check for empty inputs. |
| | | 174 | | |
| | 0 | 175 | | if (bufferLength == 0) |
| | | 176 | | { |
| | 0 | 177 | | return 0; |
| | | 178 | | } |
| | | 179 | | |
| | | 180 | | // JIT turns the below into constants |
| | | 181 | | |
| | 0 | 182 | | uint SizeOfVector128InBytes = (uint)sizeof(Vector128<byte>); |
| | 0 | 183 | | uint SizeOfVector128InChars = SizeOfVector128InBytes / sizeof(char); |
| | | 184 | | |
| | 0 | 185 | | Debug.Assert(Sse2.IsSupported, "Should've been checked by caller."); |
| | 0 | 186 | | Debug.Assert(BitConverter.IsLittleEndian, "SSE2 assumes little-endian."); |
| | | 187 | | |
| | | 188 | | Vector128<ushort> firstVector, secondVector; |
| | | 189 | | uint currentMask; |
| | 0 | 190 | | char* pOriginalBuffer = pBuffer; |
| | | 191 | | |
| | 0 | 192 | | if (bufferLength < SizeOfVector128InChars) |
| | | 193 | | { |
| | | 194 | | goto InputBufferLessThanOneVectorInLength; // can't vectorize; drain primitives instead |
| | | 195 | | } |
| | | 196 | | |
| | | 197 | | // This method is written such that control generally flows top-to-bottom, avoiding |
| | | 198 | | // jumps as much as possible in the optimistic case of "all Latin-1". If we see non-Latin-1 |
| | | 199 | | // data, we jump out of the hot paths to targets at the end of the method. |
| | | 200 | | |
| | 0 | 201 | | Vector128<ushort> latin1MaskForTestZ = Vector128.Create((ushort)0xFF00); // used for PTEST on supported hard |
| | 0 | 202 | | Vector128<ushort> latin1MaskForAddSaturate = Vector128.Create((ushort)0x7F00); // used for PADDUSW |
| | | 203 | | const uint NonLatin1DataSeenMask = 0b_1010_1010_1010_1010; // used for determining whether 'currentMask' con |
| | | 204 | | |
| | 0 | 205 | | Debug.Assert(bufferLength <= nuint.MaxValue / sizeof(char)); |
| | | 206 | | |
| | | 207 | | // Read the first vector unaligned. |
| | | 208 | | |
| | 0 | 209 | | firstVector = Sse2.LoadVector128((ushort*)pBuffer); // unaligned load |
| | | 210 | | |
| | | 211 | | // The operation below forces the 0x8000 bit of each WORD to be set iff the WORD element |
| | | 212 | | // has value >= 0x0100 (non-Latin-1). Then we'll treat the vector as a BYTE vector in order |
| | | 213 | | // to extract the mask. Reminder: the 0x0080 bit of each WORD should be ignored. |
| | | 214 | | |
| | 0 | 215 | | currentMask = (uint)Sse2.MoveMask(Sse2.AddSaturate(firstVector, latin1MaskForAddSaturate).AsByte()); |
| | | 216 | | |
| | 0 | 217 | | if ((currentMask & NonLatin1DataSeenMask) != 0) |
| | | 218 | | { |
| | | 219 | | goto FoundNonLatin1DataInCurrentMask; |
| | | 220 | | } |
| | | 221 | | |
| | | 222 | | // If we have less than 32 bytes to process, just go straight to the final unaligned |
| | | 223 | | // read. There's no need to mess with the loop logic in the middle of this method. |
| | | 224 | | |
| | | 225 | | // Adjust the remaining length to account for what we just read. |
| | | 226 | | // For the remainder of this code path, bufferLength will be in bytes, not chars. |
| | | 227 | | |
| | 0 | 228 | | bufferLength <<= 1; // chars to bytes |
| | | 229 | | |
| | 0 | 230 | | if (bufferLength < 2 * SizeOfVector128InBytes) |
| | | 231 | | { |
| | | 232 | | goto IncrementCurrentOffsetBeforeFinalUnalignedVectorRead; |
| | | 233 | | } |
| | | 234 | | |
| | | 235 | | // Now adjust the read pointer so that future reads are aligned. |
| | | 236 | | |
| | 0 | 237 | | pBuffer = (char*)(((nuint)pBuffer + SizeOfVector128InBytes) & ~(nuint)(SizeOfVector128InBytes - 1)); |
| | | 238 | | |
| | | 239 | | #if DEBUG |
| | 0 | 240 | | long numCharsRead = pBuffer - pOriginalBuffer; |
| | 0 | 241 | | Debug.Assert(0 < numCharsRead && numCharsRead <= SizeOfVector128InChars, "We should've made forward progress |
| | 0 | 242 | | Debug.Assert((nuint)numCharsRead <= bufferLength, "We shouldn't have read past the end of the input buffer." |
| | | 243 | | #endif |
| | | 244 | | |
| | | 245 | | // Adjust remaining buffer length. |
| | | 246 | | |
| | 0 | 247 | | bufferLength += (nuint)pOriginalBuffer; |
| | 0 | 248 | | bufferLength -= (nuint)pBuffer; |
| | | 249 | | |
| | | 250 | | // The buffer is now properly aligned. |
| | | 251 | | // Read 2 vectors at a time if possible. |
| | | 252 | | |
| | 0 | 253 | | if (bufferLength >= 2 * SizeOfVector128InBytes) |
| | | 254 | | { |
| | 0 | 255 | | char* pFinalVectorReadPos = (char*)((nuint)pBuffer + bufferLength - 2 * SizeOfVector128InBytes); |
| | | 256 | | |
| | | 257 | | // After this point, we no longer need to update the bufferLength value. |
| | | 258 | | |
| | | 259 | | do |
| | | 260 | | { |
| | 0 | 261 | | firstVector = Sse2.LoadAlignedVector128((ushort*)pBuffer); |
| | 0 | 262 | | secondVector = Sse2.LoadAlignedVector128((ushort*)pBuffer + SizeOfVector128InChars); |
| | 0 | 263 | | Vector128<ushort> combinedVector = firstVector | secondVector; |
| | | 264 | | |
| | | 265 | | #pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau |
| | 0 | 266 | | if (Sse41.IsSupported) |
| | | 267 | | #pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough |
| | | 268 | | { |
| | | 269 | | // If a non-Latin-1 bit is set in any WORD of the combined vector, we have seen non-Latin-1 data |
| | | 270 | | // Jump to the non-Latin-1 handler to figure out which particular vector contained non-Latin-1 d |
| | 0 | 271 | | if ((combinedVector & latin1MaskForTestZ) != Vector128<ushort>.Zero) |
| | | 272 | | { |
| | 0 | 273 | | goto FoundNonLatin1DataInFirstOrSecondVector; |
| | | 274 | | } |
| | | 275 | | } |
| | | 276 | | else |
| | | 277 | | { |
| | | 278 | | // See comment earlier in the method for an explanation of how the below logic works. |
| | 0 | 279 | | currentMask = (uint)Sse2.MoveMask(Sse2.AddSaturate(combinedVector, latin1MaskForAddSaturate).AsB |
| | 0 | 280 | | if ((currentMask & NonLatin1DataSeenMask) != 0) |
| | | 281 | | { |
| | | 282 | | goto FoundNonLatin1DataInFirstOrSecondVector; |
| | | 283 | | } |
| | | 284 | | } |
| | | 285 | | |
| | 0 | 286 | | pBuffer += 2 * SizeOfVector128InChars; |
| | 0 | 287 | | } while (pBuffer <= pFinalVectorReadPos); |
| | | 288 | | } |
| | | 289 | | |
| | | 290 | | // We have somewhere between 0 and (2 * vector length) - 1 bytes remaining to read from. |
| | | 291 | | // Since the above loop doesn't update bufferLength, we can't rely on its absolute value. |
| | | 292 | | // But we _can_ rely on it to tell us how much remaining data must be drained by looking |
| | | 293 | | // at what bits of it are set. This works because had we updated it within the loop above, |
| | | 294 | | // we would've been adding 2 * SizeOfVector128 on each iteration, but we only care about |
| | | 295 | | // bits which are less significant than those that the addition would've acted on. |
| | | 296 | | |
| | | 297 | | // If there is fewer than one vector length remaining, skip the next aligned read. |
| | | 298 | | // Remember, at this point bufferLength is measured in bytes, not chars. |
| | | 299 | | |
| | 0 | 300 | | if ((bufferLength & SizeOfVector128InBytes) == 0) |
| | | 301 | | { |
| | | 302 | | goto DoFinalUnalignedVectorRead; |
| | | 303 | | } |
| | | 304 | | |
| | | 305 | | // At least one full vector's worth of data remains, so we can safely read it. |
| | | 306 | | // Remember, at this point pBuffer is still aligned. |
| | | 307 | | |
| | 0 | 308 | | firstVector = Sse2.LoadAlignedVector128((ushort*)pBuffer); |
| | | 309 | | |
| | | 310 | | #pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau |
| | 0 | 311 | | if (Sse41.IsSupported) |
| | | 312 | | #pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough |
| | | 313 | | { |
| | | 314 | | // If a non-Latin-1 bit is set in any WORD of the combined vector, we have seen non-Latin-1 data. |
| | | 315 | | // Jump to the non-Latin-1 handler to figure out which particular vector contained non-Latin-1 data. |
| | 0 | 316 | | if ((firstVector & latin1MaskForTestZ) != Vector128<ushort>.Zero) |
| | | 317 | | { |
| | 0 | 318 | | goto FoundNonLatin1DataInFirstVector; |
| | | 319 | | } |
| | | 320 | | } |
| | | 321 | | else |
| | | 322 | | { |
| | | 323 | | // See comment earlier in the method for an explanation of how the below logic works. |
| | 0 | 324 | | currentMask = (uint)Sse2.MoveMask(Sse2.AddSaturate(firstVector, latin1MaskForAddSaturate).AsByte()); |
| | 0 | 325 | | if ((currentMask & NonLatin1DataSeenMask) != 0) |
| | | 326 | | { |
| | | 327 | | goto FoundNonLatin1DataInCurrentMask; |
| | | 328 | | } |
| | | 329 | | } |
| | | 330 | | |
| | | 331 | | IncrementCurrentOffsetBeforeFinalUnalignedVectorRead: |
| | | 332 | | |
| | 0 | 333 | | pBuffer += SizeOfVector128InChars; |
| | | 334 | | |
| | | 335 | | DoFinalUnalignedVectorRead: |
| | | 336 | | |
| | 0 | 337 | | if (((byte)bufferLength & (SizeOfVector128InBytes - 1)) != 0) |
| | | 338 | | { |
| | | 339 | | // Perform an unaligned read of the last vector. |
| | | 340 | | // We need to adjust the pointer because we're re-reading data. |
| | | 341 | | |
| | 0 | 342 | | pBuffer = (char*)((byte*)pBuffer + (bufferLength & (SizeOfVector128InBytes - 1)) - SizeOfVector128InByte |
| | 0 | 343 | | firstVector = Sse2.LoadVector128((ushort*)pBuffer); // unaligned load |
| | | 344 | | |
| | | 345 | | #pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau |
| | 0 | 346 | | if (Sse41.IsSupported) |
| | | 347 | | #pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough |
| | | 348 | | { |
| | | 349 | | // If a non-Latin-1 bit is set in any WORD of the combined vector, we have seen non-Latin-1 data. |
| | | 350 | | // Jump to the non-Latin-1 handler to figure out which particular vector contained non-Latin-1 data. |
| | 0 | 351 | | if ((firstVector & latin1MaskForTestZ) != Vector128<ushort>.Zero) |
| | | 352 | | { |
| | 0 | 353 | | goto FoundNonLatin1DataInFirstVector; |
| | | 354 | | } |
| | | 355 | | } |
| | | 356 | | else |
| | | 357 | | { |
| | | 358 | | // See comment earlier in the method for an explanation of how the below logic works. |
| | 0 | 359 | | currentMask = (uint)Sse2.MoveMask(Sse2.AddSaturate(firstVector, latin1MaskForAddSaturate).AsByte()); |
| | 0 | 360 | | if ((currentMask & NonLatin1DataSeenMask) != 0) |
| | | 361 | | { |
| | | 362 | | goto FoundNonLatin1DataInCurrentMask; |
| | | 363 | | } |
| | | 364 | | } |
| | | 365 | | |
| | 0 | 366 | | pBuffer += SizeOfVector128InChars; |
| | | 367 | | } |
| | | 368 | | |
| | | 369 | | Finish: |
| | | 370 | | |
| | 0 | 371 | | Debug.Assert(((nuint)pBuffer - (nuint)pOriginalBuffer) % 2 == 0, "Shouldn't have incremented any pointer by |
| | 0 | 372 | | return ((nuint)pBuffer - (nuint)pOriginalBuffer) / sizeof(char); // and we're done! (remember to adjust for |
| | | 373 | | |
| | | 374 | | FoundNonLatin1DataInFirstOrSecondVector: |
| | | 375 | | |
| | | 376 | | // We don't know if the first or the second vector contains non-Latin-1 data. Check the first |
| | | 377 | | // vector, and if that's all-Latin-1 then the second vector must be the culprit. Either way |
| | | 378 | | // we'll make sure the first vector local is the one that contains the non-Latin-1 data. |
| | | 379 | | |
| | | 380 | | // See comment earlier in the method for an explanation of how the below logic works. |
| | | 381 | | #pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau |
| | 0 | 382 | | if (Sse41.IsSupported) |
| | | 383 | | #pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough |
| | | 384 | | { |
| | 0 | 385 | | if ((firstVector & latin1MaskForTestZ) != Vector128<ushort>.Zero) |
| | | 386 | | { |
| | 0 | 387 | | goto FoundNonLatin1DataInFirstVector; |
| | | 388 | | } |
| | | 389 | | } |
| | | 390 | | else |
| | | 391 | | { |
| | 0 | 392 | | currentMask = (uint)Sse2.MoveMask(Sse2.AddSaturate(firstVector, latin1MaskForAddSaturate).AsByte()); |
| | 0 | 393 | | if ((currentMask & NonLatin1DataSeenMask) != 0) |
| | | 394 | | { |
| | | 395 | | goto FoundNonLatin1DataInCurrentMask; |
| | | 396 | | } |
| | | 397 | | } |
| | | 398 | | |
| | | 399 | | // Wasn't the first vector; must be the second. |
| | | 400 | | |
| | 0 | 401 | | pBuffer += SizeOfVector128InChars; |
| | 0 | 402 | | firstVector = secondVector; |
| | | 403 | | |
| | | 404 | | FoundNonLatin1DataInFirstVector: |
| | | 405 | | |
| | | 406 | | // See comment earlier in the method for an explanation of how the below logic works. |
| | 0 | 407 | | currentMask = (uint)Sse2.MoveMask(Sse2.AddSaturate(firstVector, latin1MaskForAddSaturate).AsByte()); |
| | | 408 | | |
| | | 409 | | FoundNonLatin1DataInCurrentMask: |
| | | 410 | | |
| | | 411 | | // See comment earlier in the method accounting for the 0x8000 and 0x0080 bits set after the WORD-sized oper |
| | | 412 | | |
| | 0 | 413 | | currentMask &= NonLatin1DataSeenMask; |
| | | 414 | | |
| | | 415 | | // Now, the mask contains - from the LSB - a 0b00 pair for each Latin-1 char we saw, and a 0b10 pair for eac |
| | | 416 | | // |
| | | 417 | | // (Keep endianness in mind in the below examples.) |
| | | 418 | | // A non-Latin-1 char followed by two Latin-1 chars is 0b..._00_00_10. (tzcnt = 1) |
| | | 419 | | // A Latin-1 char followed by two non-Latin-1 chars is 0b..._10_10_00. (tzcnt = 3) |
| | | 420 | | // Two Latin-1 chars followed by a non-Latin-1 char is 0b..._10_00_00. (tzcnt = 5) |
| | | 421 | | // |
| | | 422 | | // This means tzcnt = 2 * numLeadingLatin1Chars + 1. We can conveniently take advantage of the fact |
| | | 423 | | // that the 2x multiplier already matches the char* stride length, then just subtract 1 at the end to |
| | | 424 | | // compute the correct final ending pointer value. |
| | | 425 | | |
| | 0 | 426 | | Debug.Assert(currentMask != 0, "Shouldn't be here unless we see non-Latin-1 data."); |
| | 0 | 427 | | pBuffer = (char*)((byte*)pBuffer + (uint)BitOperations.TrailingZeroCount(currentMask) - 1); |
| | | 428 | | |
| | 0 | 429 | | goto Finish; |
| | | 430 | | |
| | | 431 | | FoundNonLatin1DataInCurrentDWord: |
| | | 432 | | |
| | | 433 | | uint currentDWord; |
| | 0 | 434 | | Debug.Assert(!AllCharsInUInt32AreLatin1(currentDWord), "Shouldn't be here unless we see non-Latin-1 data."); |
| | | 435 | | |
| | 0 | 436 | | if (FirstCharInUInt32IsLatin1(currentDWord)) |
| | | 437 | | { |
| | 0 | 438 | | pBuffer++; // skip past the Latin-1 char |
| | | 439 | | } |
| | | 440 | | |
| | 0 | 441 | | goto Finish; |
| | | 442 | | |
| | | 443 | | InputBufferLessThanOneVectorInLength: |
| | | 444 | | |
| | | 445 | | // These code paths get hit if the original input length was less than one vector in size. |
| | | 446 | | // We can't perform vectorized reads at this point, so we'll fall back to reading primitives |
| | | 447 | | // directly. Note that all of these reads are unaligned. |
| | | 448 | | |
| | | 449 | | // Reminder: If this code path is hit, bufferLength is still a char count, not a byte count. |
| | | 450 | | // We skipped the code path that multiplied the count by sizeof(char). |
| | | 451 | | |
| | 0 | 452 | | Debug.Assert(bufferLength < SizeOfVector128InChars); |
| | | 453 | | |
| | | 454 | | // QWORD drain |
| | | 455 | | |
| | 0 | 456 | | if ((bufferLength & 4) != 0) |
| | | 457 | | { |
| | | 458 | | #pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau |
| | 0 | 459 | | if (Bmi1.X64.IsSupported) |
| | | 460 | | #pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough |
| | | 461 | | { |
| | | 462 | | // If we can use 64-bit tzcnt to count the number of leading Latin-1 chars, prefer it. |
| | | 463 | | |
| | 0 | 464 | | ulong candidateUInt64 = Unsafe.ReadUnaligned<ulong>(pBuffer); |
| | 0 | 465 | | if (!AllCharsInUInt64AreLatin1(candidateUInt64)) |
| | | 466 | | { |
| | | 467 | | // Clear the low 8 bits (the Latin-1 bits) of each char, then tzcnt. |
| | | 468 | | // Remember the / 8 at the end to convert bit count to byte count, |
| | | 469 | | // then the & ~1 at the end to treat a match in the high byte of |
| | | 470 | | // any char the same as a match in the low byte of that same char. |
| | | 471 | | |
| | 0 | 472 | | candidateUInt64 &= 0xFF00FF00_FF00FF00ul; |
| | 0 | 473 | | pBuffer = (char*)((byte*)pBuffer + ((nuint)(Bmi1.X64.TrailingZeroCount(candidateUInt64) / 8) & ~ |
| | 0 | 474 | | goto Finish; |
| | | 475 | | } |
| | | 476 | | } |
| | | 477 | | else |
| | | 478 | | { |
| | | 479 | | // If we can't use 64-bit tzcnt, no worries. We'll just do 2x 32-bit reads instead. |
| | | 480 | | |
| | 0 | 481 | | currentDWord = Unsafe.ReadUnaligned<uint>(pBuffer); |
| | 0 | 482 | | uint nextDWord = Unsafe.ReadUnaligned<uint>(pBuffer + 4 / sizeof(char)); |
| | | 483 | | |
| | 0 | 484 | | if (!AllCharsInUInt32AreLatin1(currentDWord | nextDWord)) |
| | | 485 | | { |
| | | 486 | | // At least one of the values wasn't all-Latin-1. |
| | | 487 | | // We need to figure out which one it was and stick it in the currentMask local. |
| | | 488 | | |
| | 0 | 489 | | if (AllCharsInUInt32AreLatin1(currentDWord)) |
| | | 490 | | { |
| | 0 | 491 | | currentDWord = nextDWord; // this one is the culprit |
| | 0 | 492 | | pBuffer += 4 / sizeof(char); |
| | | 493 | | } |
| | | 494 | | |
| | 0 | 495 | | goto FoundNonLatin1DataInCurrentDWord; |
| | | 496 | | } |
| | | 497 | | } |
| | | 498 | | |
| | 0 | 499 | | pBuffer += 4; // successfully consumed 4 Latin-1 chars |
| | | 500 | | } |
| | | 501 | | |
| | | 502 | | // DWORD drain |
| | | 503 | | |
| | 0 | 504 | | if ((bufferLength & 2) != 0) |
| | | 505 | | { |
| | 0 | 506 | | currentDWord = Unsafe.ReadUnaligned<uint>(pBuffer); |
| | | 507 | | |
| | 0 | 508 | | if (!AllCharsInUInt32AreLatin1(currentDWord)) |
| | | 509 | | { |
| | | 510 | | goto FoundNonLatin1DataInCurrentDWord; |
| | | 511 | | } |
| | | 512 | | |
| | 0 | 513 | | pBuffer += 2; // successfully consumed 2 Latin-1 chars |
| | | 514 | | } |
| | | 515 | | |
| | | 516 | | // WORD drain |
| | | 517 | | // This is the final drain; there's no need for a BYTE drain since our elemental type is 16-bit char. |
| | | 518 | | |
| | 0 | 519 | | if ((bufferLength & 1) != 0) |
| | | 520 | | { |
| | 0 | 521 | | if (*pBuffer <= byte.MaxValue) |
| | | 522 | | { |
| | 0 | 523 | | pBuffer++; // successfully consumed a single char |
| | | 524 | | } |
| | | 525 | | } |
| | | 526 | | |
| | 0 | 527 | | goto Finish; |
| | | 528 | | } |
| | | 529 | | |
| | | 530 | | |
| | | 531 | | /// <summary> |
| | | 532 | | /// Copies as many Latin-1 characters (U+0000..U+00FF) as possible from <paramref name="pUtf16Buffer"/> |
| | | 533 | | /// to <paramref name="pLatin1Buffer"/>, stopping when the first non-Latin-1 character is encountered |
| | | 534 | | /// or once <paramref name="elementCount"/> elements have been converted. Returns the total number |
| | | 535 | | /// of elements that were able to be converted. |
| | | 536 | | /// </summary> |
| | | 537 | | public static unsafe nuint NarrowUtf16ToLatin1(char* pUtf16Buffer, byte* pLatin1Buffer, nuint elementCount) |
| | | 538 | | { |
| | | 539 | | nuint currentOffset = 0; |
| | | 540 | | |
| | 6806 | 541 | | uint utf16Data32BitsHigh = 0, utf16Data32BitsLow = 0; |
| | 3403 | 542 | | ulong utf16Data64Bits = 0; |
| | | 543 | | |
| | | 544 | | // If SSE2 is supported, use those specific intrinsics instead of the generic vectorized |
| | | 545 | | // code below. This has two benefits: (a) we can take advantage of specific instructions like |
| | | 546 | | // pmovmskb, ptest, vpminuw which we know are optimized, and (b) we can avoid downclocking the |
| | | 547 | | // processor while this method is running. |
| | | 548 | | |
| | 3403 | 549 | | if (Sse2.IsSupported) |
| | | 550 | | { |
| | 3403 | 551 | | Debug.Assert(BitConverter.IsLittleEndian, "Assume little endian if SSE2 is supported."); |
| | | 552 | | |
| | 3403 | 553 | | if (elementCount >= 2 * (uint)sizeof(Vector128<byte>)) |
| | | 554 | | { |
| | | 555 | | // Since there's overhead to setting up the vectorized code path, we only want to |
| | | 556 | | // call into it after a quick probe to ensure the next immediate characters really are Latin-1. |
| | | 557 | | // If we see non-Latin-1 data, we'll jump immediately to the draining logic at the end of the method |
| | | 558 | | |
| | | 559 | | if (IntPtr.Size >= 8) |
| | | 560 | | { |
| | 1719 | 561 | | utf16Data64Bits = Unsafe.ReadUnaligned<ulong>(pUtf16Buffer); |
| | 1719 | 562 | | if (!AllCharsInUInt64AreLatin1(utf16Data64Bits)) |
| | | 563 | | { |
| | 0 | 564 | | goto FoundNonLatin1DataIn64BitRead; |
| | | 565 | | } |
| | | 566 | | } |
| | | 567 | | else |
| | | 568 | | { |
| | | 569 | | utf16Data32BitsHigh = Unsafe.ReadUnaligned<uint>(pUtf16Buffer); |
| | | 570 | | utf16Data32BitsLow = Unsafe.ReadUnaligned<uint>(pUtf16Buffer + 4 / sizeof(char)); |
| | | 571 | | if (!AllCharsInUInt32AreLatin1(utf16Data32BitsHigh | utf16Data32BitsLow)) |
| | | 572 | | { |
| | | 573 | | goto FoundNonLatin1DataIn64BitRead; |
| | | 574 | | } |
| | | 575 | | } |
| | | 576 | | |
| | 1719 | 577 | | currentOffset = NarrowUtf16ToLatin1_Sse2(pUtf16Buffer, pLatin1Buffer, elementCount); |
| | | 578 | | } |
| | | 579 | | } |
| | 0 | 580 | | else if (Vector.IsHardwareAccelerated) |
| | | 581 | | { |
| | 0 | 582 | | uint SizeOfVector = (uint)sizeof(Vector<byte>); // JIT will make this a const |
| | | 583 | | |
| | | 584 | | // Only bother vectorizing if we have enough data to do so. |
| | 0 | 585 | | if (elementCount >= 2 * SizeOfVector) |
| | | 586 | | { |
| | | 587 | | // Since there's overhead to setting up the vectorized code path, we only want to |
| | | 588 | | // call into it after a quick probe to ensure the next immediate characters really are Latin-1. |
| | | 589 | | // If we see non-Latin-1 data, we'll jump immediately to the draining logic at the end of the method |
| | | 590 | | |
| | | 591 | | if (IntPtr.Size >= 8) |
| | | 592 | | { |
| | 0 | 593 | | utf16Data64Bits = Unsafe.ReadUnaligned<ulong>(pUtf16Buffer); |
| | 0 | 594 | | if (!AllCharsInUInt64AreLatin1(utf16Data64Bits)) |
| | | 595 | | { |
| | 0 | 596 | | goto FoundNonLatin1DataIn64BitRead; |
| | | 597 | | } |
| | | 598 | | } |
| | | 599 | | else |
| | | 600 | | { |
| | | 601 | | utf16Data32BitsHigh = Unsafe.ReadUnaligned<uint>(pUtf16Buffer); |
| | | 602 | | utf16Data32BitsLow = Unsafe.ReadUnaligned<uint>(pUtf16Buffer + 4 / sizeof(char)); |
| | | 603 | | if (!AllCharsInUInt32AreLatin1(utf16Data32BitsHigh | utf16Data32BitsLow)) |
| | | 604 | | { |
| | | 605 | | goto FoundNonLatin1DataIn64BitRead; |
| | | 606 | | } |
| | | 607 | | } |
| | | 608 | | |
| | 0 | 609 | | Vector<ushort> maxLatin1 = new Vector<ushort>(0x00FF); |
| | | 610 | | |
| | 0 | 611 | | nuint finalOffsetWhereCanLoop = elementCount - 2 * SizeOfVector; |
| | | 612 | | do |
| | | 613 | | { |
| | 0 | 614 | | Vector<ushort> utf16VectorHigh = Unsafe.ReadUnaligned<Vector<ushort>>(pUtf16Buffer + currentOffs |
| | 0 | 615 | | Vector<ushort> utf16VectorLow = Unsafe.ReadUnaligned<Vector<ushort>>(pUtf16Buffer + currentOffse |
| | | 616 | | |
| | 0 | 617 | | if (Vector.GreaterThanAny(Vector.BitwiseOr(utf16VectorHigh, utf16VectorLow), maxLatin1)) |
| | | 618 | | { |
| | | 619 | | break; // found non-Latin-1 data |
| | | 620 | | } |
| | | 621 | | |
| | | 622 | | // TODO: Is the below logic also valid for big-endian platforms? |
| | 0 | 623 | | Vector<byte> latin1Vector = Vector.Narrow(utf16VectorHigh, utf16VectorLow); |
| | 0 | 624 | | Unsafe.WriteUnaligned(pLatin1Buffer + currentOffset, latin1Vector); |
| | | 625 | | |
| | 0 | 626 | | currentOffset += SizeOfVector; |
| | 0 | 627 | | } while (currentOffset <= finalOffsetWhereCanLoop); |
| | | 628 | | } |
| | | 629 | | } |
| | | 630 | | |
| | 3403 | 631 | | Debug.Assert(currentOffset <= elementCount); |
| | 3403 | 632 | | nuint remainingElementCount = elementCount - currentOffset; |
| | | 633 | | |
| | | 634 | | // Try to narrow 64 bits -> 32 bits at a time. |
| | | 635 | | // We needn't update remainingElementCount after this point. |
| | | 636 | | |
| | 3403 | 637 | | if (remainingElementCount >= 4) |
| | | 638 | | { |
| | 2792 | 639 | | nuint finalOffsetWhereCanLoop = currentOffset + remainingElementCount - 4; |
| | | 640 | | do |
| | | 641 | | { |
| | | 642 | | if (IntPtr.Size >= 8) |
| | | 643 | | { |
| | | 644 | | // Only perform QWORD reads on a 64-bit platform. |
| | 7054 | 645 | | utf16Data64Bits = Unsafe.ReadUnaligned<ulong>(pUtf16Buffer + currentOffset); |
| | 7054 | 646 | | if (!AllCharsInUInt64AreLatin1(utf16Data64Bits)) |
| | | 647 | | { |
| | | 648 | | goto FoundNonLatin1DataIn64BitRead; |
| | | 649 | | } |
| | | 650 | | |
| | 7054 | 651 | | NarrowFourUtf16CharsToLatin1AndWriteToBuffer(ref pLatin1Buffer[currentOffset], utf16Data64Bits); |
| | | 652 | | } |
| | | 653 | | else |
| | | 654 | | { |
| | | 655 | | utf16Data32BitsHigh = Unsafe.ReadUnaligned<uint>(pUtf16Buffer + currentOffset); |
| | | 656 | | utf16Data32BitsLow = Unsafe.ReadUnaligned<uint>(pUtf16Buffer + currentOffset + 4 / sizeof(char)) |
| | | 657 | | if (!AllCharsInUInt32AreLatin1(utf16Data32BitsHigh | utf16Data32BitsLow)) |
| | | 658 | | { |
| | | 659 | | goto FoundNonLatin1DataIn64BitRead; |
| | | 660 | | } |
| | | 661 | | |
| | | 662 | | NarrowTwoUtf16CharsToLatin1AndWriteToBuffer(ref pLatin1Buffer[currentOffset], utf16Data32BitsHig |
| | | 663 | | NarrowTwoUtf16CharsToLatin1AndWriteToBuffer(ref pLatin1Buffer[currentOffset + 2], utf16Data32Bit |
| | | 664 | | } |
| | | 665 | | |
| | 7054 | 666 | | currentOffset += 4; |
| | 7054 | 667 | | } while (currentOffset <= finalOffsetWhereCanLoop); |
| | | 668 | | } |
| | | 669 | | |
| | | 670 | | // Try to narrow 32 bits -> 16 bits. |
| | | 671 | | |
| | 3403 | 672 | | if (((uint)remainingElementCount & 2) != 0) |
| | | 673 | | { |
| | 1725 | 674 | | utf16Data32BitsHigh = Unsafe.ReadUnaligned<uint>(pUtf16Buffer + currentOffset); |
| | 1725 | 675 | | if (!AllCharsInUInt32AreLatin1(utf16Data32BitsHigh)) |
| | | 676 | | { |
| | | 677 | | goto FoundNonLatin1DataInHigh32Bits; |
| | | 678 | | } |
| | | 679 | | |
| | 1725 | 680 | | NarrowTwoUtf16CharsToLatin1AndWriteToBuffer(ref pLatin1Buffer[currentOffset], utf16Data32BitsHigh); |
| | 1725 | 681 | | currentOffset += 2; |
| | | 682 | | } |
| | | 683 | | |
| | | 684 | | // Try to narrow 16 bits -> 8 bits. |
| | | 685 | | |
| | 3403 | 686 | | if (((uint)remainingElementCount & 1) != 0) |
| | | 687 | | { |
| | 1580 | 688 | | utf16Data32BitsHigh = pUtf16Buffer[currentOffset]; |
| | 1580 | 689 | | if (utf16Data32BitsHigh <= byte.MaxValue) |
| | | 690 | | { |
| | 1580 | 691 | | pLatin1Buffer[currentOffset] = (byte)utf16Data32BitsHigh; |
| | 1580 | 692 | | currentOffset++; |
| | | 693 | | } |
| | | 694 | | } |
| | | 695 | | |
| | | 696 | | Finish: |
| | | 697 | | |
| | 3403 | 698 | | return currentOffset; |
| | | 699 | | |
| | | 700 | | FoundNonLatin1DataIn64BitRead: |
| | | 701 | | |
| | | 702 | | if (IntPtr.Size >= 8) |
| | | 703 | | { |
| | | 704 | | // Try checking the first 32 bits of the buffer for non-Latin-1 data. |
| | | 705 | | // Regardless, we'll move the non-Latin-1 data into the utf16Data32BitsHigh local. |
| | | 706 | | |
| | 0 | 707 | | if (BitConverter.IsLittleEndian) |
| | | 708 | | { |
| | 0 | 709 | | utf16Data32BitsHigh = (uint)utf16Data64Bits; |
| | | 710 | | } |
| | | 711 | | else |
| | | 712 | | { |
| | | 713 | | utf16Data32BitsHigh = (uint)(utf16Data64Bits >> 32); |
| | | 714 | | } |
| | | 715 | | |
| | 0 | 716 | | if (AllCharsInUInt32AreLatin1(utf16Data32BitsHigh)) |
| | | 717 | | { |
| | 0 | 718 | | NarrowTwoUtf16CharsToLatin1AndWriteToBuffer(ref pLatin1Buffer[currentOffset], utf16Data32BitsHigh); |
| | | 719 | | |
| | 0 | 720 | | if (BitConverter.IsLittleEndian) |
| | | 721 | | { |
| | 0 | 722 | | utf16Data32BitsHigh = (uint)(utf16Data64Bits >> 32); |
| | | 723 | | } |
| | | 724 | | else |
| | | 725 | | { |
| | | 726 | | utf16Data32BitsHigh = (uint)utf16Data64Bits; |
| | | 727 | | } |
| | | 728 | | |
| | 0 | 729 | | currentOffset += 2; |
| | | 730 | | } |
| | | 731 | | } |
| | | 732 | | else |
| | | 733 | | { |
| | | 734 | | // Need to determine if the high or the low 32-bit value contained non-Latin-1 data. |
| | | 735 | | // Regardless, we'll move the non-Latin-1 data into the utf16Data32BitsHigh local. |
| | | 736 | | |
| | | 737 | | if (AllCharsInUInt32AreLatin1(utf16Data32BitsHigh)) |
| | | 738 | | { |
| | | 739 | | NarrowTwoUtf16CharsToLatin1AndWriteToBuffer(ref pLatin1Buffer[currentOffset], utf16Data32BitsHigh); |
| | | 740 | | utf16Data32BitsHigh = utf16Data32BitsLow; |
| | | 741 | | currentOffset += 2; |
| | | 742 | | } |
| | | 743 | | } |
| | | 744 | | |
| | | 745 | | FoundNonLatin1DataInHigh32Bits: |
| | | 746 | | |
| | 0 | 747 | | Debug.Assert(!AllCharsInUInt32AreLatin1(utf16Data32BitsHigh), "Shouldn't have reached this point if we have |
| | | 748 | | |
| | | 749 | | // There's at most one char that needs to be drained. |
| | | 750 | | |
| | 0 | 751 | | if (FirstCharInUInt32IsLatin1(utf16Data32BitsHigh)) |
| | | 752 | | { |
| | 0 | 753 | | if (!BitConverter.IsLittleEndian) |
| | | 754 | | { |
| | | 755 | | utf16Data32BitsHigh >>= 16; // move high char down to low char |
| | | 756 | | } |
| | | 757 | | |
| | 0 | 758 | | pLatin1Buffer[currentOffset] = (byte)utf16Data32BitsHigh; |
| | 0 | 759 | | currentOffset++; |
| | | 760 | | } |
| | | 761 | | |
| | 0 | 762 | | goto Finish; |
| | | 763 | | } |
| | | 764 | | |
| | | 765 | | [CompExactlyDependsOn(typeof(Sse2))] |
| | | 766 | | private static unsafe nuint NarrowUtf16ToLatin1_Sse2(char* pUtf16Buffer, byte* pLatin1Buffer, nuint elementCount |
| | | 767 | | { |
| | | 768 | | // This method contains logic optimized for both SSE2 and SSE41. Much of the logic in this method |
| | | 769 | | // will be elided by JIT once we determine which specific ISAs we support. |
| | | 770 | | |
| | | 771 | | // JIT turns the below into constants |
| | | 772 | | |
| | 1719 | 773 | | uint SizeOfVector128 = (uint)sizeof(Vector128<byte>); |
| | 1719 | 774 | | nuint MaskOfAllBitsInVector128 = SizeOfVector128 - 1; |
| | | 775 | | |
| | | 776 | | // This method is written such that control generally flows top-to-bottom, avoiding |
| | | 777 | | // jumps as much as possible in the optimistic case of "all Latin-1". If we see non-Latin-1 |
| | | 778 | | // data, we jump out of the hot paths to targets at the end of the method. |
| | | 779 | | |
| | 1719 | 780 | | Debug.Assert(Sse2.IsSupported); |
| | 1719 | 781 | | Debug.Assert(BitConverter.IsLittleEndian); |
| | 1719 | 782 | | Debug.Assert(elementCount >= 2 * SizeOfVector128); |
| | | 783 | | |
| | 1719 | 784 | | Vector128<short> latin1MaskForTestZ = Vector128.Create(unchecked((short)0xFF00)); // used for PTEST on suppo |
| | 1719 | 785 | | Vector128<ushort> latin1MaskForAddSaturate = Vector128.Create((ushort)0x7F00); // used for PADDUSW |
| | | 786 | | const int NonLatin1DataSeenMask = 0b_1010_1010_1010_1010; // used for determining whether the pmovmskb opera |
| | | 787 | | |
| | | 788 | | // First, perform an unaligned read of the first part of the input buffer. |
| | | 789 | | |
| | 1719 | 790 | | Vector128<short> utf16VectorFirst = Sse2.LoadVector128((short*)pUtf16Buffer); // unaligned load |
| | | 791 | | |
| | | 792 | | // If there's non-Latin-1 data in the first 8 elements of the vector, there's nothing we can do. |
| | | 793 | | // See comments in GetIndexOfFirstNonLatin1Char_Sse2 for information about how this works. |
| | | 794 | | |
| | | 795 | | #pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau |
| | 1719 | 796 | | if (Sse41.IsSupported) |
| | | 797 | | #pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough |
| | | 798 | | { |
| | 1719 | 799 | | if ((utf16VectorFirst & latin1MaskForTestZ) != Vector128<short>.Zero) |
| | | 800 | | { |
| | 0 | 801 | | return 0; |
| | | 802 | | } |
| | | 803 | | } |
| | | 804 | | else |
| | | 805 | | { |
| | 0 | 806 | | if ((Sse2.MoveMask(Sse2.AddSaturate(utf16VectorFirst.AsUInt16(), latin1MaskForAddSaturate).AsByte()) & N |
| | | 807 | | { |
| | 0 | 808 | | return 0; |
| | | 809 | | } |
| | | 810 | | } |
| | | 811 | | |
| | | 812 | | // Turn the 8 Latin-1 chars we just read into 8 Latin-1 bytes, then copy it to the destination. |
| | | 813 | | |
| | 1719 | 814 | | Vector128<byte> latin1Vector = Sse2.PackUnsignedSaturate(utf16VectorFirst, utf16VectorFirst); |
| | 1719 | 815 | | Sse2.StoreScalar((ulong*)pLatin1Buffer, latin1Vector.AsUInt64()); // ulong* calculated here is UNALIGNED |
| | | 816 | | |
| | 1719 | 817 | | nuint currentOffsetInElements = SizeOfVector128 / 2; // we processed 8 elements so far |
| | | 818 | | |
| | | 819 | | // We're going to get the best performance when we have aligned writes, so we'll take the |
| | | 820 | | // hit of potentially unaligned reads in order to hit this sweet spot. |
| | | 821 | | |
| | | 822 | | // pLatin1Buffer points to the start of the destination buffer, immediately before where we wrote |
| | | 823 | | // the 8 bytes previously. If the 0x08 bit is set at the pinned address, then the 8 bytes we wrote |
| | | 824 | | // previously mean that the 0x08 bit is *not* set at address &pLatin1Buffer[SizeOfVector128 / 2]. In |
| | | 825 | | // that case we can immediately back up to the previous aligned boundary and start the main loop. |
| | | 826 | | // If the 0x08 bit is *not* set at the pinned address, then it means the 0x08 bit *is* set at |
| | | 827 | | // address &pLatin1Buffer[SizeOfVector128 / 2], and we should perform one more 8-byte write to bump |
| | | 828 | | // just past the next aligned boundary address. |
| | | 829 | | |
| | 1719 | 830 | | if (((uint)pLatin1Buffer & (SizeOfVector128 / 2)) == 0) |
| | | 831 | | { |
| | | 832 | | // We need to perform one more partial vector write before we can get the alignment we want. |
| | | 833 | | |
| | 763 | 834 | | utf16VectorFirst = Sse2.LoadVector128((short*)pUtf16Buffer + currentOffsetInElements); // unaligned load |
| | | 835 | | |
| | | 836 | | // See comments earlier in this method for information about how this works. |
| | | 837 | | #pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau |
| | 763 | 838 | | if (Sse41.IsSupported) |
| | | 839 | | #pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough |
| | | 840 | | { |
| | 763 | 841 | | if ((utf16VectorFirst & latin1MaskForTestZ) != Vector128<short>.Zero) |
| | | 842 | | { |
| | 0 | 843 | | goto Finish; |
| | | 844 | | } |
| | | 845 | | } |
| | | 846 | | else |
| | | 847 | | { |
| | 0 | 848 | | if ((Sse2.MoveMask(Sse2.AddSaturate(utf16VectorFirst.AsUInt16(), latin1MaskForAddSaturate).AsByte()) |
| | | 849 | | { |
| | | 850 | | goto Finish; |
| | | 851 | | } |
| | | 852 | | } |
| | | 853 | | |
| | | 854 | | // Turn the 8 Latin-1 chars we just read into 8 Latin-1 bytes, then copy it to the destination. |
| | 763 | 855 | | latin1Vector = Sse2.PackUnsignedSaturate(utf16VectorFirst, utf16VectorFirst); |
| | 763 | 856 | | Sse2.StoreScalar((ulong*)(pLatin1Buffer + currentOffsetInElements), latin1Vector.AsUInt64()); // ulong* |
| | | 857 | | } |
| | | 858 | | |
| | | 859 | | // Calculate how many elements we wrote in order to get pLatin1Buffer to its next alignment |
| | | 860 | | // point, then use that as the base offset going forward. |
| | | 861 | | |
| | 1719 | 862 | | currentOffsetInElements = SizeOfVector128 - ((nuint)pLatin1Buffer & MaskOfAllBitsInVector128); |
| | 1719 | 863 | | Debug.Assert(0 < currentOffsetInElements && currentOffsetInElements <= SizeOfVector128, "We wrote at least 1 |
| | | 864 | | |
| | 1719 | 865 | | Debug.Assert(currentOffsetInElements <= elementCount, "Shouldn't have overrun the destination buffer."); |
| | 1719 | 866 | | Debug.Assert(elementCount - currentOffsetInElements >= SizeOfVector128, "We should be able to run at least o |
| | | 867 | | |
| | 1719 | 868 | | nuint finalOffsetWhereCanRunLoop = elementCount - SizeOfVector128; |
| | | 869 | | do |
| | | 870 | | { |
| | | 871 | | // In a loop, perform two unaligned reads, narrow to a single vector, then aligned write one vector. |
| | | 872 | | |
| | 15195 | 873 | | utf16VectorFirst = Sse2.LoadVector128((short*)pUtf16Buffer + currentOffsetInElements); // unaligned load |
| | 15195 | 874 | | Vector128<short> utf16VectorSecond = Sse2.LoadVector128((short*)pUtf16Buffer + currentOffsetInElements + |
| | 15195 | 875 | | Vector128<short> combinedVector = utf16VectorFirst | utf16VectorSecond; |
| | | 876 | | |
| | | 877 | | // See comments in GetIndexOfFirstNonLatin1Char_Sse2 for information about how this works. |
| | | 878 | | #pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau |
| | 15195 | 879 | | if (Sse41.IsSupported) |
| | | 880 | | #pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough |
| | | 881 | | { |
| | 15195 | 882 | | if ((combinedVector & latin1MaskForTestZ) != Vector128<short>.Zero) |
| | | 883 | | { |
| | 0 | 884 | | goto FoundNonLatin1DataInLoop; |
| | | 885 | | } |
| | | 886 | | } |
| | | 887 | | else |
| | | 888 | | { |
| | 0 | 889 | | if ((Sse2.MoveMask(Sse2.AddSaturate(combinedVector.AsUInt16(), latin1MaskForAddSaturate).AsByte()) & |
| | | 890 | | { |
| | | 891 | | goto FoundNonLatin1DataInLoop; |
| | | 892 | | } |
| | | 893 | | } |
| | | 894 | | |
| | | 895 | | // Build up the Latin-1 vector and perform the store. |
| | | 896 | | |
| | 15195 | 897 | | latin1Vector = Sse2.PackUnsignedSaturate(utf16VectorFirst, utf16VectorSecond); |
| | | 898 | | |
| | 15195 | 899 | | Debug.Assert(((nuint)pLatin1Buffer + currentOffsetInElements) % SizeOfVector128 == 0, "Write should be a |
| | 15195 | 900 | | Sse2.StoreAligned(pLatin1Buffer + currentOffsetInElements, latin1Vector); // aligned |
| | | 901 | | |
| | 15195 | 902 | | currentOffsetInElements += SizeOfVector128; |
| | 15195 | 903 | | } while (currentOffsetInElements <= finalOffsetWhereCanRunLoop); |
| | | 904 | | |
| | | 905 | | Finish: |
| | | 906 | | |
| | | 907 | | // There might be some Latin-1 data left over. That's fine - we'll let our caller handle the final drain. |
| | 1719 | 908 | | return currentOffsetInElements; |
| | | 909 | | |
| | | 910 | | FoundNonLatin1DataInLoop: |
| | | 911 | | |
| | | 912 | | // Can we at least narrow the high vector? |
| | | 913 | | // See comments in GetIndexOfFirstNonLatin1Char_Sse2 for information about how this works. |
| | | 914 | | #pragma warning disable IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough // In this case, we have an else clau |
| | 0 | 915 | | if (Sse41.IsSupported) |
| | | 916 | | #pragma warning restore IntrinsicsInSystemPrivateCoreLibAttributeNotSpecificEnough |
| | | 917 | | { |
| | 0 | 918 | | if ((utf16VectorFirst & latin1MaskForTestZ) != Vector128<short>.Zero) |
| | | 919 | | { |
| | 0 | 920 | | goto Finish; // found non-Latin-1 data |
| | | 921 | | } |
| | | 922 | | } |
| | | 923 | | else |
| | | 924 | | { |
| | 0 | 925 | | if ((Sse2.MoveMask(Sse2.AddSaturate(utf16VectorFirst.AsUInt16(), latin1MaskForAddSaturate).AsByte()) & N |
| | | 926 | | { |
| | | 927 | | goto Finish; // found non-Latin-1 data |
| | | 928 | | } |
| | | 929 | | } |
| | | 930 | | |
| | | 931 | | // First part was all Latin-1, narrow and aligned write. Note we're only filling in the low half of the vect |
| | 0 | 932 | | latin1Vector = Sse2.PackUnsignedSaturate(utf16VectorFirst, utf16VectorFirst); |
| | | 933 | | |
| | 0 | 934 | | Debug.Assert(((nuint)pLatin1Buffer + currentOffsetInElements) % sizeof(ulong) == 0, "Destination should be u |
| | | 935 | | |
| | 0 | 936 | | Sse2.StoreScalar((ulong*)(pLatin1Buffer + currentOffsetInElements), latin1Vector.AsUInt64()); // ulong* calc |
| | 0 | 937 | | currentOffsetInElements += SizeOfVector128 / 2; |
| | | 938 | | |
| | 0 | 939 | | goto Finish; |
| | | 940 | | } |
| | | 941 | | |
| | | 942 | | /// <summary> |
| | | 943 | | /// Copies Latin-1 (narrow character) data from <paramref name="pLatin1Buffer"/> to the UTF-16 (wide character) |
| | | 944 | | /// buffer <paramref name="pUtf16Buffer"/>, widening data while copying. <paramref name="elementCount"/> |
| | | 945 | | /// specifies the element count of both the source and destination buffers. |
| | | 946 | | /// </summary> |
| | | 947 | | public static unsafe void WidenLatin1ToUtf16(byte* pLatin1Buffer, char* pUtf16Buffer, nuint elementCount) |
| | | 948 | | { |
| | | 949 | | // If SSE2 is supported, use those specific intrinsics instead of the generic vectorized |
| | | 950 | | // code below. This has two benefits: (a) we can take advantage of specific instructions like |
| | | 951 | | // punpcklbw which we know are optimized, and (b) we can avoid downclocking the processor while |
| | | 952 | | // this method is running. |
| | | 953 | | |
| | 650291 | 954 | | if (Sse2.IsSupported) |
| | | 955 | | { |
| | 650291 | 956 | | WidenLatin1ToUtf16_Sse2(pLatin1Buffer, pUtf16Buffer, elementCount); |
| | | 957 | | } |
| | | 958 | | else |
| | | 959 | | { |
| | 0 | 960 | | WidenLatin1ToUtf16_Fallback(pLatin1Buffer, pUtf16Buffer, elementCount); |
| | | 961 | | } |
| | 0 | 962 | | } |
| | | 963 | | |
| | | 964 | | [CompExactlyDependsOn(typeof(Sse2))] |
| | | 965 | | private static unsafe void WidenLatin1ToUtf16_Sse2(byte* pLatin1Buffer, char* pUtf16Buffer, nuint elementCount) |
| | | 966 | | { |
| | | 967 | | // JIT turns the below into constants |
| | | 968 | | |
| | 650291 | 969 | | uint SizeOfVector128 = (uint)sizeof(Vector128<byte>); |
| | 650291 | 970 | | nuint MaskOfAllBitsInVector128 = SizeOfVector128 - 1; |
| | | 971 | | |
| | 650291 | 972 | | Debug.Assert(Sse2.IsSupported); |
| | 650291 | 973 | | Debug.Assert(BitConverter.IsLittleEndian); |
| | | 974 | | |
| | 650291 | 975 | | nuint currentOffset = 0; |
| | 650291 | 976 | | Vector128<byte> zeroVector = Vector128<byte>.Zero; |
| | | 977 | | Vector128<byte> latin1Vector; |
| | | 978 | | |
| | | 979 | | // We're going to get the best performance when we have aligned writes, so we'll take the |
| | | 980 | | // hit of potentially unaligned reads in order to hit this sweet spot. Our central loop |
| | | 981 | | // will perform 1x 128-bit reads followed by 2x 128-bit writes, so we want to make sure |
| | | 982 | | // we actually have 128 bits of input data before entering the loop. |
| | | 983 | | |
| | 650291 | 984 | | if (elementCount >= SizeOfVector128) |
| | | 985 | | { |
| | | 986 | | // First, perform an unaligned 1x 64-bit read from the input buffer and an unaligned |
| | | 987 | | // 1x 128-bit write to the destination buffer. |
| | | 988 | | |
| | 27934 | 989 | | latin1Vector = Sse2.LoadScalarVector128((ulong*)pLatin1Buffer).AsByte(); // unaligned load |
| | 27934 | 990 | | Sse2.Store((byte*)pUtf16Buffer, Sse2.UnpackLow(latin1Vector, zeroVector)); // unaligned write |
| | | 991 | | |
| | | 992 | | // Calculate how many elements we wrote in order to get pOutputBuffer to its next alignment |
| | | 993 | | // point, then use that as the base offset going forward. Remember the >> 1 to account for |
| | | 994 | | // that we wrote chars, not bytes. This means we may re-read data in the next iteration of |
| | | 995 | | // the loop, but this is ok. |
| | | 996 | | |
| | 27934 | 997 | | currentOffset = (SizeOfVector128 >> 1) - (((nuint)pUtf16Buffer >> 1) & (MaskOfAllBitsInVector128 >> 1)); |
| | 27934 | 998 | | Debug.Assert(0 < currentOffset && currentOffset <= SizeOfVector128 / sizeof(char)); |
| | | 999 | | |
| | | 1000 | | // Calculating the destination address outside the loop results in significant |
| | | 1001 | | // perf wins vs. relying on the JIT to fold memory addressing logic into the |
| | | 1002 | | // write instructions. See: https://github.com/dotnet/runtime/issues/33002 |
| | | 1003 | | |
| | 27934 | 1004 | | char* pCurrentWriteAddress = pUtf16Buffer + currentOffset; |
| | | 1005 | | |
| | | 1006 | | // Now run the main 1x 128-bit read + 2x 128-bit write loop. |
| | | 1007 | | |
| | 27934 | 1008 | | nuint finalOffsetWhereCanIterateLoop = elementCount - SizeOfVector128; |
| | 221241 | 1009 | | while (currentOffset <= finalOffsetWhereCanIterateLoop) |
| | | 1010 | | { |
| | 193307 | 1011 | | latin1Vector = Sse2.LoadVector128(pLatin1Buffer + currentOffset); // unaligned load |
| | | 1012 | | |
| | | 1013 | | // Calculating the destination address in the below manner results in significant |
| | | 1014 | | // performance wins vs. other patterns. See for more information: |
| | | 1015 | | // https://github.com/dotnet/runtime/issues/33002 |
| | | 1016 | | |
| | 193307 | 1017 | | Vector128<byte> low = Sse2.UnpackLow(latin1Vector, zeroVector); |
| | 193307 | 1018 | | Sse2.StoreAligned((byte*)pCurrentWriteAddress, low); |
| | | 1019 | | |
| | 193307 | 1020 | | Vector128<byte> high = Sse2.UnpackHigh(latin1Vector, zeroVector); |
| | 193307 | 1021 | | Sse2.StoreAligned((byte*)pCurrentWriteAddress + SizeOfVector128, high); |
| | | 1022 | | |
| | 193307 | 1023 | | currentOffset += SizeOfVector128; |
| | 193307 | 1024 | | pCurrentWriteAddress += SizeOfVector128; |
| | | 1025 | | } |
| | | 1026 | | } |
| | | 1027 | | |
| | 650291 | 1028 | | Debug.Assert(elementCount - currentOffset < SizeOfVector128, "Case where 2 vectors remained should've been i |
| | 650291 | 1029 | | uint remaining = (uint)elementCount - (uint)currentOffset; |
| | | 1030 | | |
| | | 1031 | | // Now handle cases where we can't process two vectors at a time. |
| | | 1032 | | |
| | 650291 | 1033 | | if ((remaining & 8) != 0) |
| | | 1034 | | { |
| | | 1035 | | // Read a single 64-bit vector; write a single 128-bit vector. |
| | | 1036 | | |
| | 19687 | 1037 | | latin1Vector = Sse2.LoadScalarVector128((ulong*)(pLatin1Buffer + currentOffset)).AsByte(); // unaligned |
| | 19687 | 1038 | | Sse2.Store((byte*)(pUtf16Buffer + currentOffset), Sse2.UnpackLow(latin1Vector, zeroVector)); // unaligne |
| | 19687 | 1039 | | currentOffset += 8; |
| | | 1040 | | } |
| | | 1041 | | |
| | 650291 | 1042 | | if ((remaining & 4) != 0) |
| | | 1043 | | { |
| | | 1044 | | // Read a single 32-bit vector; write a single 64-bit vector. |
| | | 1045 | | |
| | 82735 | 1046 | | latin1Vector = Sse2.LoadScalarVector128((uint*)(pLatin1Buffer + currentOffset)).AsByte(); // unaligned l |
| | 82735 | 1047 | | Sse2.StoreScalar((ulong*)(pUtf16Buffer + currentOffset), Sse2.UnpackLow(latin1Vector, zeroVector).AsUInt |
| | 82735 | 1048 | | currentOffset += 4; |
| | | 1049 | | } |
| | | 1050 | | |
| | 650291 | 1051 | | if ((remaining & 3) != 0) |
| | | 1052 | | { |
| | | 1053 | | // 1, 2, or 3 bytes were left over |
| | 551955 | 1054 | | pUtf16Buffer[currentOffset] = (char)pLatin1Buffer[currentOffset]; |
| | | 1055 | | |
| | 551955 | 1056 | | if ((remaining & 2) != 0) |
| | | 1057 | | { |
| | | 1058 | | // 2 or 3 bytes were left over |
| | 252125 | 1059 | | pUtf16Buffer[currentOffset + 1] = (char)pLatin1Buffer[currentOffset + 1]; |
| | | 1060 | | |
| | 252125 | 1061 | | if ((remaining & 1) != 0) |
| | | 1062 | | { |
| | | 1063 | | // 1 or 3 bytes were left over (and since '1' doesn't go down this branch, we know it was actual |
| | 102360 | 1064 | | pUtf16Buffer[currentOffset + 2] = (char)pLatin1Buffer[currentOffset + 2]; |
| | | 1065 | | } |
| | | 1066 | | } |
| | | 1067 | | } |
| | 650291 | 1068 | | } |
| | | 1069 | | |
| | | 1070 | | private static unsafe void WidenLatin1ToUtf16_Fallback(byte* pLatin1Buffer, char* pUtf16Buffer, nuint elementCou |
| | | 1071 | | { |
| | 0 | 1072 | | Debug.Assert(!Sse2.IsSupported); |
| | | 1073 | | |
| | 0 | 1074 | | nuint currentOffset = 0; |
| | | 1075 | | |
| | 0 | 1076 | | if (Vector.IsHardwareAccelerated) |
| | | 1077 | | { |
| | | 1078 | | // In a loop, read 1x vector (unaligned) and write 2x vectors (unaligned). |
| | | 1079 | | |
| | 0 | 1080 | | uint SizeOfVector = (uint)Vector<byte>.Count; // JIT will make this a const |
| | | 1081 | | |
| | | 1082 | | // Only bother vectorizing if we have enough data to do so. |
| | 0 | 1083 | | if (elementCount >= SizeOfVector) |
| | | 1084 | | { |
| | 0 | 1085 | | nuint finalOffsetWhereCanIterate = elementCount - SizeOfVector; |
| | | 1086 | | do |
| | | 1087 | | { |
| | 0 | 1088 | | Vector<byte> latin1Vector = Unsafe.ReadUnaligned<Vector<byte>>(pLatin1Buffer + currentOffset); |
| | 0 | 1089 | | Vector.Widen(Vector.AsVectorByte(latin1Vector), out Vector<ushort> utf16LowVector, out Vector<us |
| | | 1090 | | |
| | | 1091 | | // TODO: Is the below logic also valid for big-endian platforms? |
| | 0 | 1092 | | Unsafe.WriteUnaligned(pUtf16Buffer + currentOffset, utf16LowVector); |
| | 0 | 1093 | | Unsafe.WriteUnaligned(pUtf16Buffer + currentOffset + Vector<ushort>.Count, utf16HighVector); |
| | | 1094 | | |
| | 0 | 1095 | | currentOffset += SizeOfVector; |
| | 0 | 1096 | | } while (currentOffset <= finalOffsetWhereCanIterate); |
| | | 1097 | | } |
| | | 1098 | | |
| | 0 | 1099 | | Debug.Assert(elementCount - currentOffset < SizeOfVector, "Vectorized logic should result in less than a |
| | | 1100 | | } |
| | | 1101 | | |
| | | 1102 | | // Flush any remaining data. |
| | | 1103 | | |
| | 0 | 1104 | | while (currentOffset < elementCount) |
| | | 1105 | | { |
| | 0 | 1106 | | pUtf16Buffer[currentOffset] = (char)pLatin1Buffer[currentOffset]; |
| | 0 | 1107 | | currentOffset++; |
| | | 1108 | | } |
| | 0 | 1109 | | } |
| | | 1110 | | } |
| | | 1111 | | } |
| | | 1112 | | |