| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | using System.Diagnostics; |
| | | 5 | | using System.Diagnostics.CodeAnalysis; |
| | | 6 | | using System.Numerics; |
| | | 7 | | using System.Runtime.CompilerServices; |
| | | 8 | | #if NET |
| | | 9 | | using System.Runtime.Intrinsics; |
| | | 10 | | using System.Runtime.Intrinsics.Arm; |
| | | 11 | | using System.Runtime.Intrinsics.Wasm; |
| | | 12 | | using System.Runtime.Intrinsics.X86; |
| | | 13 | | #endif |
| | | 14 | | |
| | | 15 | | namespace System.Text |
| | | 16 | | { |
| | | 17 | | #if SYSTEM_PRIVATE_CORELIB |
| | | 18 | | public |
| | | 19 | | #else |
| | | 20 | | internal |
| | | 21 | | #endif |
| | | 22 | | static partial class Ascii |
| | | 23 | | { |
| | | 24 | | /// <summary> |
| | | 25 | | /// Returns <see langword="true"/> iff all bytes in <paramref name="value"/> are ASCII. |
| | | 26 | | /// </summary> |
| | | 27 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 28 | | private static bool AllBytesInUInt64AreAscii(ulong value) |
| | | 29 | | { |
| | | 30 | | // If the high bit of any byte is set, that byte is non-ASCII. |
| | | 31 | | |
| | 0 | 32 | | return (value & UInt64HighBitsOnlyMask) == 0; |
| | | 33 | | } |
| | | 34 | | |
| | | 35 | | /// <summary> |
| | | 36 | | /// Returns <see langword="true"/> iff all chars in <paramref name="value"/> are ASCII. |
| | | 37 | | /// </summary> |
| | | 38 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 39 | | private static bool AllCharsInUInt32AreAscii(uint value) |
| | | 40 | | { |
| | 11147 | 41 | | return (value & ~0x007F007Fu) == 0; |
| | | 42 | | } |
| | | 43 | | |
| | | 44 | | /// <summary> |
| | | 45 | | /// Returns <see langword="true"/> iff all chars in <paramref name="value"/> are ASCII. |
| | | 46 | | /// </summary> |
| | | 47 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 48 | | private static bool AllCharsInUInt64AreAscii(ulong value) |
| | | 49 | | { |
| | 34348 | 50 | | return (value & ~0x007F007F_007F007Ful) == 0; |
| | | 51 | | } |
| | | 52 | | |
| | | 53 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 54 | | private static bool AllCharsInUInt64AreAscii<T>(ulong value) |
| | | 55 | | where T : unmanaged |
| | | 56 | | { |
| | 1 | 57 | | Debug.Assert(typeof(T) == typeof(byte) || typeof(T) == typeof(ushort)); |
| | | 58 | | |
| | 1 | 59 | | return typeof(T) == typeof(byte) |
| | 1 | 60 | | ? AllBytesInUInt64AreAscii(value) |
| | 1 | 61 | | : AllCharsInUInt64AreAscii(value); |
| | | 62 | | } |
| | | 63 | | |
| | | 64 | | #if NET |
| | | 65 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 66 | | [CompExactlyDependsOn(typeof(AdvSimd.Arm64))] |
| | | 67 | | private static int GetIndexOfFirstNonAsciiByteInLane_AdvSimd(Vector128<byte> value, Vector128<byte> bitmask) |
| | | 68 | | { |
| | | 69 | | if (!AdvSimd.Arm64.IsSupported || !BitConverter.IsLittleEndian) |
| | | 70 | | { |
| | | 71 | | throw new PlatformNotSupportedException(); |
| | | 72 | | } |
| | | 73 | | |
| | | 74 | | // extractedBits[i] = (value[i] >> 7) & (1 << (12 * (i % 2))); |
| | | 75 | | Vector128<byte> mostSignificantBitIsSet = (value.AsSByte() >> 7).AsByte(); |
| | | 76 | | Vector128<byte> extractedBits = mostSignificantBitIsSet & bitmask; |
| | | 77 | | |
| | | 78 | | // collapse mask to lower bits |
| | | 79 | | extractedBits = AdvSimd.Arm64.AddPairwise(extractedBits, extractedBits); |
| | | 80 | | ulong mask = extractedBits.AsUInt64().ToScalar(); |
| | | 81 | | |
| | | 82 | | // calculate the index |
| | | 83 | | int index = BitOperations.TrailingZeroCount(mask) >> 2; |
| | | 84 | | Debug.Assert((mask != 0) ? index < 16 : index >= 16); |
| | | 85 | | return index; |
| | | 86 | | } |
| | | 87 | | #endif |
| | | 88 | | |
| | | 89 | | /// <summary> |
| | | 90 | | /// Given a DWORD which represents two packed chars in machine-endian order, |
| | | 91 | | /// <see langword="true"/> iff the first char (in machine-endian order) is ASCII. |
| | | 92 | | /// </summary> |
| | | 93 | | /// <param name="value"></param> |
| | | 94 | | /// <returns></returns> |
| | | 95 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 96 | | private static bool FirstCharInUInt32IsAscii(uint value) |
| | | 97 | | { |
| | | 98 | | return (BitConverter.IsLittleEndian && (value & 0xFF80u) == 0) |
| | | 99 | | || (!BitConverter.IsLittleEndian && (value & 0xFF800000u) == 0); |
| | | 100 | | } |
| | | 101 | | |
| | | 102 | | /// <summary> |
| | | 103 | | /// Returns the index in <paramref name="pBuffer"/> where the first non-ASCII byte is found. |
| | | 104 | | /// Returns <paramref name="bufferLength"/> if the buffer is empty or all-ASCII. |
| | | 105 | | /// </summary> |
| | | 106 | | /// <returns>An ASCII byte is defined as 0x00 - 0x7F, inclusive.</returns> |
| | | 107 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 108 | | internal static unsafe nuint GetIndexOfFirstNonAsciiByte(byte* pBuffer, nuint bufferLength) |
| | | 109 | | { |
| | | 110 | | // If 256/512-bit aren't supported but SSE2 is supported, use those specific intrinsics instead of |
| | | 111 | | // the generic vectorized code. This has two benefits: (a) we can take advantage of specific instructions |
| | | 112 | | // like pmovmskb which we know are optimized, and (b) we can avoid downclocking the processor while |
| | | 113 | | // this method is running. |
| | | 114 | | |
| | | 115 | | #if NET |
| | 855076 | 116 | | if (!Vector512.IsHardwareAccelerated && |
| | 855076 | 117 | | !Vector256.IsHardwareAccelerated && |
| | 855076 | 118 | | (Sse2.IsSupported || AdvSimd.IsSupported)) |
| | | 119 | | { |
| | 0 | 120 | | return GetIndexOfFirstNonAsciiByte_Intrinsified(pBuffer, bufferLength); |
| | | 121 | | } |
| | | 122 | | else |
| | | 123 | | #endif |
| | | 124 | | { |
| | | 125 | | // Handles Vector512, Vector256, Vector128, and scalar. |
| | 855076 | 126 | | return GetIndexOfFirstNonAsciiByte_Vector(pBuffer, bufferLength); |
| | | 127 | | } |
| | | 128 | | } |
| | | 129 | | |
| | | 130 | | private static unsafe nuint GetIndexOfFirstNonAsciiByte_Vector(byte* pBuffer, nuint bufferLength) |
| | | 131 | | { |
| | | 132 | | // Squirrel away the original buffer reference. This method works by determining the exact |
| | | 133 | | // byte reference where non-ASCII data begins, so we need this base value to perform the |
| | | 134 | | // final subtraction at the end of the method to get the index into the original buffer. |
| | | 135 | | |
| | | 136 | | byte* pOriginalBuffer = pBuffer; |
| | | 137 | | |
| | | 138 | | // Before we drain off byte-by-byte, try a generic vectorized loop. |
| | | 139 | | // Only run the loop if we have at least two vectors we can pull out. |
| | | 140 | | // Note use of SBYTE instead of BYTE below; we're using the two's-complement |
| | | 141 | | // representation of negative integers to act as a surrogate for "is ASCII?". |
| | | 142 | | |
| | | 143 | | #if NET |
| | 855076 | 144 | | if (Vector512.IsHardwareAccelerated && bufferLength >= 2 * (uint)Vector512<byte>.Count) |
| | | 145 | | { |
| | 295304 | 146 | | if (Vector512.Load(pBuffer).ExtractMostSignificantBits() == 0) |
| | | 147 | | { |
| | | 148 | | // The first several elements of the input buffer were ASCII. Bump up the pointer to the |
| | | 149 | | // next aligned boundary, then perform aligned reads from here on out until we find non-ASCII |
| | | 150 | | // data or we approach the end of the buffer. It's possible we'll reread data; this is ok. |
| | | 151 | | |
| | 1396 | 152 | | byte* pFinalVectorReadPos = pBuffer + bufferLength - Vector512.Size; |
| | 1396 | 153 | | pBuffer = (byte*)(((nuint)pBuffer + Vector512.Size) & ~(nuint)(Vector512.Size - 1)); |
| | | 154 | | |
| | | 155 | | #if DEBUG |
| | 1396 | 156 | | long numBytesRead = pBuffer - pOriginalBuffer; |
| | 1396 | 157 | | Debug.Assert(0 < numBytesRead && numBytesRead <= Vector512.Size, "We should've made forward progress |
| | 1396 | 158 | | Debug.Assert((nuint)numBytesRead <= bufferLength, "We shouldn't have read past the end of the input |
| | | 159 | | #endif |
| | | 160 | | |
| | 1396 | 161 | | Debug.Assert(pBuffer <= pFinalVectorReadPos, "Should be able to read at least one vector."); |
| | | 162 | | |
| | | 163 | | do |
| | | 164 | | { |
| | 3368 | 165 | | Debug.Assert((nuint)pBuffer % Vector512.Size == 0, "Vector read should be aligned."); |
| | 3368 | 166 | | if (Vector512.LoadAligned(pBuffer).ExtractMostSignificantBits() != 0) |
| | | 167 | | { |
| | | 168 | | break; // found non-ASCII data |
| | | 169 | | } |
| | | 170 | | |
| | 2361 | 171 | | pBuffer += Vector512.Size; |
| | 2361 | 172 | | } while (pBuffer <= pFinalVectorReadPos); |
| | | 173 | | |
| | | 174 | | // Adjust the remaining buffer length for the number of elements we just consumed. |
| | | 175 | | |
| | 1396 | 176 | | bufferLength -= (nuint)pBuffer; |
| | 1396 | 177 | | bufferLength += (nuint)pOriginalBuffer; |
| | | 178 | | } |
| | | 179 | | } |
| | 559772 | 180 | | else if (Vector256.IsHardwareAccelerated && bufferLength >= 2 * (uint)Vector256<byte>.Count) |
| | | 181 | | { |
| | 163342 | 182 | | if (Vector256.Load(pBuffer).ExtractMostSignificantBits() == 0) |
| | | 183 | | { |
| | | 184 | | // The first several elements of the input buffer were ASCII. Bump up the pointer to the |
| | | 185 | | // next aligned boundary, then perform aligned reads from here on out until we find non-ASCII |
| | | 186 | | // data or we approach the end of the buffer. It's possible we'll reread data; this is ok. |
| | | 187 | | |
| | 1023 | 188 | | byte* pFinalVectorReadPos = pBuffer + bufferLength - Vector256.Size; |
| | 1023 | 189 | | pBuffer = (byte*)(((nuint)pBuffer + Vector256.Size) & ~(nuint)(Vector256.Size - 1)); |
| | | 190 | | |
| | | 191 | | #if DEBUG |
| | 1023 | 192 | | long numBytesRead = pBuffer - pOriginalBuffer; |
| | 1023 | 193 | | Debug.Assert(0 < numBytesRead && numBytesRead <= Vector256.Size, "We should've made forward progress |
| | 1023 | 194 | | Debug.Assert((nuint)numBytesRead <= bufferLength, "We shouldn't have read past the end of the input |
| | | 195 | | #endif |
| | | 196 | | |
| | 1023 | 197 | | Debug.Assert(pBuffer <= pFinalVectorReadPos, "Should be able to read at least one vector."); |
| | | 198 | | |
| | | 199 | | do |
| | | 200 | | { |
| | 2147 | 201 | | Debug.Assert((nuint)pBuffer % Vector256.Size == 0, "Vector read should be aligned."); |
| | 2147 | 202 | | if (Vector256.LoadAligned(pBuffer).ExtractMostSignificantBits() != 0) |
| | | 203 | | { |
| | | 204 | | break; // found non-ASCII data |
| | | 205 | | } |
| | | 206 | | |
| | 1440 | 207 | | pBuffer += Vector256.Size; |
| | 1440 | 208 | | } while (pBuffer <= pFinalVectorReadPos); |
| | | 209 | | |
| | | 210 | | // Adjust the remaining buffer length for the number of elements we just consumed. |
| | | 211 | | |
| | 1023 | 212 | | bufferLength -= (nuint)pBuffer; |
| | 1023 | 213 | | bufferLength += (nuint)pOriginalBuffer; |
| | | 214 | | } |
| | | 215 | | } |
| | 396430 | 216 | | else if (Vector128.IsHardwareAccelerated && bufferLength >= 2 * (uint)Vector128<byte>.Count) |
| | | 217 | | { |
| | 143031 | 218 | | if (!VectorContainsNonAsciiChar(Vector128.Load(pBuffer))) |
| | | 219 | | { |
| | | 220 | | // The first several elements of the input buffer were ASCII. Bump up the pointer to the |
| | | 221 | | // next aligned boundary, then perform aligned reads from here on out until we find non-ASCII |
| | | 222 | | // data or we approach the end of the buffer. It's possible we'll reread data; this is ok. |
| | | 223 | | |
| | 1587 | 224 | | byte* pFinalVectorReadPos = pBuffer + bufferLength - Vector128.Size; |
| | 1587 | 225 | | pBuffer = (byte*)(((nuint)pBuffer + Vector128.Size) & ~(nuint)(Vector128.Size - 1)); |
| | | 226 | | |
| | | 227 | | #if DEBUG |
| | 1587 | 228 | | long numBytesRead = pBuffer - pOriginalBuffer; |
| | 1587 | 229 | | Debug.Assert(0 < numBytesRead && numBytesRead <= Vector128.Size, "We should've made forward progress |
| | 1587 | 230 | | Debug.Assert((nuint)numBytesRead <= bufferLength, "We shouldn't have read past the end of the input |
| | | 231 | | #endif |
| | | 232 | | |
| | 1587 | 233 | | Debug.Assert(pBuffer <= pFinalVectorReadPos, "Should be able to read at least one vector."); |
| | | 234 | | |
| | | 235 | | do |
| | | 236 | | { |
| | 2901 | 237 | | Debug.Assert((nuint)pBuffer % Vector128.Size == 0, "Vector read should be aligned."); |
| | 2901 | 238 | | if (VectorContainsNonAsciiChar(Vector128.LoadAligned(pBuffer))) |
| | | 239 | | { |
| | | 240 | | break; // found non-ASCII data |
| | | 241 | | } |
| | | 242 | | |
| | 1730 | 243 | | pBuffer += Vector128.Size; |
| | 1730 | 244 | | } while (pBuffer <= pFinalVectorReadPos); |
| | | 245 | | |
| | | 246 | | // Adjust the remaining buffer length for the number of elements we just consumed. |
| | | 247 | | |
| | 1587 | 248 | | bufferLength -= (nuint)pBuffer; |
| | 1587 | 249 | | bufferLength += (nuint)pOriginalBuffer; |
| | | 250 | | } |
| | | 251 | | } |
| | | 252 | | #endif |
| | | 253 | | |
| | | 254 | | // At this point, the buffer length wasn't enough to perform a vectorized search, or we did perform |
| | | 255 | | // a vectorized search and encountered non-ASCII data. In either case go down a non-vectorized code |
| | | 256 | | // path to drain any remaining ASCII bytes. |
| | | 257 | | // |
| | | 258 | | // We're going to perform unaligned reads, so prefer 32-bit reads instead of 64-bit reads. |
| | | 259 | | // This also allows us to perform more optimized bit twiddling tricks to count the number of ASCII bytes. |
| | | 260 | | |
| | | 261 | | uint currentUInt32; |
| | | 262 | | |
| | | 263 | | // Try reading 64 bits at a time in a loop. |
| | | 264 | | |
| | 905432 | 265 | | for (; bufferLength >= 8; bufferLength -= 8) |
| | | 266 | | { |
| | 796692 | 267 | | currentUInt32 = Unsafe.ReadUnaligned<uint>(pBuffer); |
| | 796692 | 268 | | uint nextUInt32 = Unsafe.ReadUnaligned<uint>(pBuffer + 4); |
| | | 269 | | |
| | 796692 | 270 | | if (!AllBytesInUInt32AreAscii(currentUInt32 | nextUInt32)) |
| | | 271 | | { |
| | | 272 | | // One of these two values contains non-ASCII bytes. |
| | | 273 | | // Figure out which one it is, then put it in 'current' so that we can drain the ASCII bytes. |
| | | 274 | | |
| | 771514 | 275 | | if (AllBytesInUInt32AreAscii(currentUInt32)) |
| | | 276 | | { |
| | 25325 | 277 | | currentUInt32 = nextUInt32; |
| | 25325 | 278 | | pBuffer += 4; |
| | | 279 | | } |
| | | 280 | | |
| | 25325 | 281 | | goto FoundNonAsciiData; |
| | | 282 | | } |
| | | 283 | | |
| | 25178 | 284 | | pBuffer += 8; // consumed 8 ASCII bytes |
| | | 285 | | } |
| | | 286 | | |
| | | 287 | | // From this point forward we don't need to update bufferLength. |
| | | 288 | | // Try reading 32 bits. |
| | | 289 | | |
| | 83562 | 290 | | if ((bufferLength & 4) != 0) |
| | | 291 | | { |
| | 44247 | 292 | | currentUInt32 = Unsafe.ReadUnaligned<uint>(pBuffer); |
| | 44247 | 293 | | if (!AllBytesInUInt32AreAscii(currentUInt32)) |
| | | 294 | | { |
| | | 295 | | goto FoundNonAsciiData; |
| | | 296 | | } |
| | | 297 | | |
| | 3765 | 298 | | pBuffer += 4; |
| | | 299 | | } |
| | | 300 | | |
| | | 301 | | // Try reading 16 bits. |
| | | 302 | | |
| | 43080 | 303 | | if ((bufferLength & 2) != 0) |
| | | 304 | | { |
| | 26894 | 305 | | currentUInt32 = Unsafe.ReadUnaligned<ushort>(pBuffer); |
| | 26894 | 306 | | if (!AllBytesInUInt32AreAscii(currentUInt32)) |
| | | 307 | | { |
| | 22665 | 308 | | if (!BitConverter.IsLittleEndian) |
| | | 309 | | { |
| | | 310 | | currentUInt32 <<= 16; |
| | | 311 | | } |
| | | 312 | | goto FoundNonAsciiData; |
| | | 313 | | } |
| | | 314 | | |
| | 4229 | 315 | | pBuffer += 2; |
| | | 316 | | } |
| | | 317 | | |
| | | 318 | | // Try reading 8 bits |
| | | 319 | | |
| | 20415 | 320 | | if ((bufferLength & 1) != 0) |
| | | 321 | | { |
| | | 322 | | // If the buffer contains non-ASCII data, the comparison below will fail, and |
| | | 323 | | // we'll end up not incrementing the buffer reference. |
| | | 324 | | |
| | 16379 | 325 | | if (*(sbyte*)pBuffer >= 0) |
| | | 326 | | { |
| | 5850 | 327 | | pBuffer++; |
| | | 328 | | } |
| | | 329 | | } |
| | | 330 | | |
| | | 331 | | Finish: |
| | | 332 | | |
| | 855076 | 333 | | nuint totalNumBytesRead = (nuint)pBuffer - (nuint)pOriginalBuffer; |
| | 855076 | 334 | | return totalNumBytesRead; |
| | | 335 | | |
| | | 336 | | FoundNonAsciiData: |
| | | 337 | | |
| | 834661 | 338 | | Debug.Assert(!AllBytesInUInt32AreAscii(currentUInt32), "Shouldn't have reached this point if we have an all- |
| | | 339 | | |
| | | 340 | | // The method being called doesn't bother looking at whether the high byte is ASCII. There are only |
| | | 341 | | // two scenarios: (a) either one of the earlier bytes is not ASCII and the search terminates before |
| | | 342 | | // we get to the high byte; or (b) all of the earlier bytes are ASCII, so the high byte must be |
| | | 343 | | // non-ASCII. In both cases we only care about the low 24 bits. |
| | | 344 | | |
| | 834661 | 345 | | pBuffer += CountNumberOfLeadingAsciiBytesFromUInt32WithSomeNonAsciiData(currentUInt32); |
| | 834661 | 346 | | goto Finish; |
| | | 347 | | } |
| | | 348 | | |
| | | 349 | | #if NET |
| | | 350 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 351 | | private static bool ContainsNonAsciiByte_Sse2(uint sseMask) |
| | | 352 | | { |
| | 0 | 353 | | Debug.Assert(sseMask != uint.MaxValue); |
| | 0 | 354 | | Debug.Assert(Sse2.IsSupported); |
| | 0 | 355 | | return sseMask != 0; |
| | | 356 | | } |
| | | 357 | | |
| | | 358 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 359 | | private static bool ContainsNonAsciiByte_AdvSimd(uint advSimdIndex) |
| | | 360 | | { |
| | | 361 | | Debug.Assert(advSimdIndex != uint.MaxValue); |
| | | 362 | | Debug.Assert(AdvSimd.IsSupported); |
| | | 363 | | return advSimdIndex < 16; |
| | | 364 | | } |
| | | 365 | | |
| | | 366 | | private static unsafe nuint GetIndexOfFirstNonAsciiByte_Intrinsified(byte* pBuffer, nuint bufferLength) |
| | | 367 | | { |
| | | 368 | | // JIT turns the below into constants |
| | | 369 | | |
| | | 370 | | uint SizeOfVector128 = (uint)sizeof(Vector128<byte>); |
| | 0 | 371 | | nuint MaskOfAllBitsInVector128 = (nuint)(SizeOfVector128 - 1); |
| | | 372 | | |
| | 0 | 373 | | Debug.Assert(Sse2.IsSupported || AdvSimd.Arm64.IsSupported, "Sse2 or AdvSimd64 required."); |
| | 0 | 374 | | Debug.Assert(BitConverter.IsLittleEndian, "This SSE2/Arm64 implementation assumes little-endian."); |
| | | 375 | | |
| | 0 | 376 | | Vector128<byte> bitmask = BitConverter.IsLittleEndian ? |
| | 0 | 377 | | Vector128.Create((ushort)0x1001).AsByte() : |
| | 0 | 378 | | Vector128.Create((ushort)0x0110).AsByte(); |
| | | 379 | | |
| | 0 | 380 | | uint currentSseMask = uint.MaxValue, secondSseMask = uint.MaxValue; |
| | 0 | 381 | | uint currentAdvSimdIndex = uint.MaxValue, secondAdvSimdIndex = uint.MaxValue; |
| | 0 | 382 | | byte* pOriginalBuffer = pBuffer; |
| | | 383 | | |
| | | 384 | | // This method is written such that control generally flows top-to-bottom, avoiding |
| | | 385 | | // jumps as much as possible in the optimistic case of a large enough buffer and |
| | | 386 | | // "all ASCII". If we see non-ASCII data, we jump out of the hot paths to targets |
| | | 387 | | // after all the main logic. |
| | | 388 | | |
| | 0 | 389 | | if (bufferLength < SizeOfVector128) |
| | | 390 | | { |
| | | 391 | | goto InputBufferLessThanOneVectorInLength; // can't vectorize; drain primitives instead |
| | | 392 | | } |
| | | 393 | | |
| | | 394 | | // Read the first vector unaligned. |
| | | 395 | | |
| | 0 | 396 | | if (Sse2.IsSupported) |
| | | 397 | | { |
| | 0 | 398 | | currentSseMask = (uint)Sse2.MoveMask(Sse2.LoadVector128(pBuffer)); // unaligned load |
| | 0 | 399 | | if (ContainsNonAsciiByte_Sse2(currentSseMask)) |
| | | 400 | | { |
| | 0 | 401 | | goto FoundNonAsciiDataInCurrentChunk; |
| | | 402 | | } |
| | | 403 | | } |
| | | 404 | | else if (AdvSimd.Arm64.IsSupported) |
| | | 405 | | { |
| | | 406 | | Vector128<byte> vector = AdvSimd.LoadVector128(pBuffer); |
| | | 407 | | if (VectorContainsNonAsciiChar(vector)) |
| | | 408 | | { |
| | | 409 | | currentAdvSimdIndex = (uint)GetIndexOfFirstNonAsciiByteInLane_AdvSimd(vector, bitmask); // unaligned |
| | | 410 | | goto FoundNonAsciiDataInCurrentChunk; |
| | | 411 | | } |
| | | 412 | | } |
| | | 413 | | else |
| | | 414 | | { |
| | 0 | 415 | | throw new PlatformNotSupportedException(); |
| | | 416 | | } |
| | | 417 | | |
| | | 418 | | // If we have less than 32 bytes to process, just go straight to the final unaligned |
| | | 419 | | // read. There's no need to mess with the loop logic in the middle of this method. |
| | | 420 | | |
| | 0 | 421 | | if (bufferLength < 2 * SizeOfVector128) |
| | | 422 | | { |
| | | 423 | | goto IncrementCurrentOffsetBeforeFinalUnalignedVectorRead; |
| | | 424 | | } |
| | | 425 | | |
| | | 426 | | // Now adjust the read pointer so that future reads are aligned. |
| | | 427 | | |
| | 0 | 428 | | pBuffer = (byte*)(((nuint)pBuffer + SizeOfVector128) & ~(nuint)MaskOfAllBitsInVector128); |
| | | 429 | | |
| | | 430 | | #if DEBUG |
| | 0 | 431 | | long numBytesRead = pBuffer - pOriginalBuffer; |
| | 0 | 432 | | Debug.Assert(0 < numBytesRead && numBytesRead <= SizeOfVector128, "We should've made forward progress of at |
| | 0 | 433 | | Debug.Assert((nuint)numBytesRead <= bufferLength, "We shouldn't have read past the end of the input buffer." |
| | | 434 | | #endif |
| | | 435 | | |
| | | 436 | | // Adjust the remaining length to account for what we just read. |
| | | 437 | | |
| | 0 | 438 | | bufferLength += (nuint)pOriginalBuffer; |
| | 0 | 439 | | bufferLength -= (nuint)pBuffer; |
| | | 440 | | |
| | | 441 | | // The buffer is now properly aligned. |
| | | 442 | | // Read 2 vectors at a time if possible. |
| | | 443 | | |
| | 0 | 444 | | if (bufferLength >= 2 * SizeOfVector128) |
| | | 445 | | { |
| | 0 | 446 | | byte* pFinalVectorReadPos = (byte*)((nuint)pBuffer + bufferLength - 2 * SizeOfVector128); |
| | | 447 | | |
| | | 448 | | // After this point, we no longer need to update the bufferLength value. |
| | | 449 | | |
| | | 450 | | do |
| | | 451 | | { |
| | 0 | 452 | | if (Sse2.IsSupported) |
| | | 453 | | { |
| | 0 | 454 | | Vector128<byte> firstVector = Sse2.LoadAlignedVector128(pBuffer); |
| | 0 | 455 | | Vector128<byte> secondVector = Sse2.LoadAlignedVector128(pBuffer + SizeOfVector128); |
| | | 456 | | |
| | 0 | 457 | | currentSseMask = (uint)Sse2.MoveMask(firstVector); |
| | 0 | 458 | | secondSseMask = (uint)Sse2.MoveMask(secondVector); |
| | 0 | 459 | | if (ContainsNonAsciiByte_Sse2(currentSseMask | secondSseMask)) |
| | | 460 | | { |
| | 0 | 461 | | goto FoundNonAsciiDataInInnerLoop; |
| | | 462 | | } |
| | | 463 | | } |
| | | 464 | | else if (AdvSimd.Arm64.IsSupported) |
| | | 465 | | { |
| | | 466 | | Vector128<byte> firstVector = AdvSimd.LoadVector128(pBuffer); |
| | | 467 | | Vector128<byte> secondVector = AdvSimd.LoadVector128(pBuffer + SizeOfVector128); |
| | | 468 | | |
| | | 469 | | if (VectorContainsNonAsciiChar(firstVector | secondVector)) |
| | | 470 | | { |
| | | 471 | | currentAdvSimdIndex = (uint)GetIndexOfFirstNonAsciiByteInLane_AdvSimd(firstVector, bitmask); |
| | | 472 | | secondAdvSimdIndex = (uint)GetIndexOfFirstNonAsciiByteInLane_AdvSimd(secondVector, bitmask); |
| | | 473 | | goto FoundNonAsciiDataInInnerLoop; |
| | | 474 | | } |
| | | 475 | | } |
| | | 476 | | else |
| | | 477 | | { |
| | 0 | 478 | | throw new PlatformNotSupportedException(); |
| | | 479 | | } |
| | | 480 | | |
| | 0 | 481 | | pBuffer += 2 * SizeOfVector128; |
| | 0 | 482 | | } while (pBuffer <= pFinalVectorReadPos); |
| | | 483 | | } |
| | | 484 | | |
| | | 485 | | // We have somewhere between 0 and (2 * vector length) - 1 bytes remaining to read from. |
| | | 486 | | // Since the above loop doesn't update bufferLength, we can't rely on its absolute value. |
| | | 487 | | // But we _can_ rely on it to tell us how much remaining data must be drained by looking |
| | | 488 | | // at what bits of it are set. This works because had we updated it within the loop above, |
| | | 489 | | // we would've been adding 2 * SizeOfVector128 on each iteration, but we only care about |
| | | 490 | | // bits which are less significant than those that the addition would've acted on. |
| | | 491 | | |
| | | 492 | | // If there is fewer than one vector length remaining, skip the next aligned read. |
| | | 493 | | |
| | 0 | 494 | | if ((bufferLength & SizeOfVector128) == 0) |
| | | 495 | | { |
| | | 496 | | goto DoFinalUnalignedVectorRead; |
| | | 497 | | } |
| | | 498 | | |
| | | 499 | | // At least one full vector's worth of data remains, so we can safely read it. |
| | | 500 | | // Remember, at this point pBuffer is still aligned. |
| | | 501 | | |
| | 0 | 502 | | if (Sse2.IsSupported) |
| | | 503 | | { |
| | 0 | 504 | | currentSseMask = (uint)Sse2.MoveMask(Sse2.LoadAlignedVector128(pBuffer)); |
| | 0 | 505 | | if (ContainsNonAsciiByte_Sse2(currentSseMask)) |
| | | 506 | | { |
| | 0 | 507 | | goto FoundNonAsciiDataInCurrentChunk; |
| | | 508 | | } |
| | | 509 | | } |
| | | 510 | | else if (AdvSimd.Arm64.IsSupported) |
| | | 511 | | { |
| | | 512 | | Vector128<byte> vector = AdvSimd.LoadVector128(pBuffer); |
| | | 513 | | if (VectorContainsNonAsciiChar(vector)) |
| | | 514 | | { |
| | | 515 | | currentAdvSimdIndex = (uint)GetIndexOfFirstNonAsciiByteInLane_AdvSimd(vector, bitmask); |
| | | 516 | | goto FoundNonAsciiDataInCurrentChunk; |
| | | 517 | | } |
| | | 518 | | } |
| | | 519 | | else |
| | | 520 | | { |
| | 0 | 521 | | throw new PlatformNotSupportedException(); |
| | | 522 | | } |
| | | 523 | | |
| | | 524 | | IncrementCurrentOffsetBeforeFinalUnalignedVectorRead: |
| | | 525 | | |
| | 0 | 526 | | pBuffer += SizeOfVector128; |
| | | 527 | | |
| | | 528 | | DoFinalUnalignedVectorRead: |
| | | 529 | | |
| | 0 | 530 | | if (((byte)bufferLength & MaskOfAllBitsInVector128) != 0) |
| | | 531 | | { |
| | | 532 | | // Perform an unaligned read of the last vector. |
| | | 533 | | // We need to adjust the pointer because we're re-reading data. |
| | | 534 | | |
| | 0 | 535 | | pBuffer += (bufferLength & MaskOfAllBitsInVector128) - SizeOfVector128; |
| | | 536 | | |
| | 0 | 537 | | if (Sse2.IsSupported) |
| | | 538 | | { |
| | 0 | 539 | | currentSseMask = (uint)Sse2.MoveMask(Sse2.LoadVector128(pBuffer)); // unaligned load |
| | 0 | 540 | | if (ContainsNonAsciiByte_Sse2(currentSseMask)) |
| | | 541 | | { |
| | 0 | 542 | | goto FoundNonAsciiDataInCurrentChunk; |
| | | 543 | | } |
| | | 544 | | |
| | | 545 | | } |
| | | 546 | | else if (AdvSimd.Arm64.IsSupported) |
| | | 547 | | { |
| | | 548 | | Vector128<byte> vector = AdvSimd.LoadVector128(pBuffer); |
| | | 549 | | if (VectorContainsNonAsciiChar(vector)) |
| | | 550 | | { |
| | | 551 | | currentAdvSimdIndex = (uint)GetIndexOfFirstNonAsciiByteInLane_AdvSimd(vector, bitmask); // unali |
| | | 552 | | goto FoundNonAsciiDataInCurrentChunk; |
| | | 553 | | } |
| | | 554 | | |
| | | 555 | | } |
| | | 556 | | else |
| | | 557 | | { |
| | 0 | 558 | | throw new PlatformNotSupportedException(); |
| | | 559 | | } |
| | | 560 | | |
| | 0 | 561 | | pBuffer += SizeOfVector128; |
| | | 562 | | } |
| | | 563 | | |
| | | 564 | | Finish: |
| | 0 | 565 | | return (nuint)pBuffer - (nuint)pOriginalBuffer; // and we're done! |
| | | 566 | | |
| | | 567 | | FoundNonAsciiDataInInnerLoop: |
| | | 568 | | |
| | | 569 | | // If the current (first) mask isn't the mask that contains non-ASCII data, then it must |
| | | 570 | | // instead be the second mask. If so, skip the entire first mask and drain ASCII bytes |
| | | 571 | | // from the second mask. |
| | | 572 | | |
| | 0 | 573 | | if (Sse2.IsSupported) |
| | | 574 | | { |
| | 0 | 575 | | if (!ContainsNonAsciiByte_Sse2(currentSseMask)) |
| | | 576 | | { |
| | 0 | 577 | | pBuffer += SizeOfVector128; |
| | 0 | 578 | | currentSseMask = secondSseMask; |
| | | 579 | | } |
| | | 580 | | } |
| | | 581 | | else if (AdvSimd.IsSupported) |
| | | 582 | | { |
| | | 583 | | if (!ContainsNonAsciiByte_AdvSimd(currentAdvSimdIndex)) |
| | | 584 | | { |
| | | 585 | | pBuffer += SizeOfVector128; |
| | | 586 | | currentAdvSimdIndex = secondAdvSimdIndex; |
| | | 587 | | } |
| | | 588 | | } |
| | | 589 | | else |
| | | 590 | | { |
| | 0 | 591 | | throw new PlatformNotSupportedException(); |
| | | 592 | | } |
| | | 593 | | FoundNonAsciiDataInCurrentChunk: |
| | | 594 | | |
| | | 595 | | |
| | 0 | 596 | | if (Sse2.IsSupported) |
| | | 597 | | { |
| | | 598 | | // The mask contains - from the LSB - a 0 for each ASCII byte we saw, and a 1 for each non-ASCII byte. |
| | | 599 | | // Tzcnt is the correct operation to count the number of zero bits quickly. If this instruction isn't |
| | | 600 | | // available, we'll fall back to a normal loop. |
| | 0 | 601 | | Debug.Assert(ContainsNonAsciiByte_Sse2(currentSseMask), "Shouldn't be here unless we see non-ASCII data. |
| | 0 | 602 | | pBuffer += (uint)BitOperations.TrailingZeroCount(currentSseMask); |
| | | 603 | | } |
| | | 604 | | else if (AdvSimd.Arm64.IsSupported) |
| | | 605 | | { |
| | | 606 | | Debug.Assert(ContainsNonAsciiByte_AdvSimd(currentAdvSimdIndex), "Shouldn't be here unless we see non-ASC |
| | | 607 | | pBuffer += currentAdvSimdIndex; |
| | | 608 | | } |
| | | 609 | | else |
| | | 610 | | { |
| | 0 | 611 | | throw new PlatformNotSupportedException(); |
| | | 612 | | } |
| | | 613 | | |
| | | 614 | | goto Finish; |
| | | 615 | | |
| | | 616 | | FoundNonAsciiDataInCurrentDWord: |
| | | 617 | | |
| | | 618 | | uint currentDWord; |
| | 0 | 619 | | Debug.Assert(!AllBytesInUInt32AreAscii(currentDWord), "Shouldn't be here unless we see non-ASCII data."); |
| | 0 | 620 | | pBuffer += CountNumberOfLeadingAsciiBytesFromUInt32WithSomeNonAsciiData(currentDWord); |
| | | 621 | | |
| | 0 | 622 | | goto Finish; |
| | | 623 | | |
| | | 624 | | InputBufferLessThanOneVectorInLength: |
| | | 625 | | |
| | | 626 | | // These code paths get hit if the original input length was less than one vector in size. |
| | | 627 | | // We can't perform vectorized reads at this point, so we'll fall back to reading primitives |
| | | 628 | | // directly. Note that all of these reads are unaligned. |
| | | 629 | | |
| | 0 | 630 | | Debug.Assert(bufferLength < SizeOfVector128); |
| | | 631 | | |
| | | 632 | | // QWORD drain |
| | | 633 | | |
| | 0 | 634 | | if ((bufferLength & 8) != 0) |
| | | 635 | | { |
| | | 636 | | if (UIntPtr.Size == sizeof(ulong)) |
| | | 637 | | { |
| | | 638 | | // If we can use 64-bit tzcnt to count the number of leading ASCII bytes, prefer it. |
| | | 639 | | |
| | 0 | 640 | | ulong candidateUInt64 = Unsafe.ReadUnaligned<ulong>(pBuffer); |
| | 0 | 641 | | if (!AllBytesInUInt64AreAscii(candidateUInt64)) |
| | | 642 | | { |
| | | 643 | | // Clear everything but the high bit of each byte, then tzcnt. |
| | | 644 | | // Remember to divide by 8 at the end to convert bit count to byte count. |
| | | 645 | | |
| | 0 | 646 | | candidateUInt64 &= UInt64HighBitsOnlyMask; |
| | 0 | 647 | | pBuffer += (nuint)(BitOperations.TrailingZeroCount(candidateUInt64) >> 3); |
| | 0 | 648 | | goto Finish; |
| | | 649 | | } |
| | | 650 | | } |
| | | 651 | | else |
| | | 652 | | { |
| | | 653 | | // If we can't use 64-bit tzcnt, no worries. We'll just do 2x 32-bit reads instead. |
| | | 654 | | |
| | | 655 | | currentDWord = Unsafe.ReadUnaligned<uint>(pBuffer); |
| | | 656 | | uint nextDWord = Unsafe.ReadUnaligned<uint>(pBuffer + 4); |
| | | 657 | | |
| | | 658 | | if (!AllBytesInUInt32AreAscii(currentDWord | nextDWord)) |
| | | 659 | | { |
| | | 660 | | // At least one of the values wasn't all-ASCII. |
| | | 661 | | // We need to figure out which one it was and stick it in the currentMask local. |
| | | 662 | | |
| | | 663 | | if (AllBytesInUInt32AreAscii(currentDWord)) |
| | | 664 | | { |
| | | 665 | | currentDWord = nextDWord; // this one is the culprit |
| | | 666 | | pBuffer += 4; |
| | | 667 | | } |
| | | 668 | | |
| | | 669 | | goto FoundNonAsciiDataInCurrentDWord; |
| | | 670 | | } |
| | | 671 | | } |
| | | 672 | | |
| | 0 | 673 | | pBuffer += 8; // successfully consumed 8 ASCII bytes |
| | | 674 | | } |
| | | 675 | | |
| | | 676 | | // DWORD drain |
| | | 677 | | |
| | 0 | 678 | | if ((bufferLength & 4) != 0) |
| | | 679 | | { |
| | 0 | 680 | | currentDWord = Unsafe.ReadUnaligned<uint>(pBuffer); |
| | | 681 | | |
| | 0 | 682 | | if (!AllBytesInUInt32AreAscii(currentDWord)) |
| | | 683 | | { |
| | | 684 | | goto FoundNonAsciiDataInCurrentDWord; |
| | | 685 | | } |
| | | 686 | | |
| | 0 | 687 | | pBuffer += 4; // successfully consumed 4 ASCII bytes |
| | | 688 | | } |
| | | 689 | | |
| | | 690 | | // WORD drain |
| | | 691 | | // (We movzx to a DWORD for ease of manipulation.) |
| | | 692 | | |
| | 0 | 693 | | if ((bufferLength & 2) != 0) |
| | | 694 | | { |
| | 0 | 695 | | currentDWord = Unsafe.ReadUnaligned<ushort>(pBuffer); |
| | | 696 | | |
| | 0 | 697 | | if (!AllBytesInUInt32AreAscii(currentDWord)) |
| | | 698 | | { |
| | | 699 | | // We only care about the 0x0080 bit of the value. If it's not set, then we |
| | | 700 | | // increment currentOffset by 1. If it's set, we don't increment it at all. |
| | | 701 | | |
| | 0 | 702 | | pBuffer += (nuint)((nint)(sbyte)currentDWord >> 7) + 1; |
| | 0 | 703 | | goto Finish; |
| | | 704 | | } |
| | | 705 | | |
| | 0 | 706 | | pBuffer += 2; // successfully consumed 2 ASCII bytes |
| | | 707 | | } |
| | | 708 | | |
| | | 709 | | // BYTE drain |
| | | 710 | | |
| | 0 | 711 | | if ((bufferLength & 1) != 0) |
| | | 712 | | { |
| | | 713 | | // sbyte has non-negative value if byte is ASCII. |
| | | 714 | | |
| | 0 | 715 | | if (*(sbyte*)(pBuffer) >= 0) |
| | | 716 | | { |
| | 0 | 717 | | pBuffer++; // successfully consumed a single byte |
| | | 718 | | } |
| | | 719 | | } |
| | | 720 | | |
| | 0 | 721 | | goto Finish; |
| | | 722 | | } |
| | | 723 | | #endif |
| | | 724 | | |
| | | 725 | | /// <summary> |
| | | 726 | | /// Returns the index in <paramref name="pBuffer"/> where the first non-ASCII char is found. |
| | | 727 | | /// Returns <paramref name="bufferLength"/> if the buffer is empty or all-ASCII. |
| | | 728 | | /// </summary> |
| | | 729 | | /// <returns>An ASCII char is defined as 0x0000 - 0x007F, inclusive.</returns> |
| | | 730 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 731 | | internal static unsafe nuint GetIndexOfFirstNonAsciiChar(char* pBuffer, nuint bufferLength /* in chars */) |
| | | 732 | | { |
| | | 733 | | // If 256/512-bit aren't supported but SSE2/ASIMD is supported, use those specific intrinsics instead of |
| | | 734 | | // the generic vectorized code. This has two benefits: (a) we can take advantage of specific instructions |
| | | 735 | | // like pmovmskb which we know are optimized, and (b) we can avoid downclocking the processor while |
| | | 736 | | // this method is running. |
| | | 737 | | |
| | | 738 | | #if NET |
| | 16 | 739 | | if (!Vector512.IsHardwareAccelerated && |
| | 16 | 740 | | !Vector256.IsHardwareAccelerated && |
| | 16 | 741 | | (Sse2.IsSupported || AdvSimd.IsSupported)) |
| | | 742 | | { |
| | 0 | 743 | | return GetIndexOfFirstNonAsciiChar_Intrinsified(pBuffer, bufferLength); |
| | | 744 | | } |
| | | 745 | | else |
| | | 746 | | #endif |
| | | 747 | | { |
| | | 748 | | // Handles Vector512, Vector256, Vector128, and scalar. |
| | 16 | 749 | | return GetIndexOfFirstNonAsciiChar_Vector(pBuffer, bufferLength); |
| | | 750 | | } |
| | | 751 | | } |
| | | 752 | | |
| | | 753 | | private static unsafe nuint GetIndexOfFirstNonAsciiChar_Vector(char* pBuffer, nuint bufferLength /* in chars */) |
| | | 754 | | { |
| | | 755 | | // Squirrel away the original buffer reference.This method works by determining the exact |
| | | 756 | | // char reference where non-ASCII data begins, so we need this base value to perform the |
| | | 757 | | // final subtraction at the end of the method to get the index into the original buffer. |
| | 16 | 758 | | char* pOriginalBuffer = pBuffer; |
| | | 759 | | |
| | | 760 | | #if SYSTEM_PRIVATE_CORELIB |
| | 16 | 761 | | Debug.Assert(bufferLength <= nuint.MaxValue / sizeof(char)); |
| | | 762 | | #endif |
| | | 763 | | |
| | | 764 | | #if NET |
| | | 765 | | // Before we drain off char-by-char, try a generic vectorized loop. |
| | | 766 | | // Only run the loop if we have at least two vectors we can pull out. |
| | 16 | 767 | | if (Vector512.IsHardwareAccelerated && bufferLength >= 2 * (uint)Vector512<ushort>.Count) |
| | | 768 | | { |
| | | 769 | | const uint SizeOfVector512InChars = Vector512.Size / sizeof(ushort); |
| | | 770 | | |
| | 0 | 771 | | if (!VectorContainsNonAsciiChar(Vector512.Load((ushort*)pBuffer))) |
| | | 772 | | { |
| | | 773 | | // The first several elements of the input buffer were ASCII. Bump up the pointer to the |
| | | 774 | | // next aligned boundary, then perform aligned reads from here on out until we find non-ASCII |
| | | 775 | | // data or we approach the end of the buffer. It's possible we'll reread data; this is ok. |
| | | 776 | | |
| | 0 | 777 | | char* pFinalVectorReadPos = pBuffer + bufferLength - SizeOfVector512InChars; |
| | 0 | 778 | | pBuffer = (char*)(((nuint)pBuffer + Vector512.Size) & ~(nuint)(Vector512.Size - 1)); |
| | | 779 | | |
| | | 780 | | #if DEBUG |
| | 0 | 781 | | long numCharsRead = pBuffer - pOriginalBuffer; |
| | 0 | 782 | | Debug.Assert(0 < numCharsRead && numCharsRead <= SizeOfVector512InChars, "We should've made forward |
| | 0 | 783 | | Debug.Assert((nuint)numCharsRead <= bufferLength, "We shouldn't have read past the end of the input |
| | | 784 | | #endif |
| | | 785 | | |
| | 0 | 786 | | Debug.Assert(pBuffer <= pFinalVectorReadPos, "Should be able to read at least one vector."); |
| | | 787 | | |
| | | 788 | | do |
| | | 789 | | { |
| | 0 | 790 | | Debug.Assert((nuint)pBuffer % Vector512.Size == 0, "Vector read should be aligned."); |
| | 0 | 791 | | if (VectorContainsNonAsciiChar(Vector512.LoadAligned((ushort*)pBuffer))) |
| | | 792 | | { |
| | | 793 | | break; // found non-ASCII data |
| | | 794 | | } |
| | 0 | 795 | | pBuffer += SizeOfVector512InChars; |
| | 0 | 796 | | } while (pBuffer <= pFinalVectorReadPos); |
| | | 797 | | |
| | | 798 | | // Adjust the remaining buffer length for the number of elements we just consumed. |
| | | 799 | | |
| | 0 | 800 | | bufferLength -= ((nuint)pBuffer - (nuint)pOriginalBuffer) / sizeof(char); |
| | | 801 | | } |
| | | 802 | | } |
| | 16 | 803 | | else if (Vector256.IsHardwareAccelerated && bufferLength >= 2 * (uint)Vector256<ushort>.Count) |
| | | 804 | | { |
| | | 805 | | const uint SizeOfVector256InChars = Vector256.Size / sizeof(ushort); |
| | | 806 | | |
| | 2 | 807 | | if (!VectorContainsNonAsciiChar(Vector256.Load((ushort*)pBuffer))) |
| | | 808 | | { |
| | | 809 | | // The first several elements of the input buffer were ASCII. Bump up the pointer to the |
| | | 810 | | // next aligned boundary, then perform aligned reads from here on out until we find non-ASCII |
| | | 811 | | // data or we approach the end of the buffer. It's possible we'll reread data; this is ok. |
| | | 812 | | |
| | 2 | 813 | | char* pFinalVectorReadPos = pBuffer + bufferLength - SizeOfVector256InChars; |
| | 2 | 814 | | pBuffer = (char*)(((nuint)pBuffer + Vector256.Size) & ~(nuint)(Vector256.Size - 1)); |
| | | 815 | | |
| | | 816 | | #if DEBUG |
| | 2 | 817 | | long numCharsRead = pBuffer - pOriginalBuffer; |
| | 2 | 818 | | Debug.Assert(0 < numCharsRead && numCharsRead <= SizeOfVector256InChars, "We should've made forward |
| | 2 | 819 | | Debug.Assert((nuint)numCharsRead <= bufferLength, "We shouldn't have read past the end of the input |
| | | 820 | | #endif |
| | | 821 | | |
| | 2 | 822 | | Debug.Assert(pBuffer <= pFinalVectorReadPos, "Should be able to read at least one vector."); |
| | | 823 | | |
| | | 824 | | do |
| | | 825 | | { |
| | 3 | 826 | | Debug.Assert((nuint)pBuffer % Vector256.Size == 0, "Vector read should be aligned."); |
| | 3 | 827 | | if (VectorContainsNonAsciiChar(Vector256.LoadAligned((ushort*)pBuffer))) |
| | | 828 | | { |
| | | 829 | | break; // found non-ASCII data |
| | | 830 | | } |
| | 3 | 831 | | pBuffer += SizeOfVector256InChars; |
| | 3 | 832 | | } while (pBuffer <= pFinalVectorReadPos); |
| | | 833 | | |
| | | 834 | | // Adjust the remaining buffer length for the number of elements we just consumed. |
| | | 835 | | |
| | 2 | 836 | | bufferLength -= ((nuint)pBuffer - (nuint)pOriginalBuffer) / sizeof(char); |
| | | 837 | | } |
| | | 838 | | } |
| | 14 | 839 | | else if (Vector128.IsHardwareAccelerated && bufferLength >= 2 * (uint)Vector128<ushort>.Count) |
| | | 840 | | { |
| | | 841 | | const uint SizeOfVector128InChars = Vector128.Size / sizeof(ushort); // JIT will make this a const |
| | | 842 | | |
| | 2 | 843 | | if (!VectorContainsNonAsciiChar(Vector128.Load((ushort*)pBuffer))) |
| | | 844 | | { |
| | | 845 | | // The first several elements of the input buffer were ASCII. Bump up the pointer to the |
| | | 846 | | // next aligned boundary, then perform aligned reads from here on out until we find non-ASCII |
| | | 847 | | // data or we approach the end of the buffer. It's possible we'll reread data; this is ok. |
| | 2 | 848 | | char* pFinalVectorReadPos = pBuffer + bufferLength - SizeOfVector128InChars; |
| | 2 | 849 | | pBuffer = (char*)(((nuint)pBuffer + Vector128.Size) & ~(nuint)(Vector128.Size - 1)); |
| | | 850 | | |
| | | 851 | | #if DEBUG |
| | 2 | 852 | | long numCharsRead = pBuffer - pOriginalBuffer; |
| | 2 | 853 | | Debug.Assert(0 < numCharsRead && numCharsRead <= SizeOfVector128InChars, "We should've made forward |
| | 2 | 854 | | Debug.Assert((nuint)numCharsRead <= bufferLength, "We shouldn't have read past the end of the input |
| | | 855 | | #endif |
| | | 856 | | |
| | 2 | 857 | | Debug.Assert(pBuffer <= pFinalVectorReadPos, "Should be able to read at least one vector."); |
| | | 858 | | |
| | | 859 | | do |
| | | 860 | | { |
| | 5 | 861 | | Debug.Assert((nuint)pBuffer % Vector128.Size == 0, "Vector read should be aligned."); |
| | 5 | 862 | | if (VectorContainsNonAsciiChar(Vector128.LoadAligned((ushort*)pBuffer))) |
| | | 863 | | { |
| | | 864 | | break; // found non-ASCII data |
| | | 865 | | } |
| | 5 | 866 | | pBuffer += SizeOfVector128InChars; |
| | 5 | 867 | | } while (pBuffer <= pFinalVectorReadPos); |
| | | 868 | | |
| | | 869 | | // Adjust the remaining buffer length for the number of elements we just consumed. |
| | | 870 | | |
| | 2 | 871 | | bufferLength -= ((nuint)pBuffer - (nuint)pOriginalBuffer) / sizeof(char); |
| | | 872 | | } |
| | | 873 | | } |
| | | 874 | | #endif |
| | | 875 | | |
| | | 876 | | // At this point, the buffer length wasn't enough to perform a vectorized search, or we did perform |
| | | 877 | | // a vectorized search and encountered non-ASCII data. In either case go down a non-vectorized code |
| | | 878 | | // path to drain any remaining ASCII chars. |
| | | 879 | | // |
| | | 880 | | // We're going to perform unaligned reads, so prefer 32-bit reads instead of 64-bit reads. |
| | | 881 | | // This also allows us to perform more optimized bit twiddling tricks to count the number of ASCII chars. |
| | | 882 | | |
| | | 883 | | uint currentUInt32; |
| | | 884 | | |
| | | 885 | | // Try reading 64 bits at a time in a loop. |
| | | 886 | | |
| | 56 | 887 | | for (; bufferLength >= 4; bufferLength -= 4) // 64 bits = 4 * 16-bit chars |
| | | 888 | | { |
| | 20 | 889 | | currentUInt32 = Unsafe.ReadUnaligned<uint>(pBuffer); |
| | 20 | 890 | | uint nextUInt32 = Unsafe.ReadUnaligned<uint>(pBuffer + 4 / sizeof(char)); |
| | | 891 | | |
| | 20 | 892 | | if (!AllCharsInUInt32AreAscii(currentUInt32 | nextUInt32)) |
| | | 893 | | { |
| | | 894 | | // One of these two values contains non-ASCII chars. |
| | | 895 | | // Figure out which one it is, then put it in 'current' so that we can drain the ASCII chars. |
| | | 896 | | |
| | 0 | 897 | | if (AllCharsInUInt32AreAscii(currentUInt32)) |
| | | 898 | | { |
| | 0 | 899 | | currentUInt32 = nextUInt32; |
| | 0 | 900 | | pBuffer += 2; |
| | | 901 | | } |
| | | 902 | | |
| | 0 | 903 | | goto FoundNonAsciiData; |
| | | 904 | | } |
| | | 905 | | |
| | 20 | 906 | | pBuffer += 4; // consumed 4 ASCII chars |
| | | 907 | | } |
| | | 908 | | |
| | | 909 | | // From this point forward we don't need to keep track of the remaining buffer length. |
| | | 910 | | // Try reading 32 bits. |
| | | 911 | | |
| | 16 | 912 | | if ((bufferLength & 2) != 0) // 32 bits = 2 * 16-bit chars |
| | | 913 | | { |
| | 6 | 914 | | currentUInt32 = Unsafe.ReadUnaligned<uint>(pBuffer); |
| | 6 | 915 | | if (!AllCharsInUInt32AreAscii(currentUInt32)) |
| | | 916 | | { |
| | | 917 | | goto FoundNonAsciiData; |
| | | 918 | | } |
| | | 919 | | |
| | 6 | 920 | | pBuffer += 2; |
| | | 921 | | } |
| | | 922 | | |
| | | 923 | | // Try reading 16 bits. |
| | | 924 | | // No need to try an 8-bit read after this since we're working with chars. |
| | | 925 | | |
| | 16 | 926 | | if ((bufferLength & 1) != 0) |
| | | 927 | | { |
| | | 928 | | // If the buffer contains non-ASCII data, the comparison below will fail, and |
| | | 929 | | // we'll end up not incrementing the buffer reference. |
| | | 930 | | |
| | 7 | 931 | | if (*pBuffer <= 0x007F) |
| | | 932 | | { |
| | 7 | 933 | | pBuffer++; |
| | | 934 | | } |
| | | 935 | | } |
| | | 936 | | |
| | | 937 | | Finish: |
| | | 938 | | |
| | 16 | 939 | | nuint totalNumBytesRead = (nuint)pBuffer - (nuint)pOriginalBuffer; |
| | 16 | 940 | | Debug.Assert(totalNumBytesRead % sizeof(char) == 0, "Total number of bytes read should be even since we're w |
| | 16 | 941 | | return totalNumBytesRead / sizeof(char); // convert byte count -> char count before returning |
| | | 942 | | |
| | | 943 | | FoundNonAsciiData: |
| | | 944 | | |
| | 0 | 945 | | Debug.Assert(!AllCharsInUInt32AreAscii(currentUInt32), "Shouldn't have reached this point if we have an all- |
| | | 946 | | |
| | | 947 | | // We don't bother looking at the second char - only the first char. |
| | | 948 | | |
| | 0 | 949 | | if (FirstCharInUInt32IsAscii(currentUInt32)) |
| | | 950 | | { |
| | 0 | 951 | | pBuffer++; |
| | | 952 | | } |
| | | 953 | | |
| | 0 | 954 | | goto Finish; |
| | | 955 | | } |
| | | 956 | | |
| | | 957 | | #if NET |
| | | 958 | | private static unsafe nuint GetIndexOfFirstNonAsciiChar_Intrinsified(char* pBuffer, nuint bufferLength /* in cha |
| | | 959 | | { |
| | | 960 | | // This method contains logic optimized using vector instructions for both x64 and Arm64. |
| | | 961 | | // Much of the logic in this method will be elided by JIT once we determine which specific ISAs we support. |
| | | 962 | | |
| | | 963 | | // Quick check for empty inputs. |
| | | 964 | | |
| | | 965 | | if (bufferLength == 0) |
| | | 966 | | { |
| | 0 | 967 | | return 0; |
| | | 968 | | } |
| | | 969 | | |
| | | 970 | | // JIT turns the below into constants |
| | | 971 | | |
| | 0 | 972 | | uint SizeOfVector128InChars = Vector128.Size / sizeof(char); |
| | | 973 | | |
| | 0 | 974 | | Debug.Assert(Sse2.IsSupported || AdvSimd.Arm64.IsSupported, "Should've been checked by caller."); |
| | 0 | 975 | | Debug.Assert(BitConverter.IsLittleEndian, "This SSE2/Arm64 assumes little-endian."); |
| | | 976 | | |
| | | 977 | | Vector128<ushort> firstVector, secondVector; |
| | | 978 | | uint currentMask; |
| | 0 | 979 | | char* pOriginalBuffer = pBuffer; |
| | | 980 | | |
| | 0 | 981 | | if (bufferLength < SizeOfVector128InChars) |
| | | 982 | | { |
| | | 983 | | goto InputBufferLessThanOneVectorInLength; // can't vectorize; drain primitives instead |
| | | 984 | | } |
| | | 985 | | |
| | | 986 | | // This method is written such that control generally flows top-to-bottom, avoiding |
| | | 987 | | // jumps as much as possible in the optimistic case of "all ASCII". If we see non-ASCII |
| | | 988 | | // data, we jump out of the hot paths to targets at the end of the method. |
| | | 989 | | |
| | | 990 | | #if SYSTEM_PRIVATE_CORELIB |
| | 0 | 991 | | Debug.Assert(bufferLength <= nuint.MaxValue / sizeof(char)); |
| | | 992 | | #endif |
| | | 993 | | |
| | | 994 | | // Read the first vector unaligned. |
| | | 995 | | |
| | 0 | 996 | | firstVector = Vector128.LoadUnsafe(ref *(ushort*)pBuffer); |
| | 0 | 997 | | if (VectorContainsNonAsciiChar(firstVector)) |
| | | 998 | | { |
| | | 999 | | goto FoundNonAsciiDataInFirstVector; |
| | | 1000 | | } |
| | | 1001 | | |
| | | 1002 | | // If we have less than 32 bytes to process, just go straight to the final unaligned |
| | | 1003 | | // read. There's no need to mess with the loop logic in the middle of this method. |
| | | 1004 | | |
| | | 1005 | | // Adjust the remaining length to account for what we just read. |
| | | 1006 | | // For the remainder of this code path, bufferLength will be in bytes, not chars. |
| | | 1007 | | |
| | 0 | 1008 | | bufferLength <<= 1; // chars to bytes |
| | | 1009 | | |
| | 0 | 1010 | | if (bufferLength < 2 * Vector128.Size) |
| | | 1011 | | { |
| | | 1012 | | goto IncrementCurrentOffsetBeforeFinalUnalignedVectorRead; |
| | | 1013 | | } |
| | | 1014 | | |
| | | 1015 | | // Now adjust the read pointer so that future reads are aligned. |
| | | 1016 | | |
| | 0 | 1017 | | pBuffer = (char*)(((nuint)pBuffer + Vector128.Size) & ~(nuint)(Vector128.Size - 1)); |
| | | 1018 | | |
| | | 1019 | | #if DEBUG |
| | 0 | 1020 | | long numCharsRead = pBuffer - pOriginalBuffer; |
| | 0 | 1021 | | Debug.Assert(0 < numCharsRead && numCharsRead <= SizeOfVector128InChars, "We should've made forward progress |
| | 0 | 1022 | | Debug.Assert((nuint)numCharsRead <= bufferLength, "We shouldn't have read past the end of the input buffer." |
| | | 1023 | | #endif |
| | | 1024 | | |
| | | 1025 | | // Adjust remaining buffer length. |
| | | 1026 | | |
| | 0 | 1027 | | nuint numBytesRead = ((nuint)pBuffer - (nuint)pOriginalBuffer); |
| | 0 | 1028 | | bufferLength -= numBytesRead; |
| | | 1029 | | |
| | | 1030 | | // The buffer is now properly aligned. |
| | | 1031 | | // Read 2 vectors at a time if possible. |
| | 0 | 1032 | | if (bufferLength >= 2 * Vector128.Size) |
| | | 1033 | | { |
| | 0 | 1034 | | char* pFinalVectorReadPos = (char*)((nuint)pBuffer + bufferLength - 2 * Vector128.Size); |
| | | 1035 | | |
| | | 1036 | | // After this point, we no longer need to update the bufferLength value. |
| | | 1037 | | do |
| | | 1038 | | { |
| | | 1039 | | |
| | 0 | 1040 | | firstVector = Vector128.LoadUnsafe(ref *(ushort*)pBuffer); |
| | 0 | 1041 | | secondVector = Vector128.LoadUnsafe(ref *(ushort*)pBuffer, SizeOfVector128InChars); |
| | 0 | 1042 | | Vector128<ushort> combinedVector = firstVector | secondVector; |
| | | 1043 | | |
| | 0 | 1044 | | if (VectorContainsNonAsciiChar(combinedVector)) |
| | | 1045 | | { |
| | | 1046 | | goto FoundNonAsciiDataInFirstOrSecondVector; |
| | | 1047 | | } |
| | | 1048 | | |
| | 0 | 1049 | | pBuffer += 2 * SizeOfVector128InChars; |
| | 0 | 1050 | | } while (pBuffer <= pFinalVectorReadPos); |
| | | 1051 | | } |
| | | 1052 | | |
| | | 1053 | | // We have somewhere between 0 and (2 * vector length) - 1 bytes remaining to read from. |
| | | 1054 | | // Since the above loop doesn't update bufferLength, we can't rely on its absolute value. |
| | | 1055 | | // But we _can_ rely on it to tell us how much remaining data must be drained by looking |
| | | 1056 | | // at what bits of it are set. This works because had we updated it within the loop above, |
| | | 1057 | | // we would've been adding 2 * SizeOfVector128 on each iteration, but we only care about |
| | | 1058 | | // bits which are less significant than those that the addition would've acted on. |
| | | 1059 | | |
| | | 1060 | | // If there is fewer than one vector length remaining, skip the next aligned read. |
| | | 1061 | | // Remember, at this point bufferLength is measured in bytes, not chars. |
| | | 1062 | | |
| | 0 | 1063 | | if ((bufferLength & Vector128.Size) == 0) |
| | | 1064 | | { |
| | | 1065 | | goto DoFinalUnalignedVectorRead; |
| | | 1066 | | } |
| | | 1067 | | |
| | | 1068 | | // At least one full vector's worth of data remains, so we can safely read it. |
| | | 1069 | | // Remember, at this point pBuffer is still aligned. |
| | | 1070 | | |
| | 0 | 1071 | | firstVector = Vector128.LoadUnsafe(ref *(ushort*)pBuffer); |
| | 0 | 1072 | | if (VectorContainsNonAsciiChar(firstVector)) |
| | | 1073 | | { |
| | | 1074 | | goto FoundNonAsciiDataInFirstVector; |
| | | 1075 | | } |
| | | 1076 | | |
| | | 1077 | | IncrementCurrentOffsetBeforeFinalUnalignedVectorRead: |
| | | 1078 | | |
| | 0 | 1079 | | pBuffer += SizeOfVector128InChars; |
| | | 1080 | | |
| | | 1081 | | DoFinalUnalignedVectorRead: |
| | | 1082 | | |
| | 0 | 1083 | | if (((byte)bufferLength & (Vector128.Size - 1)) != 0) |
| | | 1084 | | { |
| | | 1085 | | // Perform an unaligned read of the last vector. |
| | | 1086 | | // We need to adjust the pointer because we're re-reading data. |
| | | 1087 | | |
| | 0 | 1088 | | pBuffer = (char*)((byte*)pBuffer + (bufferLength & (Vector128.Size - 1)) - Vector128.Size); |
| | 0 | 1089 | | firstVector = Vector128.LoadUnsafe(ref *(ushort*)pBuffer); |
| | 0 | 1090 | | if (VectorContainsNonAsciiChar(firstVector)) |
| | | 1091 | | { |
| | | 1092 | | goto FoundNonAsciiDataInFirstVector; |
| | | 1093 | | } |
| | | 1094 | | |
| | 0 | 1095 | | pBuffer += SizeOfVector128InChars; |
| | | 1096 | | } |
| | | 1097 | | |
| | | 1098 | | Finish: |
| | | 1099 | | |
| | 0 | 1100 | | Debug.Assert(((nuint)pBuffer - (nuint)pOriginalBuffer) % 2 == 0, "Shouldn't have incremented any pointer by |
| | 0 | 1101 | | return ((nuint)pBuffer - (nuint)pOriginalBuffer) / sizeof(char); // and we're done! (remember to adjust for |
| | | 1102 | | |
| | | 1103 | | FoundNonAsciiDataInFirstOrSecondVector: |
| | | 1104 | | |
| | | 1105 | | // We don't know if the first or the second vector contains non-ASCII data. Check the first |
| | | 1106 | | // vector, and if that's all-ASCII then the second vector must be the culprit. Either way |
| | | 1107 | | // we'll make sure the first vector local is the one that contains the non-ASCII data. |
| | | 1108 | | |
| | 0 | 1109 | | if (VectorContainsNonAsciiChar(firstVector)) |
| | | 1110 | | { |
| | | 1111 | | goto FoundNonAsciiDataInFirstVector; |
| | | 1112 | | } |
| | | 1113 | | |
| | | 1114 | | // Wasn't the first vector; must be the second. |
| | | 1115 | | |
| | 0 | 1116 | | pBuffer += SizeOfVector128InChars; |
| | 0 | 1117 | | firstVector = secondVector; |
| | | 1118 | | |
| | | 1119 | | FoundNonAsciiDataInFirstVector: |
| | | 1120 | | |
| | 0 | 1121 | | if (Sse2.IsSupported) |
| | | 1122 | | { |
| | | 1123 | | // The operation below forces the 0x8000 bit of each WORD to be set iff the WORD element |
| | | 1124 | | // has value >= 0x0800 (non-ASCII). Then we'll treat the vector as a BYTE vector in order |
| | | 1125 | | // to extract the mask. Reminder: the 0x0080 bit of each WORD should be ignored. |
| | 0 | 1126 | | Vector128<ushort> asciiMaskForAddSaturate = Vector128.Create((ushort)0x7F80); |
| | | 1127 | | const uint NonAsciiDataSeenMask = 0b_1010_1010_1010_1010; // used for determining whether 'currentMask' |
| | | 1128 | | |
| | 0 | 1129 | | currentMask = (uint)Sse2.MoveMask(Sse2.AddSaturate(firstVector, asciiMaskForAddSaturate).AsByte()); |
| | 0 | 1130 | | currentMask &= NonAsciiDataSeenMask; |
| | | 1131 | | |
| | | 1132 | | // Now, the mask contains - from the LSB - a 0b00 pair for each ASCII char we saw, and a 0b10 pair for e |
| | | 1133 | | // |
| | | 1134 | | // (Keep endianness in mind in the below examples.) |
| | | 1135 | | // A non-ASCII char followed by two ASCII chars is 0b..._00_00_10. (tzcnt = 1) |
| | | 1136 | | // An ASCII char followed by two non-ASCII chars is 0b..._10_10_00. (tzcnt = 3) |
| | | 1137 | | // Two ASCII chars followed by a non-ASCII char is 0b..._10_00_00. (tzcnt = 5) |
| | | 1138 | | // |
| | | 1139 | | // This means tzcnt = 2 * numLeadingAsciiChars + 1. We can conveniently take advantage of the fact |
| | | 1140 | | // that the 2x multiplier already matches the char* stride length, then just subtract 1 at the end to |
| | | 1141 | | // compute the correct final ending pointer value. |
| | | 1142 | | |
| | 0 | 1143 | | Debug.Assert(currentMask != 0, "Shouldn't be here unless we see non-ASCII data."); |
| | 0 | 1144 | | pBuffer = (char*)((byte*)pBuffer + (uint)BitOperations.TrailingZeroCount(currentMask) - 1); |
| | | 1145 | | } |
| | | 1146 | | else if (AdvSimd.Arm64.IsSupported) |
| | | 1147 | | { |
| | | 1148 | | // The following operation sets all the bits in a WORD to 1 where a non-ASCII char is found (otherwise t |
| | | 1149 | | // in the vector. Then narrow each char to a byte by taking its top byte. Now the bottom-half (64-bits) |
| | | 1150 | | // of the vector contains 0xFFFF for non-ASCII and 0x0000 for ASCII char. We then find the index of the |
| | | 1151 | | // first non-ASCII char by counting number of trailing zeros representing ASCII chars before it. |
| | | 1152 | | |
| | | 1153 | | Vector128<ushort> largestAsciiValue = Vector128.Create((ushort)0x007F); |
| | | 1154 | | Vector128<byte> compareResult = AdvSimd.CompareGreaterThan(firstVector, largestAsciiValue).AsByte(); |
| | | 1155 | | ulong asciiCompareMask = AdvSimd.Arm64.UnzipOdd(compareResult, compareResult).AsUInt64().ToScalar(); |
| | | 1156 | | // Compare mask now contains 8 bits for each 16-bit char. Divide it by 8 to get to the first non-ASCII b |
| | | 1157 | | pBuffer += BitOperations.TrailingZeroCount(asciiCompareMask) >> 3; |
| | | 1158 | | } |
| | | 1159 | | else |
| | | 1160 | | { |
| | 0 | 1161 | | throw new PlatformNotSupportedException(); |
| | | 1162 | | } |
| | | 1163 | | goto Finish; |
| | | 1164 | | |
| | | 1165 | | FoundNonAsciiDataInCurrentDWord: |
| | | 1166 | | |
| | | 1167 | | uint currentDWord; |
| | 0 | 1168 | | Debug.Assert(!AllCharsInUInt32AreAscii(currentDWord), "Shouldn't be here unless we see non-ASCII data."); |
| | | 1169 | | |
| | 0 | 1170 | | if (FirstCharInUInt32IsAscii(currentDWord)) |
| | | 1171 | | { |
| | 0 | 1172 | | pBuffer++; // skip past the ASCII char |
| | | 1173 | | } |
| | | 1174 | | |
| | 0 | 1175 | | goto Finish; |
| | | 1176 | | |
| | | 1177 | | InputBufferLessThanOneVectorInLength: |
| | | 1178 | | |
| | | 1179 | | // These code paths get hit if the original input length was less than one vector in size. |
| | | 1180 | | // We can't perform vectorized reads at this point, so we'll fall back to reading primitives |
| | | 1181 | | // directly. Note that all of these reads are unaligned. |
| | | 1182 | | |
| | | 1183 | | // Reminder: If this code path is hit, bufferLength is still a char count, not a byte count. |
| | | 1184 | | // We skipped the code path that multiplied the count by sizeof(char). |
| | | 1185 | | |
| | 0 | 1186 | | Debug.Assert(bufferLength < SizeOfVector128InChars); |
| | | 1187 | | |
| | | 1188 | | // QWORD drain |
| | | 1189 | | |
| | 0 | 1190 | | if ((bufferLength & 4) != 0) |
| | | 1191 | | { |
| | | 1192 | | if (UIntPtr.Size == sizeof(ulong)) |
| | | 1193 | | { |
| | | 1194 | | // If we can use 64-bit tzcnt to count the number of leading ASCII chars, prefer it. |
| | | 1195 | | |
| | 0 | 1196 | | ulong candidateUInt64 = Unsafe.ReadUnaligned<ulong>(pBuffer); |
| | 0 | 1197 | | if (!AllCharsInUInt64AreAscii(candidateUInt64)) |
| | | 1198 | | { |
| | | 1199 | | // Clear the low 7 bits (the ASCII bits) of each char, then tzcnt. |
| | | 1200 | | // Remember to divide by 8 at the end to convert bit count to byte count, |
| | | 1201 | | // then the & ~1 at the end to treat a match in the high byte of |
| | | 1202 | | // any char the same as a match in the low byte of that same char. |
| | | 1203 | | |
| | 0 | 1204 | | candidateUInt64 &= 0xFF80FF80_FF80FF80ul; |
| | 0 | 1205 | | pBuffer = (char*)((byte*)pBuffer + ((nuint)(BitOperations.TrailingZeroCount(candidateUInt64) >> |
| | 0 | 1206 | | goto Finish; |
| | | 1207 | | } |
| | | 1208 | | } |
| | | 1209 | | else |
| | | 1210 | | { |
| | | 1211 | | // If we can't use 64-bit tzcnt, no worries. We'll just do 2x 32-bit reads instead. |
| | | 1212 | | |
| | | 1213 | | currentDWord = Unsafe.ReadUnaligned<uint>(pBuffer); |
| | | 1214 | | uint nextDWord = Unsafe.ReadUnaligned<uint>(pBuffer + 4 / sizeof(char)); |
| | | 1215 | | |
| | | 1216 | | if (!AllCharsInUInt32AreAscii(currentDWord | nextDWord)) |
| | | 1217 | | { |
| | | 1218 | | // At least one of the values wasn't all-ASCII. |
| | | 1219 | | // We need to figure out which one it was and stick it in the currentMask local. |
| | | 1220 | | |
| | | 1221 | | if (AllCharsInUInt32AreAscii(currentDWord)) |
| | | 1222 | | { |
| | | 1223 | | currentDWord = nextDWord; // this one is the culprit |
| | | 1224 | | pBuffer += 4 / sizeof(char); |
| | | 1225 | | } |
| | | 1226 | | |
| | | 1227 | | goto FoundNonAsciiDataInCurrentDWord; |
| | | 1228 | | } |
| | | 1229 | | } |
| | | 1230 | | |
| | 0 | 1231 | | pBuffer += 4; // successfully consumed 4 ASCII chars |
| | | 1232 | | } |
| | | 1233 | | |
| | | 1234 | | // DWORD drain |
| | | 1235 | | |
| | 0 | 1236 | | if ((bufferLength & 2) != 0) |
| | | 1237 | | { |
| | 0 | 1238 | | currentDWord = Unsafe.ReadUnaligned<uint>(pBuffer); |
| | | 1239 | | |
| | 0 | 1240 | | if (!AllCharsInUInt32AreAscii(currentDWord)) |
| | | 1241 | | { |
| | | 1242 | | goto FoundNonAsciiDataInCurrentDWord; |
| | | 1243 | | } |
| | | 1244 | | |
| | 0 | 1245 | | pBuffer += 2; // successfully consumed 2 ASCII chars |
| | | 1246 | | } |
| | | 1247 | | |
| | | 1248 | | // WORD drain |
| | | 1249 | | // This is the final drain; there's no need for a BYTE drain since our elemental type is 16-bit char. |
| | | 1250 | | |
| | 0 | 1251 | | if ((bufferLength & 1) != 0) |
| | | 1252 | | { |
| | 0 | 1253 | | if (*pBuffer <= 0x007F) |
| | | 1254 | | { |
| | 0 | 1255 | | pBuffer++; // successfully consumed a single char |
| | | 1256 | | } |
| | | 1257 | | } |
| | | 1258 | | |
| | 0 | 1259 | | goto Finish; |
| | | 1260 | | } |
| | | 1261 | | #endif |
| | | 1262 | | |
| | | 1263 | | /// <summary> |
| | | 1264 | | /// Given a QWORD which represents a buffer of 4 ASCII chars in machine-endian order, |
| | | 1265 | | /// narrows each WORD to a BYTE, then writes the 4-byte result to the output buffer |
| | | 1266 | | /// also in machine-endian order. |
| | | 1267 | | /// </summary> |
| | | 1268 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1269 | | private static void NarrowFourUtf16CharsToAsciiAndWriteToBuffer(ref byte outputBuffer, ulong value) |
| | | 1270 | | { |
| | | 1271 | | Debug.Assert(AllCharsInUInt64AreAscii(value)); |
| | | 1272 | | |
| | | 1273 | | #if NET |
| | 14382 | 1274 | | if (Sse2.X64.IsSupported) |
| | | 1275 | | { |
| | | 1276 | | // Narrows a vector of words [ w0 w1 w2 w3 ] to a vector of bytes |
| | | 1277 | | // [ b0 b1 b2 b3 b0 b1 b2 b3 ], then writes 4 bytes (32 bits) to the destination. |
| | | 1278 | | |
| | 14382 | 1279 | | Vector128<short> vecWide = Sse2.X64.ConvertScalarToVector128UInt64(value).AsInt16(); |
| | 14382 | 1280 | | Vector128<uint> vecNarrow = Sse2.PackUnsignedSaturate(vecWide, vecWide).AsUInt32(); |
| | 14382 | 1281 | | Unsafe.WriteUnaligned(ref outputBuffer, Sse2.ConvertToUInt32(vecNarrow)); |
| | | 1282 | | } |
| | | 1283 | | else if (AdvSimd.IsSupported) |
| | | 1284 | | { |
| | | 1285 | | // Narrows a vector of words [ w0 w1 w2 w3 ] to a vector of bytes |
| | | 1286 | | // [ b0 b1 b2 b3 * * * * ], then writes 4 bytes (32 bits) to the destination. |
| | | 1287 | | |
| | | 1288 | | Vector128<short> vecWide = Vector128.CreateScalarUnsafe(value).AsInt16(); |
| | | 1289 | | Vector64<byte> lower = AdvSimd.ExtractNarrowingSaturateUnsignedLower(vecWide); |
| | | 1290 | | Unsafe.WriteUnaligned(ref outputBuffer, lower.AsUInt32().ToScalar()); |
| | | 1291 | | } |
| | | 1292 | | else |
| | | 1293 | | #endif |
| | | 1294 | | { |
| | 0 | 1295 | | if (BitConverter.IsLittleEndian) |
| | | 1296 | | { |
| | 0 | 1297 | | outputBuffer = (byte)value; |
| | 0 | 1298 | | value >>= 16; |
| | 0 | 1299 | | Unsafe.Add(ref outputBuffer, 1) = (byte)value; |
| | 0 | 1300 | | value >>= 16; |
| | 0 | 1301 | | Unsafe.Add(ref outputBuffer, 2) = (byte)value; |
| | 0 | 1302 | | value >>= 16; |
| | 0 | 1303 | | Unsafe.Add(ref outputBuffer, 3) = (byte)value; |
| | | 1304 | | } |
| | | 1305 | | else |
| | | 1306 | | { |
| | | 1307 | | Unsafe.Add(ref outputBuffer, 3) = (byte)value; |
| | | 1308 | | value >>= 16; |
| | | 1309 | | Unsafe.Add(ref outputBuffer, 2) = (byte)value; |
| | | 1310 | | value >>= 16; |
| | | 1311 | | Unsafe.Add(ref outputBuffer, 1) = (byte)value; |
| | | 1312 | | value >>= 16; |
| | | 1313 | | outputBuffer = (byte)value; |
| | | 1314 | | } |
| | | 1315 | | } |
| | | 1316 | | } |
| | | 1317 | | |
| | | 1318 | | /// <summary> |
| | | 1319 | | /// Given a DWORD which represents a buffer of 2 ASCII chars in machine-endian order, |
| | | 1320 | | /// narrows each WORD to a BYTE, then writes the 2-byte result to the output buffer also in |
| | | 1321 | | /// machine-endian order. |
| | | 1322 | | /// </summary> |
| | | 1323 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1324 | | private static void NarrowTwoUtf16CharsToAsciiAndWriteToBuffer(ref byte outputBuffer, uint value) |
| | | 1325 | | { |
| | | 1326 | | Debug.Assert(AllCharsInUInt32AreAscii(value)); |
| | | 1327 | | |
| | 2417 | 1328 | | if (BitConverter.IsLittleEndian) |
| | | 1329 | | { |
| | 2417 | 1330 | | outputBuffer = (byte)value; |
| | 2417 | 1331 | | Unsafe.Add(ref outputBuffer, 1) = (byte)(value >> 16); |
| | | 1332 | | } |
| | | 1333 | | else |
| | | 1334 | | { |
| | | 1335 | | Unsafe.Add(ref outputBuffer, 1) = (byte)value; |
| | | 1336 | | outputBuffer = (byte)(value >> 16); |
| | | 1337 | | } |
| | | 1338 | | } |
| | | 1339 | | |
| | | 1340 | | /// <summary> |
| | | 1341 | | /// Copies as many ASCII characters (U+0000..U+007F) as possible from <paramref name="pUtf16Buffer"/> |
| | | 1342 | | /// to <paramref name="pAsciiBuffer"/>, stopping when the first non-ASCII character is encountered |
| | | 1343 | | /// or once <paramref name="elementCount"/> elements have been converted. Returns the total number |
| | | 1344 | | /// of elements that were able to be converted. |
| | | 1345 | | /// </summary> |
| | | 1346 | | internal static unsafe nuint NarrowUtf16ToAscii(char* pUtf16Buffer, byte* pAsciiBuffer, nuint elementCount) |
| | | 1347 | | { |
| | | 1348 | | nuint currentOffset = 0; |
| | | 1349 | | |
| | 15016 | 1350 | | uint utf16Data32BitsHigh = 0, utf16Data32BitsLow = 0; |
| | 7508 | 1351 | | ulong utf16Data64Bits = 0; |
| | | 1352 | | |
| | | 1353 | | #if NET |
| | 7508 | 1354 | | if (BitConverter.IsLittleEndian && Vector128.IsHardwareAccelerated && elementCount >= 2 * (uint)Vector128<by |
| | | 1355 | | { |
| | | 1356 | | // Since there's overhead to setting up the vectorized code path, we only want to |
| | | 1357 | | // call into it after a quick probe to ensure the next immediate characters really are ASCII. |
| | | 1358 | | // If we see non-ASCII data, we'll jump immediately to the draining logic at the end of the method. |
| | | 1359 | | |
| | | 1360 | | if (IntPtr.Size >= 8) |
| | | 1361 | | { |
| | 3862 | 1362 | | utf16Data64Bits = Unsafe.ReadUnaligned<ulong>(pUtf16Buffer); |
| | 3862 | 1363 | | if (!AllCharsInUInt64AreAscii(utf16Data64Bits)) |
| | | 1364 | | { |
| | 1435 | 1365 | | goto FoundNonAsciiDataIn64BitRead; |
| | | 1366 | | } |
| | | 1367 | | } |
| | | 1368 | | else |
| | | 1369 | | { |
| | | 1370 | | utf16Data32BitsHigh = Unsafe.ReadUnaligned<uint>(pUtf16Buffer); |
| | | 1371 | | utf16Data32BitsLow = Unsafe.ReadUnaligned<uint>(pUtf16Buffer + 4 / sizeof(char)); |
| | | 1372 | | if (!AllCharsInUInt32AreAscii(utf16Data32BitsHigh | utf16Data32BitsLow)) |
| | | 1373 | | { |
| | | 1374 | | goto FoundNonAsciiDataIn64BitRead; |
| | | 1375 | | } |
| | | 1376 | | } |
| | 2427 | 1377 | | if (Vector512.IsHardwareAccelerated && elementCount >= 2 * (uint)Vector512<byte>.Count) |
| | | 1378 | | { |
| | 836 | 1379 | | currentOffset = NarrowUtf16ToAscii_Intrinsified_512(pUtf16Buffer, pAsciiBuffer, elementCount); |
| | | 1380 | | } |
| | 1591 | 1381 | | else if (Vector256.IsHardwareAccelerated && elementCount >= 2 * (uint)Vector256<byte>.Count) |
| | | 1382 | | { |
| | 566 | 1383 | | currentOffset = NarrowUtf16ToAscii_Intrinsified_256(pUtf16Buffer, pAsciiBuffer, elementCount); |
| | | 1384 | | } |
| | | 1385 | | else |
| | | 1386 | | { |
| | 1025 | 1387 | | currentOffset = NarrowUtf16ToAscii_Intrinsified(pUtf16Buffer, pAsciiBuffer, elementCount); |
| | | 1388 | | } |
| | | 1389 | | } |
| | | 1390 | | #endif |
| | | 1391 | | |
| | 6073 | 1392 | | Debug.Assert(currentOffset <= elementCount); |
| | 6073 | 1393 | | nuint remainingElementCount = elementCount - currentOffset; |
| | | 1394 | | |
| | | 1395 | | // Try to narrow 64 bits -> 32 bits at a time. |
| | | 1396 | | // We needn't update remainingElementCount after this point. |
| | | 1397 | | |
| | 6073 | 1398 | | if (remainingElementCount >= 4) |
| | | 1399 | | { |
| | 5335 | 1400 | | nuint finalOffsetWhereCanLoop = currentOffset + remainingElementCount - 4; |
| | | 1401 | | do |
| | | 1402 | | { |
| | | 1403 | | if (IntPtr.Size >= 8) |
| | | 1404 | | { |
| | | 1405 | | // Only perform QWORD reads on a 64-bit platform. |
| | 16103 | 1406 | | utf16Data64Bits = Unsafe.ReadUnaligned<ulong>(pUtf16Buffer + currentOffset); |
| | 16103 | 1407 | | if (!AllCharsInUInt64AreAscii(utf16Data64Bits)) |
| | | 1408 | | { |
| | | 1409 | | goto FoundNonAsciiDataIn64BitRead; |
| | | 1410 | | } |
| | | 1411 | | |
| | 14382 | 1412 | | NarrowFourUtf16CharsToAsciiAndWriteToBuffer(ref pAsciiBuffer[currentOffset], utf16Data64Bits); |
| | | 1413 | | } |
| | | 1414 | | else |
| | | 1415 | | { |
| | | 1416 | | utf16Data32BitsHigh = Unsafe.ReadUnaligned<uint>(pUtf16Buffer + currentOffset); |
| | | 1417 | | utf16Data32BitsLow = Unsafe.ReadUnaligned<uint>(pUtf16Buffer + currentOffset + 4 / sizeof(char)) |
| | | 1418 | | if (!AllCharsInUInt32AreAscii(utf16Data32BitsHigh | utf16Data32BitsLow)) |
| | | 1419 | | { |
| | | 1420 | | goto FoundNonAsciiDataIn64BitRead; |
| | | 1421 | | } |
| | | 1422 | | |
| | | 1423 | | NarrowTwoUtf16CharsToAsciiAndWriteToBuffer(ref pAsciiBuffer[currentOffset], utf16Data32BitsHigh) |
| | | 1424 | | NarrowTwoUtf16CharsToAsciiAndWriteToBuffer(ref pAsciiBuffer[currentOffset + 2], utf16Data32BitsL |
| | | 1425 | | } |
| | | 1426 | | |
| | 14382 | 1427 | | currentOffset += 4; |
| | 14382 | 1428 | | } while (currentOffset <= finalOffsetWhereCanLoop); |
| | | 1429 | | } |
| | | 1430 | | |
| | | 1431 | | // Try to narrow 32 bits -> 16 bits. |
| | | 1432 | | |
| | 4352 | 1433 | | if (((uint)remainingElementCount & 2) != 0) |
| | | 1434 | | { |
| | 2166 | 1435 | | utf16Data32BitsHigh = Unsafe.ReadUnaligned<uint>(pUtf16Buffer + currentOffset); |
| | 2166 | 1436 | | if (!AllCharsInUInt32AreAscii(utf16Data32BitsHigh)) |
| | | 1437 | | { |
| | | 1438 | | goto FoundNonAsciiDataInHigh32Bits; |
| | | 1439 | | } |
| | | 1440 | | |
| | 1940 | 1441 | | NarrowTwoUtf16CharsToAsciiAndWriteToBuffer(ref pAsciiBuffer[currentOffset], utf16Data32BitsHigh); |
| | 1940 | 1442 | | currentOffset += 2; |
| | | 1443 | | } |
| | | 1444 | | |
| | | 1445 | | // Try to narrow 16 bits -> 8 bits. |
| | | 1446 | | |
| | 4126 | 1447 | | if (((uint)remainingElementCount & 1) != 0) |
| | | 1448 | | { |
| | 1923 | 1449 | | utf16Data32BitsHigh = pUtf16Buffer[currentOffset]; |
| | 1923 | 1450 | | if (utf16Data32BitsHigh <= 0x007Fu) |
| | | 1451 | | { |
| | 1876 | 1452 | | pAsciiBuffer[currentOffset] = (byte)utf16Data32BitsHigh; |
| | 1876 | 1453 | | currentOffset++; |
| | | 1454 | | } |
| | | 1455 | | } |
| | | 1456 | | |
| | | 1457 | | Finish: |
| | | 1458 | | |
| | 7508 | 1459 | | return currentOffset; |
| | | 1460 | | |
| | | 1461 | | FoundNonAsciiDataIn64BitRead: |
| | | 1462 | | |
| | | 1463 | | if (IntPtr.Size >= 8) |
| | | 1464 | | { |
| | | 1465 | | // Try checking the first 32 bits of the buffer for non-ASCII data. |
| | | 1466 | | // Regardless, we'll move the non-ASCII data into the utf16Data32BitsHigh local. |
| | | 1467 | | |
| | 3156 | 1468 | | if (BitConverter.IsLittleEndian) |
| | | 1469 | | { |
| | 3156 | 1470 | | utf16Data32BitsHigh = (uint)utf16Data64Bits; |
| | | 1471 | | } |
| | | 1472 | | else |
| | | 1473 | | { |
| | | 1474 | | utf16Data32BitsHigh = (uint)(utf16Data64Bits >> 32); |
| | | 1475 | | } |
| | | 1476 | | |
| | 3156 | 1477 | | if (AllCharsInUInt32AreAscii(utf16Data32BitsHigh)) |
| | | 1478 | | { |
| | 477 | 1479 | | NarrowTwoUtf16CharsToAsciiAndWriteToBuffer(ref pAsciiBuffer[currentOffset], utf16Data32BitsHigh); |
| | | 1480 | | |
| | 477 | 1481 | | if (BitConverter.IsLittleEndian) |
| | | 1482 | | { |
| | 477 | 1483 | | utf16Data32BitsHigh = (uint)(utf16Data64Bits >> 32); |
| | | 1484 | | } |
| | | 1485 | | else |
| | | 1486 | | { |
| | | 1487 | | utf16Data32BitsHigh = (uint)utf16Data64Bits; |
| | | 1488 | | } |
| | | 1489 | | |
| | 477 | 1490 | | currentOffset += 2; |
| | | 1491 | | } |
| | | 1492 | | } |
| | | 1493 | | else |
| | | 1494 | | { |
| | | 1495 | | // Need to determine if the high or the low 32-bit value contained non-ASCII data. |
| | | 1496 | | // Regardless, we'll move the non-ASCII data into the utf16Data32BitsHigh local. |
| | | 1497 | | |
| | | 1498 | | if (AllCharsInUInt32AreAscii(utf16Data32BitsHigh)) |
| | | 1499 | | { |
| | | 1500 | | NarrowTwoUtf16CharsToAsciiAndWriteToBuffer(ref pAsciiBuffer[currentOffset], utf16Data32BitsHigh); |
| | | 1501 | | utf16Data32BitsHigh = utf16Data32BitsLow; |
| | | 1502 | | currentOffset += 2; |
| | | 1503 | | } |
| | | 1504 | | } |
| | | 1505 | | |
| | | 1506 | | FoundNonAsciiDataInHigh32Bits: |
| | | 1507 | | |
| | 3382 | 1508 | | Debug.Assert(!AllCharsInUInt32AreAscii(utf16Data32BitsHigh), "Shouldn't have reached this point if we have a |
| | | 1509 | | |
| | | 1510 | | // There's at most one char that needs to be drained. |
| | | 1511 | | |
| | 3382 | 1512 | | if (FirstCharInUInt32IsAscii(utf16Data32BitsHigh)) |
| | | 1513 | | { |
| | 685 | 1514 | | if (!BitConverter.IsLittleEndian) |
| | | 1515 | | { |
| | | 1516 | | utf16Data32BitsHigh >>= 16; // move high char down to low char |
| | | 1517 | | } |
| | | 1518 | | |
| | 685 | 1519 | | pAsciiBuffer[currentOffset] = (byte)utf16Data32BitsHigh; |
| | 685 | 1520 | | currentOffset++; |
| | | 1521 | | } |
| | | 1522 | | |
| | 685 | 1523 | | goto Finish; |
| | | 1524 | | } |
| | | 1525 | | |
| | | 1526 | | #if NET |
| | | 1527 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1528 | | private static bool VectorContainsNonAsciiChar(Vector128<byte> asciiVector) |
| | | 1529 | | { |
| | | 1530 | | // max ASCII character is 0b_0111_1111, so the most significant bit (0x80) tells whether it contains non asc |
| | | 1531 | | |
| | | 1532 | | // For performance, prefer architecture specific implementation |
| | | 1533 | | if (Sse41.IsSupported) |
| | | 1534 | | { |
| | 145932 | 1535 | | return (asciiVector & Vector128.Create((byte)0x80)) != Vector128<byte>.Zero; |
| | | 1536 | | } |
| | | 1537 | | else if (AdvSimd.Arm64.IsSupported) |
| | | 1538 | | { |
| | | 1539 | | Vector128<byte> maxBytes = AdvSimd.Arm64.MaxPairwise(asciiVector, asciiVector); |
| | | 1540 | | return (maxBytes.AsUInt64().ToScalar() & 0x8080808080808080) != 0; |
| | | 1541 | | } |
| | | 1542 | | else |
| | | 1543 | | { |
| | 0 | 1544 | | return asciiVector.ExtractMostSignificantBits() != 0; |
| | | 1545 | | } |
| | | 1546 | | } |
| | | 1547 | | |
| | | 1548 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1549 | | internal static bool VectorContainsNonAsciiChar(Vector128<ushort> utf16Vector) |
| | | 1550 | | { |
| | | 1551 | | // For performance, prefer architecture specific implementation |
| | | 1552 | | if (Sse41.IsSupported) |
| | | 1553 | | { |
| | | 1554 | | const ushort asciiMask = ushort.MaxValue - 127; // 0xFF80 |
| | 3245 | 1555 | | Vector128<ushort> zeroIsAscii = utf16Vector & Vector128.Create(asciiMask); |
| | | 1556 | | // If a non-ASCII bit is set in any WORD of the vector, we have seen non-ASCII data. |
| | 3245 | 1557 | | return zeroIsAscii != Vector128<ushort>.Zero; |
| | | 1558 | | } |
| | 0 | 1559 | | else if (Sse2.IsSupported) |
| | | 1560 | | { |
| | 0 | 1561 | | Vector128<ushort> asciiMaskForAddSaturate = Vector128.Create((ushort)0x7F80); |
| | | 1562 | | // The operation below forces the 0x8000 bit of each WORD to be set iff the WORD element |
| | | 1563 | | // has value >= 0x0800 (non-ASCII). Then we'll treat the vector as a BYTE vector in order |
| | | 1564 | | // to extract the mask. Reminder: the 0x0080 bit of each WORD should be ignored. |
| | 0 | 1565 | | return (Sse2.MoveMask(Sse2.AddSaturate(utf16Vector, asciiMaskForAddSaturate).AsByte()) & 0b_1010_1010_10 |
| | | 1566 | | } |
| | | 1567 | | else if (AdvSimd.Arm64.IsSupported) |
| | | 1568 | | { |
| | | 1569 | | // First we pick four chars, a larger one from all four pairs of adjecent chars in the vector. |
| | | 1570 | | // If any of those four chars has a non-ASCII bit set, we have seen non-ASCII data. |
| | | 1571 | | Vector128<ushort> maxChars = AdvSimd.Arm64.MaxPairwise(utf16Vector, utf16Vector); |
| | | 1572 | | return (maxChars.AsUInt64().ToScalar() & 0xFF80FF80FF80FF80) != 0; |
| | | 1573 | | } |
| | | 1574 | | else |
| | | 1575 | | { |
| | | 1576 | | const ushort asciiMask = ushort.MaxValue - 127; // 0xFF80 |
| | 0 | 1577 | | Vector128<ushort> zeroIsAscii = utf16Vector & Vector128.Create(asciiMask); |
| | | 1578 | | // If a non-ASCII bit is set in any WORD of the vector, we have seen non-ASCII data. |
| | 0 | 1579 | | return zeroIsAscii != Vector128<ushort>.Zero; |
| | | 1580 | | } |
| | | 1581 | | } |
| | | 1582 | | |
| | | 1583 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1584 | | internal static bool VectorContainsNonAsciiChar(Vector256<ushort> utf16Vector) |
| | | 1585 | | { |
| | | 1586 | | const ushort asciiMask = ushort.MaxValue - 127; // 0xFF80 |
| | 1790 | 1587 | | Vector256<ushort> zeroIsAscii = utf16Vector & Vector256.Create(asciiMask); |
| | | 1588 | | // If a non-ASCII bit is set in any WORD of the vector, we have seen non-ASCII data. |
| | 1790 | 1589 | | return zeroIsAscii != Vector256<ushort>.Zero; |
| | | 1590 | | } |
| | | 1591 | | |
| | | 1592 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1593 | | internal static bool VectorContainsNonAsciiChar(Vector512<ushort> utf16Vector) |
| | | 1594 | | { |
| | | 1595 | | const ushort asciiMask = ushort.MaxValue - 127; // 0xFF80 |
| | 4034 | 1596 | | Vector512<ushort> zeroIsAscii = utf16Vector & Vector512.Create(asciiMask); |
| | | 1597 | | // If a non-ASCII bit is set in any WORD of the vector, we have seen non-ASCII data. |
| | 4034 | 1598 | | return zeroIsAscii != Vector512<ushort>.Zero; |
| | | 1599 | | } |
| | | 1600 | | |
| | | 1601 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1602 | | private static bool VectorContainsNonAsciiChar<T>(Vector128<T> vector) |
| | | 1603 | | where T : unmanaged |
| | | 1604 | | { |
| | 2 | 1605 | | Debug.Assert(typeof(T) == typeof(byte) || typeof(T) == typeof(ushort)); |
| | | 1606 | | |
| | 2 | 1607 | | return typeof(T) == typeof(byte) |
| | 2 | 1608 | | ? VectorContainsNonAsciiChar(vector.AsByte()) |
| | 2 | 1609 | | : VectorContainsNonAsciiChar(vector.AsUInt16()); |
| | | 1610 | | } |
| | | 1611 | | |
| | | 1612 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1613 | | private static bool AllCharsInVectorAreAscii<T>(Vector128<T> vector) |
| | | 1614 | | where T : unmanaged |
| | | 1615 | | { |
| | | 1616 | | Debug.Assert(typeof(T) == typeof(byte) || typeof(T) == typeof(ushort)); |
| | | 1617 | | |
| | | 1618 | | // This is a copy of VectorContainsNonAsciiChar with an inverted condition. |
| | 0 | 1619 | | if (typeof(T) == typeof(byte)) |
| | | 1620 | | { |
| | 0 | 1621 | | return |
| | 0 | 1622 | | Sse41.IsSupported ? (vector.AsByte() & Vector128.Create((byte)0x80)) == Vector128<byte>.Zero : |
| | 0 | 1623 | | AdvSimd.Arm64.IsSupported ? AllBytesInUInt64AreAscii(AdvSimd.Arm64.MaxPairwise(vector.AsByte(), vect |
| | 0 | 1624 | | vector.AsByte().ExtractMostSignificantBits() == 0; |
| | | 1625 | | } |
| | | 1626 | | else |
| | | 1627 | | { |
| | | 1628 | | return |
| | | 1629 | | AdvSimd.Arm64.IsSupported ? AllCharsInUInt64AreAscii(AdvSimd.Arm64.MaxPairwise(vector.AsUInt16(), ve |
| | | 1630 | | (vector.AsUInt16() & Vector128.Create((ushort)0xFF80)) == Vector128<ushort>.Zero; |
| | | 1631 | | } |
| | | 1632 | | } |
| | | 1633 | | |
| | | 1634 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1635 | | [CompExactlyDependsOn(typeof(Avx))] |
| | | 1636 | | [CompHasFallback] |
| | | 1637 | | private static bool AllCharsInVectorAreAscii<T>(Vector256<T> vector) |
| | | 1638 | | where T : unmanaged |
| | | 1639 | | { |
| | 0 | 1640 | | Debug.Assert(typeof(T) == typeof(byte) || typeof(T) == typeof(ushort)); |
| | | 1641 | | |
| | 0 | 1642 | | if (typeof(T) == typeof(byte)) |
| | | 1643 | | { |
| | 0 | 1644 | | return |
| | 0 | 1645 | | Avx.IsSupported ? (vector.AsByte() & Vector256.Create((byte)0x80)) == Vector256<byte>.Zero: |
| | 0 | 1646 | | vector.AsByte().ExtractMostSignificantBits() == 0; |
| | | 1647 | | } |
| | | 1648 | | else |
| | | 1649 | | { |
| | 0 | 1650 | | return (vector.AsUInt16() & Vector256.Create((ushort)0xFF80)) == Vector256<ushort>.Zero; |
| | | 1651 | | } |
| | | 1652 | | } |
| | | 1653 | | |
| | | 1654 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1655 | | private static bool AllCharsInVectorAreAscii<T>(Vector512<T> vector) |
| | | 1656 | | where T : unmanaged |
| | | 1657 | | { |
| | 0 | 1658 | | Debug.Assert(typeof(T) == typeof(byte) || typeof(T) == typeof(ushort)); |
| | | 1659 | | |
| | 0 | 1660 | | if (typeof(T) == typeof(byte)) |
| | | 1661 | | { |
| | 0 | 1662 | | return vector.AsByte().ExtractMostSignificantBits() == 0; |
| | | 1663 | | } |
| | | 1664 | | else |
| | | 1665 | | { |
| | 0 | 1666 | | return (vector.AsUInt16() & Vector512.Create((ushort)0xFF80)) == Vector512<ushort>.Zero; |
| | | 1667 | | } |
| | | 1668 | | } |
| | | 1669 | | |
| | | 1670 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1671 | | internal static Vector128<byte> ExtractAsciiVector(Vector128<ushort> vectorFirst, Vector128<ushort> vectorSecond |
| | | 1672 | | { |
| | | 1673 | | // Narrows two vectors of words [ w7 w6 w5 w4 w3 w2 w1 w0 ] and [ w7' w6' w5' w4' w3' w2' w1' w0' ] |
| | | 1674 | | // to a vector of bytes [ b7 ... b0 b7' ... b0']. |
| | | 1675 | | |
| | | 1676 | | // prefer architecture specific intrinsic as they don't perform additional AND like Vector128.Narrow does |
| | | 1677 | | if (Sse2.IsSupported) |
| | | 1678 | | { |
| | 3154 | 1679 | | return Sse2.PackUnsignedSaturate(vectorFirst.AsInt16(), vectorSecond.AsInt16()); |
| | | 1680 | | } |
| | | 1681 | | else if (AdvSimd.Arm64.IsSupported) |
| | | 1682 | | { |
| | | 1683 | | return AdvSimd.Arm64.UnzipEven(vectorFirst.AsByte(), vectorSecond.AsByte()); |
| | | 1684 | | } |
| | | 1685 | | else if (PackedSimd.IsSupported) |
| | | 1686 | | { |
| | | 1687 | | return PackedSimd.ConvertNarrowingSaturateUnsigned(vectorFirst.AsInt16(), vectorSecond.AsInt16()); |
| | | 1688 | | } |
| | | 1689 | | else |
| | | 1690 | | { |
| | 0 | 1691 | | return Vector128.Narrow(vectorFirst, vectorSecond); |
| | | 1692 | | } |
| | | 1693 | | } |
| | | 1694 | | |
| | | 1695 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1696 | | internal static Vector256<byte> ExtractAsciiVector(Vector256<ushort> vectorFirst, Vector256<ushort> vectorSecond |
| | | 1697 | | { |
| | 1718 | 1698 | | return Avx2.IsSupported |
| | 1718 | 1699 | | ? PackedSpanHelpers.FixUpPackedVector256Result(Avx2.PackUnsignedSaturate(vectorFirst.AsInt16(), vectorSe |
| | 1718 | 1700 | | : Vector256.Narrow(vectorFirst, vectorSecond); |
| | | 1701 | | } |
| | | 1702 | | |
| | | 1703 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1704 | | internal static Vector512<byte> ExtractAsciiVector(Vector512<ushort> vectorFirst, Vector512<ushort> vectorSecond |
| | | 1705 | | { |
| | 3934 | 1706 | | return Avx512BW.IsSupported |
| | 3934 | 1707 | | ? PackedSpanHelpers.FixUpPackedVector512Result(Avx512BW.PackUnsignedSaturate(vectorFirst.AsInt16(), vect |
| | 3934 | 1708 | | : Vector512.Narrow(vectorFirst, vectorSecond); |
| | | 1709 | | } |
| | | 1710 | | |
| | | 1711 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1712 | | private static unsafe nuint NarrowUtf16ToAscii_Intrinsified(char* pUtf16Buffer, byte* pAsciiBuffer, nuint elemen |
| | | 1713 | | { |
| | | 1714 | | // This method contains logic optimized using vector instructions for both x64 and Arm64. |
| | | 1715 | | // Much of the logic in this method will be elided by JIT once we determine which specific ISAs we support. |
| | | 1716 | | |
| | | 1717 | | // JIT turns the below into constants |
| | | 1718 | | |
| | 1025 | 1719 | | uint SizeOfVector128 = (uint)Vector128<byte>.Count; |
| | 1025 | 1720 | | nuint MaskOfAllBitsInVector128 = (nuint)(SizeOfVector128 - 1); |
| | | 1721 | | |
| | | 1722 | | // This method is written such that control generally flows top-to-bottom, avoiding |
| | | 1723 | | // jumps as much as possible in the optimistic case of "all ASCII". If we see non-ASCII |
| | | 1724 | | // data, we jump out of the hot paths to targets at the end of the method. |
| | | 1725 | | |
| | 1025 | 1726 | | Debug.Assert(Vector128.IsHardwareAccelerated, "Vector128 is required."); |
| | 1025 | 1727 | | Debug.Assert(BitConverter.IsLittleEndian, "This implementation assumes little-endian."); |
| | 1025 | 1728 | | Debug.Assert(elementCount >= 2 * SizeOfVector128); |
| | | 1729 | | |
| | | 1730 | | // First, perform an unaligned read of the first part of the input buffer. |
| | 1025 | 1731 | | ref ushort utf16Buffer = ref *(ushort*)pUtf16Buffer; |
| | 1025 | 1732 | | Vector128<ushort> utf16VectorFirst = Vector128.LoadUnsafe(ref utf16Buffer); |
| | | 1733 | | |
| | | 1734 | | // If there's non-ASCII data in the first 8 elements of the vector, there's nothing we can do. |
| | 1025 | 1735 | | if (VectorContainsNonAsciiChar(utf16VectorFirst)) |
| | | 1736 | | { |
| | 18 | 1737 | | return 0; |
| | | 1738 | | } |
| | | 1739 | | |
| | | 1740 | | // Turn the 8 ASCII chars we just read into 8 ASCII bytes, then copy it to the destination. |
| | | 1741 | | |
| | 1007 | 1742 | | ref byte asciiBuffer = ref *pAsciiBuffer; |
| | 1007 | 1743 | | Vector128<byte> asciiVector = ExtractAsciiVector(utf16VectorFirst, utf16VectorFirst); |
| | 1007 | 1744 | | asciiVector.StoreLowerUnsafe(ref asciiBuffer, 0); |
| | 1007 | 1745 | | nuint currentOffsetInElements = SizeOfVector128 / 2; // we processed 8 elements so far |
| | | 1746 | | |
| | | 1747 | | // We're going to get the best performance when we have aligned writes, so we'll take the |
| | | 1748 | | // hit of potentially unaligned reads in order to hit this sweet spot. |
| | | 1749 | | |
| | | 1750 | | // pAsciiBuffer points to the start of the destination buffer, immediately before where we wrote |
| | | 1751 | | // the 8 bytes previously. If the 0x08 bit is set at the pinned address, then the 8 bytes we wrote |
| | | 1752 | | // previously mean that the 0x08 bit is *not* set at address &pAsciiBuffer[SizeOfVector128 / 2]. In |
| | | 1753 | | // that case we can immediately back up to the previous aligned boundary and start the main loop. |
| | | 1754 | | // If the 0x08 bit is *not* set at the pinned address, then it means the 0x08 bit *is* set at |
| | | 1755 | | // address &pAsciiBuffer[SizeOfVector128 / 2], and we should perform one more 8-byte write to bump |
| | | 1756 | | // just past the next aligned boundary address. |
| | | 1757 | | |
| | 1007 | 1758 | | if (((uint)pAsciiBuffer & (SizeOfVector128 / 2)) == 0) |
| | | 1759 | | { |
| | | 1760 | | // We need to perform one more partial vector write before we can get the alignment we want. |
| | | 1761 | | |
| | 636 | 1762 | | utf16VectorFirst = Vector128.LoadUnsafe(ref utf16Buffer, currentOffsetInElements); |
| | | 1763 | | |
| | 636 | 1764 | | if (VectorContainsNonAsciiChar(utf16VectorFirst)) |
| | | 1765 | | { |
| | | 1766 | | goto Finish; |
| | | 1767 | | } |
| | | 1768 | | |
| | | 1769 | | // Turn the 8 ASCII chars we just read into 8 ASCII bytes, then copy it to the destination. |
| | 621 | 1770 | | asciiVector = ExtractAsciiVector(utf16VectorFirst, utf16VectorFirst); |
| | 621 | 1771 | | asciiVector.StoreLowerUnsafe(ref asciiBuffer, currentOffsetInElements); |
| | | 1772 | | } |
| | | 1773 | | |
| | | 1774 | | // Calculate how many elements we wrote in order to get pAsciiBuffer to its next alignment |
| | | 1775 | | // point, then use that as the base offset going forward. |
| | | 1776 | | |
| | 992 | 1777 | | currentOffsetInElements = SizeOfVector128 - ((nuint)pAsciiBuffer & MaskOfAllBitsInVector128); |
| | | 1778 | | |
| | 992 | 1779 | | Debug.Assert(0 < currentOffsetInElements && currentOffsetInElements <= SizeOfVector128, "We wrote at least 1 |
| | 992 | 1780 | | Debug.Assert(currentOffsetInElements <= elementCount, "Shouldn't have overrun the destination buffer."); |
| | 992 | 1781 | | Debug.Assert(elementCount - currentOffsetInElements >= SizeOfVector128, "We should be able to run at least o |
| | | 1782 | | |
| | 992 | 1783 | | nuint finalOffsetWhereCanRunLoop = elementCount - SizeOfVector128; |
| | | 1784 | | do |
| | | 1785 | | { |
| | | 1786 | | // In a loop, perform two unaligned reads, narrow to a single vector, then aligned write one vector. |
| | | 1787 | | |
| | 1539 | 1788 | | utf16VectorFirst = Vector128.LoadUnsafe(ref utf16Buffer, currentOffsetInElements); |
| | 1539 | 1789 | | Vector128<ushort> utf16VectorSecond = Vector128.LoadUnsafe(ref utf16Buffer, currentOffsetInElements + Si |
| | 1539 | 1790 | | Vector128<ushort> combinedVector = utf16VectorFirst | utf16VectorSecond; |
| | | 1791 | | |
| | 1539 | 1792 | | if (VectorContainsNonAsciiChar(combinedVector)) |
| | | 1793 | | { |
| | | 1794 | | goto FoundNonAsciiDataInLoop; |
| | | 1795 | | } |
| | | 1796 | | |
| | | 1797 | | // Build up the ASCII vector and perform the store. |
| | | 1798 | | |
| | 1503 | 1799 | | Debug.Assert(((nuint)pAsciiBuffer + currentOffsetInElements) % SizeOfVector128 == 0, "Write should be al |
| | 1503 | 1800 | | asciiVector = ExtractAsciiVector(utf16VectorFirst, utf16VectorSecond); |
| | 1503 | 1801 | | asciiVector.StoreUnsafe(ref asciiBuffer, currentOffsetInElements); |
| | | 1802 | | |
| | 1503 | 1803 | | currentOffsetInElements += SizeOfVector128; |
| | 1503 | 1804 | | } while (currentOffsetInElements <= finalOffsetWhereCanRunLoop); |
| | | 1805 | | |
| | | 1806 | | Finish: |
| | | 1807 | | |
| | | 1808 | | // There might be some ASCII data left over. That's fine - we'll let our caller handle the final drain. |
| | 1007 | 1809 | | return currentOffsetInElements; |
| | | 1810 | | |
| | | 1811 | | FoundNonAsciiDataInLoop: |
| | | 1812 | | |
| | | 1813 | | // Can we at least narrow the high vector? |
| | | 1814 | | // See comments in GetIndexOfFirstNonAsciiChar_Intrinsified for information about how this works. |
| | 36 | 1815 | | if (VectorContainsNonAsciiChar(utf16VectorFirst)) |
| | | 1816 | | { |
| | | 1817 | | goto Finish; |
| | | 1818 | | } |
| | | 1819 | | |
| | | 1820 | | // First part was all ASCII, narrow and aligned write. Note we're only filling in the low half of the vector |
| | | 1821 | | |
| | 23 | 1822 | | Debug.Assert(((nuint)pAsciiBuffer + currentOffsetInElements) % sizeof(ulong) == 0, "Destination should be ul |
| | 23 | 1823 | | asciiVector = ExtractAsciiVector(utf16VectorFirst, utf16VectorFirst); |
| | 23 | 1824 | | asciiVector.StoreLowerUnsafe(ref asciiBuffer, currentOffsetInElements); |
| | 23 | 1825 | | currentOffsetInElements += SizeOfVector128 / 2; |
| | | 1826 | | |
| | 23 | 1827 | | goto Finish; |
| | | 1828 | | } |
| | | 1829 | | |
| | | 1830 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1831 | | private static unsafe nuint NarrowUtf16ToAscii_Intrinsified_256(char* pUtf16Buffer, byte* pAsciiBuffer, nuint el |
| | | 1832 | | { |
| | | 1833 | | // This method contains logic optimized using vector instructions for x64 only. |
| | | 1834 | | // Much of the logic in this method will be elided by JIT once we determine which specific ISAs we support. |
| | | 1835 | | |
| | | 1836 | | // JIT turns the below into constants |
| | | 1837 | | |
| | | 1838 | | const nuint MaskOfAllBitsInVector256 = (nuint)(Vector256.Size - 1); |
| | | 1839 | | |
| | | 1840 | | // This method is written such that control generally flows top-to-bottom, avoiding |
| | | 1841 | | // jumps as much as possible in the optimistic case of "all ASCII". If we see non-ASCII |
| | | 1842 | | // data, we jump out of the hot paths to targets at the end of the method. |
| | | 1843 | | |
| | 566 | 1844 | | Debug.Assert(Vector256.IsHardwareAccelerated, "Vector256 is required."); |
| | 566 | 1845 | | Debug.Assert(BitConverter.IsLittleEndian, "This implementation assumes little-endian."); |
| | 566 | 1846 | | Debug.Assert(elementCount >= 2 * Vector256.Size); |
| | | 1847 | | |
| | | 1848 | | // First, perform an unaligned read of the first part of the input buffer. |
| | 566 | 1849 | | ref ushort utf16Buffer = ref *(ushort*)pUtf16Buffer; |
| | 566 | 1850 | | Vector256<ushort> utf16VectorFirst = Vector256.LoadUnsafe(ref utf16Buffer); |
| | | 1851 | | |
| | | 1852 | | // If there's non-ASCII data in the first 16 elements of the vector, there's nothing we can do. |
| | 566 | 1853 | | if (VectorContainsNonAsciiChar(utf16VectorFirst)) |
| | | 1854 | | { |
| | 25 | 1855 | | return 0; |
| | | 1856 | | } |
| | | 1857 | | |
| | | 1858 | | // Turn the 16 ASCII chars we just read into 16 ASCII bytes, then copy it to the destination. |
| | | 1859 | | |
| | 541 | 1860 | | ref byte asciiBuffer = ref *pAsciiBuffer; |
| | 541 | 1861 | | Vector256<byte> asciiVector = ExtractAsciiVector(utf16VectorFirst, utf16VectorFirst); |
| | 541 | 1862 | | asciiVector.GetLower().StoreUnsafe(ref asciiBuffer, 0); |
| | 541 | 1863 | | nuint currentOffsetInElements = Vector256.Size / 2; // we processed 16 elements so far |
| | | 1864 | | |
| | | 1865 | | // We're going to get the best performance when we have aligned writes, so we'll take the |
| | | 1866 | | // hit of potentially unaligned reads in order to hit this sweet spot. |
| | | 1867 | | |
| | | 1868 | | // pAsciiBuffer points to the start of the destination buffer, immediately before where we wrote |
| | | 1869 | | // the 16 bytes previously. If the 0x10 bit is set at the pinned address, then the 16 bytes we wrote |
| | | 1870 | | // previously mean that the 0x10 bit is *not* set at address &pAsciiBuffer[SizeOfVector256 / 2]. In |
| | | 1871 | | // that case we can immediately back up to the previous aligned boundary and start the main loop. |
| | | 1872 | | // If the 0x10 bit is *not* set at the pinned address, then it means the 0x10 bit *is* set at |
| | | 1873 | | // address &pAsciiBuffer[SizeOfVector256 / 2], and we should perform one more 16-byte write to bump |
| | | 1874 | | // just past the next aligned boundary address. |
| | 541 | 1875 | | if (((uint)pAsciiBuffer & (Vector256.Size / 2)) == 0) |
| | | 1876 | | { |
| | | 1877 | | // We need to perform one more partial vector write before we can get the alignment we want. |
| | | 1878 | | |
| | 248 | 1879 | | utf16VectorFirst = Vector256.LoadUnsafe(ref utf16Buffer, currentOffsetInElements); |
| | | 1880 | | |
| | 248 | 1881 | | if (VectorContainsNonAsciiChar(utf16VectorFirst)) |
| | | 1882 | | { |
| | | 1883 | | goto Finish; |
| | | 1884 | | } |
| | | 1885 | | |
| | | 1886 | | // Turn the 16 ASCII chars we just read into 16 ASCII bytes, then copy it to the destination. |
| | 231 | 1887 | | asciiVector = ExtractAsciiVector(utf16VectorFirst, utf16VectorFirst); |
| | 231 | 1888 | | asciiVector.GetLower().StoreUnsafe(ref asciiBuffer, currentOffsetInElements); |
| | | 1889 | | } |
| | | 1890 | | |
| | | 1891 | | // Calculate how many elements we wrote in order to get pAsciiBuffer to its next alignment |
| | | 1892 | | // point, then use that as the base offset going forward. |
| | | 1893 | | |
| | 524 | 1894 | | currentOffsetInElements = Vector256.Size - ((nuint)pAsciiBuffer & MaskOfAllBitsInVector256); |
| | | 1895 | | |
| | 524 | 1896 | | Debug.Assert(0 < currentOffsetInElements && currentOffsetInElements <= Vector256.Size, "We wrote at least 1 |
| | 524 | 1897 | | Debug.Assert(currentOffsetInElements <= elementCount, "Shouldn't have overrun the destination buffer."); |
| | 524 | 1898 | | Debug.Assert(elementCount - currentOffsetInElements >= Vector256.Size, "We should be able to run at least on |
| | | 1899 | | |
| | 524 | 1900 | | nuint finalOffsetWhereCanRunLoop = elementCount - Vector256.Size; |
| | | 1901 | | do |
| | | 1902 | | { |
| | | 1903 | | // In a loop, perform two unaligned reads, narrow to a single vector, then aligned write one vector. |
| | | 1904 | | |
| | 955 | 1905 | | utf16VectorFirst = Vector256.LoadUnsafe(ref utf16Buffer, currentOffsetInElements); |
| | 955 | 1906 | | Vector256<ushort> utf16VectorSecond = Vector256.LoadUnsafe(ref utf16Buffer, currentOffsetInElements + Ve |
| | 955 | 1907 | | Vector256<ushort> combinedVector = utf16VectorFirst | utf16VectorSecond; |
| | | 1908 | | |
| | 955 | 1909 | | if (VectorContainsNonAsciiChar(combinedVector)) |
| | | 1910 | | { |
| | | 1911 | | goto FoundNonAsciiDataInLoop; |
| | | 1912 | | } |
| | | 1913 | | |
| | | 1914 | | // Build up the ASCII vector and perform the store. |
| | | 1915 | | |
| | 939 | 1916 | | Debug.Assert(((nuint)pAsciiBuffer + currentOffsetInElements) % Vector256.Size == 0, "Write should be ali |
| | 939 | 1917 | | asciiVector = ExtractAsciiVector(utf16VectorFirst, utf16VectorSecond); |
| | 939 | 1918 | | asciiVector.StoreUnsafe(ref asciiBuffer, currentOffsetInElements); |
| | | 1919 | | |
| | 939 | 1920 | | currentOffsetInElements += Vector256.Size; |
| | 939 | 1921 | | } while (currentOffsetInElements <= finalOffsetWhereCanRunLoop); |
| | | 1922 | | |
| | | 1923 | | Finish: |
| | | 1924 | | |
| | | 1925 | | // There might be some ASCII data left over. That's fine - we'll let our caller handle the final drain. |
| | 541 | 1926 | | return currentOffsetInElements; |
| | | 1927 | | |
| | | 1928 | | FoundNonAsciiDataInLoop: |
| | | 1929 | | |
| | | 1930 | | // Can we at least narrow the high vector? |
| | | 1931 | | // See comments in GetIndexOfFirstNonAsciiChar_Intrinsified for information about how this works. |
| | 16 | 1932 | | if (VectorContainsNonAsciiChar(utf16VectorFirst)) |
| | | 1933 | | { |
| | | 1934 | | goto Finish; |
| | | 1935 | | } |
| | | 1936 | | |
| | | 1937 | | // First part was all ASCII, narrow and aligned write. Note we're only filling in the low half of the vector |
| | | 1938 | | |
| | 7 | 1939 | | Debug.Assert(((nuint)pAsciiBuffer + currentOffsetInElements) % Vector128.Size == 0, "Destination should be 1 |
| | 7 | 1940 | | asciiVector = ExtractAsciiVector(utf16VectorFirst, utf16VectorFirst); |
| | 7 | 1941 | | asciiVector.GetLower().StoreUnsafe(ref asciiBuffer, currentOffsetInElements); |
| | 7 | 1942 | | currentOffsetInElements += Vector256.Size / 2; |
| | | 1943 | | |
| | 7 | 1944 | | goto Finish; |
| | | 1945 | | } |
| | | 1946 | | |
| | | 1947 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 1948 | | private static unsafe nuint NarrowUtf16ToAscii_Intrinsified_512(char* pUtf16Buffer, byte* pAsciiBuffer, nuint el |
| | | 1949 | | { |
| | | 1950 | | // This method contains logic optimized using vector instructions for x64 only. |
| | | 1951 | | // Much of the logic in this method will be elided by JIT once we determine which specific ISAs we support. |
| | | 1952 | | |
| | | 1953 | | // JIT turns the below into constants |
| | | 1954 | | |
| | | 1955 | | const nuint MaskOfAllBitsInVector512 = (nuint)(Vector512.Size - 1); |
| | | 1956 | | |
| | | 1957 | | // This method is written such that control generally flows top-to-bottom, avoiding |
| | | 1958 | | // jumps as much as possible in the optimistic case of "all ASCII". If we see non-ASCII |
| | | 1959 | | // data, we jump out of the hot paths to targets at the end of the method. |
| | | 1960 | | |
| | 836 | 1961 | | Debug.Assert(Vector512.IsHardwareAccelerated, "Vector512 is required."); |
| | 836 | 1962 | | Debug.Assert(BitConverter.IsLittleEndian, "This implementation assumes little-endian."); |
| | 836 | 1963 | | Debug.Assert(elementCount >= 2 * Vector512.Size); |
| | | 1964 | | |
| | | 1965 | | // First, perform an unaligned read of the first part of the input buffer. |
| | 836 | 1966 | | ref ushort utf16Buffer = ref *(ushort*)pUtf16Buffer; |
| | 836 | 1967 | | Vector512<ushort> utf16VectorFirst = Vector512.LoadUnsafe(ref utf16Buffer); |
| | | 1968 | | |
| | | 1969 | | // If there's non-ASCII data in the first 32 elements of the vector, there's nothing we can do. |
| | 836 | 1970 | | if (VectorContainsNonAsciiChar(utf16VectorFirst)) |
| | | 1971 | | { |
| | 39 | 1972 | | return 0; |
| | | 1973 | | } |
| | | 1974 | | |
| | | 1975 | | // Turn the 32 ASCII chars we just read into 32 ASCII bytes, then copy it to the destination. |
| | | 1976 | | |
| | 797 | 1977 | | ref byte asciiBuffer = ref *pAsciiBuffer; |
| | 797 | 1978 | | Vector512<byte> asciiVector = ExtractAsciiVector(utf16VectorFirst, utf16VectorFirst); |
| | 797 | 1979 | | asciiVector.GetLower().StoreUnsafe(ref asciiBuffer, 0); // how to store the lower part of a avx512 |
| | 797 | 1980 | | nuint currentOffsetInElements = Vector512.Size / 2; // we processed 32 elements so far |
| | | 1981 | | |
| | | 1982 | | // We're going to get the best performance when we have aligned writes, so we'll take the |
| | | 1983 | | // hit of potentially unaligned reads in order to hit this sweet spot. |
| | | 1984 | | |
| | | 1985 | | // pAsciiBuffer points to the start of the destination buffer, immediately before where we wrote |
| | | 1986 | | // the 32 bytes previously. If the 0x20 bit is set at the pinned address, then the 32 bytes we wrote |
| | | 1987 | | // previously mean that the 0x20 bit is *not* set at address &pAsciiBuffer[SizeOfVector512 / 2]. In |
| | | 1988 | | // that case we can immediately back up to the previous aligned boundary and start the main loop. |
| | | 1989 | | // If the 0x20 bit is *not* set at the pinned address, then it means the 0x20 bit *is* set at |
| | | 1990 | | // address &pAsciiBuffer[SizeOfVector512 / 2], and we should perform one more 32-byte write to bump |
| | | 1991 | | // just past the next aligned boundary address. |
| | | 1992 | | |
| | 797 | 1993 | | if (((uint)pAsciiBuffer & (Vector512.Size / 2)) == 0) |
| | | 1994 | | { |
| | | 1995 | | // We need to perform one more partial vector write before we can get the alignment we want. |
| | | 1996 | | |
| | 360 | 1997 | | utf16VectorFirst = Vector512.LoadUnsafe(ref utf16Buffer, currentOffsetInElements); |
| | | 1998 | | |
| | 360 | 1999 | | if (VectorContainsNonAsciiChar(utf16VectorFirst)) |
| | | 2000 | | { |
| | | 2001 | | goto Finish; |
| | | 2002 | | } |
| | | 2003 | | |
| | | 2004 | | // Turn the 32 ASCII chars we just read into 32 ASCII bytes, then copy it to the destination. |
| | 358 | 2005 | | asciiVector = ExtractAsciiVector(utf16VectorFirst, utf16VectorFirst); |
| | 358 | 2006 | | asciiVector.GetLower().StoreUnsafe(ref asciiBuffer, currentOffsetInElements); |
| | | 2007 | | } |
| | | 2008 | | |
| | | 2009 | | // Calculate how many elements we wrote in order to get pAsciiBuffer to its next alignment |
| | | 2010 | | // point, then use that as the base offset going forward. |
| | | 2011 | | |
| | 795 | 2012 | | currentOffsetInElements = Vector512.Size - ((nuint)pAsciiBuffer & MaskOfAllBitsInVector512); |
| | | 2013 | | |
| | 795 | 2014 | | Debug.Assert(0 < currentOffsetInElements && currentOffsetInElements <= Vector512.Size, "We wrote at least 1 |
| | 795 | 2015 | | Debug.Assert(currentOffsetInElements <= elementCount, "Shouldn't have overrun the destination buffer."); |
| | 795 | 2016 | | Debug.Assert(elementCount - currentOffsetInElements >= Vector512.Size, "We should be able to run at least on |
| | | 2017 | | |
| | 795 | 2018 | | nuint finalOffsetWhereCanRunLoop = elementCount - Vector512.Size; |
| | | 2019 | | do |
| | | 2020 | | { |
| | | 2021 | | // In a loop, perform two unaligned reads, narrow to a single vector, then aligned write one vector. |
| | | 2022 | | |
| | 2799 | 2023 | | utf16VectorFirst = Vector512.LoadUnsafe(ref utf16Buffer, currentOffsetInElements); |
| | 2799 | 2024 | | Vector512<ushort> utf16VectorSecond = Vector512.LoadUnsafe(ref utf16Buffer, currentOffsetInElements + Ve |
| | 2799 | 2025 | | Vector512<ushort> combinedVector = utf16VectorFirst | utf16VectorSecond; |
| | | 2026 | | |
| | 2799 | 2027 | | if (VectorContainsNonAsciiChar(combinedVector)) |
| | | 2028 | | { |
| | | 2029 | | goto FoundNonAsciiDataInLoop; |
| | | 2030 | | } |
| | | 2031 | | |
| | | 2032 | | // Build up the ASCII vector and perform the store. |
| | | 2033 | | |
| | 2760 | 2034 | | Debug.Assert(((nuint)pAsciiBuffer + currentOffsetInElements) % Vector512.Size == 0, "Write should be ali |
| | 2760 | 2035 | | asciiVector = ExtractAsciiVector(utf16VectorFirst, utf16VectorSecond); |
| | 2760 | 2036 | | asciiVector.StoreUnsafe(ref asciiBuffer, currentOffsetInElements); |
| | | 2037 | | |
| | 2760 | 2038 | | currentOffsetInElements += Vector512.Size; |
| | 2760 | 2039 | | } while (currentOffsetInElements <= finalOffsetWhereCanRunLoop); |
| | | 2040 | | |
| | | 2041 | | Finish: |
| | | 2042 | | |
| | | 2043 | | // There might be some ASCII data left over. That's fine - we'll let our caller handle the final drain. |
| | 797 | 2044 | | return currentOffsetInElements; |
| | | 2045 | | |
| | | 2046 | | FoundNonAsciiDataInLoop: |
| | | 2047 | | |
| | | 2048 | | // Can we at least narrow the high vector? |
| | | 2049 | | // See comments in GetIndexOfFirstNonAsciiChar_Intrinsified for information about how this works. |
| | 39 | 2050 | | if (VectorContainsNonAsciiChar(utf16VectorFirst)) |
| | | 2051 | | { |
| | | 2052 | | goto Finish; |
| | | 2053 | | } |
| | | 2054 | | |
| | | 2055 | | // First part was all ASCII, narrow and aligned write. Note we're only filling in the low half of the vector |
| | | 2056 | | |
| | 19 | 2057 | | Debug.Assert(((nuint)pAsciiBuffer + currentOffsetInElements) % Vector256.Size == 0, "Destination should be 2 |
| | 19 | 2058 | | asciiVector = ExtractAsciiVector(utf16VectorFirst, utf16VectorFirst); |
| | 19 | 2059 | | asciiVector.GetLower().StoreUnsafe(ref asciiBuffer, currentOffsetInElements); |
| | 19 | 2060 | | currentOffsetInElements += Vector512.Size / 2; |
| | | 2061 | | |
| | 19 | 2062 | | goto Finish; |
| | | 2063 | | } |
| | | 2064 | | #endif |
| | | 2065 | | |
| | | 2066 | | /// <summary> |
| | | 2067 | | /// Copies as many ASCII bytes (00..7F) as possible from <paramref name="pAsciiBuffer"/> |
| | | 2068 | | /// to <paramref name="pUtf16Buffer"/>, stopping when the first non-ASCII byte is encountered |
| | | 2069 | | /// or once <paramref name="elementCount"/> elements have been converted. Returns the total number |
| | | 2070 | | /// of elements that were able to be converted. |
| | | 2071 | | /// </summary> |
| | | 2072 | | internal static unsafe nuint WidenAsciiToUtf16(byte* pAsciiBuffer, char* pUtf16Buffer, nuint elementCount) |
| | | 2073 | | { |
| | | 2074 | | // Intrinsified in mono interpreter |
| | | 2075 | | nuint currentOffset = 0; |
| | | 2076 | | |
| | | 2077 | | #if NET |
| | 5571920 | 2078 | | if (BitConverter.IsLittleEndian && Vector128.IsHardwareAccelerated && elementCount >= (uint)Vector128<byte>. |
| | | 2079 | | { |
| | 2453399 | 2080 | | if (Vector512.IsHardwareAccelerated && (elementCount - currentOffset) >= (uint)Vector512<byte>.Count) |
| | | 2081 | | { |
| | 1577977 | 2082 | | WidenAsciiToUtf1_Vector<Vector512<byte>, Vector512<ushort>>(pAsciiBuffer, pUtf16Buffer, ref currentO |
| | | 2083 | | } |
| | 875422 | 2084 | | else if (Vector256.IsHardwareAccelerated && (elementCount - currentOffset) >= (uint)Vector256<byte>.Coun |
| | | 2085 | | { |
| | 503469 | 2086 | | WidenAsciiToUtf1_Vector<Vector256<byte>, Vector256<ushort>>(pAsciiBuffer, pUtf16Buffer, ref currentO |
| | | 2087 | | } |
| | 371953 | 2088 | | else if (Vector128.IsHardwareAccelerated && (elementCount - currentOffset) >= (uint)Vector128<byte>.Coun |
| | | 2089 | | { |
| | 371953 | 2090 | | WidenAsciiToUtf1_Vector<Vector128<byte>, Vector128<ushort>>(pAsciiBuffer, pUtf16Buffer, ref currentO |
| | | 2091 | | } |
| | | 2092 | | } |
| | | 2093 | | #endif |
| | | 2094 | | |
| | 5571920 | 2095 | | Debug.Assert(currentOffset <= elementCount); |
| | 5571920 | 2096 | | nuint remainingElementCount = elementCount - currentOffset; |
| | | 2097 | | |
| | | 2098 | | // Try to widen 32 bits -> 64 bits at a time. |
| | | 2099 | | // We needn't update remainingElementCount after this point. |
| | | 2100 | | |
| | | 2101 | | uint asciiData; |
| | | 2102 | | |
| | 5571920 | 2103 | | if (remainingElementCount >= 4) |
| | | 2104 | | { |
| | 3073122 | 2105 | | nuint finalOffsetWhereCanLoop = currentOffset + remainingElementCount - 4; |
| | | 2106 | | do |
| | | 2107 | | { |
| | 3332871 | 2108 | | asciiData = Unsafe.ReadUnaligned<uint>(pAsciiBuffer + currentOffset); |
| | 3332871 | 2109 | | if (!AllBytesInUInt32AreAscii(asciiData)) |
| | | 2110 | | { |
| | | 2111 | | goto FoundNonAsciiData; |
| | | 2112 | | } |
| | | 2113 | | |
| | 319016 | 2114 | | WidenFourAsciiBytesToUtf16AndWriteToBuffer(ref pUtf16Buffer[currentOffset], asciiData); |
| | 319016 | 2115 | | currentOffset += 4; |
| | 319016 | 2116 | | } while (currentOffset <= finalOffsetWhereCanLoop); |
| | | 2117 | | } |
| | | 2118 | | |
| | | 2119 | | // Try to widen 16 bits -> 32 bits. |
| | | 2120 | | |
| | 2558065 | 2121 | | if (((uint)remainingElementCount & 2) != 0) |
| | | 2122 | | { |
| | 1107383 | 2123 | | asciiData = Unsafe.ReadUnaligned<ushort>(pAsciiBuffer + currentOffset); |
| | 1107383 | 2124 | | if (!AllBytesInUInt32AreAscii(asciiData)) |
| | | 2125 | | { |
| | 905262 | 2126 | | if (!BitConverter.IsLittleEndian) |
| | | 2127 | | { |
| | | 2128 | | asciiData <<= 16; |
| | | 2129 | | } |
| | | 2130 | | goto FoundNonAsciiData; |
| | | 2131 | | } |
| | | 2132 | | |
| | 202121 | 2133 | | if (BitConverter.IsLittleEndian) |
| | | 2134 | | { |
| | 202121 | 2135 | | pUtf16Buffer[currentOffset] = (char)(byte)asciiData; |
| | 202121 | 2136 | | pUtf16Buffer[currentOffset + 1] = (char)(asciiData >> 8); |
| | | 2137 | | } |
| | | 2138 | | else |
| | | 2139 | | { |
| | | 2140 | | pUtf16Buffer[currentOffset + 1] = (char)(byte)asciiData; |
| | | 2141 | | pUtf16Buffer[currentOffset] = (char)(asciiData >> 8); |
| | | 2142 | | } |
| | | 2143 | | |
| | 202121 | 2144 | | currentOffset += 2; |
| | | 2145 | | } |
| | | 2146 | | |
| | | 2147 | | // Try to widen 8 bits -> 16 bits. |
| | | 2148 | | |
| | 1652803 | 2149 | | if (((uint)remainingElementCount & 1) != 0) |
| | | 2150 | | { |
| | 1443610 | 2151 | | asciiData = pAsciiBuffer[currentOffset]; |
| | 1443610 | 2152 | | if (((byte)asciiData & 0x80) != 0) |
| | | 2153 | | { |
| | | 2154 | | goto Finish; |
| | | 2155 | | } |
| | | 2156 | | |
| | 402296 | 2157 | | pUtf16Buffer[currentOffset] = (char)asciiData; |
| | 402296 | 2158 | | currentOffset++; |
| | | 2159 | | } |
| | | 2160 | | |
| | | 2161 | | Finish: |
| | | 2162 | | |
| | 5571920 | 2163 | | return currentOffset; |
| | | 2164 | | |
| | | 2165 | | FoundNonAsciiData: |
| | | 2166 | | |
| | 3919117 | 2167 | | Debug.Assert(!AllBytesInUInt32AreAscii(asciiData), "Shouldn't have reached this point if we have an all-ASCI |
| | | 2168 | | |
| | | 2169 | | // Drain ASCII bytes one at a time. |
| | | 2170 | | |
| | 3919117 | 2171 | | if (BitConverter.IsLittleEndian) |
| | | 2172 | | { |
| | 4702172 | 2173 | | while (((byte)asciiData & 0x80) == 0) |
| | | 2174 | | { |
| | 783055 | 2175 | | pUtf16Buffer[currentOffset] = (char)(byte)asciiData; |
| | 783055 | 2176 | | currentOffset++; |
| | 783055 | 2177 | | asciiData >>= 8; |
| | | 2178 | | } |
| | | 2179 | | } |
| | | 2180 | | else |
| | | 2181 | | { |
| | | 2182 | | while ((asciiData & 0x80000000) == 0) |
| | | 2183 | | { |
| | | 2184 | | asciiData = BitOperations.RotateLeft(asciiData, 8); |
| | | 2185 | | pUtf16Buffer[currentOffset] = (char)(byte)asciiData; |
| | | 2186 | | currentOffset++; |
| | | 2187 | | } |
| | | 2188 | | } |
| | | 2189 | | |
| | | 2190 | | goto Finish; |
| | | 2191 | | } |
| | | 2192 | | |
| | | 2193 | | #if NET |
| | | 2194 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 2195 | | private static unsafe void WidenAsciiToUtf1_Vector<TVectorByte, TVectorUInt16>(byte* pAsciiBuffer, char* pUtf16B |
| | | 2196 | | where TVectorByte : unmanaged, ISimdVector<TVectorByte, byte> |
| | | 2197 | | where TVectorUInt16 : unmanaged, ISimdVector<TVectorUInt16, ushort> |
| | | 2198 | | { |
| | 2453399 | 2199 | | ushort* pCurrentWriteAddress = (ushort*)pUtf16Buffer; |
| | | 2200 | | // Calculating the destination address outside the loop results in significant |
| | | 2201 | | // perf wins vs. relying on the JIT to fold memory addressing logic into the |
| | | 2202 | | // write instructions. See: https://github.com/dotnet/runtime/issues/33002 |
| | 2453399 | 2203 | | nuint finalOffsetWhereCanRunLoop = elementCount - (nuint)TVectorByte.ElementCount; |
| | 2453399 | 2204 | | TVectorByte asciiVector = TVectorByte.Load(pAsciiBuffer + currentOffset); |
| | 2453399 | 2205 | | if (!HasMatch<TVectorByte>(asciiVector)) |
| | | 2206 | | { |
| | 14584 | 2207 | | (TVectorUInt16 utf16LowVector, TVectorUInt16 utf16HighVector) = Widen<TVectorByte, TVectorUInt16>(asciiV |
| | 14584 | 2208 | | utf16LowVector.Store(pCurrentWriteAddress); |
| | 14584 | 2209 | | utf16HighVector.Store(pCurrentWriteAddress + TVectorUInt16.ElementCount); |
| | 14584 | 2210 | | pCurrentWriteAddress += (nuint)(TVectorUInt16.ElementCount * 2); |
| | 14584 | 2211 | | if (((nuint)pCurrentWriteAddress % sizeof(char)) == 0) |
| | | 2212 | | { |
| | | 2213 | | // Bump write buffer up to the next aligned boundary |
| | 14584 | 2214 | | pCurrentWriteAddress = (ushort*)((nuint)pCurrentWriteAddress & ~(nuint)(TVectorUInt16.Alignment - 1) |
| | 14584 | 2215 | | nuint numBytesWritten = (nuint)pCurrentWriteAddress - (nuint)pUtf16Buffer; |
| | 14584 | 2216 | | currentOffset += (nuint)numBytesWritten / 2; |
| | | 2217 | | } |
| | | 2218 | | else |
| | | 2219 | | { |
| | | 2220 | | // If input isn't char aligned, we won't be able to align it to a Vector |
| | 0 | 2221 | | currentOffset += (nuint)TVectorByte.ElementCount; |
| | | 2222 | | } |
| | 24113 | 2223 | | while (currentOffset <= finalOffsetWhereCanRunLoop) |
| | | 2224 | | { |
| | 13323 | 2225 | | asciiVector = TVectorByte.Load(pAsciiBuffer + currentOffset); |
| | 13323 | 2226 | | if (HasMatch<TVectorByte>(asciiVector)) |
| | | 2227 | | { |
| | | 2228 | | break; |
| | | 2229 | | } |
| | 9529 | 2230 | | (utf16LowVector, utf16HighVector) = Widen<TVectorByte, TVectorUInt16>(asciiVector); |
| | 9529 | 2231 | | utf16LowVector.Store(pCurrentWriteAddress); |
| | 9529 | 2232 | | utf16HighVector.Store(pCurrentWriteAddress + TVectorUInt16.ElementCount); |
| | | 2233 | | |
| | 9529 | 2234 | | currentOffset += (nuint)TVectorByte.ElementCount; |
| | 9529 | 2235 | | pCurrentWriteAddress += (nuint)(TVectorUInt16.ElementCount * 2); |
| | | 2236 | | } |
| | | 2237 | | } |
| | 2453399 | 2238 | | return; |
| | | 2239 | | } |
| | | 2240 | | |
| | | 2241 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 2242 | | private static bool HasMatch<TVectorByte>(TVectorByte vector) |
| | | 2243 | | where TVectorByte : unmanaged, ISimdVector<TVectorByte, byte> |
| | | 2244 | | { |
| | 2466722 | 2245 | | return !(vector & TVectorByte.Create((byte)0x80)).Equals(TVectorByte.Zero); |
| | | 2246 | | } |
| | | 2247 | | |
| | | 2248 | | |
| | | 2249 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 2250 | | private static (TVectorUInt16 Lower, TVectorUInt16 Upper) Widen<TVectorByte, TVectorUInt16>(TVectorByte vector) |
| | | 2251 | | where TVectorByte : unmanaged, ISimdVector<TVectorByte, byte> |
| | | 2252 | | where TVectorUInt16 : unmanaged, ISimdVector<TVectorUInt16, ushort> |
| | | 2253 | | { |
| | 24113 | 2254 | | if (typeof(TVectorByte) == typeof(Vector256<byte>)) |
| | | 2255 | | { |
| | 4475 | 2256 | | (Vector256<ushort> Lower256, Vector256<ushort> Upper256) = Vector256.Widen((Vector256<byte>)(object)vect |
| | 4475 | 2257 | | return ((TVectorUInt16)(object)Lower256, (TVectorUInt16)(object)Upper256); |
| | | 2258 | | } |
| | 19638 | 2259 | | else if (typeof(TVectorByte) == typeof(Vector512<byte>)) |
| | | 2260 | | { |
| | 15400 | 2261 | | (Vector512<ushort> Lower512, Vector512<ushort> Upper512) = Vector512.Widen((Vector512<byte>)(object)vect |
| | 15400 | 2262 | | return ((TVectorUInt16)(object)Lower512, (TVectorUInt16)(object)Upper512); |
| | | 2263 | | } |
| | | 2264 | | else |
| | | 2265 | | { |
| | 4238 | 2266 | | Debug.Assert(typeof(TVectorByte) == typeof(Vector128<byte>)); |
| | 4238 | 2267 | | (Vector128<ushort> Lower128, Vector128<ushort> Upper128) = Vector128.Widen((Vector128<byte>)(object)vect |
| | 4238 | 2268 | | return ((TVectorUInt16)(object)Lower128, (TVectorUInt16)(object)Upper128); |
| | | 2269 | | } |
| | | 2270 | | } |
| | | 2271 | | #endif |
| | | 2272 | | |
| | | 2273 | | /// <summary> |
| | | 2274 | | /// Given a DWORD which represents a buffer of 4 bytes, widens the buffer into 4 WORDs and |
| | | 2275 | | /// writes them to the output buffer with machine endianness. |
| | | 2276 | | /// </summary> |
| | | 2277 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 2278 | | internal static void WidenFourAsciiBytesToUtf16AndWriteToBuffer(ref char outputBuffer, uint value) |
| | | 2279 | | { |
| | | 2280 | | Debug.Assert(AllBytesInUInt32AreAscii(value)); |
| | | 2281 | | |
| | | 2282 | | #if NET |
| | | 2283 | | if (AdvSimd.Arm64.IsSupported) |
| | | 2284 | | { |
| | | 2285 | | Vector128<byte> vecNarrow = AdvSimd.DuplicateToVector128(value).AsByte(); |
| | | 2286 | | Vector128<ulong> vecWide = AdvSimd.Arm64.ZipLow(vecNarrow, Vector128<byte>.Zero).AsUInt64(); |
| | | 2287 | | Unsafe.WriteUnaligned(ref Unsafe.As<char, byte>(ref outputBuffer), vecWide.ToScalar()); |
| | | 2288 | | } |
| | 380730 | 2289 | | else if (Vector128.IsHardwareAccelerated) |
| | | 2290 | | { |
| | 380730 | 2291 | | Vector128<byte> vecNarrow = Vector128.CreateScalar(value).AsByte(); |
| | 380730 | 2292 | | Vector128<ulong> vecWide = Vector128.WidenLower(vecNarrow).AsUInt64(); |
| | 380730 | 2293 | | Unsafe.WriteUnaligned(ref Unsafe.As<char, byte>(ref outputBuffer), vecWide.ToScalar()); |
| | | 2294 | | } |
| | | 2295 | | else |
| | | 2296 | | #endif |
| | | 2297 | | { |
| | 0 | 2298 | | if (BitConverter.IsLittleEndian) |
| | | 2299 | | { |
| | 0 | 2300 | | outputBuffer = (char)(byte)value; |
| | 0 | 2301 | | value >>= 8; |
| | 0 | 2302 | | Unsafe.Add(ref outputBuffer, 1) = (char)(byte)value; |
| | 0 | 2303 | | value >>= 8; |
| | 0 | 2304 | | Unsafe.Add(ref outputBuffer, 2) = (char)(byte)value; |
| | 0 | 2305 | | value >>= 8; |
| | 0 | 2306 | | Unsafe.Add(ref outputBuffer, 3) = (char)value; |
| | | 2307 | | } |
| | | 2308 | | else |
| | | 2309 | | { |
| | | 2310 | | Unsafe.Add(ref outputBuffer, 3) = (char)(byte)value; |
| | | 2311 | | value >>= 8; |
| | | 2312 | | Unsafe.Add(ref outputBuffer, 2) = (char)(byte)value; |
| | | 2313 | | value >>= 8; |
| | | 2314 | | Unsafe.Add(ref outputBuffer, 1) = (char)(byte)value; |
| | | 2315 | | value >>= 8; |
| | | 2316 | | outputBuffer = (char)value; |
| | | 2317 | | } |
| | | 2318 | | } |
| | | 2319 | | } |
| | | 2320 | | } |
| | | 2321 | | } |
| | | 2322 | | |