| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | using System.Buffers; |
| | | 5 | | using System.Buffers.Text; |
| | | 6 | | using System.Diagnostics; |
| | | 7 | | using System.Diagnostics.CodeAnalysis; |
| | | 8 | | using System.Numerics; |
| | | 9 | | using System.Runtime.CompilerServices; |
| | | 10 | | #if NET |
| | | 11 | | using System.Runtime.Intrinsics; |
| | | 12 | | using System.Runtime.Intrinsics.Arm; |
| | | 13 | | using System.Runtime.Intrinsics.Wasm; |
| | | 14 | | using System.Runtime.Intrinsics.X86; |
| | | 15 | | #endif |
| | | 16 | | |
| | | 17 | | namespace System.Text.Unicode |
| | | 18 | | { |
| | | 19 | | internal static unsafe partial class Utf8Utility |
| | | 20 | | { |
| | | 21 | | // On method return, pInputBufferRemaining and pOutputBufferRemaining will both point to where |
| | | 22 | | // the next byte would have been consumed from / the next char would have been written to. |
| | | 23 | | // inputLength in bytes, outputCharsRemaining in chars. |
| | | 24 | | public static OperationStatus TranscodeToUtf16(byte* pInputBuffer, int inputLength, char* pOutputBuffer, int out |
| | | 25 | | { |
| | | 26 | | Debug.Assert(inputLength >= 0, "Input length must not be negative."); |
| | 2495231 | 27 | | Debug.Assert(pInputBuffer != null || inputLength == 0, "Input length must be zero if input buffer pointer is |
| | | 28 | | |
| | 2495231 | 29 | | Debug.Assert(outputCharsRemaining >= 0, "Destination length must not be negative."); |
| | 2495231 | 30 | | Debug.Assert(pOutputBuffer != null || outputCharsRemaining == 0, "Destination length must be zero if destina |
| | | 31 | | |
| | | 32 | | // First, try vectorized conversion. |
| | | 33 | | { |
| | 2495231 | 34 | | nuint numElementsConverted = Ascii.WidenAsciiToUtf16(pInputBuffer, pOutputBuffer, (uint)Math.Min(inputLe |
| | | 35 | | |
| | 2495231 | 36 | | pInputBuffer += numElementsConverted; |
| | 2495231 | 37 | | pOutputBuffer += numElementsConverted; |
| | | 38 | | |
| | | 39 | | // Quick check - did we just end up consuming the entire input buffer? |
| | | 40 | | // If so, short-circuit the remainder of the method. |
| | | 41 | | |
| | 2495231 | 42 | | if ((int)numElementsConverted == inputLength) |
| | | 43 | | { |
| | 320289 | 44 | | pInputBufferRemaining = pInputBuffer; |
| | 320289 | 45 | | pOutputBufferRemaining = pOutputBuffer; |
| | 320289 | 46 | | return OperationStatus.Done; |
| | | 47 | | } |
| | | 48 | | |
| | 2174942 | 49 | | inputLength -= (int)numElementsConverted; |
| | 2174942 | 50 | | outputCharsRemaining -= (int)numElementsConverted; |
| | | 51 | | } |
| | | 52 | | |
| | 2174942 | 53 | | if (inputLength < sizeof(uint)) |
| | | 54 | | { |
| | | 55 | | goto ProcessInputOfLessThanDWordSize; |
| | | 56 | | } |
| | | 57 | | |
| | 1205073 | 58 | | byte* pFinalPosWhereCanReadDWordFromInputBuffer = pInputBuffer + (uint)inputLength - 4; |
| | | 59 | | |
| | | 60 | | // Begin the main loop. |
| | | 61 | | |
| | | 62 | | #if DEBUG |
| | 1205073 | 63 | | byte* pLastBufferPosProcessed = null; // used for invariant checking in debug builds |
| | | 64 | | #endif |
| | | 65 | | |
| | 1205073 | 66 | | Debug.Assert(pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer); |
| | | 67 | | do |
| | | 68 | | { |
| | | 69 | | // Read 32 bits at a time. This is enough to hold any possible UTF8-encoded scalar. |
| | | 70 | | |
| | 1252680 | 71 | | uint thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); |
| | | 72 | | |
| | | 73 | | AfterReadDWord: |
| | | 74 | | |
| | | 75 | | #if DEBUG |
| | 1310994 | 76 | | Debug.Assert(pLastBufferPosProcessed < pInputBuffer, "Algorithm should've made forward progress since la |
| | 1310994 | 77 | | pLastBufferPosProcessed = pInputBuffer; |
| | | 78 | | #endif |
| | | 79 | | // First, check for the common case of all-ASCII bytes. |
| | | 80 | | |
| | 1310994 | 81 | | if (Ascii.AllBytesInUInt32AreAscii(thisDWord)) |
| | | 82 | | { |
| | | 83 | | // We read an all-ASCII sequence. |
| | | 84 | | |
| | 22927 | 85 | | if (outputCharsRemaining < sizeof(uint)) |
| | | 86 | | { |
| | | 87 | | goto ProcessRemainingBytesSlow; // running out of space, but may be able to write some data |
| | | 88 | | } |
| | | 89 | | |
| | 22927 | 90 | | Ascii.WidenFourAsciiBytesToUtf16AndWriteToBuffer(ref *pOutputBuffer, thisDWord); |
| | 22927 | 91 | | pInputBuffer += 4; |
| | 22927 | 92 | | pOutputBuffer += 4; |
| | 22927 | 93 | | outputCharsRemaining -= 4; |
| | | 94 | | |
| | | 95 | | // If we saw a sequence of all ASCII, there's a good chance a significant amount of following data i |
| | | 96 | | // Below is basically unrolled loops with poor man's vectorization. |
| | | 97 | | |
| | 22927 | 98 | | uint remainingInputBytes = (uint)(void*)Unsafe.ByteOffset(ref *pInputBuffer, ref *pFinalPosWhereCanR |
| | 22927 | 99 | | uint maxIters = Math.Min(remainingInputBytes, (uint)outputCharsRemaining) / (2 * sizeof(uint)); |
| | | 100 | | uint secondDWord; |
| | | 101 | | int i; |
| | 80474 | 102 | | for (i = 0; (uint)i < maxIters; i++) |
| | | 103 | | { |
| | | 104 | | // Reading two DWORDs in parallel benchmarked faster than reading a single QWORD. |
| | | 105 | | |
| | 36408 | 106 | | thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); |
| | 36408 | 107 | | secondDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer + sizeof(uint)); |
| | | 108 | | |
| | 36408 | 109 | | if (!Ascii.AllBytesInUInt32AreAscii(thisDWord | secondDWord)) |
| | | 110 | | { |
| | | 111 | | goto LoopTerminatedEarlyDueToNonAsciiData; |
| | | 112 | | } |
| | | 113 | | |
| | 17310 | 114 | | pInputBuffer += 8; |
| | | 115 | | |
| | 17310 | 116 | | Ascii.WidenFourAsciiBytesToUtf16AndWriteToBuffer(ref pOutputBuffer[0], thisDWord); |
| | 17310 | 117 | | Ascii.WidenFourAsciiBytesToUtf16AndWriteToBuffer(ref pOutputBuffer[4], secondDWord); |
| | | 118 | | |
| | 17310 | 119 | | pOutputBuffer += 8; |
| | | 120 | | } |
| | | 121 | | |
| | 3829 | 122 | | outputCharsRemaining -= 8 * i; |
| | | 123 | | |
| | 3829 | 124 | | continue; // need to perform a bounds check because we might be running out of data |
| | | 125 | | |
| | | 126 | | LoopTerminatedEarlyDueToNonAsciiData: |
| | | 127 | | |
| | 19098 | 128 | | if (Ascii.AllBytesInUInt32AreAscii(thisDWord)) |
| | | 129 | | { |
| | | 130 | | // The first DWORD contained all-ASCII bytes, so expand it. |
| | | 131 | | |
| | 4167 | 132 | | Ascii.WidenFourAsciiBytesToUtf16AndWriteToBuffer(ref *pOutputBuffer, thisDWord); |
| | | 133 | | |
| | | 134 | | // continue the outer loop from the second DWORD |
| | | 135 | | |
| | 4167 | 136 | | Debug.Assert(!Ascii.AllBytesInUInt32AreAscii(secondDWord)); |
| | 4167 | 137 | | thisDWord = secondDWord; |
| | | 138 | | |
| | 4167 | 139 | | pInputBuffer += 4; |
| | 4167 | 140 | | pOutputBuffer += 4; |
| | 4167 | 141 | | outputCharsRemaining -= 4; |
| | | 142 | | } |
| | | 143 | | |
| | 19098 | 144 | | outputCharsRemaining -= 8 * i; |
| | | 145 | | |
| | | 146 | | // We know that there's *at least* one DWORD of data remaining in the buffer. |
| | | 147 | | // We also know that it's not all-ASCII. We can skip the logic at the beginning of the main loop. |
| | | 148 | | |
| | | 149 | | goto AfterReadDWordSkipAllBytesAsciiCheck; |
| | | 150 | | } |
| | | 151 | | |
| | | 152 | | AfterReadDWordSkipAllBytesAsciiCheck: |
| | | 153 | | |
| | 1307165 | 154 | | Debug.Assert(!Ascii.AllBytesInUInt32AreAscii(thisDWord)); // this should have been handled earlier |
| | | 155 | | |
| | | 156 | | // Next, try stripping off ASCII bytes one at a time. |
| | | 157 | | // We only handle up to three ASCII bytes here since we handled the four ASCII byte case above. |
| | | 158 | | |
| | 1307165 | 159 | | if (UInt32FirstByteIsAscii(thisDWord)) |
| | | 160 | | { |
| | 57108 | 161 | | if (outputCharsRemaining >= 3) |
| | | 162 | | { |
| | | 163 | | // Fast-track: we don't need to check the destination length for subsequent |
| | | 164 | | // ASCII bytes since we know we can write them all now. |
| | | 165 | | |
| | 56935 | 166 | | uint thisDWordLittleEndian = ToLittleEndian(thisDWord); |
| | | 167 | | |
| | 56935 | 168 | | nuint adjustment = 1; |
| | 56935 | 169 | | pOutputBuffer[0] = (char)(byte)thisDWordLittleEndian; |
| | | 170 | | |
| | 56935 | 171 | | if (UInt32SecondByteIsAscii(thisDWord)) |
| | | 172 | | { |
| | 27265 | 173 | | adjustment++; |
| | 27265 | 174 | | thisDWordLittleEndian >>= 8; |
| | 27265 | 175 | | pOutputBuffer[1] = (char)(byte)thisDWordLittleEndian; |
| | | 176 | | |
| | 27265 | 177 | | if (UInt32ThirdByteIsAscii(thisDWord)) |
| | | 178 | | { |
| | 13290 | 179 | | adjustment++; |
| | 13290 | 180 | | thisDWordLittleEndian >>= 8; |
| | 13290 | 181 | | pOutputBuffer[2] = (char)(byte)thisDWordLittleEndian; |
| | | 182 | | } |
| | | 183 | | } |
| | | 184 | | |
| | 56935 | 185 | | pInputBuffer += adjustment; |
| | 56935 | 186 | | pOutputBuffer += adjustment; |
| | 56935 | 187 | | outputCharsRemaining -= (int)adjustment; |
| | | 188 | | } |
| | | 189 | | else |
| | | 190 | | { |
| | | 191 | | // Slow-track: we need to make sure each individual write has enough |
| | | 192 | | // of a buffer so that we don't overrun the destination. |
| | | 193 | | |
| | 173 | 194 | | if (outputCharsRemaining == 0) |
| | | 195 | | { |
| | | 196 | | goto OutputBufferTooSmall; |
| | | 197 | | } |
| | | 198 | | |
| | 173 | 199 | | uint thisDWordLittleEndian = ToLittleEndian(thisDWord); |
| | | 200 | | |
| | 173 | 201 | | pInputBuffer++; |
| | 173 | 202 | | *pOutputBuffer++ = (char)(byte)thisDWordLittleEndian; |
| | 173 | 203 | | outputCharsRemaining--; |
| | | 204 | | |
| | 173 | 205 | | if (UInt32SecondByteIsAscii(thisDWord)) |
| | | 206 | | { |
| | 0 | 207 | | if (outputCharsRemaining == 0) |
| | | 208 | | { |
| | | 209 | | goto OutputBufferTooSmall; |
| | | 210 | | } |
| | | 211 | | |
| | 0 | 212 | | pInputBuffer++; |
| | 0 | 213 | | thisDWordLittleEndian >>= 8; |
| | 0 | 214 | | *pOutputBuffer++ = (char)(byte)thisDWordLittleEndian; |
| | | 215 | | |
| | | 216 | | // We can perform a small optimization here. We know at this point that |
| | | 217 | | // the output buffer is fully consumed (we read two ASCII bytes and wrote |
| | | 218 | | // two ASCII chars, and we checked earlier that the destination buffer |
| | | 219 | | // can't store a third byte). If the next byte is ASCII, we can jump straight |
| | | 220 | | // to the return statement since the end-of-method logic only relies on the |
| | | 221 | | // destination buffer pointer -- NOT the output chars remaining count -- being |
| | | 222 | | // correct. If the next byte is not ASCII, we'll need to continue with the |
| | | 223 | | // rest of the main loop, but we can set the buffer length directly to zero |
| | | 224 | | // rather than decrementing it from 1 to 0. |
| | | 225 | | |
| | 0 | 226 | | Debug.Assert(outputCharsRemaining == 1); |
| | | 227 | | |
| | 0 | 228 | | if (UInt32ThirdByteIsAscii(thisDWord)) |
| | | 229 | | { |
| | | 230 | | goto OutputBufferTooSmall; |
| | | 231 | | } |
| | | 232 | | else |
| | | 233 | | { |
| | 0 | 234 | | outputCharsRemaining = 0; |
| | | 235 | | } |
| | | 236 | | } |
| | | 237 | | } |
| | | 238 | | |
| | 57108 | 239 | | if (pInputBuffer > pFinalPosWhereCanReadDWordFromInputBuffer) |
| | | 240 | | { |
| | | 241 | | goto ProcessRemainingBytesSlow; // input buffer doesn't contain enough data to read a DWORD |
| | | 242 | | } |
| | | 243 | | else |
| | | 244 | | { |
| | | 245 | | // The input buffer at the current offset contains a non-ASCII byte. |
| | | 246 | | // Read an entire DWORD and fall through to multi-byte consumption logic. |
| | 56356 | 247 | | thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); |
| | | 248 | | } |
| | | 249 | | } |
| | | 250 | | |
| | | 251 | | BeforeProcessTwoByteSequence: |
| | | 252 | | |
| | | 253 | | // At this point, we know we're working with a multi-byte code unit, |
| | | 254 | | // but we haven't yet validated it. |
| | | 255 | | |
| | | 256 | | // The masks and comparands are derived from the Unicode Standard, Table 3-6. |
| | | 257 | | // Additionally, we need to check for valid byte sequences per Table 3-7. |
| | | 258 | | |
| | | 259 | | // Check the 2-byte case. |
| | | 260 | | |
| | 1350865 | 261 | | if (UInt32BeginsWithUtf8TwoByteMask(thisDWord)) |
| | | 262 | | { |
| | | 263 | | // Per Table 3-7, valid sequences are: |
| | | 264 | | // [ C2..DF ] [ 80..BF ] |
| | | 265 | | |
| | 148203 | 266 | | if (UInt32BeginsWithOverlongUtf8TwoByteSequence(thisDWord)) |
| | | 267 | | { |
| | | 268 | | goto Error; |
| | | 269 | | } |
| | | 270 | | |
| | | 271 | | ProcessTwoByteSequenceSkipOverlongFormCheck: |
| | | 272 | | |
| | | 273 | | // Optimization: If this is a two-byte-per-character language like Cyrillic or Hebrew, |
| | | 274 | | // there's a good chance that if we see one two-byte run then there's another two-byte |
| | | 275 | | // run immediately after. Let's check that now. |
| | | 276 | | |
| | | 277 | | // On little-endian platforms, we can check for the two-byte UTF8 mask *and* validate that |
| | | 278 | | // the value isn't overlong using a single comparison. On big-endian platforms, we'll need |
| | | 279 | | // to validate the mask and validate that the sequence isn't overlong as two separate comparisons. |
| | | 280 | | |
| | 167777 | 281 | | if ((BitConverter.IsLittleEndian && UInt32EndsWithValidUtf8TwoByteSequenceLittleEndian(thisDWord)) |
| | 167777 | 282 | | || (!BitConverter.IsLittleEndian && (UInt32EndsWithUtf8TwoByteMask(thisDWord) && !UInt32EndsWith |
| | | 283 | | { |
| | | 284 | | // We have two runs of two bytes each. |
| | | 285 | | |
| | 46603 | 286 | | if (outputCharsRemaining < 2) |
| | | 287 | | { |
| | | 288 | | goto ProcessRemainingBytesSlow; // running out of output buffer |
| | | 289 | | } |
| | | 290 | | |
| | 46603 | 291 | | Unsafe.WriteUnaligned(pOutputBuffer, ExtractTwoCharsPackedFromTwoAdjacentTwoByteSequences(thisDW |
| | | 292 | | |
| | 46603 | 293 | | pInputBuffer += 4; |
| | 46603 | 294 | | pOutputBuffer += 2; |
| | 46603 | 295 | | outputCharsRemaining -= 2; |
| | | 296 | | |
| | 46603 | 297 | | if (pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer) |
| | | 298 | | { |
| | | 299 | | // Optimization: If we read a long run of two-byte sequences, the next sequence is probably |
| | | 300 | | // also two bytes. Check for that first before going back to the beginning of the loop. |
| | | 301 | | |
| | 44027 | 302 | | thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); |
| | | 303 | | |
| | 44027 | 304 | | if (BitConverter.IsLittleEndian) |
| | | 305 | | { |
| | 44027 | 306 | | if (UInt32BeginsWithValidUtf8TwoByteSequenceLittleEndian(thisDWord)) |
| | | 307 | | { |
| | | 308 | | // The next sequence is a valid two-byte sequence. |
| | 21674 | 309 | | goto ProcessTwoByteSequenceSkipOverlongFormCheck; |
| | | 310 | | } |
| | | 311 | | } |
| | | 312 | | else |
| | | 313 | | { |
| | | 314 | | if (UInt32BeginsWithUtf8TwoByteMask(thisDWord)) |
| | | 315 | | { |
| | | 316 | | if (UInt32BeginsWithOverlongUtf8TwoByteSequence(thisDWord)) |
| | | 317 | | { |
| | | 318 | | goto Error; // The next sequence purports to be a 2-byte sequence but is overlon |
| | | 319 | | } |
| | | 320 | | |
| | | 321 | | goto ProcessTwoByteSequenceSkipOverlongFormCheck; |
| | | 322 | | } |
| | | 323 | | } |
| | | 324 | | |
| | | 325 | | // If we reached this point, the next sequence is something other than a valid |
| | | 326 | | // two-byte sequence, so go back to the beginning of the loop. |
| | | 327 | | goto AfterReadDWord; |
| | | 328 | | } |
| | | 329 | | else |
| | | 330 | | { |
| | | 331 | | goto ProcessRemainingBytesSlow; // Running out of data - go down slow path |
| | | 332 | | } |
| | | 333 | | } |
| | | 334 | | |
| | | 335 | | // The buffer contains a 2-byte sequence followed by 2 bytes that aren't a 2-byte sequence. |
| | | 336 | | // Unlikely that a 3-byte sequence would follow a 2-byte sequence, so perhaps remaining |
| | | 337 | | // bytes are ASCII? |
| | | 338 | | |
| | 121174 | 339 | | uint charToWrite = ExtractCharFromFirstTwoByteSequence(thisDWord); // optimistically compute this no |
| | | 340 | | |
| | 121174 | 341 | | if (UInt32ThirdByteIsAscii(thisDWord)) |
| | | 342 | | { |
| | 85664 | 343 | | if (UInt32FourthByteIsAscii(thisDWord)) |
| | | 344 | | { |
| | 39131 | 345 | | if (outputCharsRemaining < 3) |
| | | 346 | | { |
| | | 347 | | goto ProcessRemainingBytesSlow; // running out of output buffer |
| | | 348 | | } |
| | | 349 | | |
| | 39131 | 350 | | pOutputBuffer[0] = (char)charToWrite; |
| | 39131 | 351 | | if (BitConverter.IsLittleEndian) |
| | | 352 | | { |
| | 39131 | 353 | | thisDWord >>= 16; |
| | 39131 | 354 | | pOutputBuffer[1] = (char)(byte)thisDWord; |
| | 39131 | 355 | | thisDWord >>= 8; |
| | 39131 | 356 | | pOutputBuffer[2] = (char)thisDWord; |
| | | 357 | | } |
| | | 358 | | else |
| | | 359 | | { |
| | | 360 | | pOutputBuffer[2] = (char)(byte)thisDWord; |
| | | 361 | | pOutputBuffer[1] = (char)(byte)(thisDWord >> 8); |
| | | 362 | | } |
| | 39131 | 363 | | pInputBuffer += 4; |
| | 39131 | 364 | | pOutputBuffer += 3; |
| | 39131 | 365 | | outputCharsRemaining -= 3; |
| | | 366 | | |
| | 39131 | 367 | | continue; // go back to original bounds check and check for ASCII |
| | | 368 | | } |
| | | 369 | | else |
| | | 370 | | { |
| | 46533 | 371 | | if (outputCharsRemaining < 2) |
| | | 372 | | { |
| | | 373 | | goto ProcessRemainingBytesSlow; // running out of output buffer |
| | | 374 | | } |
| | | 375 | | |
| | 46533 | 376 | | pOutputBuffer[0] = (char)charToWrite; |
| | 46533 | 377 | | pOutputBuffer[1] = (char)(byte)(thisDWord >> (BitConverter.IsLittleEndian ? 16 : 8)); |
| | 46533 | 378 | | pInputBuffer += 3; |
| | 46533 | 379 | | pOutputBuffer += 2; |
| | 46533 | 380 | | outputCharsRemaining -= 2; |
| | | 381 | | |
| | | 382 | | // A two-byte sequence followed by an ASCII byte followed by a non-ASCII byte. |
| | | 383 | | // Read in the next DWORD and jump directly to the start of the multi-byte processing block. |
| | | 384 | | |
| | 46533 | 385 | | if (pFinalPosWhereCanReadDWordFromInputBuffer < pInputBuffer) |
| | | 386 | | { |
| | | 387 | | goto ProcessRemainingBytesSlow; // Running out of data - go down slow path |
| | | 388 | | } |
| | | 389 | | else |
| | | 390 | | { |
| | 44452 | 391 | | thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); |
| | 44452 | 392 | | goto BeforeProcessTwoByteSequence; |
| | | 393 | | } |
| | | 394 | | } |
| | | 395 | | } |
| | | 396 | | else |
| | | 397 | | { |
| | 35510 | 398 | | if (outputCharsRemaining == 0) |
| | | 399 | | { |
| | | 400 | | goto ProcessRemainingBytesSlow; // running out of output buffer |
| | | 401 | | } |
| | | 402 | | |
| | 35510 | 403 | | pOutputBuffer[0] = (char)charToWrite; |
| | 35510 | 404 | | pInputBuffer += 2; |
| | 35510 | 405 | | pOutputBuffer++; |
| | 35510 | 406 | | outputCharsRemaining--; |
| | | 407 | | |
| | 35510 | 408 | | if (pFinalPosWhereCanReadDWordFromInputBuffer < pInputBuffer) |
| | | 409 | | { |
| | | 410 | | goto ProcessRemainingBytesSlow; // Running out of data - go down slow path |
| | | 411 | | } |
| | | 412 | | else |
| | | 413 | | { |
| | 33292 | 414 | | thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); |
| | | 415 | | goto BeforeProcessThreeByteSequence; // we know the next byte isn't ASCII, and it's not the |
| | | 416 | | } |
| | | 417 | | } |
| | | 418 | | } |
| | | 419 | | |
| | | 420 | | // Check the 3-byte case. |
| | | 421 | | |
| | | 422 | | BeforeProcessThreeByteSequence: |
| | | 423 | | |
| | 1235954 | 424 | | if (UInt32BeginsWithUtf8ThreeByteMask(thisDWord)) |
| | | 425 | | { |
| | | 426 | | ProcessThreeByteSequenceWithCheck: |
| | | 427 | | |
| | | 428 | | // We need to check for overlong or surrogate three-byte sequences. |
| | | 429 | | // |
| | | 430 | | // Per Table 3-7, valid sequences are: |
| | | 431 | | // [ E0 ] [ A0..BF ] [ 80..BF ] |
| | | 432 | | // [ E1..EC ] [ 80..BF ] [ 80..BF ] |
| | | 433 | | // [ ED ] [ 80..9F ] [ 80..BF ] |
| | | 434 | | // [ EE..EF ] [ 80..BF ] [ 80..BF ] |
| | | 435 | | // |
| | | 436 | | // Big-endian examples of using the above validation table: |
| | | 437 | | // E0A0 = 1110 0000 1010 0000 => invalid (overlong ) patterns are 1110 0000 100# #### |
| | | 438 | | // ED9F = 1110 1101 1001 1111 => invalid (surrogate) patterns are 1110 1101 101# #### |
| | | 439 | | // If using the bitmask ......................................... 0000 1111 0010 0000 (=0F20), |
| | | 440 | | // Then invalid (overlong) patterns match the comparand ......... 0000 0000 0000 0000 (=0000), |
| | | 441 | | // And invalid (surrogate) patterns match the comparand ......... 0000 1101 0010 0000 (=0D20). |
| | | 442 | | |
| | 119306 | 443 | | if (BitConverter.IsLittleEndian) |
| | | 444 | | { |
| | | 445 | | // The "overlong or surrogate" check can be implemented using a single jump, but there's |
| | | 446 | | // some overhead to moving the bits into the correct locations in order to perform the |
| | | 447 | | // correct comparison, and in practice the processor's branch prediction capability is |
| | | 448 | | // good enough that we shouldn't bother. So we'll use two jumps instead. |
| | | 449 | | |
| | | 450 | | // Can't extract this check into its own helper method because JITter produces suboptimal |
| | | 451 | | // assembly, even with aggressive inlining. |
| | | 452 | | |
| | | 453 | | // Code below becomes 5 instructions: test, jz, lea, test, jz |
| | | 454 | | |
| | 119306 | 455 | | if (((thisDWord & 0x0000_200Fu) == 0) || (((thisDWord - 0x0000_200Du) & 0x0000_200Fu) == 0)) |
| | | 456 | | { |
| | 6076 | 457 | | goto Error; // overlong or surrogate |
| | | 458 | | } |
| | | 459 | | } |
| | | 460 | | else |
| | | 461 | | { |
| | | 462 | | if (((thisDWord & 0x0F20_0000u) == 0) || (((thisDWord - 0x0D20_0000u) & 0x0F20_0000u) == 0)) |
| | | 463 | | { |
| | | 464 | | goto Error; // overlong or surrogate |
| | | 465 | | } |
| | | 466 | | } |
| | | 467 | | |
| | | 468 | | // At this point, we know the incoming scalar is well-formed. |
| | | 469 | | |
| | 99936 | 470 | | if (outputCharsRemaining == 0) |
| | | 471 | | { |
| | | 472 | | goto OutputBufferTooSmall; // not enough space in the destination buffer to write |
| | | 473 | | } |
| | | 474 | | |
| | | 475 | | // As an optimization, on compatible platforms check if a second three-byte sequence immediately |
| | | 476 | | // follows the one we just read, and if so extract them together. |
| | | 477 | | |
| | 99936 | 478 | | if (BitConverter.IsLittleEndian) |
| | | 479 | | { |
| | | 480 | | // First, check that the leftover byte from the original DWORD is in the range [ E0..EF ], which |
| | | 481 | | // would indicate the potential start of a second three-byte sequence. |
| | | 482 | | |
| | 99936 | 483 | | if (((thisDWord - 0xE000_0000u) & 0xF000_0000u) == 0) |
| | | 484 | | { |
| | | 485 | | // The const '3' below is correct because pFinalPosWhereCanReadDWordFromInputBuffer represen |
| | | 486 | | // the final place where we can safely perform a DWORD read, and we want to probe whether it |
| | | 487 | | // safe to read a DWORD beginning at address &pInputBuffer[3]. |
| | | 488 | | |
| | 74241 | 489 | | if (outputCharsRemaining > 1 && (nint)(void*)Unsafe.ByteOffset(ref *pInputBuffer, ref *pFina |
| | | 490 | | { |
| | | 491 | | // We're going to attempt to read a second 3-byte sequence and write them both out one a |
| | | 492 | | // We need to check the continuation bit mask on the remaining two bytes (and we may as |
| | | 493 | | // byte mask again since it's free), then perform overlong + surrogate checks. If the ov |
| | | 494 | | // checks fail, we'll fall through to the remainder of the logic which will transcode th |
| | | 495 | | // 3-byte UTF-8 sequence we read; and on the next iteration of the loop the validation r |
| | | 496 | | // fail, and redirect control flow to the error handling logic at the very end of this m |
| | | 497 | | |
| | 72298 | 498 | | uint secondDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer + 3); |
| | | 499 | | |
| | 72298 | 500 | | if (UInt32BeginsWithUtf8ThreeByteMask(secondDWord) |
| | 72298 | 501 | | && ((secondDWord & 0x0000_200Fu) != 0) |
| | 72298 | 502 | | && (((secondDWord - 0x0000_200Du) & 0x0000_200Fu) != 0)) |
| | | 503 | | { |
| | 56829 | 504 | | pOutputBuffer[0] = (char)ExtractCharFromFirstThreeByteSequence(thisDWord); |
| | 56829 | 505 | | pOutputBuffer[1] = (char)ExtractCharFromFirstThreeByteSequence(secondDWord); |
| | 56829 | 506 | | pInputBuffer += 6; |
| | 56829 | 507 | | pOutputBuffer += 2; |
| | 56829 | 508 | | outputCharsRemaining -= 2; |
| | | 509 | | |
| | | 510 | | // Drain any ASCII data following the second three-byte sequence. |
| | | 511 | | |
| | 56829 | 512 | | goto CheckForAsciiByteAfterThreeByteSequence; |
| | | 513 | | } |
| | | 514 | | } |
| | | 515 | | } |
| | | 516 | | } |
| | | 517 | | |
| | | 518 | | // Couldn't extract 2x three-byte sequences together, just do this one by itself. |
| | | 519 | | |
| | 43107 | 520 | | *pOutputBuffer = (char)ExtractCharFromFirstThreeByteSequence(thisDWord); |
| | 43107 | 521 | | pInputBuffer += 3; |
| | 43107 | 522 | | pOutputBuffer++; |
| | 43107 | 523 | | outputCharsRemaining--; |
| | | 524 | | |
| | | 525 | | CheckForAsciiByteAfterThreeByteSequence: |
| | | 526 | | |
| | | 527 | | // Occasionally one-off ASCII characters like spaces, periods, or newlines will make their way |
| | | 528 | | // in to the text. If this happens strip it off now before seeing if the next character |
| | | 529 | | // consists of three code units. |
| | | 530 | | |
| | 99936 | 531 | | if (UInt32FourthByteIsAscii(thisDWord)) |
| | | 532 | | { |
| | 17488 | 533 | | if (outputCharsRemaining == 0) |
| | | 534 | | { |
| | | 535 | | goto OutputBufferTooSmall; |
| | | 536 | | } |
| | | 537 | | |
| | 17488 | 538 | | if (BitConverter.IsLittleEndian) |
| | | 539 | | { |
| | 17488 | 540 | | *pOutputBuffer = (char)(thisDWord >> 24); |
| | | 541 | | } |
| | | 542 | | else |
| | | 543 | | { |
| | | 544 | | *pOutputBuffer = (char)(byte)thisDWord; |
| | | 545 | | } |
| | | 546 | | |
| | 17488 | 547 | | pInputBuffer++; |
| | 17488 | 548 | | pOutputBuffer++; |
| | 17488 | 549 | | outputCharsRemaining--; |
| | | 550 | | } |
| | | 551 | | |
| | 99936 | 552 | | if (pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer) |
| | | 553 | | { |
| | 95410 | 554 | | thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); |
| | | 555 | | |
| | | 556 | | // Optimization: A three-byte character could indicate CJK text, which makes it likely |
| | | 557 | | // that the character following this one is also CJK. We'll check for a three-byte sequence |
| | | 558 | | // marker now and jump directly to three-byte sequence processing if we see one, skipping |
| | | 559 | | // all of the logic at the beginning of the loop. |
| | | 560 | | |
| | 95410 | 561 | | if (UInt32BeginsWithUtf8ThreeByteMask(thisDWord)) |
| | | 562 | | { |
| | 59449 | 563 | | goto ProcessThreeByteSequenceWithCheck; // found a three-byte sequence marker; validate and |
| | | 564 | | } |
| | | 565 | | else |
| | | 566 | | { |
| | | 567 | | goto AfterReadDWord; // probably ASCII punctuation or whitespace |
| | | 568 | | } |
| | | 569 | | } |
| | | 570 | | else |
| | | 571 | | { |
| | | 572 | | goto ProcessRemainingBytesSlow; // Running out of data - go down slow path |
| | | 573 | | } |
| | | 574 | | } |
| | | 575 | | |
| | | 576 | | // Assume the 4-byte case, but we need to validate. |
| | | 577 | | |
| | | 578 | | { |
| | | 579 | | // We need to check for overlong or invalid (over U+10FFFF) four-byte sequences. |
| | | 580 | | // |
| | | 581 | | // Per Table 3-7, valid sequences are: |
| | | 582 | | // [ F0 ] [ 90..BF ] [ 80..BF ] [ 80..BF ] |
| | | 583 | | // [ F1..F3 ] [ 80..BF ] [ 80..BF ] [ 80..BF ] |
| | | 584 | | // [ F4 ] [ 80..8F ] [ 80..BF ] [ 80..BF ] |
| | | 585 | | |
| | 1176097 | 586 | | if (!UInt32BeginsWithUtf8FourByteMask(thisDWord)) |
| | | 587 | | { |
| | | 588 | | goto Error; |
| | | 589 | | } |
| | | 590 | | |
| | | 591 | | // Now check for overlong / out-of-range sequences. |
| | | 592 | | |
| | 10658 | 593 | | if (BitConverter.IsLittleEndian) |
| | | 594 | | { |
| | | 595 | | // The DWORD we read is [ 10xxxxxx 10yyyyyy 10zzzzzz 11110www ]. |
| | | 596 | | // We want to get the 'w' byte in front of the 'z' byte so that we can perform |
| | | 597 | | // a single range comparison. We'll take advantage of the fact that the JITter |
| | | 598 | | // can detect a ROR / ROL operation, then we'll just zero out the bytes that |
| | | 599 | | // aren't involved in the range check. |
| | | 600 | | |
| | 10658 | 601 | | uint toCheck = thisDWord & 0x0000_FFFFu; |
| | | 602 | | |
| | | 603 | | // At this point, toCheck = [ 00000000 00000000 10zzzzzz 11110www ]. |
| | | 604 | | |
| | 10658 | 605 | | toCheck = BitOperations.RotateRight(toCheck, 8); |
| | | 606 | | |
| | | 607 | | // At this point, toCheck = [ 11110www 00000000 00000000 10zzzzzz ]. |
| | | 608 | | |
| | 10658 | 609 | | if (!UnicodeUtility.IsInRangeInclusive(toCheck, 0xF000_0090u, 0xF400_008Fu)) |
| | | 610 | | { |
| | 448 | 611 | | goto Error; |
| | | 612 | | } |
| | | 613 | | } |
| | | 614 | | else |
| | | 615 | | { |
| | | 616 | | if (!UnicodeUtility.IsInRangeInclusive(thisDWord, 0xF090_0000u, 0xF48F_FFFFu)) |
| | | 617 | | { |
| | | 618 | | goto Error; |
| | | 619 | | } |
| | | 620 | | } |
| | | 621 | | |
| | | 622 | | // Validation complete. |
| | | 623 | | |
| | 10210 | 624 | | if (outputCharsRemaining < 2) |
| | | 625 | | { |
| | | 626 | | // There's no point to falling back to the "drain the input buffer" logic, since we know |
| | | 627 | | // we can't write anything to the destination. So we'll just exit immediately. |
| | | 628 | | goto OutputBufferTooSmall; |
| | | 629 | | } |
| | | 630 | | |
| | 10210 | 631 | | Unsafe.WriteUnaligned(pOutputBuffer, ExtractCharsFromFourByteSequence(thisDWord)); |
| | | 632 | | |
| | 10210 | 633 | | pInputBuffer += 4; |
| | 10210 | 634 | | pOutputBuffer += 2; |
| | 10210 | 635 | | outputCharsRemaining -= 2; |
| | | 636 | | |
| | | 637 | | continue; // go back to beginning of loop for processing |
| | | 638 | | } |
| | 53170 | 639 | | } while (pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer); |
| | | 640 | | |
| | | 641 | | ProcessRemainingBytesSlow: |
| | 17716 | 642 | | inputLength = (int)(void*)Unsafe.ByteOffset(ref *pInputBuffer, ref *pFinalPosWhereCanReadDWordFromInputBuffe |
| | | 643 | | |
| | | 644 | | ProcessInputOfLessThanDWordSize: |
| | 1033555 | 645 | | while (inputLength > 0) |
| | | 646 | | { |
| | 997076 | 647 | | uint firstByte = pInputBuffer[0]; |
| | 997076 | 648 | | if (firstByte <= 0x7Fu) |
| | | 649 | | { |
| | 10488 | 650 | | if (outputCharsRemaining == 0) |
| | | 651 | | { |
| | | 652 | | goto OutputBufferTooSmall; // we have no hope of writing anything to the output |
| | | 653 | | } |
| | | 654 | | |
| | | 655 | | // 1-byte (ASCII) case |
| | 10488 | 656 | | *pOutputBuffer = (char)firstByte; |
| | | 657 | | |
| | 10488 | 658 | | pInputBuffer++; |
| | 10488 | 659 | | pOutputBuffer++; |
| | 10488 | 660 | | inputLength--; |
| | 10488 | 661 | | outputCharsRemaining--; |
| | 10488 | 662 | | continue; |
| | | 663 | | } |
| | | 664 | | |
| | | 665 | | // Potentially the start of a multi-byte sequence? |
| | | 666 | | |
| | 986588 | 667 | | firstByte -= 0xC2u; |
| | 986588 | 668 | | if ((byte)firstByte <= (0xDFu - 0xC2u)) |
| | | 669 | | { |
| | | 670 | | // Potentially a 2-byte sequence? |
| | 417142 | 671 | | if (inputLength < 2) |
| | | 672 | | { |
| | | 673 | | goto InputBufferTooSmall; // out of data |
| | | 674 | | } |
| | | 675 | | |
| | 161076 | 676 | | uint secondByte = pInputBuffer[1]; |
| | 161076 | 677 | | if (!IsLowByteUtf8ContinuationByte(secondByte)) |
| | | 678 | | { |
| | | 679 | | goto Error; // 2-byte marker not followed by continuation byte |
| | | 680 | | } |
| | | 681 | | |
| | 30301 | 682 | | if (outputCharsRemaining == 0) |
| | | 683 | | { |
| | | 684 | | goto OutputBufferTooSmall; // we have no hope of writing anything to the output |
| | | 685 | | } |
| | | 686 | | |
| | 30301 | 687 | | uint asChar = (firstByte << 6) + secondByte + ((0xC2u - 0xC0u) << 6) - 0x80u; // remove UTF-8 marker |
| | 30301 | 688 | | *pOutputBuffer = (char)asChar; |
| | | 689 | | |
| | 30301 | 690 | | pInputBuffer += 2; |
| | 30301 | 691 | | pOutputBuffer++; |
| | 30301 | 692 | | inputLength -= 2; |
| | 30301 | 693 | | outputCharsRemaining--; |
| | 30301 | 694 | | continue; |
| | | 695 | | } |
| | 569446 | 696 | | else if ((byte)firstByte <= (0xEFu - 0xC2u)) |
| | | 697 | | { |
| | | 698 | | // Potentially a 3-byte sequence? |
| | 121760 | 699 | | if (inputLength >= 3) |
| | | 700 | | { |
| | 20541 | 701 | | uint secondByte = pInputBuffer[1]; |
| | 20541 | 702 | | uint thirdByte = pInputBuffer[2]; |
| | 20541 | 703 | | if (!IsLowByteUtf8ContinuationByte(secondByte) || !IsLowByteUtf8ContinuationByte(thirdByte)) |
| | | 704 | | { |
| | | 705 | | goto Error; // 3-byte marker not followed by 2 continuation bytes |
| | | 706 | | } |
| | | 707 | | |
| | | 708 | | // To speed up the validation logic below, we're not going to remove the UTF-8 markers from the |
| | | 709 | | // We account for this in the comparisons below. |
| | | 710 | | |
| | 8029 | 711 | | uint partialChar = (firstByte << 12) + (secondByte << 6); |
| | 8029 | 712 | | if (partialChar < ((0xE0u - 0xC2u) << 12) + (0xA0u << 6)) |
| | | 713 | | { |
| | | 714 | | goto Error; // this is an overlong encoding; fail |
| | | 715 | | } |
| | | 716 | | |
| | 6185 | 717 | | partialChar -= ((0xEDu - 0xC2u) << 12) + (0xA0u << 6); // if partialChar = 0, we're at beginning |
| | 6185 | 718 | | if (partialChar < 0x0800u /* number of code points in UTF-16 surrogate code point range */) |
| | | 719 | | { |
| | | 720 | | goto Error; // attempted to encode a UTF-16 surrogate code point; fail |
| | | 721 | | } |
| | | 722 | | |
| | 5181 | 723 | | if (outputCharsRemaining == 0) |
| | | 724 | | { |
| | | 725 | | goto OutputBufferTooSmall; // we have no hope of writing anything to the output |
| | | 726 | | } |
| | | 727 | | |
| | | 728 | | // Now restore the full scalar value. |
| | | 729 | | |
| | 5181 | 730 | | partialChar += thirdByte; |
| | 5181 | 731 | | partialChar += 0xD800; // undo "move to beginning of UTF-16 surrogate code point range" from ear |
| | 5181 | 732 | | partialChar -= 0x80u; // remove third byte continuation marker |
| | | 733 | | |
| | 5181 | 734 | | *pOutputBuffer = (char)partialChar; |
| | | 735 | | |
| | 5181 | 736 | | pInputBuffer += 3; |
| | 5181 | 737 | | pOutputBuffer++; |
| | 5181 | 738 | | inputLength -= 3; |
| | 5181 | 739 | | outputCharsRemaining--; |
| | 5181 | 740 | | continue; |
| | | 741 | | } |
| | 101219 | 742 | | else if (inputLength >= 2) |
| | | 743 | | { |
| | 34433 | 744 | | uint secondByte = pInputBuffer[1]; |
| | 34433 | 745 | | if (!IsLowByteUtf8ContinuationByte(secondByte)) |
| | | 746 | | { |
| | | 747 | | goto Error; // 3-byte marker not followed by continuation byte |
| | | 748 | | } |
| | | 749 | | |
| | | 750 | | // We can't build up the entire scalar value now, but we can check for overlong / surrogate repr |
| | | 751 | | // from just the first two bytes. |
| | | 752 | | |
| | 19523 | 753 | | uint partialChar = (firstByte << 6) + secondByte; // don't worry about fixing up the UTF-8 marke |
| | 19523 | 754 | | if (partialChar < ((0xE0u - 0xC2u) << 6) + 0xA0u) |
| | | 755 | | { |
| | | 756 | | goto Error; // failed overlong check |
| | | 757 | | } |
| | 16243 | 758 | | if (UnicodeUtility.IsInRangeInclusive(partialChar, ((0xEDu - 0xC2u) << 6) + 0xA0u, ((0xEEu - 0xC |
| | | 759 | | { |
| | 1631 | 760 | | goto Error; // failed surrogate check |
| | | 761 | | } |
| | | 762 | | } |
| | | 763 | | |
| | | 764 | | goto InputBufferTooSmall; // out of data |
| | | 765 | | } |
| | 447686 | 766 | | else if ((byte)firstByte <= (0xF4u - 0xC2u)) |
| | | 767 | | { |
| | | 768 | | // Potentially a 4-byte sequence? |
| | | 769 | | |
| | 61199 | 770 | | if (inputLength < 2) |
| | | 771 | | { |
| | | 772 | | goto InputBufferTooSmall; // ran out of data |
| | | 773 | | } |
| | | 774 | | |
| | 28795 | 775 | | uint nextByte = pInputBuffer[1]; |
| | 28795 | 776 | | if (!IsLowByteUtf8ContinuationByte(nextByte)) |
| | | 777 | | { |
| | | 778 | | goto Error; // 4-byte marker not followed by a continuation byte |
| | | 779 | | } |
| | | 780 | | |
| | 13234 | 781 | | uint asPartialChar = (firstByte << 6) + nextByte; // don't worry about fixing up the UTF-8 markers; |
| | 13234 | 782 | | if (!UnicodeUtility.IsInRangeInclusive(asPartialChar, ((0xF0u - 0xC2u) << 6) + 0x90u, ((0xF4u - 0xC2 |
| | | 783 | | { |
| | | 784 | | goto Error; // failed overlong / out-of-range check |
| | | 785 | | } |
| | | 786 | | |
| | 12775 | 787 | | if (inputLength < 3) |
| | | 788 | | { |
| | | 789 | | goto InputBufferTooSmall; // ran out of data |
| | | 790 | | } |
| | | 791 | | |
| | 5864 | 792 | | if (!IsLowByteUtf8ContinuationByte(pInputBuffer[2])) |
| | | 793 | | { |
| | | 794 | | goto Error; // third byte in 4-byte sequence not a continuation byte |
| | | 795 | | } |
| | | 796 | | |
| | 3721 | 797 | | if (inputLength < 4) |
| | | 798 | | { |
| | | 799 | | goto InputBufferTooSmall; // ran out of data |
| | | 800 | | } |
| | | 801 | | |
| | 0 | 802 | | if (!IsLowByteUtf8ContinuationByte(pInputBuffer[3])) |
| | | 803 | | { |
| | 0 | 804 | | goto Error; // fourth byte in 4-byte sequence not a continuation byte |
| | | 805 | | } |
| | | 806 | | |
| | | 807 | | // If we read a valid astral scalar value, the only way we could've fallen down this code path |
| | | 808 | | // is that we didn't have enough output buffer to write the result. |
| | | 809 | | |
| | | 810 | | goto OutputBufferTooSmall; |
| | | 811 | | } |
| | | 812 | | else |
| | | 813 | | { |
| | | 814 | | goto Error; // didn't begin with [ C2 .. F4 ], so invalid multi-byte sequence header byte |
| | | 815 | | } |
| | | 816 | | } |
| | | 817 | | |
| | 36479 | 818 | | OperationStatus retVal = OperationStatus.Done; |
| | 36479 | 819 | | goto ReturnCommon; |
| | | 820 | | |
| | | 821 | | InputBufferTooSmall: |
| | 380500 | 822 | | retVal = OperationStatus.NeedMoreData; |
| | 380500 | 823 | | goto ReturnCommon; |
| | | 824 | | |
| | | 825 | | OutputBufferTooSmall: |
| | 0 | 826 | | retVal = OperationStatus.DestinationTooSmall; |
| | 0 | 827 | | goto ReturnCommon; |
| | | 828 | | |
| | | 829 | | Error: |
| | 1757963 | 830 | | retVal = OperationStatus.InvalidData; |
| | | 831 | | goto ReturnCommon; |
| | | 832 | | |
| | | 833 | | ReturnCommon: |
| | 2174942 | 834 | | pInputBufferRemaining = pInputBuffer; |
| | 2174942 | 835 | | pOutputBufferRemaining = pOutputBuffer; |
| | 2174942 | 836 | | return retVal; |
| | | 837 | | } |
| | | 838 | | |
| | | 839 | | // On method return, pInputBufferRemaining and pOutputBufferRemaining will both point to where |
| | | 840 | | // the next char would have been consumed from / the next byte would have been written to. |
| | | 841 | | // inputLength in chars, outputBytesRemaining in bytes. |
| | | 842 | | public static OperationStatus TranscodeToUtf8(char* pInputBuffer, int inputLength, byte* pOutputBuffer, int outp |
| | | 843 | | { |
| | | 844 | | const int CharsPerDWord = sizeof(uint) / sizeof(char); |
| | | 845 | | |
| | | 846 | | Debug.Assert(inputLength >= 0, "Input length must not be negative."); |
| | 4105 | 847 | | Debug.Assert(pInputBuffer != null || inputLength == 0, "Input length must be zero if input buffer pointer is |
| | | 848 | | |
| | 4105 | 849 | | Debug.Assert(outputBytesRemaining >= 0, "Destination length must not be negative."); |
| | 4105 | 850 | | Debug.Assert(pOutputBuffer != null || outputBytesRemaining == 0, "Destination length must be zero if destina |
| | | 851 | | |
| | | 852 | | // First, try vectorized conversion. |
| | | 853 | | |
| | | 854 | | { |
| | 4105 | 855 | | nuint numElementsConverted = Ascii.NarrowUtf16ToAscii(pInputBuffer, pOutputBuffer, (uint)Math.Min(inputL |
| | | 856 | | |
| | 4105 | 857 | | pInputBuffer += numElementsConverted; |
| | 4105 | 858 | | pOutputBuffer += numElementsConverted; |
| | | 859 | | |
| | | 860 | | // Quick check - did we just end up consuming the entire input buffer? |
| | | 861 | | // If so, short-circuit the remainder of the method. |
| | | 862 | | |
| | 4105 | 863 | | if ((int)numElementsConverted == inputLength) |
| | | 864 | | { |
| | 676 | 865 | | pInputBufferRemaining = pInputBuffer; |
| | 676 | 866 | | pOutputBufferRemaining = pOutputBuffer; |
| | 676 | 867 | | return OperationStatus.Done; |
| | | 868 | | } |
| | | 869 | | |
| | 3429 | 870 | | inputLength -= (int)numElementsConverted; |
| | 3429 | 871 | | outputBytesRemaining -= (int)numElementsConverted; |
| | | 872 | | } |
| | | 873 | | |
| | 3429 | 874 | | if (inputLength < CharsPerDWord) |
| | | 875 | | { |
| | | 876 | | goto ProcessInputOfLessThanDWordSize; |
| | | 877 | | } |
| | | 878 | | |
| | 3349 | 879 | | char* pFinalPosWhereCanReadDWordFromInputBuffer = pInputBuffer + (uint)inputLength - CharsPerDWord; |
| | | 880 | | |
| | | 881 | | // We have paths for SSE4.1 vectorization inside the inner loop. Since the below |
| | | 882 | | // vector is only used in those code paths, we leave it uninitialized if SSE4.1 |
| | | 883 | | // is not enabled. |
| | | 884 | | |
| | | 885 | | #if NET |
| | | 886 | | Vector128<short> nonAsciiUtf16DataMask; |
| | | 887 | | |
| | 3349 | 888 | | if (Sse41.X64.IsSupported || (AdvSimd.Arm64.IsSupported && BitConverter.IsLittleEndian) || PackedSimd.IsSupp |
| | | 889 | | { |
| | 3349 | 890 | | nonAsciiUtf16DataMask = Vector128.Create(unchecked((short)0xFF80)); // mask of non-ASCII bits in a UTF-1 |
| | | 891 | | } |
| | | 892 | | #endif |
| | | 893 | | |
| | | 894 | | // Begin the main loop. |
| | | 895 | | |
| | | 896 | | #if DEBUG |
| | 3349 | 897 | | char* pLastBufferPosProcessed = null; // used for invariant checking in debug builds |
| | | 898 | | #endif |
| | | 899 | | |
| | | 900 | | uint thisDWord; |
| | | 901 | | |
| | 3349 | 902 | | Debug.Assert(pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer); |
| | | 903 | | do |
| | | 904 | | { |
| | | 905 | | // Read 32 bits at a time. This is enough to hold any possible UTF16-encoded scalar. |
| | | 906 | | |
| | 12251 | 907 | | thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); |
| | | 908 | | |
| | | 909 | | AfterReadDWord: |
| | | 910 | | |
| | | 911 | | #if DEBUG |
| | 29736 | 912 | | Debug.Assert(pLastBufferPosProcessed < pInputBuffer, "Algorithm should've made forward progress since la |
| | 29736 | 913 | | pLastBufferPosProcessed = pInputBuffer; |
| | | 914 | | #endif |
| | | 915 | | |
| | | 916 | | // First, check for the common case of all-ASCII chars. |
| | | 917 | | |
| | 29736 | 918 | | if (Utf16Utility.AllCharsInUInt32AreAscii(thisDWord)) |
| | | 919 | | { |
| | | 920 | | // We read an all-ASCII sequence (2 chars). |
| | | 921 | | |
| | 10475 | 922 | | if (outputBytesRemaining < 2) |
| | | 923 | | { |
| | | 924 | | goto ProcessOneCharFromCurrentDWordAndFinish; // running out of space, but may be able to write |
| | | 925 | | } |
| | | 926 | | |
| | | 927 | | // The high WORD of the local declared below might be populated with garbage |
| | | 928 | | // as a result of our shifts below, but that's ok since we're only going to |
| | | 929 | | // write the low WORD. |
| | | 930 | | // |
| | | 931 | | // [ 00000000 0bbbbbbb | 00000000 0aaaaaaa ] -> [ 00000000 0bbbbbbb | 0bbbbbbb 0aaaaaaa ] |
| | | 932 | | // (Same logic works regardless of endianness.) |
| | 10475 | 933 | | uint valueToWrite = thisDWord | (thisDWord >> 8); |
| | | 934 | | |
| | 10475 | 935 | | Unsafe.WriteUnaligned(pOutputBuffer, (ushort)valueToWrite); |
| | | 936 | | |
| | 10475 | 937 | | pInputBuffer += 2; |
| | 10475 | 938 | | pOutputBuffer += 2; |
| | 10475 | 939 | | outputBytesRemaining -= 2; |
| | | 940 | | |
| | | 941 | | // If we saw a sequence of all ASCII, there's a good chance a significant amount of following data i |
| | | 942 | | // Below is basically unrolled loops with poor man's vectorization. |
| | | 943 | | |
| | 10475 | 944 | | uint inputCharsRemaining = (uint)(pFinalPosWhereCanReadDWordFromInputBuffer - pInputBuffer) + 2; |
| | 10475 | 945 | | uint minElementsRemaining = (uint)Math.Min(inputCharsRemaining, outputBytesRemaining); |
| | | 946 | | |
| | | 947 | | #if NET |
| | 10475 | 948 | | if (Sse41.X64.IsSupported || (AdvSimd.Arm64.IsSupported && BitConverter.IsLittleEndian) || PackedSim |
| | | 949 | | { |
| | | 950 | | // Try reading and writing 8 elements per iteration. |
| | 10475 | 951 | | uint maxIters = minElementsRemaining / 8; |
| | | 952 | | ulong possibleNonAsciiQWord; |
| | | 953 | | int i; |
| | | 954 | | Vector128<short> utf16Data; |
| | 32798 | 955 | | for (i = 0; (uint)i < maxIters; i++) |
| | | 956 | | { |
| | | 957 | | // The trimmer won't trim out nonAsciiUtf16DataMask unless this is in the loop. |
| | | 958 | | // Luckily, this is a nop and will be elided by the JIT |
| | 14672 | 959 | | Unsafe.SkipInit(out nonAsciiUtf16DataMask); |
| | | 960 | | |
| | 14672 | 961 | | utf16Data = Unsafe.ReadUnaligned<Vector128<short>>(pInputBuffer); |
| | | 962 | | |
| | | 963 | | if (AdvSimd.Arm64.IsSupported) |
| | | 964 | | { |
| | | 965 | | Vector128<short> isUtf16DataNonAscii = AdvSimd.CompareTest(utf16Data, nonAsciiUtf16DataM |
| | | 966 | | bool hasNonAsciiDataInVector = AdvSimd.Arm64.MinPairwise(isUtf16DataNonAscii, isUtf16Dat |
| | | 967 | | |
| | | 968 | | if (hasNonAsciiDataInVector) |
| | | 969 | | { |
| | | 970 | | goto LoopTerminatedDueToNonAsciiDataInVectorLocal; |
| | | 971 | | } |
| | | 972 | | |
| | | 973 | | Vector64<byte> lower = AdvSimd.ExtractNarrowingSaturateUnsignedLower(utf16Data); |
| | | 974 | | AdvSimd.Store(pOutputBuffer, lower); |
| | | 975 | | } |
| | 14672 | 976 | | else if (Sse41.IsSupported) |
| | | 977 | | { |
| | 14672 | 978 | | if ((utf16Data & nonAsciiUtf16DataMask) != Vector128<short>.Zero) |
| | | 979 | | { |
| | | 980 | | goto LoopTerminatedDueToNonAsciiDataInVectorLocal; |
| | | 981 | | } |
| | | 982 | | |
| | | 983 | | // narrow and write |
| | 5924 | 984 | | Sse2.StoreScalar((ulong*)pOutputBuffer /* unaligned */, Sse2.PackUnsignedSaturate(utf16D |
| | | 985 | | } |
| | | 986 | | else if (PackedSimd.IsSupported) |
| | | 987 | | { |
| | | 988 | | if ((utf16Data & nonAsciiUtf16DataMask) != Vector128<short>.Zero) |
| | | 989 | | { |
| | | 990 | | goto LoopTerminatedDueToNonAsciiDataInVectorLocal; |
| | | 991 | | } |
| | | 992 | | |
| | | 993 | | // narrow and write low 8 bytes |
| | | 994 | | Vector128<byte> narrowed = PackedSimd.ConvertNarrowingSaturateUnsigned(utf16Data, utf16D |
| | | 995 | | Unsafe.WriteUnaligned<ulong>(pOutputBuffer, narrowed.AsUInt64().ToScalar()); |
| | | 996 | | } |
| | | 997 | | else |
| | | 998 | | { |
| | | 999 | | // We explicitly recheck each IsSupported query to ensure that the trimmer can see which |
| | 0 | 1000 | | ThrowHelper.ThrowUnreachableException(); |
| | | 1001 | | } |
| | | 1002 | | |
| | 5924 | 1003 | | pInputBuffer += 8; |
| | 5924 | 1004 | | pOutputBuffer += 8; |
| | | 1005 | | } |
| | | 1006 | | |
| | 1727 | 1007 | | outputBytesRemaining -= 8 * i; |
| | | 1008 | | |
| | | 1009 | | // Can we perform one more iteration, but reading & writing 4 elements instead of 8? |
| | | 1010 | | |
| | 1727 | 1011 | | if ((minElementsRemaining & 4) != 0) |
| | | 1012 | | { |
| | 713 | 1013 | | possibleNonAsciiQWord = Unsafe.ReadUnaligned<ulong>(pInputBuffer); |
| | 713 | 1014 | | if (!Utf16Utility.AllCharsInUInt64AreAscii(possibleNonAsciiQWord)) |
| | | 1015 | | { |
| | | 1016 | | goto LoopTerminatedDueToNonAsciiDataInPossibleNonAsciiQWordLocal; |
| | | 1017 | | } |
| | | 1018 | | |
| | 344 | 1019 | | utf16Data = Vector128.CreateScalarUnsafe(possibleNonAsciiQWord).AsInt16(); |
| | | 1020 | | |
| | | 1021 | | if (AdvSimd.IsSupported) |
| | | 1022 | | { |
| | | 1023 | | Vector64<byte> lower = AdvSimd.ExtractNarrowingSaturateUnsignedLower(utf16Data); |
| | | 1024 | | AdvSimd.StoreSelectedScalar((uint*)pOutputBuffer, lower.AsUInt32(), 0); |
| | | 1025 | | } |
| | 344 | 1026 | | else if (Sse2.IsSupported) |
| | | 1027 | | { |
| | 344 | 1028 | | Unsafe.WriteUnaligned(pOutputBuffer, Sse2.ConvertToUInt32(Sse2.PackUnsignedSaturate(utf1 |
| | | 1029 | | } |
| | | 1030 | | else if (PackedSimd.IsSupported) |
| | | 1031 | | { |
| | | 1032 | | Vector128<byte> narrowed = PackedSimd.ConvertNarrowingSaturateUnsigned(utf16Data, utf16D |
| | | 1033 | | Unsafe.WriteUnaligned<uint>(pOutputBuffer, narrowed.AsUInt32().ToScalar()); |
| | | 1034 | | } |
| | | 1035 | | else |
| | | 1036 | | { |
| | | 1037 | | // We explicitly recheck each IsSupported query to ensure that the trimmer can see which |
| | 0 | 1038 | | ThrowHelper.ThrowUnreachableException(); |
| | | 1039 | | } |
| | | 1040 | | |
| | 344 | 1041 | | pInputBuffer += 4; |
| | 344 | 1042 | | pOutputBuffer += 4; |
| | 344 | 1043 | | outputBytesRemaining -= 4; |
| | | 1044 | | } |
| | | 1045 | | |
| | 344 | 1046 | | continue; // Go back to beginning of main loop, read data, check for ASCII |
| | | 1047 | | |
| | | 1048 | | LoopTerminatedDueToNonAsciiDataInVectorLocal: |
| | | 1049 | | |
| | 8748 | 1050 | | outputBytesRemaining -= 8 * i; |
| | | 1051 | | |
| | 8748 | 1052 | | if (Sse2.X64.IsSupported) |
| | | 1053 | | { |
| | 8748 | 1054 | | possibleNonAsciiQWord = Sse2.X64.ConvertToUInt64(utf16Data.AsUInt64()); |
| | | 1055 | | } |
| | | 1056 | | else |
| | | 1057 | | { |
| | 0 | 1058 | | possibleNonAsciiQWord = utf16Data.AsUInt64().ToScalar(); |
| | | 1059 | | } |
| | | 1060 | | |
| | | 1061 | | // Temporarily set 'possibleNonAsciiQWord' to be the low 64 bits of the vector, |
| | | 1062 | | // then check whether it's all-ASCII. If so, narrow and write to the destination |
| | | 1063 | | // buffer. Since we know that either the high 64 bits or the low 64 bits of the |
| | | 1064 | | // vector contains non-ASCII data, by the end of the following block the |
| | | 1065 | | // 'possibleNonAsciiQWord' local is guaranteed to contain the non-ASCII segment. |
| | | 1066 | | |
| | 8748 | 1067 | | if (Utf16Utility.AllCharsInUInt64AreAscii(possibleNonAsciiQWord)) // all chars in first QWORD ar |
| | | 1068 | | { |
| | | 1069 | | if (AdvSimd.IsSupported) |
| | | 1070 | | { |
| | | 1071 | | Vector64<byte> lower = AdvSimd.ExtractNarrowingSaturateUnsignedLower(utf16Data); |
| | | 1072 | | AdvSimd.StoreSelectedScalar((uint*)pOutputBuffer, lower.AsUInt32(), 0); |
| | | 1073 | | } |
| | 1757 | 1074 | | else if (Sse2.IsSupported) |
| | | 1075 | | { |
| | 1757 | 1076 | | Unsafe.WriteUnaligned(pOutputBuffer, Sse2.ConvertToUInt32(Sse2.PackUnsignedSaturate(utf1 |
| | | 1077 | | } |
| | | 1078 | | else if (PackedSimd.IsSupported) |
| | | 1079 | | { |
| | | 1080 | | Vector128<byte> narrowed = PackedSimd.ConvertNarrowingSaturateUnsigned(utf16Data, utf16D |
| | | 1081 | | Unsafe.WriteUnaligned<uint>(pOutputBuffer, narrowed.AsUInt32().ToScalar()); |
| | | 1082 | | } |
| | | 1083 | | else |
| | | 1084 | | { |
| | | 1085 | | // We explicitly recheck each IsSupported query to ensure that the trimmer can see which |
| | 0 | 1086 | | ThrowHelper.ThrowUnreachableException(); |
| | | 1087 | | } |
| | 1757 | 1088 | | pInputBuffer += 4; |
| | 1757 | 1089 | | pOutputBuffer += 4; |
| | 1757 | 1090 | | outputBytesRemaining -= 4; |
| | 1757 | 1091 | | possibleNonAsciiQWord = utf16Data.AsUInt64().GetElement(1); |
| | | 1092 | | } |
| | | 1093 | | |
| | | 1094 | | LoopTerminatedDueToNonAsciiDataInPossibleNonAsciiQWordLocal: |
| | | 1095 | | |
| | 9117 | 1096 | | Debug.Assert(!Utf16Utility.AllCharsInUInt64AreAscii(possibleNonAsciiQWord)); // this condition s |
| | | 1097 | | |
| | 9117 | 1098 | | thisDWord = (uint)possibleNonAsciiQWord; |
| | 9117 | 1099 | | if (Utf16Utility.AllCharsInUInt32AreAscii(thisDWord)) |
| | | 1100 | | { |
| | | 1101 | | // [ 00000000 0bbbbbbb | 00000000 0aaaaaaa ] -> [ 00000000 0bbbbbbb | 0bbbbbbb 0aaaaaaa ] |
| | 3605 | 1102 | | Unsafe.WriteUnaligned(pOutputBuffer, (ushort)(thisDWord | (thisDWord >> 8))); |
| | 3605 | 1103 | | pInputBuffer += 2; |
| | 3605 | 1104 | | pOutputBuffer += 2; |
| | 3605 | 1105 | | outputBytesRemaining -= 2; |
| | 3605 | 1106 | | thisDWord = (uint)(possibleNonAsciiQWord >> 32); |
| | | 1107 | | } |
| | | 1108 | | |
| | 3605 | 1109 | | goto AfterReadDWordSkipAllCharsAsciiCheck; |
| | | 1110 | | } |
| | | 1111 | | else |
| | | 1112 | | #endif |
| | | 1113 | | { |
| | | 1114 | | // Can't use SSE41 x64, so we'll only read and write 4 elements per iteration. |
| | 0 | 1115 | | uint maxIters = minElementsRemaining / 4; |
| | | 1116 | | uint secondDWord; |
| | | 1117 | | int i; |
| | 0 | 1118 | | for (i = 0; (uint)i < maxIters; i++) |
| | | 1119 | | { |
| | 0 | 1120 | | thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); |
| | 0 | 1121 | | secondDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer + 2); |
| | | 1122 | | |
| | 0 | 1123 | | if (!Utf16Utility.AllCharsInUInt32AreAscii(thisDWord | secondDWord)) |
| | | 1124 | | { |
| | | 1125 | | goto LoopTerminatedDueToNonAsciiData; |
| | | 1126 | | } |
| | | 1127 | | |
| | | 1128 | | // [ 00000000 0bbbbbbb | 00000000 0aaaaaaa ] -> [ 00000000 0bbbbbbb | 0bbbbbbb 0aaaaaaa ] |
| | | 1129 | | // (Same logic works regardless of endianness.) |
| | 0 | 1130 | | Unsafe.WriteUnaligned(pOutputBuffer, (ushort)(thisDWord | (thisDWord >> 8))); |
| | 0 | 1131 | | Unsafe.WriteUnaligned(pOutputBuffer + 2, (ushort)(secondDWord | (secondDWord >> 8))); |
| | | 1132 | | |
| | 0 | 1133 | | pInputBuffer += 4; |
| | 0 | 1134 | | pOutputBuffer += 4; |
| | | 1135 | | } |
| | | 1136 | | |
| | 0 | 1137 | | outputBytesRemaining -= 4 * i; |
| | | 1138 | | |
| | 0 | 1139 | | continue; // Go back to beginning of main loop, read data, check for ASCII |
| | | 1140 | | |
| | | 1141 | | LoopTerminatedDueToNonAsciiData: |
| | | 1142 | | |
| | 0 | 1143 | | outputBytesRemaining -= 4 * i; |
| | | 1144 | | |
| | | 1145 | | // First, see if we can drain any ASCII data from the first DWORD. |
| | | 1146 | | |
| | 0 | 1147 | | if (Utf16Utility.AllCharsInUInt32AreAscii(thisDWord)) |
| | | 1148 | | { |
| | | 1149 | | // [ 00000000 0bbbbbbb | 00000000 0aaaaaaa ] -> [ 00000000 0bbbbbbb | 0bbbbbbb 0aaaaaaa ] |
| | | 1150 | | // (Same logic works regardless of endianness.) |
| | 0 | 1151 | | Unsafe.WriteUnaligned(pOutputBuffer, (ushort)(thisDWord | (thisDWord >> 8))); |
| | 0 | 1152 | | pInputBuffer += 2; |
| | 0 | 1153 | | pOutputBuffer += 2; |
| | 0 | 1154 | | outputBytesRemaining -= 2; |
| | 0 | 1155 | | thisDWord = secondDWord; |
| | | 1156 | | } |
| | | 1157 | | |
| | | 1158 | | goto AfterReadDWordSkipAllCharsAsciiCheck; |
| | | 1159 | | } |
| | | 1160 | | } |
| | | 1161 | | |
| | | 1162 | | AfterReadDWordSkipAllCharsAsciiCheck: |
| | | 1163 | | |
| | 31982 | 1164 | | Debug.Assert(!Utf16Utility.AllCharsInUInt32AreAscii(thisDWord)); // this should have been handled earlie |
| | | 1165 | | |
| | | 1166 | | // Next, try stripping off the first ASCII char if it exists. |
| | | 1167 | | // We don't check for a second ASCII char since that should have been handled above. |
| | | 1168 | | |
| | 31982 | 1169 | | if (IsFirstCharAscii(thisDWord)) |
| | | 1170 | | { |
| | 13238 | 1171 | | if (outputBytesRemaining == 0) |
| | | 1172 | | { |
| | | 1173 | | goto OutputBufferTooSmall; |
| | | 1174 | | } |
| | | 1175 | | |
| | 13238 | 1176 | | if (BitConverter.IsLittleEndian) |
| | | 1177 | | { |
| | 13238 | 1178 | | pOutputBuffer[0] = (byte)thisDWord; // extract [ ## ## 00 AA ] |
| | | 1179 | | } |
| | | 1180 | | else |
| | | 1181 | | { |
| | | 1182 | | pOutputBuffer[0] = (byte)(thisDWord >> 16); // extract [ 00 AA ## ## ] |
| | | 1183 | | } |
| | | 1184 | | |
| | 13238 | 1185 | | pInputBuffer++; |
| | 13238 | 1186 | | pOutputBuffer++; |
| | 13238 | 1187 | | outputBytesRemaining--; |
| | | 1188 | | |
| | 13238 | 1189 | | if (pInputBuffer > pFinalPosWhereCanReadDWordFromInputBuffer) |
| | | 1190 | | { |
| | | 1191 | | goto ProcessNextCharAndFinish; // input buffer doesn't contain enough data to read a DWORD |
| | | 1192 | | } |
| | | 1193 | | else |
| | | 1194 | | { |
| | | 1195 | | // The input buffer at the current offset contains a non-ASCII char. |
| | | 1196 | | // Read an entire DWORD and fall through to non-ASCII consumption logic. |
| | 13024 | 1197 | | thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); |
| | | 1198 | | } |
| | | 1199 | | } |
| | | 1200 | | |
| | | 1201 | | // At this point, we know the first char in the buffer is non-ASCII, but we haven't yet validated it. |
| | | 1202 | | |
| | 31768 | 1203 | | if (!IsFirstCharAtLeastThreeUtf8Bytes(thisDWord)) |
| | | 1204 | | { |
| | | 1205 | | TryConsumeMultipleTwoByteSequences: |
| | | 1206 | | |
| | | 1207 | | // For certain text (Greek, Cyrillic, ...), 2-byte sequences tend to be clustered. We'll try transco |
| | | 1208 | | // a tight loop without falling back to the main loop. |
| | | 1209 | | |
| | 14507 | 1210 | | if (IsSecondCharTwoUtf8Bytes(thisDWord)) |
| | | 1211 | | { |
| | | 1212 | | // We have two runs of two bytes each. |
| | | 1213 | | |
| | 3811 | 1214 | | if (outputBytesRemaining < 4) |
| | | 1215 | | { |
| | | 1216 | | goto ProcessOneCharFromCurrentDWordAndFinish; // running out of output buffer |
| | | 1217 | | } |
| | | 1218 | | |
| | 3811 | 1219 | | Unsafe.WriteUnaligned(pOutputBuffer, ExtractTwoUtf8TwoByteSequencesFromTwoPackedUtf16Chars(thisD |
| | | 1220 | | |
| | 3811 | 1221 | | pInputBuffer += 2; |
| | 3811 | 1222 | | pOutputBuffer += 4; |
| | 3811 | 1223 | | outputBytesRemaining -= 4; |
| | | 1224 | | |
| | 3811 | 1225 | | if (pInputBuffer > pFinalPosWhereCanReadDWordFromInputBuffer) |
| | | 1226 | | { |
| | | 1227 | | goto ProcessNextCharAndFinish; // Running out of data - go down slow path |
| | | 1228 | | } |
| | | 1229 | | else |
| | | 1230 | | { |
| | | 1231 | | // Optimization: If we read a long run of two-byte sequences, the next sequence is probably |
| | | 1232 | | // also two bytes. Check for that first before going back to the beginning of the loop. |
| | | 1233 | | |
| | 3748 | 1234 | | thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); |
| | | 1235 | | |
| | 3748 | 1236 | | if (IsFirstCharTwoUtf8Bytes(thisDWord)) |
| | | 1237 | | { |
| | | 1238 | | // Validated we have a two-byte sequence coming up |
| | 1840 | 1239 | | goto TryConsumeMultipleTwoByteSequences; |
| | | 1240 | | } |
| | | 1241 | | |
| | | 1242 | | // If we reached this point, the next sequence is something other than a valid |
| | | 1243 | | // two-byte sequence, so go back to the beginning of the loop. |
| | | 1244 | | goto AfterReadDWord; |
| | | 1245 | | } |
| | | 1246 | | } |
| | | 1247 | | |
| | 10696 | 1248 | | if (outputBytesRemaining < 2) |
| | | 1249 | | { |
| | | 1250 | | goto OutputBufferTooSmall; |
| | | 1251 | | } |
| | | 1252 | | |
| | 10696 | 1253 | | Unsafe.WriteUnaligned(pOutputBuffer, (ushort)ExtractUtf8TwoByteSequenceFromFirstUtf16Char(thisDWord) |
| | | 1254 | | |
| | | 1255 | | // The buffer contains a 2-byte sequence followed by 2 bytes that aren't a 2-byte sequence. |
| | | 1256 | | // Unlikely that a 3-byte sequence would follow a 2-byte sequence, so perhaps remaining |
| | | 1257 | | // char is ASCII? |
| | | 1258 | | |
| | 10696 | 1259 | | if (IsSecondCharAscii(thisDWord)) |
| | | 1260 | | { |
| | 7683 | 1261 | | if (outputBytesRemaining >= 3) |
| | | 1262 | | { |
| | 7683 | 1263 | | if (BitConverter.IsLittleEndian) |
| | | 1264 | | { |
| | 7683 | 1265 | | thisDWord >>= 16; |
| | | 1266 | | } |
| | 7683 | 1267 | | pOutputBuffer[2] = (byte)thisDWord; |
| | | 1268 | | |
| | 7683 | 1269 | | pInputBuffer += 2; |
| | 7683 | 1270 | | pOutputBuffer += 3; |
| | 7683 | 1271 | | outputBytesRemaining -= 3; |
| | | 1272 | | |
| | 7683 | 1273 | | continue; // go back to original bounds check and check for ASCII |
| | | 1274 | | } |
| | | 1275 | | else |
| | | 1276 | | { |
| | 0 | 1277 | | pInputBuffer++; |
| | 0 | 1278 | | pOutputBuffer += 2; |
| | 0 | 1279 | | goto OutputBufferTooSmall; |
| | | 1280 | | } |
| | | 1281 | | } |
| | | 1282 | | else |
| | | 1283 | | { |
| | 3013 | 1284 | | pInputBuffer++; |
| | 3013 | 1285 | | pOutputBuffer += 2; |
| | 3013 | 1286 | | outputBytesRemaining -= 2; |
| | | 1287 | | |
| | 3013 | 1288 | | if (pInputBuffer > pFinalPosWhereCanReadDWordFromInputBuffer) |
| | | 1289 | | { |
| | | 1290 | | goto ProcessNextCharAndFinish; // Running out of data - go down slow path |
| | | 1291 | | } |
| | | 1292 | | else |
| | | 1293 | | { |
| | 2891 | 1294 | | thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); |
| | | 1295 | | goto BeforeProcessThreeByteSequence; // we know the next byte isn't ASCII, and it's not the |
| | | 1296 | | } |
| | | 1297 | | } |
| | | 1298 | | } |
| | | 1299 | | |
| | | 1300 | | // Check the 3-byte case. |
| | | 1301 | | |
| | | 1302 | | BeforeProcessThreeByteSequence: |
| | | 1303 | | |
| | 62180 | 1304 | | if (!IsFirstCharSurrogate(thisDWord)) |
| | | 1305 | | { |
| | | 1306 | | // Optimization: A three-byte character could indicate CJK text, which makes it likely |
| | | 1307 | | // that the character following this one is also CJK. We'll perform the check now |
| | | 1308 | | // rather than jumping to the beginning of the main loop. |
| | | 1309 | | |
| | 61329 | 1310 | | if (IsSecondCharAtLeastThreeUtf8Bytes(thisDWord)) |
| | | 1311 | | { |
| | 44773 | 1312 | | if (!IsSecondCharSurrogate(thisDWord)) |
| | | 1313 | | { |
| | 44196 | 1314 | | if (outputBytesRemaining < 6) |
| | | 1315 | | { |
| | | 1316 | | goto ConsumeSingleThreeByteRun; // not enough space - try consuming as much as we can |
| | | 1317 | | } |
| | | 1318 | | |
| | 44196 | 1319 | | WriteTwoUtf16CharsAsTwoUtf8ThreeByteSequences(ref *pOutputBuffer, thisDWord); |
| | | 1320 | | |
| | 44196 | 1321 | | pInputBuffer += 2; |
| | 44196 | 1322 | | pOutputBuffer += 6; |
| | 44196 | 1323 | | outputBytesRemaining -= 6; |
| | | 1324 | | |
| | | 1325 | | // Try to remain in the 3-byte processing loop if at all possible. |
| | | 1326 | | |
| | 44196 | 1327 | | if (pInputBuffer > pFinalPosWhereCanReadDWordFromInputBuffer) |
| | | 1328 | | { |
| | | 1329 | | goto ProcessNextCharAndFinish; // Running out of data - go down slow path |
| | | 1330 | | } |
| | | 1331 | | else |
| | | 1332 | | { |
| | 42804 | 1333 | | thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); |
| | | 1334 | | |
| | 42804 | 1335 | | if (IsFirstCharAtLeastThreeUtf8Bytes(thisDWord)) |
| | | 1336 | | { |
| | 34790 | 1337 | | goto BeforeProcessThreeByteSequence; |
| | | 1338 | | } |
| | | 1339 | | else |
| | | 1340 | | { |
| | | 1341 | | // Fall back to standard processing loop since we don't know how to optimize this. |
| | | 1342 | | goto AfterReadDWord; |
| | | 1343 | | } |
| | | 1344 | | } |
| | | 1345 | | } |
| | | 1346 | | } |
| | | 1347 | | |
| | | 1348 | | ConsumeSingleThreeByteRun: |
| | | 1349 | | |
| | 17133 | 1350 | | if (outputBytesRemaining < 3) |
| | | 1351 | | { |
| | | 1352 | | goto OutputBufferTooSmall; |
| | | 1353 | | } |
| | | 1354 | | |
| | 17133 | 1355 | | WriteFirstUtf16CharAsUtf8ThreeByteSequence(ref *pOutputBuffer, thisDWord); |
| | | 1356 | | |
| | 17133 | 1357 | | pInputBuffer++; |
| | 17133 | 1358 | | pOutputBuffer += 3; |
| | 17133 | 1359 | | outputBytesRemaining -= 3; |
| | | 1360 | | |
| | | 1361 | | // Occasionally one-off ASCII characters like spaces, periods, or newlines will make their way |
| | | 1362 | | // in to the text. If this happens strip it off now before seeing if the next character |
| | | 1363 | | // consists of three code units. |
| | | 1364 | | |
| | 17133 | 1365 | | if (IsSecondCharAscii(thisDWord)) |
| | | 1366 | | { |
| | 13483 | 1367 | | if (outputBytesRemaining == 0) |
| | | 1368 | | { |
| | | 1369 | | goto OutputBufferTooSmall; |
| | | 1370 | | } |
| | | 1371 | | |
| | 13483 | 1372 | | if (BitConverter.IsLittleEndian) |
| | | 1373 | | { |
| | 13483 | 1374 | | *pOutputBuffer = (byte)(thisDWord >> 16); |
| | | 1375 | | } |
| | | 1376 | | else |
| | | 1377 | | { |
| | | 1378 | | *pOutputBuffer = (byte)(thisDWord); |
| | | 1379 | | } |
| | | 1380 | | |
| | 13483 | 1381 | | pInputBuffer++; |
| | 13483 | 1382 | | pOutputBuffer++; |
| | 13483 | 1383 | | outputBytesRemaining--; |
| | | 1384 | | |
| | 13483 | 1385 | | if (pInputBuffer > pFinalPosWhereCanReadDWordFromInputBuffer) |
| | | 1386 | | { |
| | | 1387 | | goto ProcessNextCharAndFinish; // Running out of data - go down slow path |
| | | 1388 | | } |
| | | 1389 | | else |
| | | 1390 | | { |
| | 12961 | 1391 | | thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); |
| | | 1392 | | |
| | 12961 | 1393 | | if (IsFirstCharAtLeastThreeUtf8Bytes(thisDWord)) |
| | | 1394 | | { |
| | 5398 | 1395 | | goto BeforeProcessThreeByteSequence; |
| | | 1396 | | } |
| | | 1397 | | else |
| | | 1398 | | { |
| | | 1399 | | // Fall back to standard processing loop since we don't know how to optimize this. |
| | | 1400 | | goto AfterReadDWord; |
| | | 1401 | | } |
| | | 1402 | | } |
| | | 1403 | | } |
| | | 1404 | | |
| | 3650 | 1405 | | if (pInputBuffer > pFinalPosWhereCanReadDWordFromInputBuffer) |
| | | 1406 | | { |
| | | 1407 | | goto ProcessNextCharAndFinish; // Running out of data - go down slow path |
| | | 1408 | | } |
| | | 1409 | | else |
| | | 1410 | | { |
| | 3604 | 1411 | | thisDWord = Unsafe.ReadUnaligned<uint>(pInputBuffer); |
| | 3604 | 1412 | | goto AfterReadDWordSkipAllCharsAsciiCheck; // we just checked above that this value isn't ASCII |
| | | 1413 | | } |
| | | 1414 | | } |
| | | 1415 | | |
| | | 1416 | | // Four byte sequence processing |
| | | 1417 | | |
| | 851 | 1418 | | if (IsWellFormedUtf16SurrogatePair(thisDWord)) |
| | | 1419 | | { |
| | 851 | 1420 | | if (outputBytesRemaining < 4) |
| | | 1421 | | { |
| | | 1422 | | goto OutputBufferTooSmall; |
| | | 1423 | | } |
| | | 1424 | | |
| | 851 | 1425 | | Unsafe.WriteUnaligned(pOutputBuffer, ExtractFourUtf8BytesFromSurrogatePair(thisDWord)); |
| | | 1426 | | |
| | 851 | 1427 | | pInputBuffer += 2; |
| | 851 | 1428 | | pOutputBuffer += 4; |
| | 851 | 1429 | | outputBytesRemaining -= 4; |
| | | 1430 | | |
| | | 1431 | | continue; // go back to beginning of loop for processing |
| | | 1432 | | } |
| | | 1433 | | |
| | | 1434 | | goto Error; // an ill-formed surrogate sequence: high not followed by low, or low not preceded by high |
| | 9892 | 1435 | | } while (pInputBuffer <= pFinalPosWhereCanReadDWordFromInputBuffer); |
| | | 1436 | | |
| | | 1437 | | ProcessNextCharAndFinish: |
| | 3349 | 1438 | | inputLength = (int)(pFinalPosWhereCanReadDWordFromInputBuffer - pInputBuffer) + CharsPerDWord; |
| | | 1439 | | |
| | | 1440 | | ProcessInputOfLessThanDWordSize: |
| | 3429 | 1441 | | Debug.Assert(inputLength < CharsPerDWord); |
| | | 1442 | | |
| | 3429 | 1443 | | if (inputLength == 0) |
| | | 1444 | | { |
| | | 1445 | | goto InputBufferFullyConsumed; |
| | | 1446 | | } |
| | | 1447 | | |
| | 1745 | 1448 | | uint thisChar = *pInputBuffer; |
| | 1745 | 1449 | | goto ProcessFinalChar; |
| | | 1450 | | |
| | | 1451 | | ProcessOneCharFromCurrentDWordAndFinish: |
| | 0 | 1452 | | if (BitConverter.IsLittleEndian) |
| | | 1453 | | { |
| | 0 | 1454 | | thisChar = thisDWord & 0xFFFFu; // preserve only the first char |
| | | 1455 | | } |
| | | 1456 | | else |
| | | 1457 | | { |
| | | 1458 | | thisChar = thisDWord >> 16; // preserve only the first char |
| | | 1459 | | } |
| | | 1460 | | |
| | | 1461 | | ProcessFinalChar: |
| | | 1462 | | { |
| | 1745 | 1463 | | if (thisChar <= 0x7Fu) |
| | | 1464 | | { |
| | 505 | 1465 | | if (outputBytesRemaining == 0) |
| | | 1466 | | { |
| | | 1467 | | goto OutputBufferTooSmall; // we have no hope of writing anything to the output |
| | | 1468 | | } |
| | | 1469 | | |
| | | 1470 | | // 1-byte (ASCII) case |
| | 505 | 1471 | | *pOutputBuffer = (byte)thisChar; |
| | | 1472 | | |
| | 505 | 1473 | | pInputBuffer++; |
| | 505 | 1474 | | pOutputBuffer++; |
| | | 1475 | | } |
| | 1240 | 1476 | | else if (thisChar < 0x0800u) |
| | | 1477 | | { |
| | 108 | 1478 | | if (outputBytesRemaining < 2) |
| | | 1479 | | { |
| | | 1480 | | goto OutputBufferTooSmall; // we have no hope of writing anything to the output |
| | | 1481 | | } |
| | | 1482 | | |
| | | 1483 | | // 2-byte case |
| | 108 | 1484 | | pOutputBuffer[1] = (byte)((thisChar & 0x3Fu) | unchecked((uint)(sbyte)0x80)); // [ 10xxxxxx ] |
| | 108 | 1485 | | pOutputBuffer[0] = (byte)((thisChar >> 6) | unchecked((uint)(sbyte)0xC0)); // [ 110yyyyy ] |
| | | 1486 | | |
| | 108 | 1487 | | pInputBuffer++; |
| | 108 | 1488 | | pOutputBuffer += 2; |
| | | 1489 | | } |
| | 1132 | 1490 | | else if (!UnicodeUtility.IsSurrogateCodePoint(thisChar)) |
| | | 1491 | | { |
| | 1132 | 1492 | | if (outputBytesRemaining < 3) |
| | | 1493 | | { |
| | | 1494 | | goto OutputBufferTooSmall; // we have no hope of writing anything to the output |
| | | 1495 | | } |
| | | 1496 | | |
| | | 1497 | | // 3-byte case |
| | 1132 | 1498 | | pOutputBuffer[2] = (byte)((thisChar & 0x3Fu) | unchecked((uint)(sbyte)0x80)); // [ 10xxxxxx ] |
| | 1132 | 1499 | | pOutputBuffer[1] = (byte)(((thisChar >> 6) & 0x3Fu) | unchecked((uint)(sbyte)0x80)); // [ 10yyyyyy ] |
| | 1132 | 1500 | | pOutputBuffer[0] = (byte)((thisChar >> 12) | unchecked((uint)(sbyte)0xE0)); // [ 1110zzzz ] |
| | | 1501 | | |
| | 1132 | 1502 | | pInputBuffer++; |
| | 1132 | 1503 | | pOutputBuffer += 3; |
| | | 1504 | | } |
| | 0 | 1505 | | else if (thisChar <= 0xDBFFu) |
| | | 1506 | | { |
| | | 1507 | | // UTF-16 high surrogate code point with no trailing data, report incomplete input buffer |
| | 0 | 1508 | | goto InputBufferTooSmall; |
| | | 1509 | | } |
| | | 1510 | | else |
| | | 1511 | | { |
| | | 1512 | | // UTF-16 low surrogate code point with no leading data, report error |
| | | 1513 | | goto Error; |
| | | 1514 | | } |
| | | 1515 | | } |
| | | 1516 | | |
| | | 1517 | | // There are two ways we can end up here. Either we were running low on input data, |
| | | 1518 | | // or we were running low on space in the destination buffer. If we're running low on |
| | | 1519 | | // input data (label targets ProcessInputOfLessThanDWordSize and ProcessNextCharAndFinish), |
| | | 1520 | | // then the inputLength value is guaranteed to be between 0 and 1, and we should return Done. |
| | | 1521 | | // If we're running low on destination buffer space (label target ProcessOneCharFromCurrentDWordAndFinish), |
| | | 1522 | | // then we didn't modify inputLength since entering the main loop, which means it should |
| | | 1523 | | // still have a value of >= 2. So checking the value of inputLength is all we need to do to determine |
| | | 1524 | | // which of the two scenarios we're in. |
| | | 1525 | | |
| | 1745 | 1526 | | if (inputLength > 1) |
| | | 1527 | | { |
| | | 1528 | | goto OutputBufferTooSmall; |
| | | 1529 | | } |
| | | 1530 | | |
| | | 1531 | | InputBufferFullyConsumed: |
| | 3429 | 1532 | | OperationStatus retVal = OperationStatus.Done; |
| | 3429 | 1533 | | goto ReturnCommon; |
| | | 1534 | | |
| | | 1535 | | InputBufferTooSmall: |
| | 0 | 1536 | | retVal = OperationStatus.NeedMoreData; |
| | 0 | 1537 | | goto ReturnCommon; |
| | | 1538 | | |
| | | 1539 | | OutputBufferTooSmall: |
| | 0 | 1540 | | retVal = OperationStatus.DestinationTooSmall; |
| | 0 | 1541 | | goto ReturnCommon; |
| | | 1542 | | |
| | | 1543 | | Error: |
| | 0 | 1544 | | retVal = OperationStatus.InvalidData; |
| | | 1545 | | goto ReturnCommon; |
| | | 1546 | | |
| | | 1547 | | ReturnCommon: |
| | 3429 | 1548 | | pInputBufferRemaining = pInputBuffer; |
| | 3429 | 1549 | | pOutputBufferRemaining = pOutputBuffer; |
| | 3429 | 1550 | | return retVal; |
| | | 1551 | | } |
| | | 1552 | | } |
| | | 1553 | | } |
| | | 1554 | | |