| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | // The worker functions in this file was optimized for performance. If you make changes |
| | | 5 | | // you should use care to consider all of the interesting cases. |
| | | 6 | | |
| | | 7 | | // The code of all worker functions in this file is written twice: Once as a slow loop, and the |
| | | 8 | | // second time as a fast loop. The slow loops handles all special cases, throws exceptions, etc. |
| | | 9 | | // The fast loops attempts to blaze through as fast as possible with optimistic range checks, |
| | | 10 | | // processing multiple characters at a time, and falling back to the slow loop for all special cases. |
| | | 11 | | |
| | | 12 | | using System.Buffers; |
| | | 13 | | using System.Diagnostics; |
| | | 14 | | using System.Diagnostics.CodeAnalysis; |
| | | 15 | | using System.Runtime.CompilerServices; |
| | | 16 | | using System.Runtime.InteropServices; |
| | | 17 | | using System.Text.Unicode; |
| | | 18 | | |
| | | 19 | | namespace System.Text |
| | | 20 | | { |
| | | 21 | | // Encodes text into and out of UTF-8. UTF-8 is a way of writing |
| | | 22 | | // Unicode characters with variable numbers of bytes per character, |
| | | 23 | | // optimized for the lower 127 ASCII characters. It's an efficient way |
| | | 24 | | // of encoding US English in an internationalizable way. |
| | | 25 | | // |
| | | 26 | | // Don't override IsAlwaysNormalized because it is just a Unicode Transformation and could be confused. |
| | | 27 | | // |
| | | 28 | | // The UTF-8 byte order mark is simply the Unicode byte order mark |
| | | 29 | | // (0xFEFF) written in UTF-8 (0xEF 0xBB 0xBF). The byte order mark is |
| | | 30 | | // used mostly to distinguish UTF-8 text from other encodings, and doesn't |
| | | 31 | | // switch the byte orderings. |
| | | 32 | | |
| | | 33 | | public partial class UTF8Encoding : Encoding |
| | | 34 | | { |
| | | 35 | | /* |
| | | 36 | | bytes bits UTF-8 representation |
| | | 37 | | ----- ---- ----------------------------------- |
| | | 38 | | 1 7 0vvvvvvv |
| | | 39 | | 2 11 110vvvvv 10vvvvvv |
| | | 40 | | 3 16 1110vvvv 10vvvvvv 10vvvvvv |
| | | 41 | | 4 21 11110vvv 10vvvvvv 10vvvvvv 10vvvvvv |
| | | 42 | | ----- ---- ----------------------------------- |
| | | 43 | | |
| | | 44 | | Surrogate: |
| | | 45 | | Real Unicode value = (HighSurrogate - 0xD800) * 0x400 + (LowSurrogate - 0xDC00) + 0x10000 |
| | | 46 | | */ |
| | | 47 | | |
| | | 48 | | private const int UTF8_CODEPAGE = 65001; |
| | | 49 | | |
| | | 50 | | /// <summary> |
| | | 51 | | /// Transcoding to UTF-8 bytes from UTF-16 input chars will result in a maximum 3:1 expansion. |
| | | 52 | | /// </summary> |
| | | 53 | | /// <remarks> |
| | | 54 | | /// Supplementary code points are expanded to UTF-8 from UTF-16 at a 4:2 ratio, |
| | | 55 | | /// so 3:1 is still the correct value for maximum expansion. |
| | | 56 | | /// </remarks> |
| | | 57 | | private const int MaxUtf8BytesPerChar = 3; |
| | | 58 | | |
| | | 59 | | |
| | | 60 | | // Used by Encoding.UTF8 for lazy initialization |
| | | 61 | | // The initialization code will not be run until a static member of the class is referenced |
| | 1 | 62 | | internal static readonly UTF8EncodingSealed s_default = new UTF8EncodingSealed(encoderShouldEmitUTF8Identifier: |
| | | 63 | | |
| | 1 | 64 | | internal static ReadOnlySpan<byte> PreambleSpan => [0xEF, 0xBB, 0xBF]; |
| | | 65 | | |
| | | 66 | | // Yes, the idea of emitting U+FEFF as a UTF-8 identifier has made it into |
| | | 67 | | // the standard. |
| | | 68 | | private readonly bool _emitUTF8Identifier; |
| | | 69 | | |
| | | 70 | | private readonly bool _isThrowException; |
| | | 71 | | |
| | | 72 | | |
| | | 73 | | public UTF8Encoding() : |
| | 10211 | 74 | | base(UTF8_CODEPAGE) |
| | | 75 | | { |
| | 10211 | 76 | | } |
| | | 77 | | |
| | | 78 | | |
| | | 79 | | public UTF8Encoding(bool encoderShouldEmitUTF8Identifier) : |
| | 3405 | 80 | | this() |
| | | 81 | | { |
| | 3405 | 82 | | _emitUTF8Identifier = encoderShouldEmitUTF8Identifier; |
| | 3405 | 83 | | } |
| | | 84 | | |
| | | 85 | | |
| | | 86 | | public UTF8Encoding(bool encoderShouldEmitUTF8Identifier, bool throwOnInvalidBytes) : |
| | 3403 | 87 | | this(encoderShouldEmitUTF8Identifier) |
| | | 88 | | { |
| | 3403 | 89 | | _isThrowException = throwOnInvalidBytes; |
| | | 90 | | |
| | | 91 | | // Encoding's constructor already did this, but it'll be wrong if we're throwing exceptions |
| | 3403 | 92 | | if (_isThrowException) |
| | 3403 | 93 | | SetDefaultFallbacks(); |
| | 3403 | 94 | | } |
| | | 95 | | |
| | | 96 | | internal sealed override void SetDefaultFallbacks() |
| | | 97 | | { |
| | | 98 | | // For UTF-X encodings, we use a replacement fallback with an empty string |
| | 13614 | 99 | | if (_isThrowException) |
| | | 100 | | { |
| | 3403 | 101 | | this.encoderFallback = EncoderFallback.ExceptionFallback; |
| | 3403 | 102 | | this.decoderFallback = DecoderFallback.ExceptionFallback; |
| | | 103 | | } |
| | | 104 | | else |
| | | 105 | | { |
| | 10211 | 106 | | this.encoderFallback = new EncoderReplacementFallback("\xFFFD"); |
| | 10211 | 107 | | this.decoderFallback = new DecoderReplacementFallback("\xFFFD"); |
| | | 108 | | } |
| | 10211 | 109 | | } |
| | | 110 | | |
| | | 111 | | |
| | | 112 | | // WARNING: GetByteCount(string chars) |
| | | 113 | | // WARNING: has different variable names than EncodingNLS.cs, so this can't just be cut & pasted, |
| | | 114 | | // WARNING: otherwise it'll break VB's way of declaring these. |
| | | 115 | | // |
| | | 116 | | // The following methods are copied from EncodingNLS.cs. |
| | | 117 | | // Unfortunately EncodingNLS.cs is internal and we're public, so we have to re-implement them here. |
| | | 118 | | // These should be kept in sync for the following classes: |
| | | 119 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 120 | | |
| | | 121 | | // Returns the number of bytes required to encode a range of characters in |
| | | 122 | | // a character array. |
| | | 123 | | // |
| | | 124 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 125 | | // So if you fix this, fix the others. Currently those include: |
| | | 126 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 127 | | // parent method is safe |
| | | 128 | | |
| | | 129 | | public override unsafe int GetByteCount(char[] chars, int index, int count) |
| | | 130 | | { |
| | 0 | 131 | | if (chars is null) |
| | | 132 | | { |
| | 0 | 133 | | ThrowHelper.ThrowArgumentNullException(ExceptionArgument.chars, ExceptionResource.ArgumentNull_Array); |
| | | 134 | | } |
| | | 135 | | |
| | 0 | 136 | | if ((index | count) < 0) |
| | | 137 | | { |
| | 0 | 138 | | ThrowHelper.ThrowArgumentOutOfRangeException((index < 0) ? ExceptionArgument.index : ExceptionArgument.c |
| | | 139 | | } |
| | | 140 | | |
| | 0 | 141 | | if (chars.Length - index < count) |
| | | 142 | | { |
| | 0 | 143 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.chars, ExceptionResource.ArgumentOutOfRan |
| | | 144 | | } |
| | | 145 | | |
| | 0 | 146 | | fixed (char* pChars = chars) |
| | | 147 | | { |
| | 0 | 148 | | return GetByteCountCommon(pChars + index, count); |
| | | 149 | | } |
| | | 150 | | } |
| | | 151 | | |
| | | 152 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 153 | | // So if you fix this, fix the others. Currently those include: |
| | | 154 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 155 | | // parent method is safe |
| | | 156 | | |
| | | 157 | | public override unsafe int GetByteCount(string chars) |
| | | 158 | | { |
| | 16 | 159 | | if (chars is null) |
| | | 160 | | { |
| | 0 | 161 | | ThrowHelper.ThrowArgumentNullException(ExceptionArgument.chars); |
| | | 162 | | } |
| | | 163 | | |
| | 16 | 164 | | fixed (char* pChars = chars) |
| | | 165 | | { |
| | 16 | 166 | | return GetByteCountCommon(pChars, chars.Length); |
| | | 167 | | } |
| | | 168 | | } |
| | | 169 | | |
| | | 170 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 171 | | // So if you fix this, fix the others. Currently those include: |
| | | 172 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 173 | | |
| | | 174 | | [CLSCompliant(false)] |
| | | 175 | | public override unsafe int GetByteCount(char* chars, int count) |
| | | 176 | | { |
| | 0 | 177 | | if (chars is null) |
| | | 178 | | { |
| | 0 | 179 | | ThrowHelper.ThrowArgumentNullException(ExceptionArgument.chars); |
| | | 180 | | } |
| | | 181 | | |
| | 0 | 182 | | if (count < 0) |
| | | 183 | | { |
| | 0 | 184 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.count, ExceptionResource.ArgumentOutOfRan |
| | | 185 | | } |
| | | 186 | | |
| | 0 | 187 | | return GetByteCountCommon(chars, count); |
| | | 188 | | } |
| | | 189 | | |
| | | 190 | | public override unsafe int GetByteCount(ReadOnlySpan<char> chars) |
| | 0 | 191 | | { |
| | | 192 | | // It's ok for us to pass null pointers down to the workhorse below. |
| | | 193 | | |
| | 0 | 194 | | fixed (char* charsPtr = &MemoryMarshal.GetReference(chars)) |
| | | 195 | | { |
| | 0 | 196 | | return GetByteCountCommon(charsPtr, chars.Length); |
| | | 197 | | } |
| | | 198 | | } |
| | | 199 | | |
| | | 200 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 201 | | private unsafe int GetByteCountCommon(char* pChars, int charCount) |
| | | 202 | | { |
| | | 203 | | // Common helper method for all non-EncoderNLS entry points to GetByteCount. |
| | | 204 | | // A modification of this method should be copied in to each of the supported encodings: ASCII, UTF8, UTF16, |
| | | 205 | | |
| | 16 | 206 | | Debug.Assert(charCount >= 0, "Caller shouldn't specify negative length buffer."); |
| | 16 | 207 | | Debug.Assert(pChars is not null || charCount == 0, "Input pointer shouldn't be null if non-zero length speci |
| | | 208 | | |
| | | 209 | | // First call into the fast path. |
| | | 210 | | // Don't bother providing a fallback mechanism; our fast path doesn't use it. |
| | | 211 | | |
| | 16 | 212 | | int totalByteCount = GetByteCountFast(pChars, charCount, fallback: null, out int charsConsumed); |
| | | 213 | | |
| | 16 | 214 | | if (charsConsumed != charCount) |
| | | 215 | | { |
| | | 216 | | // If there's still data remaining in the source buffer, go down the fallback path. |
| | | 217 | | // We need to check for integer overflow since the fallback could change the required |
| | | 218 | | // output count in unexpected ways. |
| | | 219 | | |
| | 0 | 220 | | totalByteCount += GetByteCountWithFallback(pChars, charCount, charsConsumed); |
| | 0 | 221 | | if (totalByteCount < 0) |
| | | 222 | | { |
| | 0 | 223 | | ThrowConversionOverflow(); |
| | | 224 | | } |
| | | 225 | | } |
| | | 226 | | |
| | 16 | 227 | | return totalByteCount; |
| | | 228 | | } |
| | | 229 | | |
| | | 230 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] // called directly by GetCharCountCommon |
| | | 231 | | private protected sealed override unsafe int GetByteCountFast(char* pChars, int charsLength, EncoderFallback? fa |
| | | 232 | | { |
| | | 233 | | // The number of UTF-8 code units may exceed the number of UTF-16 code units, |
| | | 234 | | // so we'll need to check for overflow before casting to Int32. |
| | | 235 | | |
| | 16 | 236 | | char* ptrToFirstInvalidChar = Utf16Utility.GetPointerToFirstInvalidChar(pChars, charsLength, out long utf8Co |
| | | 237 | | |
| | 16 | 238 | | int tempCharsConsumed = (int)(ptrToFirstInvalidChar - pChars); |
| | 16 | 239 | | charsConsumed = tempCharsConsumed; |
| | | 240 | | |
| | 16 | 241 | | long totalUtf8Bytes = tempCharsConsumed + utf8CodeUnitCountAdjustment; |
| | 16 | 242 | | if ((ulong)totalUtf8Bytes > int.MaxValue) |
| | | 243 | | { |
| | 0 | 244 | | ThrowConversionOverflow(); |
| | | 245 | | } |
| | | 246 | | |
| | 16 | 247 | | return (int)totalUtf8Bytes; |
| | | 248 | | } |
| | | 249 | | |
| | | 250 | | // Parent method is safe. |
| | | 251 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 252 | | // So if you fix this, fix the others. Currently those include: |
| | | 253 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 254 | | |
| | | 255 | | public override unsafe int GetBytes(string s, int charIndex, int charCount, |
| | | 256 | | byte[] bytes, int byteIndex) |
| | | 257 | | { |
| | 5 | 258 | | if (s is null || bytes is null) |
| | | 259 | | { |
| | 0 | 260 | | ThrowHelper.ThrowArgumentNullException( |
| | 0 | 261 | | argument: (s is null) ? ExceptionArgument.s : ExceptionArgument.bytes, |
| | 0 | 262 | | resource: ExceptionResource.ArgumentNull_Array); |
| | | 263 | | } |
| | | 264 | | |
| | 5 | 265 | | if ((charIndex | charCount) < 0) |
| | | 266 | | { |
| | 0 | 267 | | ThrowHelper.ThrowArgumentOutOfRangeException( |
| | 0 | 268 | | argument: (charIndex < 0) ? ExceptionArgument.charIndex : ExceptionArgument.charCount, |
| | 0 | 269 | | resource: ExceptionResource.ArgumentOutOfRange_NeedNonNegNum); |
| | | 270 | | } |
| | | 271 | | |
| | 5 | 272 | | if (s.Length - charIndex < charCount) |
| | | 273 | | { |
| | 0 | 274 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.s, ExceptionResource.ArgumentOutOfRange_I |
| | | 275 | | } |
| | | 276 | | |
| | 5 | 277 | | if ((uint)byteIndex > bytes.Length) |
| | | 278 | | { |
| | 0 | 279 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.byteIndex, ExceptionResource.ArgumentOutO |
| | | 280 | | } |
| | | 281 | | |
| | 5 | 282 | | fixed (char* pChars = s) |
| | 5 | 283 | | fixed (byte* pBytes = bytes) |
| | | 284 | | { |
| | 5 | 285 | | return GetBytesCommon(pChars + charIndex, charCount, pBytes + byteIndex, bytes.Length - byteIndex); |
| | | 286 | | } |
| | | 287 | | } |
| | | 288 | | |
| | | 289 | | // Encodes a range of characters in a character array into a range of bytes |
| | | 290 | | // in a byte array. An exception occurs if the byte array is not large |
| | | 291 | | // enough to hold the complete encoding of the characters. The |
| | | 292 | | // GetByteCount method can be used to determine the exact number of |
| | | 293 | | // bytes that will be produced for a given range of characters. |
| | | 294 | | // Alternatively, the GetMaxByteCount method can be used to |
| | | 295 | | // determine the maximum number of bytes that will be produced for a given |
| | | 296 | | // number of characters, regardless of the actual character values. |
| | | 297 | | // |
| | | 298 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 299 | | // So if you fix this, fix the others. Currently those include: |
| | | 300 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 301 | | // parent method is safe |
| | | 302 | | |
| | | 303 | | public override unsafe int GetBytes(char[] chars, int charIndex, int charCount, |
| | | 304 | | byte[] bytes, int byteIndex) |
| | | 305 | | { |
| | 0 | 306 | | if (chars is null || bytes is null) |
| | | 307 | | { |
| | 0 | 308 | | ThrowHelper.ThrowArgumentNullException( |
| | 0 | 309 | | argument: (chars is null) ? ExceptionArgument.chars : ExceptionArgument.bytes, |
| | 0 | 310 | | resource: ExceptionResource.ArgumentNull_Array); |
| | | 311 | | } |
| | | 312 | | |
| | 0 | 313 | | if ((charIndex | charCount) < 0) |
| | | 314 | | { |
| | 0 | 315 | | ThrowHelper.ThrowArgumentOutOfRangeException( |
| | 0 | 316 | | argument: (charIndex < 0) ? ExceptionArgument.charIndex : ExceptionArgument.charCount, |
| | 0 | 317 | | resource: ExceptionResource.ArgumentOutOfRange_NeedNonNegNum); |
| | | 318 | | } |
| | | 319 | | |
| | 0 | 320 | | if (chars.Length - charIndex < charCount) |
| | | 321 | | { |
| | 0 | 322 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.chars, ExceptionResource.ArgumentOutOfRan |
| | | 323 | | } |
| | | 324 | | |
| | 0 | 325 | | if ((uint)byteIndex > bytes.Length) |
| | | 326 | | { |
| | 0 | 327 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.byteIndex, ExceptionResource.ArgumentOutO |
| | | 328 | | } |
| | | 329 | | |
| | 0 | 330 | | fixed (char* pChars = chars) |
| | 0 | 331 | | fixed (byte* pBytes = bytes) |
| | | 332 | | { |
| | 0 | 333 | | return GetBytesCommon(pChars + charIndex, charCount, pBytes + byteIndex, bytes.Length - byteIndex); |
| | | 334 | | } |
| | | 335 | | } |
| | | 336 | | |
| | | 337 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 338 | | // So if you fix this, fix the others. Currently those include: |
| | | 339 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 340 | | |
| | | 341 | | [CLSCompliant(false)] |
| | | 342 | | public override unsafe int GetBytes(char* chars, int charCount, byte* bytes, int byteCount) |
| | | 343 | | { |
| | 0 | 344 | | if (chars is null || bytes is null) |
| | | 345 | | { |
| | 0 | 346 | | ThrowHelper.ThrowArgumentNullException( |
| | 0 | 347 | | argument: (chars is null) ? ExceptionArgument.chars : ExceptionArgument.bytes, |
| | 0 | 348 | | resource: ExceptionResource.ArgumentNull_Array); |
| | | 349 | | } |
| | | 350 | | |
| | 0 | 351 | | if ((charCount | byteCount) < 0) |
| | | 352 | | { |
| | 0 | 353 | | ThrowHelper.ThrowArgumentOutOfRangeException( |
| | 0 | 354 | | argument: (charCount < 0) ? ExceptionArgument.charCount : ExceptionArgument.byteCount, |
| | 0 | 355 | | resource: ExceptionResource.ArgumentOutOfRange_NeedNonNegNum); |
| | | 356 | | } |
| | | 357 | | |
| | 0 | 358 | | return GetBytesCommon(chars, charCount, bytes, byteCount); |
| | | 359 | | } |
| | | 360 | | |
| | | 361 | | public override unsafe int GetBytes(ReadOnlySpan<char> chars, Span<byte> bytes) |
| | 350 | 362 | | { |
| | | 363 | | // It's ok for us to operate on null / empty spans. |
| | | 364 | | |
| | 350 | 365 | | fixed (char* charsPtr = &MemoryMarshal.GetReference(chars)) |
| | 350 | 366 | | fixed (byte* bytesPtr = &MemoryMarshal.GetReference(bytes)) |
| | | 367 | | { |
| | 350 | 368 | | return GetBytesCommon(charsPtr, chars.Length, bytesPtr, bytes.Length); |
| | | 369 | | } |
| | | 370 | | } |
| | | 371 | | |
| | | 372 | | /// <inheritdoc/> |
| | | 373 | | public override unsafe bool TryGetBytes(ReadOnlySpan<char> chars, Span<byte> bytes, out int bytesWritten) |
| | 0 | 374 | | { |
| | 0 | 375 | | fixed (char* charsPtr = &MemoryMarshal.GetReference(chars)) |
| | 0 | 376 | | fixed (byte* bytesPtr = &MemoryMarshal.GetReference(bytes)) |
| | | 377 | | { |
| | 0 | 378 | | int written = GetBytesCommon(charsPtr, chars.Length, bytesPtr, bytes.Length, throwForDestinationOverflow |
| | 0 | 379 | | if (written >= 0) |
| | | 380 | | { |
| | 0 | 381 | | bytesWritten = written; |
| | 0 | 382 | | return true; |
| | | 383 | | } |
| | | 384 | | |
| | 0 | 385 | | bytesWritten = 0; |
| | 0 | 386 | | return false; |
| | | 387 | | } |
| | | 388 | | } |
| | | 389 | | |
| | | 390 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 391 | | private unsafe int GetBytesCommon(char* pChars, int charCount, byte* pBytes, int byteCount, bool throwForDestina |
| | | 392 | | { |
| | | 393 | | // Common helper method for all non-EncoderNLS entry points to GetBytes. |
| | | 394 | | // A modification of this method should be copied in to each of the supported encodings: ASCII, UTF8, UTF16, |
| | | 395 | | |
| | 355 | 396 | | Debug.Assert(charCount >= 0, "Caller shouldn't specify negative length buffer."); |
| | 355 | 397 | | Debug.Assert(pChars is not null || charCount == 0, "Input pointer shouldn't be null if non-zero length speci |
| | 355 | 398 | | Debug.Assert(byteCount >= 0, "Caller shouldn't specify negative length buffer."); |
| | 355 | 399 | | Debug.Assert(pBytes is not null || byteCount == 0, "Input pointer shouldn't be null if non-zero length speci |
| | | 400 | | |
| | | 401 | | // First call into the fast path. |
| | | 402 | | |
| | 355 | 403 | | int bytesWritten = GetBytesFast(pChars, charCount, pBytes, byteCount, out int charsConsumed); |
| | | 404 | | |
| | 355 | 405 | | if (charsConsumed == charCount) |
| | | 406 | | { |
| | | 407 | | // All elements converted - return immediately. |
| | | 408 | | |
| | 355 | 409 | | return bytesWritten; |
| | | 410 | | } |
| | | 411 | | else |
| | | 412 | | { |
| | | 413 | | // Simple narrowing conversion couldn't operate on entire buffer - invoke fallback. |
| | | 414 | | |
| | 0 | 415 | | return GetBytesWithFallback(pChars, charCount, pBytes, byteCount, charsConsumed, bytesWritten, throwForD |
| | | 416 | | } |
| | | 417 | | } |
| | | 418 | | |
| | | 419 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] // called directly by GetBytesCommon |
| | | 420 | | private protected sealed override unsafe int GetBytesFast(char* pChars, int charsLength, byte* pBytes, int bytes |
| | | 421 | | { |
| | | 422 | | // We don't care about the exact OperationStatus value returned by the workhorse routine; we only |
| | | 423 | | // care if the workhorse was able to consume the entire input payload. If we're unable to do so, |
| | | 424 | | // we'll handle the remainder in the fallback routine. |
| | | 425 | | |
| | 4105 | 426 | | Utf8Utility.TranscodeToUtf8(pChars, charsLength, pBytes, bytesLength, out char* pInputBufferRemaining, out b |
| | | 427 | | |
| | 4105 | 428 | | charsConsumed = (int)(pInputBufferRemaining - pChars); |
| | 4105 | 429 | | return (int)(pOutputBufferRemaining - pBytes); |
| | | 430 | | } |
| | | 431 | | |
| | | 432 | | // Returns the number of characters produced by decoding a range of bytes |
| | | 433 | | // in a byte array. |
| | | 434 | | // |
| | | 435 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 436 | | // So if you fix this, fix the others. Currently those include: |
| | | 437 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 438 | | // parent method is safe |
| | | 439 | | |
| | | 440 | | public override unsafe int GetCharCount(byte[] bytes, int index, int count) |
| | | 441 | | { |
| | 0 | 442 | | if (bytes is null) |
| | | 443 | | { |
| | 0 | 444 | | ThrowHelper.ThrowArgumentNullException(ExceptionArgument.bytes, ExceptionResource.ArgumentNull_Array); |
| | | 445 | | } |
| | | 446 | | |
| | 0 | 447 | | if ((index | count) < 0) |
| | | 448 | | { |
| | 0 | 449 | | ThrowHelper.ThrowArgumentOutOfRangeException((index < 0) ? ExceptionArgument.index : ExceptionArgument.c |
| | | 450 | | } |
| | | 451 | | |
| | 0 | 452 | | if (bytes.Length - index < count) |
| | | 453 | | { |
| | 0 | 454 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.bytes, ExceptionResource.ArgumentOutOfRan |
| | | 455 | | } |
| | | 456 | | |
| | 0 | 457 | | fixed (byte* pBytes = bytes) |
| | | 458 | | { |
| | 0 | 459 | | return GetCharCountCommon(pBytes + index, count); |
| | | 460 | | } |
| | | 461 | | } |
| | | 462 | | |
| | | 463 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 464 | | // So if you fix this, fix the others. Currently those include: |
| | | 465 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 466 | | |
| | | 467 | | [CLSCompliant(false)] |
| | | 468 | | public override unsafe int GetCharCount(byte* bytes, int count) |
| | | 469 | | { |
| | 1452 | 470 | | if (bytes is null) |
| | | 471 | | { |
| | 0 | 472 | | ThrowHelper.ThrowArgumentNullException(ExceptionArgument.bytes, ExceptionResource.ArgumentNull_Array); |
| | | 473 | | } |
| | | 474 | | |
| | 1452 | 475 | | if (count < 0) |
| | | 476 | | { |
| | 0 | 477 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.count, ExceptionResource.ArgumentOutOfRan |
| | | 478 | | } |
| | | 479 | | |
| | 1452 | 480 | | return GetCharCountCommon(bytes, count); |
| | | 481 | | } |
| | | 482 | | |
| | | 483 | | public override unsafe int GetCharCount(ReadOnlySpan<byte> bytes) |
| | 0 | 484 | | { |
| | | 485 | | // It's ok for us to pass null pointers down to the workhorse routine. |
| | | 486 | | |
| | 0 | 487 | | fixed (byte* bytesPtr = &MemoryMarshal.GetReference(bytes)) |
| | | 488 | | { |
| | 0 | 489 | | return GetCharCountCommon(bytesPtr, bytes.Length); |
| | | 490 | | } |
| | | 491 | | } |
| | | 492 | | |
| | | 493 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 494 | | // So if you fix this, fix the others. Currently those include: |
| | | 495 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 496 | | // parent method is safe |
| | | 497 | | |
| | | 498 | | public override unsafe int GetChars(byte[] bytes, int byteIndex, int byteCount, |
| | | 499 | | char[] chars, int charIndex) |
| | | 500 | | { |
| | 0 | 501 | | if (bytes is null || chars is null) |
| | | 502 | | { |
| | 0 | 503 | | ThrowHelper.ThrowArgumentNullException( |
| | 0 | 504 | | argument: (bytes is null) ? ExceptionArgument.bytes : ExceptionArgument.chars, |
| | 0 | 505 | | resource: ExceptionResource.ArgumentNull_Array); |
| | | 506 | | } |
| | | 507 | | |
| | 0 | 508 | | if ((byteIndex | byteCount) < 0) |
| | | 509 | | { |
| | 0 | 510 | | ThrowHelper.ThrowArgumentOutOfRangeException( |
| | 0 | 511 | | argument: (byteIndex < 0) ? ExceptionArgument.byteIndex : ExceptionArgument.byteCount, |
| | 0 | 512 | | resource: ExceptionResource.ArgumentOutOfRange_NeedNonNegNum); |
| | | 513 | | } |
| | | 514 | | |
| | 0 | 515 | | if (bytes.Length - byteIndex < byteCount) |
| | | 516 | | { |
| | 0 | 517 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.bytes, ExceptionResource.ArgumentOutOfRan |
| | | 518 | | } |
| | | 519 | | |
| | 0 | 520 | | if ((uint)charIndex > (uint)chars.Length) |
| | | 521 | | { |
| | 0 | 522 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.charIndex, ExceptionResource.ArgumentOutO |
| | | 523 | | } |
| | | 524 | | |
| | 0 | 525 | | fixed (byte* pBytes = bytes) |
| | 0 | 526 | | fixed (char* pChars = chars) |
| | | 527 | | { |
| | 0 | 528 | | return GetCharsCommon(pBytes + byteIndex, byteCount, pChars + charIndex, chars.Length - charIndex); |
| | | 529 | | } |
| | | 530 | | } |
| | | 531 | | |
| | | 532 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 533 | | // So if you fix this, fix the others. Currently those include: |
| | | 534 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 535 | | |
| | | 536 | | [CLSCompliant(false)] |
| | | 537 | | public override unsafe int GetChars(byte* bytes, int byteCount, char* chars, int charCount) |
| | | 538 | | { |
| | 1452 | 539 | | if (bytes is null || chars is null) |
| | | 540 | | { |
| | 0 | 541 | | ThrowHelper.ThrowArgumentNullException( |
| | 0 | 542 | | argument: (bytes is null) ? ExceptionArgument.bytes : ExceptionArgument.chars, |
| | 0 | 543 | | resource: ExceptionResource.ArgumentNull_Array); |
| | | 544 | | } |
| | | 545 | | |
| | 1452 | 546 | | if ((byteCount | charCount) < 0) |
| | | 547 | | { |
| | 0 | 548 | | ThrowHelper.ThrowArgumentOutOfRangeException( |
| | 0 | 549 | | argument: (byteCount < 0) ? ExceptionArgument.byteCount : ExceptionArgument.charCount, |
| | 0 | 550 | | resource: ExceptionResource.ArgumentOutOfRange_NeedNonNegNum); |
| | | 551 | | } |
| | | 552 | | |
| | 1452 | 553 | | return GetCharsCommon(bytes, byteCount, chars, charCount); |
| | | 554 | | } |
| | | 555 | | |
| | | 556 | | public override unsafe int GetChars(ReadOnlySpan<byte> bytes, Span<char> chars) |
| | 339 | 557 | | { |
| | | 558 | | // It's ok for us to pass null pointers down to the workhorse below. |
| | | 559 | | |
| | 339 | 560 | | fixed (byte* bytesPtr = &MemoryMarshal.GetReference(bytes)) |
| | 339 | 561 | | fixed (char* charsPtr = &MemoryMarshal.GetReference(chars)) |
| | | 562 | | { |
| | 339 | 563 | | return GetCharsCommon(bytesPtr, bytes.Length, charsPtr, chars.Length); |
| | | 564 | | } |
| | | 565 | | } |
| | | 566 | | |
| | | 567 | | /// <inheritdoc/> |
| | | 568 | | public override unsafe bool TryGetChars(ReadOnlySpan<byte> bytes, Span<char> chars, out int charsWritten) |
| | 0 | 569 | | { |
| | 0 | 570 | | fixed (byte* bytesPtr = &MemoryMarshal.GetReference(bytes)) |
| | 0 | 571 | | fixed (char* charsPtr = &MemoryMarshal.GetReference(chars)) |
| | | 572 | | { |
| | 0 | 573 | | int written = GetCharsCommon(bytesPtr, bytes.Length, charsPtr, chars.Length, throwForDestinationOverflow |
| | 0 | 574 | | if (written >= 0) |
| | | 575 | | { |
| | 0 | 576 | | charsWritten = written; |
| | 0 | 577 | | return true; |
| | | 578 | | } |
| | | 579 | | |
| | 0 | 580 | | charsWritten = 0; |
| | 0 | 581 | | return false; |
| | | 582 | | } |
| | | 583 | | } |
| | | 584 | | |
| | | 585 | | // WARNING: If we throw an error, then System.Resources.ResourceReader calls this method. |
| | | 586 | | // So if we're really broken, then that could also throw an error... recursively. |
| | | 587 | | // So try to make sure GetChars can at least process all uses by |
| | | 588 | | // System.Resources.ResourceReader! |
| | | 589 | | // |
| | | 590 | | // Note: We throw exceptions on individually encoded surrogates and other non-shortest forms. |
| | | 591 | | // If exceptions aren't turned on, then we drop all non-shortest &individual surrogates. |
| | | 592 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 593 | | private unsafe int GetCharsCommon(byte* pBytes, int byteCount, char* pChars, int charCount, bool throwForDestina |
| | | 594 | | { |
| | | 595 | | // Common helper method for all non-DecoderNLS entry points to GetChars. |
| | | 596 | | // A modification of this method should be copied in to each of the supported encodings: ASCII, UTF8, UTF16, |
| | | 597 | | |
| | 1791 | 598 | | Debug.Assert(byteCount >= 0, "Caller shouldn't specify negative length buffer."); |
| | 1791 | 599 | | Debug.Assert(pBytes is not null || byteCount == 0, "Input pointer shouldn't be null if non-zero length speci |
| | 1791 | 600 | | Debug.Assert(charCount >= 0, "Caller shouldn't specify negative length buffer."); |
| | 1791 | 601 | | Debug.Assert(pChars is not null || charCount == 0, "Input pointer shouldn't be null if non-zero length speci |
| | | 602 | | |
| | | 603 | | // First call into the fast path. |
| | | 604 | | |
| | 1791 | 605 | | int charsWritten = GetCharsFast(pBytes, byteCount, pChars, charCount, out int bytesConsumed); |
| | | 606 | | |
| | 1791 | 607 | | if (bytesConsumed == byteCount) |
| | | 608 | | { |
| | | 609 | | // All elements converted - return immediately. |
| | | 610 | | |
| | 1791 | 611 | | return charsWritten; |
| | | 612 | | } |
| | | 613 | | else |
| | | 614 | | { |
| | | 615 | | // Simple narrowing conversion couldn't operate on entire buffer - invoke fallback. |
| | | 616 | | |
| | 0 | 617 | | return GetCharsWithFallback(pBytes, byteCount, pChars, charCount, bytesConsumed, charsWritten, throwForD |
| | | 618 | | } |
| | | 619 | | } |
| | | 620 | | |
| | | 621 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] // called directly by GetCharsCommon |
| | | 622 | | private protected sealed override unsafe int GetCharsFast(byte* pBytes, int bytesLength, char* pChars, int chars |
| | | 623 | | { |
| | | 624 | | // We don't care about the exact OperationStatus value returned by the workhorse routine; we only |
| | | 625 | | // care if the workhorse was able to consume the entire input payload. If we're unable to do so, |
| | | 626 | | // we'll handle the remainder in the fallback routine. |
| | | 627 | | |
| | 875986 | 628 | | Utf8Utility.TranscodeToUtf16(pBytes, bytesLength, pChars, charsLength, out byte* pInputBufferRemaining, out |
| | | 629 | | |
| | 875986 | 630 | | bytesConsumed = (int)(pInputBufferRemaining - pBytes); |
| | 875986 | 631 | | return (int)(pOutputBufferRemaining - pChars); |
| | | 632 | | } |
| | | 633 | | |
| | | 634 | | private protected sealed override int GetCharsWithFallback(ReadOnlySpan<byte> bytes, int originalBytesLength, Sp |
| | | 635 | | { |
| | | 636 | | // We special-case DecoderReplacementFallback if it's telling us to write a single U+FFFD char, |
| | | 637 | | // since we believe this to be relatively common and we can handle it more efficiently than |
| | | 638 | | // the base implementation. |
| | | 639 | | |
| | 338562 | 640 | | if (((decoder is null) ? this.DecoderFallback : decoder.Fallback) is DecoderReplacementFallback replacementF |
| | 338562 | 641 | | && replacementFallback.MaxCharCount == 1 |
| | 338562 | 642 | | && replacementFallback.DefaultString[0] == UnicodeUtility.ReplacementChar) |
| | | 643 | | { |
| | | 644 | | // Don't care about the exact OperationStatus, just how much of the payload we were able |
| | | 645 | | // to process. |
| | | 646 | | |
| | 338562 | 647 | | Utf8.ToUtf16(bytes, chars, out int bytesRead, out int charsWritten, replaceInvalidSequences: true, isFin |
| | | 648 | | |
| | | 649 | | // Slice off how much we consumed / wrote. |
| | | 650 | | |
| | 338562 | 651 | | bytes = bytes.Slice(bytesRead); |
| | 338562 | 652 | | chars = chars.Slice(charsWritten); |
| | | 653 | | } |
| | | 654 | | |
| | | 655 | | // If we couldn't go through our fast fallback mechanism, or if we still have leftover |
| | | 656 | | // data because we couldn't consume everything in the loop above, we need to go down the |
| | | 657 | | // slow fallback path. |
| | | 658 | | |
| | 338562 | 659 | | if (bytes.IsEmpty) |
| | | 660 | | { |
| | 162682 | 661 | | return originalCharsLength - chars.Length; // total number of chars written |
| | | 662 | | } |
| | | 663 | | else |
| | | 664 | | { |
| | 175880 | 665 | | return base.GetCharsWithFallback(bytes, originalBytesLength, chars, originalCharsLength, decoder, throwF |
| | | 666 | | } |
| | | 667 | | } |
| | | 668 | | |
| | | 669 | | // Returns a string containing the decoded representation of a range of |
| | | 670 | | // bytes in a byte array. |
| | | 671 | | // |
| | | 672 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 673 | | // So if you fix this, fix the others. Currently those include: |
| | | 674 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 675 | | // parent method is safe |
| | | 676 | | |
| | | 677 | | public override unsafe string GetString(byte[] bytes, int index, int count) |
| | | 678 | | { |
| | 0 | 679 | | if (bytes is null) |
| | | 680 | | { |
| | 0 | 681 | | ThrowHelper.ThrowArgumentNullException(ExceptionArgument.bytes, ExceptionResource.ArgumentNull_Array); |
| | | 682 | | } |
| | | 683 | | |
| | 0 | 684 | | if ((index | count) < 0) |
| | | 685 | | { |
| | 0 | 686 | | ThrowHelper.ThrowArgumentOutOfRangeException( |
| | 0 | 687 | | argument: (index < 0) ? ExceptionArgument.index : ExceptionArgument.count, |
| | 0 | 688 | | resource: ExceptionResource.ArgumentOutOfRange_NeedNonNegNum); |
| | | 689 | | } |
| | | 690 | | |
| | 0 | 691 | | if (bytes.Length - index < count) |
| | | 692 | | { |
| | 0 | 693 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.bytes, ExceptionResource.ArgumentOutOfRan |
| | | 694 | | } |
| | | 695 | | |
| | | 696 | | // Avoid problems with empty input buffer |
| | 0 | 697 | | if (count == 0) |
| | 0 | 698 | | return string.Empty; |
| | | 699 | | |
| | 0 | 700 | | fixed (byte* pBytes = bytes) |
| | | 701 | | { |
| | 0 | 702 | | return string.CreateStringFromEncoding(pBytes + index, count, this); |
| | | 703 | | } |
| | | 704 | | } |
| | | 705 | | |
| | | 706 | | // |
| | | 707 | | // End of standard methods copied from EncodingNLS.cs |
| | | 708 | | // |
| | | 709 | | |
| | | 710 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 711 | | private unsafe int GetCharCountCommon(byte* pBytes, int byteCount) |
| | | 712 | | { |
| | | 713 | | // Common helper method for all non-DecoderNLS entry points to GetCharCount. |
| | | 714 | | // A modification of this method should be copied in to each of the supported encodings: ASCII, UTF8, UTF16, |
| | | 715 | | |
| | 1452 | 716 | | Debug.Assert(byteCount >= 0, "Caller shouldn't specify negative length buffer."); |
| | 1452 | 717 | | Debug.Assert(pBytes is not null || byteCount == 0, "Input pointer shouldn't be null if non-zero length speci |
| | | 718 | | |
| | | 719 | | // First call into the fast path. |
| | | 720 | | // Don't bother providing a fallback mechanism; our fast path doesn't use it. |
| | | 721 | | |
| | 1452 | 722 | | int totalCharCount = GetCharCountFast(pBytes, byteCount, fallback: null, out int bytesConsumed); |
| | | 723 | | |
| | 1452 | 724 | | if (bytesConsumed != byteCount) |
| | | 725 | | { |
| | | 726 | | // If there's still data remaining in the source buffer, go down the fallback path. |
| | | 727 | | // We need to check for integer overflow since the fallback could change the required |
| | | 728 | | // output count in unexpected ways. |
| | | 729 | | |
| | 0 | 730 | | totalCharCount += GetCharCountWithFallback(pBytes, byteCount, bytesConsumed); |
| | 0 | 731 | | if (totalCharCount < 0) |
| | | 732 | | { |
| | 0 | 733 | | ThrowConversionOverflow(); |
| | | 734 | | } |
| | | 735 | | } |
| | | 736 | | |
| | 1452 | 737 | | return totalCharCount; |
| | | 738 | | } |
| | | 739 | | |
| | | 740 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] // called directly by GetCharCountCommon |
| | | 741 | | private protected sealed override unsafe int GetCharCountFast(byte* pBytes, int bytesLength, DecoderFallback? fa |
| | | 742 | | { |
| | | 743 | | // The number of UTF-16 code units will never exceed the number of UTF-8 code units, |
| | | 744 | | // so the addition at the end of this method will not overflow. |
| | | 745 | | |
| | 855076 | 746 | | byte* ptrToFirstInvalidByte = Utf8Utility.GetPointerToFirstInvalidByte(pBytes, bytesLength, out int utf16Cod |
| | | 747 | | |
| | 855076 | 748 | | int tempBytesConsumed = (int)(ptrToFirstInvalidByte - pBytes); |
| | 855076 | 749 | | bytesConsumed = tempBytesConsumed; |
| | | 750 | | |
| | 855076 | 751 | | return tempBytesConsumed + utf16CodeUnitCountAdjustment; |
| | | 752 | | } |
| | | 753 | | |
| | | 754 | | public override Decoder GetDecoder() |
| | | 755 | | { |
| | 29587 | 756 | | return new DecoderNLS(this); |
| | | 757 | | } |
| | | 758 | | |
| | | 759 | | |
| | | 760 | | public override Encoder GetEncoder() |
| | | 761 | | { |
| | 3748 | 762 | | return new EncoderNLS(this); |
| | | 763 | | } |
| | | 764 | | |
| | | 765 | | // |
| | | 766 | | // Beginning of methods used by shared fallback logic. |
| | | 767 | | // |
| | | 768 | | |
| | | 769 | | internal sealed override bool TryGetByteCount(Rune value, out int byteCount) |
| | | 770 | | { |
| | | 771 | | // All well-formed Rune instances can be converted to 1..4 UTF-8 code units. |
| | | 772 | | |
| | 0 | 773 | | byteCount = value.Utf8SequenceLength; |
| | 0 | 774 | | return true; |
| | | 775 | | } |
| | | 776 | | |
| | | 777 | | internal sealed override OperationStatus EncodeRune(Rune value, Span<byte> bytes, out int bytesWritten) |
| | | 778 | | { |
| | | 779 | | // All well-formed Rune instances can be encoded as 1..4 UTF-8 code units. |
| | | 780 | | // If there's an error, it's because the destination was too small. |
| | | 781 | | |
| | 0 | 782 | | return value.TryEncodeToUtf8(bytes, out bytesWritten) ? OperationStatus.Done : OperationStatus.DestinationTo |
| | | 783 | | } |
| | | 784 | | |
| | | 785 | | internal sealed override OperationStatus DecodeFirstRune(ReadOnlySpan<byte> bytes, out Rune value, out int bytes |
| | | 786 | | { |
| | 1178590 | 787 | | return Rune.DecodeFromUtf8(bytes, out value, out bytesConsumed); |
| | | 788 | | } |
| | | 789 | | |
| | | 790 | | // |
| | | 791 | | // End of methods used by shared fallback logic. |
| | | 792 | | // |
| | | 793 | | |
| | | 794 | | public override int GetMaxByteCount(int charCount) |
| | | 795 | | { |
| | 0 | 796 | | ArgumentOutOfRangeException.ThrowIfNegative(charCount); |
| | | 797 | | |
| | | 798 | | // GetMaxByteCount assumes that the caller might have a stateful Encoder instance. If the |
| | | 799 | | // Encoder instance already has a captured high surrogate, then one of two things will |
| | | 800 | | // happen: |
| | | 801 | | // |
| | | 802 | | // - The next char is a low surrogate, at which point the two chars together result in 4 |
| | | 803 | | // UTF-8 bytes in the output; or |
| | | 804 | | // - The next char is not a low surrogate (or the input reaches EOF), at which point the |
| | | 805 | | // standalone captured surrogate will go through the fallback routine. |
| | | 806 | | // |
| | | 807 | | // The second case is the worst-case scenario for expansion, so it's what we use for any |
| | | 808 | | // pessimistic "max byte count" calculation: assume there's a captured surrogate and that |
| | | 809 | | // it must fall back. |
| | | 810 | | |
| | 0 | 811 | | long byteCount = (long)charCount + 1; // +1 to account for captured surrogate, per above |
| | | 812 | | |
| | 0 | 813 | | if (EncoderFallback.MaxCharCount > 1) |
| | 0 | 814 | | byteCount *= EncoderFallback.MaxCharCount; |
| | | 815 | | |
| | 0 | 816 | | byteCount *= MaxUtf8BytesPerChar; |
| | | 817 | | |
| | 0 | 818 | | if (byteCount > 0x7fffffff) |
| | 0 | 819 | | throw new ArgumentOutOfRangeException(nameof(charCount), SR.ArgumentOutOfRange_GetByteCountOverflow); |
| | | 820 | | |
| | 0 | 821 | | return (int)byteCount; |
| | | 822 | | } |
| | | 823 | | |
| | | 824 | | |
| | | 825 | | public override int GetMaxCharCount(int byteCount) |
| | | 826 | | { |
| | 0 | 827 | | ArgumentOutOfRangeException.ThrowIfNegative(byteCount); |
| | | 828 | | |
| | | 829 | | // GetMaxCharCount assumes that the caller might have a stateful Decoder instance. If the |
| | | 830 | | // Decoder instance already has a captured partial UTF-8 subsequence, then one of two |
| | | 831 | | // thngs will happen: |
| | | 832 | | // |
| | | 833 | | // - The next byte(s) won't complete the subsequence but will instead be consumed into |
| | | 834 | | // the Decoder's internal state, resulting in no character output; or |
| | | 835 | | // - The next byte(s) will complete the subsequence, and the previously captured |
| | | 836 | | // subsequence and the next byte(s) will result in 1 - 2 chars output; or |
| | | 837 | | // - The captured subsequence will be treated as a singular ill-formed subsequence, at |
| | | 838 | | // which point the captured subsequence will go through the fallback routine. |
| | | 839 | | // (See The Unicode Standard, Sec. 3.9 for more information on this.) |
| | | 840 | | // |
| | | 841 | | // The third case is the worst-case scenario for expansion, since it means 0 bytes of |
| | | 842 | | // new input could cause any existing captured state to expand via fallback. So it's |
| | | 843 | | // what we'll use for any pessimistic "max char count" calculation. |
| | | 844 | | |
| | 0 | 845 | | long charCount = ((long)byteCount + 1); // +1 to account for captured subsequence, as above |
| | | 846 | | |
| | | 847 | | // Non-shortest form would fall back, so get max count from fallback. |
| | | 848 | | // So would 11... followed by 11..., so you could fall back every byte |
| | 0 | 849 | | if (DecoderFallback.MaxCharCount > 1) |
| | | 850 | | { |
| | 0 | 851 | | charCount *= DecoderFallback.MaxCharCount; |
| | | 852 | | } |
| | | 853 | | |
| | 0 | 854 | | if (charCount > 0x7fffffff) |
| | 0 | 855 | | throw new ArgumentOutOfRangeException(nameof(byteCount), SR.ArgumentOutOfRange_GetCharCountOverflow); |
| | | 856 | | |
| | 0 | 857 | | return (int)charCount; |
| | | 858 | | } |
| | | 859 | | |
| | | 860 | | |
| | | 861 | | public override byte[] GetPreamble() |
| | | 862 | | { |
| | 0 | 863 | | if (_emitUTF8Identifier) |
| | | 864 | | { |
| | | 865 | | // Allocate new array to prevent users from modifying it. |
| | 0 | 866 | | return [0xEF, 0xBB, 0xBF]; |
| | | 867 | | } |
| | | 868 | | else |
| | 0 | 869 | | return []; |
| | | 870 | | } |
| | | 871 | | |
| | | 872 | | public override ReadOnlySpan<byte> Preamble => |
| | 0 | 873 | | GetType() != typeof(UTF8Encoding) ? new ReadOnlySpan<byte>(GetPreamble()) : // in case a derived UTF8Encodin |
| | 0 | 874 | | _emitUTF8Identifier ? PreambleSpan : |
| | 0 | 875 | | default; |
| | | 876 | | |
| | | 877 | | public override bool Equals([NotNullWhen(true)] object? value) |
| | | 878 | | { |
| | 0 | 879 | | if (value is UTF8Encoding that) |
| | | 880 | | { |
| | 0 | 881 | | return (_emitUTF8Identifier == that._emitUTF8Identifier) && |
| | 0 | 882 | | (EncoderFallback.Equals(that.EncoderFallback)) && |
| | 0 | 883 | | (DecoderFallback.Equals(that.DecoderFallback)); |
| | | 884 | | } |
| | 0 | 885 | | return false; |
| | | 886 | | } |
| | | 887 | | |
| | | 888 | | |
| | | 889 | | public override int GetHashCode() |
| | | 890 | | { |
| | | 891 | | // Not great distribution, but this is relatively unlikely to be used as the key in a hashtable. |
| | 0 | 892 | | return this.EncoderFallback.GetHashCode() + this.DecoderFallback.GetHashCode() + |
| | 0 | 893 | | UTF8_CODEPAGE + (_emitUTF8Identifier ? 1 : 0); |
| | | 894 | | } |
| | | 895 | | } |
| | | 896 | | } |
| | | 897 | | |