| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | // |
| | | 5 | | // Don't override IsAlwaysNormalized because it is just a Unicode Transformation and could be confused. |
| | | 6 | | // |
| | | 7 | | |
| | | 8 | | // This define can be used to turn off the fast loops. Useful for finding whether |
| | | 9 | | // the problem is fastloop-specific. |
| | | 10 | | #define FASTLOOP |
| | | 11 | | |
| | | 12 | | using System.Diagnostics; |
| | | 13 | | using System.Diagnostics.CodeAnalysis; |
| | | 14 | | using System.Runtime.CompilerServices; |
| | | 15 | | using System.Runtime.InteropServices; |
| | | 16 | | |
| | | 17 | | namespace System.Text |
| | | 18 | | { |
| | | 19 | | public class UnicodeEncoding : Encoding |
| | | 20 | | { |
| | | 21 | | // Used by Encoding.BigEndianUnicode/Unicode for lazy initialization |
| | | 22 | | // The initialization code will not be run until a static member of the class is referenced |
| | 1 | 23 | | internal static readonly UnicodeEncoding s_bigEndianDefault = new UnicodeEncoding(bigEndian: true, byteOrderMark |
| | 1 | 24 | | internal static readonly UnicodeEncoding s_littleEndianDefault = new UnicodeEncoding(bigEndian: false, byteOrder |
| | | 25 | | |
| | | 26 | | private readonly bool isThrowException; |
| | | 27 | | |
| | | 28 | | private readonly bool bigEndian; |
| | | 29 | | private readonly bool byteOrderMark; |
| | | 30 | | |
| | | 31 | | // Unicode version 2.0 character size in bytes |
| | | 32 | | public const int CharSize = 2; |
| | | 33 | | |
| | | 34 | | public UnicodeEncoding() |
| | 6806 | 35 | | : this(false, true) |
| | | 36 | | { |
| | 6806 | 37 | | } |
| | | 38 | | |
| | | 39 | | public UnicodeEncoding(bool bigEndian, bool byteOrderMark) |
| | 10211 | 40 | | : base(bigEndian ? 1201 : 1200) // Set the data item. |
| | | 41 | | { |
| | 10211 | 42 | | this.bigEndian = bigEndian; |
| | 10211 | 43 | | this.byteOrderMark = byteOrderMark; |
| | 10211 | 44 | | } |
| | | 45 | | |
| | | 46 | | public UnicodeEncoding(bool bigEndian, bool byteOrderMark, bool throwOnInvalidBytes) |
| | 3403 | 47 | | : this(bigEndian, byteOrderMark) |
| | | 48 | | { |
| | 3403 | 49 | | this.isThrowException = throwOnInvalidBytes; |
| | | 50 | | |
| | | 51 | | // Encoding constructor already did this, but it'll be wrong if we're throwing exceptions |
| | 3403 | 52 | | if (this.isThrowException) |
| | 3403 | 53 | | SetDefaultFallbacks(); |
| | 3403 | 54 | | } |
| | | 55 | | |
| | | 56 | | internal sealed override void SetDefaultFallbacks() |
| | | 57 | | { |
| | | 58 | | // For UTF-X encodings, we use a replacement fallback with an empty string |
| | 13614 | 59 | | if (this.isThrowException) |
| | | 60 | | { |
| | 3403 | 61 | | this.encoderFallback = EncoderFallback.ExceptionFallback; |
| | 3403 | 62 | | this.decoderFallback = DecoderFallback.ExceptionFallback; |
| | | 63 | | } |
| | | 64 | | else |
| | | 65 | | { |
| | 10211 | 66 | | this.encoderFallback = new EncoderReplacementFallback("\xFFFD"); |
| | 10211 | 67 | | this.decoderFallback = new DecoderReplacementFallback("\xFFFD"); |
| | | 68 | | } |
| | 10211 | 69 | | } |
| | | 70 | | |
| | | 71 | | // The following methods are copied from EncodingNLS.cs. |
| | | 72 | | // Unfortunately EncodingNLS.cs is internal and we're public, so we have to re-implement them here. |
| | | 73 | | // These should be kept in sync for the following classes: |
| | | 74 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 75 | | // |
| | | 76 | | |
| | | 77 | | // Returns the number of bytes required to encode a range of characters in |
| | | 78 | | // a character array. |
| | | 79 | | // |
| | | 80 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 81 | | // So if you fix this, fix the others. Currently those include: |
| | | 82 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 83 | | // parent method is safe |
| | | 84 | | |
| | | 85 | | public override unsafe int GetByteCount(char[] chars, int index, int count) |
| | | 86 | | { |
| | 0 | 87 | | ArgumentNullException.ThrowIfNull(chars); |
| | | 88 | | |
| | 0 | 89 | | ArgumentOutOfRangeException.ThrowIfNegative(index); |
| | 0 | 90 | | ArgumentOutOfRangeException.ThrowIfNegative(count); |
| | | 91 | | |
| | 0 | 92 | | if (chars.Length - index < count) |
| | 0 | 93 | | throw new ArgumentOutOfRangeException(nameof(chars), SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 94 | | |
| | | 95 | | // If no input, return 0, avoid fixed empty array problem |
| | 0 | 96 | | if (count == 0) |
| | 0 | 97 | | return 0; |
| | | 98 | | |
| | | 99 | | // Just call the pointer version |
| | 0 | 100 | | fixed (char* pChars = chars) |
| | 0 | 101 | | return GetByteCount(pChars + index, count, null); |
| | | 102 | | } |
| | | 103 | | |
| | | 104 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 105 | | // So if you fix this, fix the others. Currently those include: |
| | | 106 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 107 | | // parent method is safe |
| | | 108 | | |
| | | 109 | | public override unsafe int GetByteCount(string s) |
| | | 110 | | { |
| | 0 | 111 | | if (s is null) |
| | | 112 | | { |
| | 0 | 113 | | ThrowHelper.ThrowArgumentNullException(ExceptionArgument.s); |
| | | 114 | | } |
| | | 115 | | |
| | 0 | 116 | | fixed (char* pChars = s) |
| | 0 | 117 | | return GetByteCount(pChars, s.Length, null); |
| | | 118 | | } |
| | | 119 | | |
| | | 120 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 121 | | // So if you fix this, fix the others. Currently those include: |
| | | 122 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 123 | | |
| | | 124 | | [CLSCompliant(false)] |
| | | 125 | | public override unsafe int GetByteCount(char* chars, int count) |
| | | 126 | | { |
| | 0 | 127 | | ArgumentNullException.ThrowIfNull(chars); |
| | | 128 | | |
| | 0 | 129 | | ArgumentOutOfRangeException.ThrowIfNegative(count); |
| | | 130 | | |
| | | 131 | | // Call it with empty encoder |
| | 0 | 132 | | return GetByteCount(chars, count, null); |
| | | 133 | | } |
| | | 134 | | |
| | | 135 | | // Parent method is safe. |
| | | 136 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 137 | | // So if you fix this, fix the others. Currently those include: |
| | | 138 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 139 | | |
| | | 140 | | public override unsafe int GetBytes(string s, int charIndex, int charCount, |
| | | 141 | | byte[] bytes, int byteIndex) |
| | | 142 | | { |
| | 0 | 143 | | ArgumentNullException.ThrowIfNull(s); |
| | 0 | 144 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 145 | | |
| | 0 | 146 | | ArgumentOutOfRangeException.ThrowIfNegative(charIndex); |
| | 0 | 147 | | ArgumentOutOfRangeException.ThrowIfNegative(charCount); |
| | | 148 | | |
| | 0 | 149 | | if (s.Length - charIndex < charCount) |
| | 0 | 150 | | throw new ArgumentOutOfRangeException(nameof(s), SR.ArgumentOutOfRange_IndexCount); |
| | | 151 | | |
| | 0 | 152 | | if (byteIndex < 0 || byteIndex > bytes.Length) |
| | 0 | 153 | | throw new ArgumentOutOfRangeException(nameof(byteIndex), SR.ArgumentOutOfRange_IndexMustBeLessOrEqual); |
| | | 154 | | |
| | 0 | 155 | | int byteCount = bytes.Length - byteIndex; |
| | | 156 | | |
| | 0 | 157 | | fixed (char* pChars = s) |
| | 0 | 158 | | fixed (byte* pBytes = &MemoryMarshal.GetArrayDataReference(bytes)) |
| | | 159 | | { |
| | 0 | 160 | | return GetBytes(pChars + charIndex, charCount, pBytes + byteIndex, byteCount, null); |
| | | 161 | | } |
| | | 162 | | } |
| | | 163 | | |
| | | 164 | | // Encodes a range of characters in a character array into a range of bytes |
| | | 165 | | // in a byte array. An exception occurs if the byte array is not large |
| | | 166 | | // enough to hold the complete encoding of the characters. The |
| | | 167 | | // GetByteCount method can be used to determine the exact number of |
| | | 168 | | // bytes that will be produced for a given range of characters. |
| | | 169 | | // Alternatively, the GetMaxByteCount method can be used to |
| | | 170 | | // determine the maximum number of bytes that will be produced for a given |
| | | 171 | | // number of characters, regardless of the actual character values. |
| | | 172 | | // |
| | | 173 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 174 | | // So if you fix this, fix the others. Currently those include: |
| | | 175 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 176 | | // parent method is safe |
| | | 177 | | |
| | | 178 | | public override unsafe int GetBytes(char[] chars, int charIndex, int charCount, |
| | | 179 | | byte[] bytes, int byteIndex) |
| | | 180 | | { |
| | 0 | 181 | | ArgumentNullException.ThrowIfNull(chars); |
| | 0 | 182 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 183 | | |
| | 0 | 184 | | ArgumentOutOfRangeException.ThrowIfNegative(charIndex); |
| | 0 | 185 | | ArgumentOutOfRangeException.ThrowIfNegative(charCount); |
| | | 186 | | |
| | 0 | 187 | | if (chars.Length - charIndex < charCount) |
| | 0 | 188 | | throw new ArgumentOutOfRangeException(nameof(chars), SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 189 | | |
| | 0 | 190 | | if (byteIndex < 0 || byteIndex > bytes.Length) |
| | 0 | 191 | | throw new ArgumentOutOfRangeException(nameof(byteIndex), SR.ArgumentOutOfRange_IndexMustBeLessOrEqual); |
| | | 192 | | |
| | | 193 | | // If nothing to encode return 0, avoid fixed problem |
| | 0 | 194 | | if (charCount == 0) |
| | 0 | 195 | | return 0; |
| | | 196 | | |
| | | 197 | | // Just call pointer version |
| | 0 | 198 | | int byteCount = bytes.Length - byteIndex; |
| | | 199 | | |
| | 0 | 200 | | fixed (char* pChars = chars) |
| | 0 | 201 | | fixed (byte* pBytes = &MemoryMarshal.GetArrayDataReference(bytes)) |
| | | 202 | | { |
| | | 203 | | // Remember that byteCount is # to decode, not size of array. |
| | 0 | 204 | | return GetBytes(pChars + charIndex, charCount, pBytes + byteIndex, byteCount, null); |
| | | 205 | | } |
| | | 206 | | } |
| | | 207 | | |
| | | 208 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 209 | | // So if you fix this, fix the others. Currently those include: |
| | | 210 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 211 | | |
| | | 212 | | [CLSCompliant(false)] |
| | | 213 | | public override unsafe int GetBytes(char* chars, int charCount, byte* bytes, int byteCount) |
| | | 214 | | { |
| | 0 | 215 | | ArgumentNullException.ThrowIfNull(chars); |
| | 0 | 216 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 217 | | |
| | 0 | 218 | | ArgumentOutOfRangeException.ThrowIfNegative(charCount); |
| | 0 | 219 | | ArgumentOutOfRangeException.ThrowIfNegative(byteCount); |
| | | 220 | | |
| | 0 | 221 | | return GetBytes(chars, charCount, bytes, byteCount, null); |
| | | 222 | | } |
| | | 223 | | |
| | | 224 | | // Returns the number of characters produced by decoding a range of bytes |
| | | 225 | | // in a byte array. |
| | | 226 | | // |
| | | 227 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 228 | | // So if you fix this, fix the others. Currently those include: |
| | | 229 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 230 | | // parent method is safe |
| | | 231 | | |
| | | 232 | | public override unsafe int GetCharCount(byte[] bytes, int index, int count) |
| | | 233 | | { |
| | 0 | 234 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 235 | | |
| | 0 | 236 | | ArgumentOutOfRangeException.ThrowIfNegative(index); |
| | 0 | 237 | | ArgumentOutOfRangeException.ThrowIfNegative(count); |
| | | 238 | | |
| | 0 | 239 | | if (bytes.Length - index < count) |
| | 0 | 240 | | throw new ArgumentOutOfRangeException(nameof(bytes), SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 241 | | |
| | | 242 | | // If no input just return 0, fixed doesn't like 0 length arrays |
| | 0 | 243 | | if (count == 0) |
| | 0 | 244 | | return 0; |
| | | 245 | | |
| | | 246 | | // Just call pointer version |
| | 0 | 247 | | fixed (byte* pBytes = bytes) |
| | 0 | 248 | | return GetCharCount(pBytes + index, count, null); |
| | | 249 | | } |
| | | 250 | | |
| | | 251 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 252 | | // So if you fix this, fix the others. Currently those include: |
| | | 253 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 254 | | |
| | | 255 | | [CLSCompliant(false)] |
| | | 256 | | public override unsafe int GetCharCount(byte* bytes, int count) |
| | | 257 | | { |
| | 0 | 258 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 259 | | |
| | 0 | 260 | | ArgumentOutOfRangeException.ThrowIfNegative(count); |
| | | 261 | | |
| | 0 | 262 | | return GetCharCount(bytes, count, null); |
| | | 263 | | } |
| | | 264 | | |
| | | 265 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 266 | | // So if you fix this, fix the others. Currently those include: |
| | | 267 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 268 | | // parent method is safe |
| | | 269 | | |
| | | 270 | | public override unsafe int GetChars(byte[] bytes, int byteIndex, int byteCount, |
| | | 271 | | char[] chars, int charIndex) |
| | | 272 | | { |
| | 0 | 273 | | ArgumentNullException.ThrowIfNull(bytes); |
| | 0 | 274 | | ArgumentNullException.ThrowIfNull(chars); |
| | | 275 | | |
| | 0 | 276 | | ArgumentOutOfRangeException.ThrowIfNegative(byteIndex); |
| | 0 | 277 | | ArgumentOutOfRangeException.ThrowIfNegative(byteCount); |
| | | 278 | | |
| | 0 | 279 | | if (bytes.Length - byteIndex < byteCount) |
| | 0 | 280 | | throw new ArgumentOutOfRangeException(nameof(bytes), SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 281 | | |
| | 0 | 282 | | if (charIndex < 0 || charIndex > chars.Length) |
| | 0 | 283 | | throw new ArgumentOutOfRangeException(nameof(charIndex), SR.ArgumentOutOfRange_IndexMustBeLessOrEqual); |
| | | 284 | | |
| | | 285 | | // If no input, return 0 & avoid fixed problem |
| | 0 | 286 | | if (byteCount == 0) |
| | 0 | 287 | | return 0; |
| | | 288 | | |
| | | 289 | | // Just call pointer version |
| | 0 | 290 | | int charCount = chars.Length - charIndex; |
| | | 291 | | |
| | 0 | 292 | | fixed (byte* pBytes = bytes) |
| | 0 | 293 | | fixed (char* pChars = &MemoryMarshal.GetArrayDataReference(chars)) |
| | | 294 | | { |
| | | 295 | | // Remember that charCount is # to decode, not size of array |
| | 0 | 296 | | return GetChars(pBytes + byteIndex, byteCount, pChars + charIndex, charCount, null); |
| | | 297 | | } |
| | | 298 | | } |
| | | 299 | | |
| | | 300 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 301 | | // So if you fix this, fix the others. Currently those include: |
| | | 302 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 303 | | |
| | | 304 | | [CLSCompliant(false)] |
| | | 305 | | public override unsafe int GetChars(byte* bytes, int byteCount, char* chars, int charCount) |
| | | 306 | | { |
| | 0 | 307 | | ArgumentNullException.ThrowIfNull(bytes); |
| | 0 | 308 | | ArgumentNullException.ThrowIfNull(chars); |
| | | 309 | | |
| | 0 | 310 | | ArgumentOutOfRangeException.ThrowIfNegative(charCount); |
| | 0 | 311 | | ArgumentOutOfRangeException.ThrowIfNegative(byteCount); |
| | | 312 | | |
| | 0 | 313 | | return GetChars(bytes, byteCount, chars, charCount, null); |
| | | 314 | | } |
| | | 315 | | |
| | | 316 | | // Returns a string containing the decoded representation of a range of |
| | | 317 | | // bytes in a byte array. |
| | | 318 | | // |
| | | 319 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 320 | | // So if you fix this, fix the others. Currently those include: |
| | | 321 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 322 | | // parent method is safe |
| | | 323 | | |
| | | 324 | | public override unsafe string GetString(byte[] bytes, int index, int count) |
| | | 325 | | { |
| | 0 | 326 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 327 | | |
| | 0 | 328 | | ArgumentOutOfRangeException.ThrowIfNegative(index); |
| | 0 | 329 | | ArgumentOutOfRangeException.ThrowIfNegative(count); |
| | | 330 | | |
| | 0 | 331 | | if (bytes.Length - index < count) |
| | 0 | 332 | | throw new ArgumentOutOfRangeException(nameof(bytes), SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 333 | | |
| | | 334 | | // Avoid problems with empty input buffer |
| | 0 | 335 | | if (count == 0) return string.Empty; |
| | | 336 | | |
| | 0 | 337 | | fixed (byte* pBytes = bytes) |
| | 0 | 338 | | return string.CreateStringFromEncoding( |
| | 0 | 339 | | pBytes + index, count, this); |
| | | 340 | | } |
| | | 341 | | |
| | | 342 | | // |
| | | 343 | | // End of standard methods copied from EncodingNLS.cs |
| | | 344 | | // |
| | | 345 | | internal sealed override unsafe int GetByteCount(char* chars, int count, EncoderNLS? encoder) |
| | | 346 | | { |
| | | 347 | | Debug.Assert(chars is not null, "[UnicodeEncoding.GetByteCount]chars!=null"); |
| | 0 | 348 | | Debug.Assert(count >= 0, "[UnicodeEncoding.GetByteCount]count >=0"); |
| | | 349 | | |
| | | 350 | | // Start by assuming each char gets 2 bytes |
| | 0 | 351 | | int byteCount = count << 1; |
| | | 352 | | |
| | | 353 | | // Check for overflow in byteCount |
| | | 354 | | // (If they were all invalid chars, this would actually be wrong, |
| | | 355 | | // but that's a ridiculously large # so we're not concerned about that case) |
| | 0 | 356 | | if (byteCount < 0) |
| | 0 | 357 | | throw new ArgumentOutOfRangeException(nameof(count), SR.ArgumentOutOfRange_GetByteCountOverflow); |
| | | 358 | | |
| | 0 | 359 | | char* charStart = chars; |
| | 0 | 360 | | char* charEnd = chars + count; |
| | 0 | 361 | | char charLeftOver = (char)0; |
| | | 362 | | |
| | 0 | 363 | | bool wasHereBefore = false; |
| | | 364 | | |
| | | 365 | | // For fallback we may need a fallback buffer |
| | 0 | 366 | | EncoderFallbackBuffer? fallbackBuffer = null; |
| | | 367 | | char* charsForFallback; |
| | | 368 | | |
| | 0 | 369 | | if (encoder is not null) |
| | | 370 | | { |
| | 0 | 371 | | charLeftOver = encoder._charLeftOver; |
| | | 372 | | |
| | | 373 | | // Assume extra bytes to encode charLeftOver if it existed |
| | 0 | 374 | | if (charLeftOver > 0) |
| | 0 | 375 | | byteCount += 2; |
| | | 376 | | |
| | | 377 | | // We mustn't have left over fallback data when counting |
| | 0 | 378 | | if (encoder.InternalHasFallbackBuffer) |
| | | 379 | | { |
| | 0 | 380 | | fallbackBuffer = encoder.FallbackBuffer; |
| | 0 | 381 | | if (fallbackBuffer.Remaining > 0) |
| | 0 | 382 | | throw new ArgumentException(SR.Format(SR.Argument_EncoderFallbackNotEmpty, this.EncodingName, en |
| | | 383 | | |
| | | 384 | | // Set our internal fallback interesting things. |
| | 0 | 385 | | fallbackBuffer.InternalInitialize(charStart, charEnd, encoder, false); |
| | | 386 | | } |
| | | 387 | | } |
| | | 388 | | |
| | | 389 | | char ch; |
| | | 390 | | TryAgain: |
| | | 391 | | |
| | 0 | 392 | | while (((ch = (fallbackBuffer is null) ? (char)0 : fallbackBuffer.InternalGetNextChar()) != 0) || chars < ch |
| | | 393 | | { |
| | | 394 | | // First unwind any fallback |
| | 0 | 395 | | if (ch == 0) |
| | | 396 | | { |
| | | 397 | | // No fallback, maybe we can do it fast |
| | | 398 | | #if FASTLOOP |
| | | 399 | | // If endianness is backwards then each pair of bytes would be backwards. |
| | 0 | 400 | | if ((bigEndian ^ BitConverter.IsLittleEndian) && |
| | 0 | 401 | | #if TARGET_64BIT |
| | 0 | 402 | | (unchecked((long)chars) & 7) == 0 && |
| | 0 | 403 | | #else |
| | 0 | 404 | | (unchecked((int)chars) & 3) == 0 && |
| | 0 | 405 | | #endif |
| | 0 | 406 | | charLeftOver == 0) |
| | | 407 | | { |
| | | 408 | | // Need -1 to check 2 at a time. If we have an even #, longChars will go |
| | | 409 | | // from longEnd - 1/2 long to longEnd + 1/2 long. If we're odd, longChars |
| | | 410 | | // will go from longEnd - 1 long to longEnd. (Might not get to use this) |
| | 0 | 411 | | ulong* longEnd = (ulong*)(charEnd - 3); |
| | | 412 | | |
| | | 413 | | // Need new char* so we can check 4 at a time |
| | 0 | 414 | | ulong* longChars = (ulong*)chars; |
| | | 415 | | |
| | 0 | 416 | | while (longChars < longEnd) |
| | | 417 | | { |
| | | 418 | | // See if we potentially have surrogates (0x8000 bit set) |
| | | 419 | | // (We're either big endian on a big endian machine or little endian on |
| | | 420 | | // a little endian machine so that'll work) |
| | 0 | 421 | | if ((0x8000800080008000 & *longChars) != 0) |
| | | 422 | | { |
| | | 423 | | // See if any of these are high or low surrogates (0xd800 - 0xdfff). If the high |
| | | 424 | | // 5 bits looks like 11011, then its a high or low surrogate. |
| | | 425 | | // We do the & f800 to filter the 5 bits, then ^ d800 to ensure the 0 isn't set. |
| | | 426 | | // Note that we expect BMP characters to be more common than surrogates |
| | | 427 | | // & each char with 11111... then ^ with 11011. Zeroes then indicate surrogates |
| | 0 | 428 | | ulong uTemp = (0xf800f800f800f800 & *longChars) ^ 0xd800d800d800d800; |
| | | 429 | | |
| | | 430 | | // Check each of the 4 chars. 0 for those 16 bits means it was a surrogate |
| | | 431 | | // but no clue if they're high or low. |
| | | 432 | | // If each of the 4 characters are non-zero, then none are surrogates. |
| | 0 | 433 | | if ((uTemp & 0xFFFF000000000000) == 0 || |
| | 0 | 434 | | (uTemp & 0x0000FFFF00000000) == 0 || |
| | 0 | 435 | | (uTemp & 0x00000000FFFF0000) == 0 || |
| | 0 | 436 | | (uTemp & 0x000000000000FFFF) == 0) |
| | | 437 | | { |
| | | 438 | | // It has at least 1 surrogate, but we don't know if they're high or low surrogates, |
| | | 439 | | // or if there's 1 or 4 surrogates |
| | | 440 | | |
| | | 441 | | // If they happen to be high/low/high/low, we may as well continue. Check the next |
| | | 442 | | // bit to see if its set (low) or not (high) in the right pattern |
| | 0 | 443 | | if ((0xfc00fc00fc00fc00 & *longChars) != |
| | 0 | 444 | | (BitConverter.IsLittleEndian ? (ulong)0xdc00d800dc00d800 : (ulong)0xd800dc00 |
| | | 445 | | { |
| | | 446 | | // Either there weren't 4 surrogates, or the 0x0400 bit was set when a high |
| | | 447 | | // was hoped for or the 0x0400 bit wasn't set where a low was hoped for. |
| | | 448 | | |
| | | 449 | | // Drop out to the slow loop to resolve the surrogates |
| | | 450 | | break; |
| | | 451 | | } |
| | | 452 | | // else they are all surrogates in High/Low/High/Low order, so we can use them. |
| | | 453 | | } |
| | | 454 | | // else none are surrogates, so we can use them. |
| | | 455 | | } |
| | | 456 | | // else all < 0x8000 so we can use them |
| | | 457 | | |
| | | 458 | | // We already counted these four chars, go to next long. |
| | 0 | 459 | | longChars++; |
| | | 460 | | } |
| | | 461 | | |
| | 0 | 462 | | chars = (char*)longChars; |
| | | 463 | | |
| | 0 | 464 | | if (chars >= charEnd) |
| | | 465 | | break; |
| | | 466 | | } |
| | | 467 | | #endif // FASTLOOP |
| | | 468 | | |
| | | 469 | | // No fallback, just get next char |
| | 0 | 470 | | ch = *chars; |
| | 0 | 471 | | chars++; |
| | | 472 | | } |
| | | 473 | | else |
| | | 474 | | { |
| | | 475 | | // We weren't preallocating fallback space. |
| | 0 | 476 | | byteCount += 2; |
| | | 477 | | } |
| | | 478 | | |
| | | 479 | | // Check for high or low surrogates |
| | 0 | 480 | | if (ch >= 0xd800 && ch <= 0xdfff) |
| | | 481 | | { |
| | | 482 | | // Was it a high surrogate? |
| | 0 | 483 | | if (ch <= 0xdbff) |
| | | 484 | | { |
| | | 485 | | // Its a high surrogate, if we already had a high surrogate do its fallback |
| | 0 | 486 | | if (charLeftOver > 0) |
| | | 487 | | { |
| | | 488 | | // Unwind the current character, this should be safe because we |
| | | 489 | | // don't have leftover data in the fallback, so chars must have |
| | | 490 | | // advanced already. |
| | 0 | 491 | | Debug.Assert(chars > charStart, |
| | 0 | 492 | | "[UnicodeEncoding.GetByteCount]Expected chars to have advanced in unexpected high surrog |
| | 0 | 493 | | chars--; |
| | | 494 | | |
| | | 495 | | // If previous high surrogate deallocate 2 bytes |
| | 0 | 496 | | byteCount -= 2; |
| | | 497 | | |
| | | 498 | | // Fallback the previous surrogate |
| | | 499 | | // Need to initialize fallback buffer? |
| | 0 | 500 | | if (fallbackBuffer is null) |
| | | 501 | | { |
| | 0 | 502 | | fallbackBuffer = encoder is null ? |
| | 0 | 503 | | this.encoderFallback.CreateFallbackBuffer() : |
| | 0 | 504 | | encoder.FallbackBuffer; |
| | | 505 | | |
| | | 506 | | // Set our internal fallback interesting things. |
| | 0 | 507 | | fallbackBuffer.InternalInitialize(charStart, charEnd, encoder, false); |
| | | 508 | | } |
| | | 509 | | |
| | 0 | 510 | | charsForFallback = chars; // Avoid passing chars by reference to allow it to be enregistered |
| | 0 | 511 | | fallbackBuffer.InternalFallback(charLeftOver, ref charsForFallback); |
| | 0 | 512 | | chars = charsForFallback; |
| | | 513 | | |
| | | 514 | | // Now no high surrogate left over |
| | 0 | 515 | | charLeftOver = (char)0; |
| | 0 | 516 | | continue; |
| | | 517 | | } |
| | | 518 | | |
| | | 519 | | // Remember this high surrogate |
| | 0 | 520 | | charLeftOver = ch; |
| | 0 | 521 | | continue; |
| | | 522 | | } |
| | | 523 | | |
| | | 524 | | // Its a low surrogate |
| | 0 | 525 | | if (charLeftOver == 0) |
| | | 526 | | { |
| | | 527 | | // Expected a previous high surrogate. |
| | | 528 | | // Don't count this one (we'll count its fallback if necessary) |
| | 0 | 529 | | byteCount -= 2; |
| | | 530 | | |
| | | 531 | | // fallback this one |
| | | 532 | | // Need to initialize fallback buffer? |
| | 0 | 533 | | if (fallbackBuffer is null) |
| | | 534 | | { |
| | 0 | 535 | | fallbackBuffer = encoder is null ? |
| | 0 | 536 | | this.encoderFallback.CreateFallbackBuffer() : |
| | 0 | 537 | | encoder.FallbackBuffer; |
| | | 538 | | |
| | | 539 | | // Set our internal fallback interesting things. |
| | 0 | 540 | | fallbackBuffer.InternalInitialize(charStart, charEnd, encoder, false); |
| | | 541 | | } |
| | 0 | 542 | | charsForFallback = chars; // Avoid passing chars by reference to allow it to be en-registered |
| | 0 | 543 | | fallbackBuffer.InternalFallback(ch, ref charsForFallback); |
| | 0 | 544 | | chars = charsForFallback; |
| | 0 | 545 | | continue; |
| | | 546 | | } |
| | | 547 | | |
| | | 548 | | // Valid surrogate pair, add our charLeftOver |
| | 0 | 549 | | charLeftOver = (char)0; |
| | 0 | 550 | | continue; |
| | | 551 | | } |
| | 0 | 552 | | else if (charLeftOver > 0) |
| | | 553 | | { |
| | | 554 | | // Expected a low surrogate, but this char is normal |
| | | 555 | | |
| | | 556 | | // Rewind the current character, fallback previous character. |
| | | 557 | | // this should be safe because we don't have leftover data in the |
| | | 558 | | // fallback, so chars must have advanced already. |
| | 0 | 559 | | Debug.Assert(chars > charStart, |
| | 0 | 560 | | "[UnicodeEncoding.GetByteCount]Expected chars to have advanced when expected low surrogate"); |
| | 0 | 561 | | chars--; |
| | | 562 | | |
| | | 563 | | // fallback previous chars |
| | | 564 | | // Need to initialize fallback buffer? |
| | 0 | 565 | | if (fallbackBuffer is null) |
| | | 566 | | { |
| | 0 | 567 | | fallbackBuffer = encoder is null ? |
| | 0 | 568 | | this.encoderFallback.CreateFallbackBuffer() : |
| | 0 | 569 | | encoder.FallbackBuffer; |
| | | 570 | | |
| | | 571 | | // Set our internal fallback interesting things. |
| | 0 | 572 | | fallbackBuffer.InternalInitialize(charStart, charEnd, encoder, false); |
| | | 573 | | } |
| | 0 | 574 | | charsForFallback = chars; // Avoid passing chars by reference to allow it to be en-registered |
| | 0 | 575 | | fallbackBuffer.InternalFallback(charLeftOver, ref charsForFallback); |
| | 0 | 576 | | chars = charsForFallback; |
| | | 577 | | |
| | | 578 | | // Ignore charLeftOver or throw |
| | 0 | 579 | | byteCount -= 2; |
| | 0 | 580 | | charLeftOver = (char)0; |
| | | 581 | | |
| | | 582 | | continue; |
| | | 583 | | } |
| | | 584 | | |
| | | 585 | | // Ok we had something to add (already counted) |
| | | 586 | | } |
| | | 587 | | |
| | | 588 | | // Don't allocate space for left over char |
| | 0 | 589 | | if (charLeftOver > 0) |
| | | 590 | | { |
| | 0 | 591 | | byteCount -= 2; |
| | | 592 | | |
| | | 593 | | // If we have to flush, stick it in fallback and try again |
| | 0 | 594 | | if (encoder is null || encoder.MustFlush) |
| | | 595 | | { |
| | 0 | 596 | | if (wasHereBefore) |
| | | 597 | | { |
| | | 598 | | // Throw it, using our complete character |
| | 0 | 599 | | throw new ArgumentException( |
| | 0 | 600 | | SR.Format(SR.Argument_RecursiveFallback, charLeftOver), nameof(chars)); |
| | | 601 | | } |
| | | 602 | | else |
| | | 603 | | { |
| | | 604 | | // Need to initialize fallback buffer? |
| | 0 | 605 | | if (fallbackBuffer is null) |
| | | 606 | | { |
| | 0 | 607 | | fallbackBuffer = encoder is null ? |
| | 0 | 608 | | this.encoderFallback.CreateFallbackBuffer() : |
| | 0 | 609 | | encoder.FallbackBuffer; |
| | | 610 | | |
| | | 611 | | // Set our internal fallback interesting things. |
| | 0 | 612 | | fallbackBuffer.InternalInitialize(charStart, charEnd, encoder, false); |
| | | 613 | | } |
| | 0 | 614 | | charsForFallback = chars; // Avoid passing chars by reference to allow it to be en-registered |
| | 0 | 615 | | fallbackBuffer.InternalFallback(charLeftOver, ref charsForFallback); |
| | 0 | 616 | | chars = charsForFallback; |
| | 0 | 617 | | charLeftOver = (char)0; |
| | 0 | 618 | | wasHereBefore = true; |
| | 0 | 619 | | goto TryAgain; |
| | | 620 | | } |
| | | 621 | | } |
| | | 622 | | } |
| | | 623 | | |
| | | 624 | | // Shouldn't have anything in fallback buffer for GetByteCount |
| | | 625 | | // (don't have to check _throwOnOverflow for count) |
| | 0 | 626 | | Debug.Assert(fallbackBuffer is null || fallbackBuffer.Remaining == 0, |
| | 0 | 627 | | "[UnicodeEncoding.GetByteCount]Expected empty fallback buffer at end"); |
| | | 628 | | |
| | | 629 | | // Don't remember fallbackBuffer.encoder for counting |
| | 0 | 630 | | return byteCount; |
| | | 631 | | } |
| | | 632 | | |
| | | 633 | | internal sealed override unsafe int GetBytes( |
| | | 634 | | char* chars, int charCount, byte* bytes, int byteCount, EncoderNLS? encoder) |
| | | 635 | | { |
| | | 636 | | Debug.Assert(chars is not null, "[UnicodeEncoding.GetBytes]chars!=null"); |
| | 4517 | 637 | | Debug.Assert(byteCount >= 0, "[UnicodeEncoding.GetBytes]byteCount >=0"); |
| | 4517 | 638 | | Debug.Assert(charCount >= 0, "[UnicodeEncoding.GetBytes]charCount >=0"); |
| | 4517 | 639 | | Debug.Assert(bytes is not null, "[UnicodeEncoding.GetBytes]bytes!=null"); |
| | | 640 | | |
| | 4517 | 641 | | char charLeftOver = (char)0; |
| | | 642 | | char ch; |
| | 4517 | 643 | | bool wasHereBefore = false; |
| | | 644 | | |
| | 4517 | 645 | | byte* byteEnd = bytes + byteCount; |
| | 4517 | 646 | | char* charEnd = chars + charCount; |
| | 4517 | 647 | | byte* byteStart = bytes; |
| | 4517 | 648 | | char* charStart = chars; |
| | | 649 | | |
| | | 650 | | // For fallback we may need a fallback buffer |
| | 4517 | 651 | | EncoderFallbackBuffer? fallbackBuffer = null; |
| | | 652 | | char* charsForFallback; |
| | | 653 | | |
| | | 654 | | // Get our encoder, but don't clear it yet. |
| | 4517 | 655 | | if (encoder is not null) |
| | | 656 | | { |
| | 4517 | 657 | | charLeftOver = encoder._charLeftOver; |
| | | 658 | | |
| | | 659 | | // We mustn't have left over fallback data when counting |
| | 4517 | 660 | | if (encoder.InternalHasFallbackBuffer) |
| | | 661 | | { |
| | | 662 | | // We always need the fallback buffer in get bytes so we can flush any remaining ones if necessary |
| | 0 | 663 | | fallbackBuffer = encoder.FallbackBuffer; |
| | 0 | 664 | | if (fallbackBuffer.Remaining > 0 && encoder._throwOnOverflow) |
| | 0 | 665 | | throw new ArgumentException(SR.Format(SR.Argument_EncoderFallbackNotEmpty, this.EncodingName, en |
| | | 666 | | |
| | | 667 | | // Set our internal fallback interesting things. |
| | 0 | 668 | | fallbackBuffer.InternalInitialize(charStart, charEnd, encoder, false); |
| | | 669 | | } |
| | | 670 | | } |
| | | 671 | | |
| | | 672 | | TryAgain: |
| | 20067 | 673 | | while (((ch = (fallbackBuffer is null) ? |
| | 20067 | 674 | | (char)0 : fallbackBuffer.InternalGetNextChar()) != 0) || |
| | 20067 | 675 | | chars < charEnd) |
| | | 676 | | { |
| | | 677 | | // First unwind any fallback |
| | 19077 | 678 | | if (ch == 0) |
| | | 679 | | { |
| | | 680 | | // No fallback, maybe we can do it fast |
| | | 681 | | #if FASTLOOP |
| | | 682 | | // If endianness is backwards then each pair of bytes would be backwards. |
| | 19077 | 683 | | if ((bigEndian ^ BitConverter.IsLittleEndian) && |
| | 19077 | 684 | | #if TARGET_64BIT |
| | 19077 | 685 | | (unchecked((long)chars) & 7) == 0 && |
| | 19077 | 686 | | #else |
| | 19077 | 687 | | (unchecked((int)chars) & 3) == 0 && |
| | 19077 | 688 | | #endif |
| | 19077 | 689 | | charLeftOver == 0) |
| | | 690 | | { |
| | | 691 | | // Need -1 to check 2 at a time. If we have an even #, longChars will go |
| | | 692 | | // from longEnd - 1/2 long to longEnd + 1/2 long. If we're odd, longChars |
| | | 693 | | // will go from longEnd - 1 long to longEnd. (Might not get to use this) |
| | | 694 | | // We can only go iCount units (limited by shorter of char or byte buffers. |
| | 5249 | 695 | | ulong* longEnd = (ulong*)(chars - 3 + |
| | 5249 | 696 | | (((byteEnd - bytes) >> 1 < charEnd - chars) ? |
| | 5249 | 697 | | (byteEnd - bytes) >> 1 : charEnd - chars)); |
| | | 698 | | |
| | | 699 | | // Need new char* so we can check 4 at a time |
| | 5249 | 700 | | ulong* longChars = (ulong*)chars; |
| | 5249 | 701 | | ulong* longBytes = (ulong*)bytes; |
| | | 702 | | |
| | 46488 | 703 | | while (longChars < longEnd) |
| | | 704 | | { |
| | | 705 | | // See if we potentially have surrogates (0x8000 bit set) |
| | | 706 | | // (We're either big endian on a big endian machine or little endian on |
| | | 707 | | // a little endian machine so that'll work) |
| | 42961 | 708 | | if ((0x8000800080008000 & *longChars) != 0) |
| | | 709 | | { |
| | | 710 | | // See if any of these are high or low surrogates (0xd800 - 0xdfff). If the high |
| | | 711 | | // 5 bits looks like 11011, then its a high or low surrogate. |
| | | 712 | | // We do the & f800 to filter the 5 bits, then ^ d800 to ensure the 0 isn't set. |
| | | 713 | | // Note that we expect BMP characters to be more common than surrogates |
| | | 714 | | // & each char with 11111... then ^ with 11011. Zeroes then indicate surrogates |
| | 31329 | 715 | | ulong uTemp = (0xf800f800f800f800 & *longChars) ^ 0xd800d800d800d800; |
| | | 716 | | |
| | | 717 | | // Check each of the 4 chars. 0 for those 16 bits means it was a surrogate |
| | | 718 | | // but no clue if they're high or low. |
| | | 719 | | // If each of the 4 characters are non-zero, then none are surrogates. |
| | 31329 | 720 | | if ((uTemp & 0xFFFF000000000000) == 0 || |
| | 31329 | 721 | | (uTemp & 0x0000FFFF00000000) == 0 || |
| | 31329 | 722 | | (uTemp & 0x00000000FFFF0000) == 0 || |
| | 31329 | 723 | | (uTemp & 0x000000000000FFFF) == 0) |
| | | 724 | | { |
| | | 725 | | // It has at least 1 surrogate, but we don't know if they're high or low surrogates, |
| | | 726 | | // or if there's 1 or 4 surrogates |
| | | 727 | | |
| | | 728 | | // If they happen to be high/low/high/low, we may as well continue. Check the next |
| | | 729 | | // bit to see if its set (low) or not (high) in the right pattern |
| | 1779 | 730 | | if ((0xfc00fc00fc00fc00 & *longChars) != |
| | 1779 | 731 | | (BitConverter.IsLittleEndian ? (ulong)0xdc00d800dc00d800 : (ulong)0xd800dc00 |
| | | 732 | | { |
| | | 733 | | // Either there weren't 4 surrogates, or the 0x0400 bit was set when a high |
| | | 734 | | // was hoped for or the 0x0400 bit wasn't set where a low was hoped for. |
| | | 735 | | |
| | | 736 | | // Drop out to the slow loop to resolve the surrogates |
| | | 737 | | break; |
| | | 738 | | } |
| | | 739 | | // else they are all surrogates in High/Low/High/Low order, so we can use them. |
| | | 740 | | } |
| | | 741 | | // else none are surrogates, so we can use them. |
| | | 742 | | } |
| | | 743 | | // else all < 0x8000 so we can use them |
| | | 744 | | |
| | | 745 | | // We can use these 4 chars. |
| | 41239 | 746 | | Unsafe.WriteUnaligned(longBytes, *longChars); |
| | 41239 | 747 | | longChars++; |
| | 41239 | 748 | | longBytes++; |
| | | 749 | | } |
| | | 750 | | |
| | 5249 | 751 | | chars = (char*)longChars; |
| | 5249 | 752 | | bytes = (byte*)longBytes; |
| | | 753 | | |
| | 5249 | 754 | | if (chars >= charEnd) |
| | | 755 | | break; |
| | | 756 | | } |
| | | 757 | | #endif // FASTLOOP |
| | | 758 | | |
| | | 759 | | // No fallback, just get next char |
| | 15550 | 760 | | ch = *chars; |
| | 15550 | 761 | | chars++; |
| | | 762 | | } |
| | | 763 | | |
| | | 764 | | // Check for high or low surrogates |
| | 15550 | 765 | | if (ch >= 0xd800 && ch <= 0xdfff) |
| | | 766 | | { |
| | | 767 | | // Was it a high surrogate? |
| | 3984 | 768 | | if (ch <= 0xdbff) |
| | | 769 | | { |
| | | 770 | | // Its a high surrogate, see if we already had a high surrogate |
| | 1992 | 771 | | if (charLeftOver > 0) |
| | | 772 | | { |
| | | 773 | | // Unwind the current character, this should be safe because we |
| | | 774 | | // don't have leftover data in the fallback, so chars must have |
| | | 775 | | // advanced already. |
| | 0 | 776 | | Debug.Assert(chars > charStart, |
| | 0 | 777 | | "[UnicodeEncoding.GetBytes]Expected chars to have advanced in unexpected high surrogate" |
| | 0 | 778 | | chars--; |
| | | 779 | | |
| | | 780 | | // Fallback the previous surrogate |
| | | 781 | | // Might need to create our fallback buffer |
| | 0 | 782 | | if (fallbackBuffer is null) |
| | | 783 | | { |
| | 0 | 784 | | fallbackBuffer = encoder is null ? |
| | 0 | 785 | | this.encoderFallback.CreateFallbackBuffer() : |
| | 0 | 786 | | encoder.FallbackBuffer; |
| | | 787 | | |
| | | 788 | | // Set our internal fallback interesting things. |
| | 0 | 789 | | fallbackBuffer.InternalInitialize(charStart, charEnd, encoder, true); |
| | | 790 | | } |
| | | 791 | | |
| | 0 | 792 | | charsForFallback = chars; // Avoid passing chars by reference to allow it to be en-registere |
| | 0 | 793 | | fallbackBuffer.InternalFallback(charLeftOver, ref charsForFallback); |
| | 0 | 794 | | chars = charsForFallback; |
| | | 795 | | |
| | 0 | 796 | | charLeftOver = (char)0; |
| | 0 | 797 | | continue; |
| | | 798 | | } |
| | | 799 | | |
| | | 800 | | // Remember this high surrogate |
| | 1992 | 801 | | charLeftOver = ch; |
| | 1992 | 802 | | continue; |
| | | 803 | | } |
| | | 804 | | |
| | | 805 | | // Its a low surrogate |
| | 1992 | 806 | | if (charLeftOver == 0) |
| | | 807 | | { |
| | | 808 | | // We'll fall back this one |
| | | 809 | | // Might need to create our fallback buffer |
| | 0 | 810 | | if (fallbackBuffer is null) |
| | | 811 | | { |
| | 0 | 812 | | fallbackBuffer = encoder is null ? |
| | 0 | 813 | | this.encoderFallback.CreateFallbackBuffer() : |
| | 0 | 814 | | encoder.FallbackBuffer; |
| | | 815 | | |
| | | 816 | | // Set our internal fallback interesting things. |
| | 0 | 817 | | fallbackBuffer.InternalInitialize(charStart, charEnd, encoder, true); |
| | | 818 | | } |
| | | 819 | | |
| | 0 | 820 | | charsForFallback = chars; // Avoid passing chars by reference to allow it to be en-registered |
| | 0 | 821 | | fallbackBuffer.InternalFallback(ch, ref charsForFallback); |
| | 0 | 822 | | chars = charsForFallback; |
| | 0 | 823 | | continue; |
| | | 824 | | } |
| | | 825 | | |
| | | 826 | | // Valid surrogate pair, add our charLeftOver |
| | 1992 | 827 | | if (bytes + 3 >= byteEnd) |
| | | 828 | | { |
| | | 829 | | // Not enough room to add this surrogate pair |
| | 0 | 830 | | if (fallbackBuffer is not null && fallbackBuffer.bFallingBack) |
| | | 831 | | { |
| | | 832 | | // These must have both been from the fallbacks. |
| | | 833 | | // Both of these MUST have been from a fallback because if the 1st wasn't |
| | | 834 | | // from a fallback, then a high surrogate followed by an illegal char |
| | | 835 | | // would've caused the high surrogate to fall back. If a high surrogate |
| | | 836 | | // fell back, then it was consumed and both chars came from the fallback. |
| | 0 | 837 | | fallbackBuffer.MovePrevious(); // Didn't use either fallback surrogate |
| | 0 | 838 | | fallbackBuffer.MovePrevious(); |
| | | 839 | | } |
| | | 840 | | else |
| | | 841 | | { |
| | | 842 | | // If we don't have enough room, then either we should've advanced a while |
| | | 843 | | // or we should have bytes==byteStart and throw below |
| | 0 | 844 | | Debug.Assert(chars > charStart + 1 || bytes == byteStart, |
| | 0 | 845 | | "[UnicodeEncoding.GetBytes]Expected chars to have when no room to add surrogate pair"); |
| | 0 | 846 | | chars -= 2; // Didn't use either surrogate |
| | | 847 | | } |
| | 0 | 848 | | ThrowBytesOverflow(encoder, bytes == byteStart); // Throw maybe (if no bytes written) |
| | 0 | 849 | | charLeftOver = (char)0; // we'll retry it later |
| | 0 | 850 | | break; // Didn't throw, but stop 'til next time. |
| | | 851 | | } |
| | | 852 | | |
| | 1992 | 853 | | if (bigEndian) |
| | | 854 | | { |
| | 0 | 855 | | *(bytes++) = (byte)(charLeftOver >> 8); |
| | 0 | 856 | | *(bytes++) = (byte)charLeftOver; |
| | | 857 | | } |
| | | 858 | | else |
| | | 859 | | { |
| | 1992 | 860 | | *(bytes++) = (byte)charLeftOver; |
| | 1992 | 861 | | *(bytes++) = (byte)(charLeftOver >> 8); |
| | | 862 | | } |
| | | 863 | | |
| | 1992 | 864 | | charLeftOver = (char)0; |
| | | 865 | | } |
| | 11566 | 866 | | else if (charLeftOver > 0) |
| | | 867 | | { |
| | | 868 | | // Expected a low surrogate, but this char is normal |
| | | 869 | | |
| | | 870 | | // Rewind the current character, fallback previous character. |
| | | 871 | | // this should be safe because we don't have leftover data in the |
| | | 872 | | // fallback, so chars must have advanced already. |
| | 0 | 873 | | Debug.Assert(chars > charStart, |
| | 0 | 874 | | "[UnicodeEncoding.GetBytes]Expected chars to have advanced after expecting low surrogate"); |
| | 0 | 875 | | chars--; |
| | | 876 | | |
| | | 877 | | // fallback previous chars |
| | | 878 | | // Might need to create our fallback buffer |
| | 0 | 879 | | if (fallbackBuffer is null) |
| | | 880 | | { |
| | 0 | 881 | | fallbackBuffer = encoder is null ? |
| | 0 | 882 | | this.encoderFallback.CreateFallbackBuffer() : |
| | 0 | 883 | | encoder.FallbackBuffer; |
| | | 884 | | |
| | | 885 | | // Set our internal fallback interesting things. |
| | 0 | 886 | | fallbackBuffer.InternalInitialize(charStart, charEnd, encoder, true); |
| | | 887 | | } |
| | | 888 | | |
| | 0 | 889 | | charsForFallback = chars; // Avoid passing chars by reference to allow it to be en-registered |
| | 0 | 890 | | fallbackBuffer.InternalFallback(charLeftOver, ref charsForFallback); |
| | 0 | 891 | | chars = charsForFallback; |
| | | 892 | | |
| | | 893 | | // Ignore charLeftOver or throw |
| | 0 | 894 | | charLeftOver = (char)0; |
| | 0 | 895 | | continue; |
| | | 896 | | } |
| | | 897 | | |
| | | 898 | | // Ok, we have a char to add |
| | 13558 | 899 | | if (bytes + 1 >= byteEnd) |
| | | 900 | | { |
| | | 901 | | // Couldn't add this char |
| | 0 | 902 | | if (fallbackBuffer is not null && fallbackBuffer.bFallingBack) |
| | 0 | 903 | | fallbackBuffer.MovePrevious(); // Not using this fallback char |
| | | 904 | | else |
| | | 905 | | { |
| | | 906 | | // Lonely charLeftOver (from previous call) would've been caught up above, |
| | | 907 | | // so this must be a case where we've already read an input char. |
| | 0 | 908 | | Debug.Assert(chars > charStart, |
| | 0 | 909 | | "[UnicodeEncoding.GetBytes]Expected chars to have advanced for failed fallback"); |
| | 0 | 910 | | chars--; // Not using this char |
| | | 911 | | } |
| | 0 | 912 | | ThrowBytesOverflow(encoder, bytes == byteStart); // Throw maybe (if no bytes written) |
| | 0 | 913 | | break; // didn't throw, just stop |
| | | 914 | | } |
| | | 915 | | |
| | 13558 | 916 | | if (bigEndian) |
| | | 917 | | { |
| | 0 | 918 | | *(bytes++) = (byte)(ch >> 8); |
| | 0 | 919 | | *(bytes++) = (byte)ch; |
| | | 920 | | } |
| | | 921 | | else |
| | | 922 | | { |
| | 13558 | 923 | | *(bytes++) = (byte)ch; |
| | 13558 | 924 | | *(bytes++) = (byte)(ch >> 8); |
| | | 925 | | } |
| | | 926 | | } |
| | | 927 | | |
| | | 928 | | // Don't allocate space for left over char |
| | 4517 | 929 | | if (charLeftOver > 0) |
| | | 930 | | { |
| | | 931 | | // If we aren't flushing we need to fall this back |
| | 0 | 932 | | if (encoder is null || encoder.MustFlush) |
| | | 933 | | { |
| | 0 | 934 | | if (wasHereBefore) |
| | | 935 | | { |
| | | 936 | | // Throw it, using our complete character |
| | 0 | 937 | | throw new ArgumentException( |
| | 0 | 938 | | SR.Format(SR.Argument_RecursiveFallback, charLeftOver), nameof(chars)); |
| | | 939 | | } |
| | | 940 | | else |
| | | 941 | | { |
| | | 942 | | // If we have to flush, stick it in fallback and try again |
| | | 943 | | // Might need to create our fallback buffer |
| | 0 | 944 | | if (fallbackBuffer is null) |
| | | 945 | | { |
| | 0 | 946 | | fallbackBuffer = encoder is null ? |
| | 0 | 947 | | this.encoderFallback.CreateFallbackBuffer() : |
| | 0 | 948 | | encoder.FallbackBuffer; |
| | | 949 | | |
| | | 950 | | // Set our internal fallback interesting things. |
| | 0 | 951 | | fallbackBuffer.InternalInitialize(charStart, charEnd, encoder, true); |
| | | 952 | | } |
| | | 953 | | |
| | | 954 | | // If we're not flushing, that'll remember the left over character. |
| | 0 | 955 | | charsForFallback = chars; // Avoid passing chars by reference to allow it to be en-registered |
| | 0 | 956 | | fallbackBuffer.InternalFallback(charLeftOver, ref charsForFallback); |
| | 0 | 957 | | chars = charsForFallback; |
| | | 958 | | |
| | 0 | 959 | | charLeftOver = (char)0; |
| | 0 | 960 | | wasHereBefore = true; |
| | 0 | 961 | | goto TryAgain; |
| | | 962 | | } |
| | | 963 | | } |
| | | 964 | | } |
| | | 965 | | |
| | | 966 | | // Not flushing, remember it in the encoder |
| | 4517 | 967 | | if (encoder is not null) |
| | | 968 | | { |
| | 4517 | 969 | | encoder._charLeftOver = charLeftOver; |
| | 4517 | 970 | | encoder._charsUsed = (int)(chars - charStart); |
| | | 971 | | } |
| | | 972 | | |
| | | 973 | | // Remember charLeftOver if we must, or clear it if we're flushing |
| | | 974 | | // (charLeftOver should be 0 if we're flushing) |
| | 4517 | 975 | | Debug.Assert((encoder is not null && !encoder.MustFlush) || charLeftOver == (char)0, |
| | 4517 | 976 | | "[UnicodeEncoding.GetBytes] Expected no left over characters if flushing"); |
| | | 977 | | |
| | 4517 | 978 | | Debug.Assert(fallbackBuffer is null || fallbackBuffer.Remaining == 0 || |
| | 4517 | 979 | | encoder is null || !encoder._throwOnOverflow, |
| | 4517 | 980 | | "[UnicodeEncoding.GetBytes]Expected empty fallback buffer if not converting"); |
| | | 981 | | |
| | 4517 | 982 | | return (int)(bytes - byteStart); |
| | | 983 | | } |
| | | 984 | | |
| | | 985 | | internal sealed override unsafe int GetCharCount(byte* bytes, int count, DecoderNLS? baseDecoder) |
| | | 986 | | { |
| | | 987 | | Debug.Assert(bytes is not null, "[UnicodeEncoding.GetCharCount]bytes!=null"); |
| | 30357 | 988 | | Debug.Assert(count >= 0, "[UnicodeEncoding.GetCharCount]count >=0"); |
| | | 989 | | |
| | 30357 | 990 | | Decoder? decoder = (Decoder?)baseDecoder; |
| | | 991 | | |
| | 30357 | 992 | | byte* byteEnd = bytes + count; |
| | 30357 | 993 | | byte* byteStart = bytes; |
| | | 994 | | |
| | | 995 | | // Need last vars |
| | 30357 | 996 | | int lastByte = -1; |
| | 30357 | 997 | | char lastChar = (char)0; |
| | | 998 | | |
| | | 999 | | // Start by assuming same # of chars as bytes |
| | 30357 | 1000 | | int charCount = count >> 1; |
| | | 1001 | | |
| | | 1002 | | // For fallback we may need a fallback buffer |
| | 30357 | 1003 | | DecoderFallbackBuffer? fallbackBuffer = null; |
| | | 1004 | | |
| | 30357 | 1005 | | if (decoder is not null) |
| | | 1006 | | { |
| | 30357 | 1007 | | lastByte = decoder.lastByte; |
| | 30357 | 1008 | | lastChar = decoder.lastChar; |
| | | 1009 | | |
| | | 1010 | | // Assume extra char if last char was around |
| | 30357 | 1011 | | if (lastChar > 0) |
| | 0 | 1012 | | charCount++; |
| | | 1013 | | |
| | | 1014 | | // Assume extra char if extra last byte makes up odd # of input bytes |
| | 30357 | 1015 | | if (lastByte >= 0 && (count & 1) == 1) |
| | | 1016 | | { |
| | 0 | 1017 | | charCount++; |
| | | 1018 | | } |
| | | 1019 | | |
| | | 1020 | | // Shouldn't have anything in fallback buffer for GetCharCount |
| | | 1021 | | // (don't have to check _throwOnOverflow for count) |
| | 30357 | 1022 | | Debug.Assert(!decoder.InternalHasFallbackBuffer || decoder.FallbackBuffer.Remaining == 0, |
| | 30357 | 1023 | | "[UnicodeEncoding.GetCharCount]Expected empty fallback buffer at start"); |
| | | 1024 | | } |
| | | 1025 | | |
| | 801162 | 1026 | | while (bytes < byteEnd) |
| | | 1027 | | { |
| | | 1028 | | // If we're aligned then maybe we can do it fast |
| | | 1029 | | // That'll hurt if we're unaligned because we'll always test but never be aligned |
| | | 1030 | | #if FASTLOOP |
| | 794814 | 1031 | | if ((bigEndian ^ BitConverter.IsLittleEndian) && |
| | 794814 | 1032 | | #if TARGET_64BIT |
| | 794814 | 1033 | | (unchecked((long)bytes) & 7) == 0 && |
| | 794814 | 1034 | | #else |
| | 794814 | 1035 | | (unchecked((int)bytes) & 3) == 0 && |
| | 794814 | 1036 | | #endif // TARGET_64BIT |
| | 794814 | 1037 | | lastByte == -1 && lastChar == 0) |
| | | 1038 | | { |
| | | 1039 | | // Need -1 to check 2 at a time. If we have an even #, longBytes will go |
| | | 1040 | | // from longEnd - 1/2 long to longEnd + 1/2 long. If we're odd, longBytes |
| | | 1041 | | // will go from longEnd - 1 long to longEnd. (Might not get to use this) |
| | 42904 | 1042 | | ulong* longEnd = (ulong*)(byteEnd - 7); |
| | | 1043 | | |
| | | 1044 | | // Need new char* so we can check 4 at a time |
| | 42904 | 1045 | | ulong* longBytes = (ulong*)bytes; |
| | | 1046 | | |
| | 163678 | 1047 | | while (longBytes < longEnd) |
| | | 1048 | | { |
| | | 1049 | | // See if we potentially have surrogates (0x8000 bit set) |
| | | 1050 | | // (We're either big endian on a big endian machine or little endian on |
| | | 1051 | | // a little endian machine so that'll work) |
| | 153958 | 1052 | | if ((0x8000800080008000 & *longBytes) != 0) |
| | | 1053 | | { |
| | | 1054 | | // See if any of these are high or low surrogates (0xd800 - 0xdfff). If the high |
| | | 1055 | | // 5 bits looks like 11011, then its a high or low surrogate. |
| | | 1056 | | // We do the & f800 to filter the 5 bits, then ^ d800 to ensure the 0 isn't set. |
| | | 1057 | | // Note that we expect BMP characters to be more common than surrogates |
| | | 1058 | | // & each char with 11111... then ^ with 11011. Zeroes then indicate surrogates |
| | 108118 | 1059 | | ulong uTemp = (0xf800f800f800f800 & *longBytes) ^ 0xd800d800d800d800; |
| | | 1060 | | |
| | | 1061 | | // Check each of the 4 chars. 0 for those 16 bits means it was a surrogate |
| | | 1062 | | // but no clue if they're high or low. |
| | | 1063 | | // If each of the 4 characters are non-zero, then none are surrogates. |
| | 108118 | 1064 | | if ((uTemp & 0xFFFF000000000000) == 0 || |
| | 108118 | 1065 | | (uTemp & 0x0000FFFF00000000) == 0 || |
| | 108118 | 1066 | | (uTemp & 0x00000000FFFF0000) == 0 || |
| | 108118 | 1067 | | (uTemp & 0x000000000000FFFF) == 0) |
| | | 1068 | | { |
| | | 1069 | | // It has at least 1 surrogate, but we don't know if they're high or low surrogates, |
| | | 1070 | | // or if there's 1 or 4 surrogates |
| | | 1071 | | |
| | | 1072 | | // If they happen to be high/low/high/low, we may as well continue. Check the next |
| | | 1073 | | // bit to see if its set (low) or not (high) in the right pattern |
| | 33482 | 1074 | | if ((0xfc00fc00fc00fc00 & *longBytes) != |
| | 33482 | 1075 | | (BitConverter.IsLittleEndian ? (ulong)0xdc00d800dc00d800 : (ulong)0xd800dc00d800 |
| | | 1076 | | { |
| | | 1077 | | // Either there weren't 4 surrogates, or the 0x0400 bit was set when a high |
| | | 1078 | | // was hoped for or the 0x0400 bit wasn't set where a low was hoped for. |
| | | 1079 | | |
| | | 1080 | | // Drop out to the slow loop to resolve the surrogates |
| | | 1081 | | break; |
| | | 1082 | | } |
| | | 1083 | | // else they are all surrogates in High/Low/High/Low order, so we can use them. |
| | | 1084 | | } |
| | | 1085 | | // else none are surrogates, so we can use them. |
| | | 1086 | | } |
| | | 1087 | | // else all < 0x8000 so we can use them |
| | | 1088 | | |
| | | 1089 | | // We can use these 4 chars. |
| | 120774 | 1090 | | longBytes++; |
| | | 1091 | | } |
| | | 1092 | | |
| | 42904 | 1093 | | bytes = (byte*)longBytes; |
| | | 1094 | | |
| | 42904 | 1095 | | if (bytes >= byteEnd) |
| | | 1096 | | break; |
| | | 1097 | | } |
| | | 1098 | | #endif // FASTLOOP |
| | | 1099 | | |
| | | 1100 | | // Get 1st byte |
| | 785094 | 1101 | | if (lastByte < 0) |
| | | 1102 | | { |
| | 785094 | 1103 | | lastByte = *bytes++; |
| | 785094 | 1104 | | if (bytes >= byteEnd) break; |
| | | 1105 | | } |
| | | 1106 | | |
| | | 1107 | | // Get full char |
| | | 1108 | | char ch; |
| | 772165 | 1109 | | if (bigEndian) |
| | | 1110 | | { |
| | 0 | 1111 | | ch = (char)(lastByte << 8 | *(bytes++)); |
| | | 1112 | | } |
| | | 1113 | | else |
| | | 1114 | | { |
| | 772165 | 1115 | | ch = (char)(*(bytes++) << 8 | lastByte); |
| | | 1116 | | } |
| | 772165 | 1117 | | lastByte = -1; |
| | | 1118 | | |
| | | 1119 | | // See if the char's valid |
| | 772165 | 1120 | | if (ch >= 0xd800 && ch <= 0xdfff) |
| | | 1121 | | { |
| | | 1122 | | // Was it a high surrogate? |
| | 166162 | 1123 | | if (ch <= 0xdbff) |
| | | 1124 | | { |
| | | 1125 | | // Its a high surrogate, if we had one then do fallback for previous one |
| | 95057 | 1126 | | if (lastChar > 0) |
| | | 1127 | | { |
| | | 1128 | | // Ignore previous bad high surrogate |
| | 34928 | 1129 | | charCount--; |
| | | 1130 | | |
| | | 1131 | | // Get fallback for previous high surrogate |
| | | 1132 | | // Note we have to reconstruct bytes because some may have been in decoder |
| | 34928 | 1133 | | byte[]? byteBuffer = null; |
| | 34928 | 1134 | | if (bigEndian) |
| | | 1135 | | { |
| | 0 | 1136 | | byteBuffer = [unchecked((byte)(lastChar >> 8)), unchecked((byte)lastChar)]; |
| | | 1137 | | } |
| | | 1138 | | else |
| | | 1139 | | { |
| | 34928 | 1140 | | byteBuffer = [unchecked((byte)lastChar), unchecked((byte)(lastChar >> 8))]; |
| | | 1141 | | } |
| | | 1142 | | |
| | 34928 | 1143 | | if (fallbackBuffer is null) |
| | | 1144 | | { |
| | 2186 | 1145 | | fallbackBuffer = decoder is null ? |
| | 2186 | 1146 | | this.decoderFallback.CreateFallbackBuffer() : |
| | 2186 | 1147 | | decoder.FallbackBuffer; |
| | | 1148 | | |
| | | 1149 | | // Set our internal fallback interesting things. |
| | 2186 | 1150 | | fallbackBuffer.InternalInitialize(byteStart, null); |
| | | 1151 | | } |
| | | 1152 | | |
| | | 1153 | | // Get fallback. |
| | 34928 | 1154 | | charCount += fallbackBuffer.InternalFallback(byteBuffer, bytes); |
| | | 1155 | | } |
| | | 1156 | | |
| | | 1157 | | // Ignore the last one which fell back already, |
| | | 1158 | | // and remember the new high surrogate |
| | 94811 | 1159 | | lastChar = ch; |
| | 94811 | 1160 | | continue; |
| | | 1161 | | } |
| | | 1162 | | |
| | | 1163 | | // Its a low surrogate |
| | 71105 | 1164 | | if (lastChar == 0) |
| | | 1165 | | { |
| | | 1166 | | // Expected a previous high surrogate |
| | 57080 | 1167 | | charCount--; |
| | | 1168 | | |
| | | 1169 | | // Get fallback for this low surrogate |
| | | 1170 | | // Note we have to reconstruct bytes because some may have been in decoder |
| | 57080 | 1171 | | byte[]? byteBuffer = null; |
| | 57080 | 1172 | | if (bigEndian) |
| | | 1173 | | { |
| | 0 | 1174 | | byteBuffer = [unchecked((byte)(ch >> 8)), unchecked((byte)ch)]; |
| | | 1175 | | } |
| | | 1176 | | else |
| | | 1177 | | { |
| | 57080 | 1178 | | byteBuffer = [unchecked((byte)ch), unchecked((byte)(ch >> 8))]; |
| | | 1179 | | } |
| | | 1180 | | |
| | 57080 | 1181 | | if (fallbackBuffer is null) |
| | | 1182 | | { |
| | 5066 | 1183 | | fallbackBuffer = decoder is null ? |
| | 5066 | 1184 | | this.decoderFallback.CreateFallbackBuffer() : |
| | 5066 | 1185 | | decoder.FallbackBuffer; |
| | | 1186 | | |
| | | 1187 | | // Set our internal fallback interesting things. |
| | 5066 | 1188 | | fallbackBuffer.InternalInitialize(byteStart, null); |
| | | 1189 | | } |
| | | 1190 | | |
| | 57080 | 1191 | | charCount += fallbackBuffer.InternalFallback(byteBuffer, bytes); |
| | | 1192 | | |
| | | 1193 | | // Ignore this one (we already did its fallback) |
| | 56506 | 1194 | | continue; |
| | | 1195 | | } |
| | | 1196 | | |
| | | 1197 | | // Valid surrogate pair, already counted. |
| | 14025 | 1198 | | lastChar = (char)0; |
| | | 1199 | | } |
| | 606003 | 1200 | | else if (lastChar > 0) |
| | | 1201 | | { |
| | | 1202 | | // Had a high surrogate, expected a low surrogate |
| | | 1203 | | // Un-count the last high surrogate |
| | 44443 | 1204 | | charCount--; |
| | | 1205 | | |
| | | 1206 | | // fall back the high surrogate. |
| | 44443 | 1207 | | byte[]? byteBuffer = null; |
| | 44443 | 1208 | | if (bigEndian) |
| | | 1209 | | { |
| | 0 | 1210 | | byteBuffer = [unchecked((byte)(lastChar >> 8)), unchecked((byte)lastChar)]; |
| | | 1211 | | } |
| | | 1212 | | else |
| | | 1213 | | { |
| | 44443 | 1214 | | byteBuffer = [unchecked((byte)lastChar), unchecked((byte)(lastChar >> 8))]; |
| | | 1215 | | } |
| | | 1216 | | |
| | 44443 | 1217 | | if (fallbackBuffer is null) |
| | | 1218 | | { |
| | 4798 | 1219 | | fallbackBuffer = decoder is null ? |
| | 4798 | 1220 | | this.decoderFallback.CreateFallbackBuffer() : |
| | 4798 | 1221 | | decoder.FallbackBuffer; |
| | | 1222 | | |
| | | 1223 | | // Set our internal fallback interesting things. |
| | 4798 | 1224 | | fallbackBuffer.InternalInitialize(byteStart, null); |
| | | 1225 | | } |
| | | 1226 | | |
| | | 1227 | | // Already subtracted high surrogate |
| | 44443 | 1228 | | charCount += fallbackBuffer.InternalFallback(byteBuffer, bytes); |
| | | 1229 | | |
| | | 1230 | | // Not left over now, clear previous high surrogate and continue to add current char |
| | 43903 | 1231 | | lastChar = (char)0; |
| | | 1232 | | } |
| | | 1233 | | |
| | | 1234 | | // Valid char, already counted |
| | | 1235 | | } |
| | | 1236 | | |
| | | 1237 | | // Extra space if we can't use decoder |
| | 28997 | 1238 | | if (decoder is null || decoder.MustFlush) |
| | | 1239 | | { |
| | 28997 | 1240 | | if (lastChar > 0) |
| | | 1241 | | { |
| | | 1242 | | // No hanging high surrogates allowed, do fallback and remove count for it |
| | 1415 | 1243 | | charCount--; |
| | 1415 | 1244 | | byte[]? byteBuffer = null; |
| | 1415 | 1245 | | if (bigEndian) |
| | | 1246 | | { |
| | 0 | 1247 | | byteBuffer = [unchecked((byte)(lastChar >> 8)), unchecked((byte)lastChar)]; |
| | | 1248 | | } |
| | | 1249 | | else |
| | | 1250 | | { |
| | 1415 | 1251 | | byteBuffer = [unchecked((byte)lastChar), unchecked((byte)(lastChar >> 8))]; |
| | | 1252 | | } |
| | | 1253 | | |
| | 1415 | 1254 | | if (fallbackBuffer is null) |
| | | 1255 | | { |
| | 203 | 1256 | | fallbackBuffer = decoder is null ? |
| | 203 | 1257 | | this.decoderFallback.CreateFallbackBuffer() : |
| | 203 | 1258 | | decoder.FallbackBuffer; |
| | | 1259 | | |
| | | 1260 | | // Set our internal fallback interesting things. |
| | 203 | 1261 | | fallbackBuffer.InternalInitialize(byteStart, null); |
| | | 1262 | | } |
| | | 1263 | | |
| | 1415 | 1264 | | charCount += fallbackBuffer.InternalFallback(byteBuffer, bytes); |
| | | 1265 | | |
| | 1389 | 1266 | | lastChar = (char)0; |
| | | 1267 | | } |
| | | 1268 | | |
| | 28971 | 1269 | | if (lastByte >= 0) |
| | | 1270 | | { |
| | 12919 | 1271 | | if (fallbackBuffer is null) |
| | | 1272 | | { |
| | 7580 | 1273 | | fallbackBuffer = decoder is null ? |
| | 7580 | 1274 | | this.decoderFallback.CreateFallbackBuffer() : |
| | 7580 | 1275 | | decoder.FallbackBuffer; |
| | | 1276 | | |
| | | 1277 | | // Set our internal fallback interesting things. |
| | 7580 | 1278 | | fallbackBuffer.InternalInitialize(byteStart, null); |
| | | 1279 | | } |
| | | 1280 | | |
| | | 1281 | | // No hanging odd bytes allowed if must flush |
| | 12919 | 1282 | | charCount += fallbackBuffer.InternalFallback([unchecked((byte)lastByte)], bytes); |
| | 12016 | 1283 | | lastByte = -1; |
| | | 1284 | | } |
| | | 1285 | | } |
| | | 1286 | | |
| | | 1287 | | // If we had a high surrogate left over, we can't count it |
| | 28068 | 1288 | | if (lastChar > 0) |
| | 0 | 1289 | | charCount--; |
| | | 1290 | | |
| | | 1291 | | // Shouldn't have anything in fallback buffer for GetCharCount |
| | | 1292 | | // (don't have to check _throwOnOverflow for count) |
| | 28068 | 1293 | | Debug.Assert(fallbackBuffer is null || fallbackBuffer.Remaining == 0, |
| | 28068 | 1294 | | "[UnicodeEncoding.GetCharCount]Expected empty fallback buffer at end"); |
| | | 1295 | | |
| | 28068 | 1296 | | return charCount; |
| | | 1297 | | } |
| | | 1298 | | |
| | | 1299 | | internal sealed override unsafe int GetChars( |
| | | 1300 | | byte* bytes, int byteCount, char* chars, int charCount, DecoderNLS? baseDecoder) |
| | | 1301 | | { |
| | | 1302 | | Debug.Assert(chars is not null, "[UnicodeEncoding.GetChars]chars!=null"); |
| | 652519 | 1303 | | Debug.Assert(byteCount >= 0, "[UnicodeEncoding.GetChars]byteCount >=0"); |
| | 652519 | 1304 | | Debug.Assert(charCount >= 0, "[UnicodeEncoding.GetChars]charCount >=0"); |
| | 652519 | 1305 | | Debug.Assert(bytes is not null, "[UnicodeEncoding.GetChars]bytes!=null"); |
| | | 1306 | | |
| | 652519 | 1307 | | Decoder? decoder = (Decoder?)baseDecoder; |
| | | 1308 | | |
| | | 1309 | | // Need last vars |
| | 652519 | 1310 | | int lastByte = -1; |
| | 652519 | 1311 | | char lastChar = (char)0; |
| | | 1312 | | |
| | | 1313 | | // Get our decoder (but don't clear it yet) |
| | 652519 | 1314 | | if (decoder is not null) |
| | | 1315 | | { |
| | 652519 | 1316 | | lastByte = decoder.lastByte; |
| | 652519 | 1317 | | lastChar = decoder.lastChar; |
| | | 1318 | | |
| | | 1319 | | // Shouldn't have anything in fallback buffer for GetChars |
| | | 1320 | | // (don't have to check _throwOnOverflow for chars) |
| | 652519 | 1321 | | Debug.Assert(!decoder.InternalHasFallbackBuffer || decoder.FallbackBuffer.Remaining == 0, |
| | 652519 | 1322 | | "[UnicodeEncoding.GetChars]Expected empty fallback buffer at start"); |
| | | 1323 | | } |
| | | 1324 | | |
| | | 1325 | | // For fallback we may need a fallback buffer |
| | 652519 | 1326 | | DecoderFallbackBuffer? fallbackBuffer = null; |
| | | 1327 | | char* charsForFallback; |
| | | 1328 | | |
| | 652519 | 1329 | | byte* byteEnd = bytes + byteCount; |
| | 652519 | 1330 | | char* charEnd = chars + charCount; |
| | 652519 | 1331 | | byte* byteStart = bytes; |
| | 652519 | 1332 | | char* charStart = chars; |
| | | 1333 | | |
| | 2968495 | 1334 | | while (bytes < byteEnd) |
| | | 1335 | | { |
| | | 1336 | | // If we're aligned then maybe we can do it fast |
| | | 1337 | | // That'll hurt if we're unaligned because we'll always test but never be aligned |
| | | 1338 | | #if FASTLOOP |
| | 2330514 | 1339 | | if ((bigEndian ^ BitConverter.IsLittleEndian) && |
| | 2330514 | 1340 | | #if TARGET_64BIT |
| | 2330514 | 1341 | | (unchecked((long)chars) & 7) == 0 && |
| | 2330514 | 1342 | | #else |
| | 2330514 | 1343 | | (unchecked((int)chars) & 3) == 0 && |
| | 2330514 | 1344 | | #endif |
| | 2330514 | 1345 | | lastByte == -1 && lastChar == 0) |
| | | 1346 | | { |
| | | 1347 | | // Need -1 to check 2 at a time. If we have an even #, longChars will go |
| | | 1348 | | // from longEnd - 1/2 long to longEnd + 1/2 long. If we're odd, longChars |
| | | 1349 | | // will go from longEnd - 1 long to longEnd. (Might not get to use this) |
| | | 1350 | | // We can only go iCount units (limited by shorter of char or byte buffers. |
| | 244496 | 1351 | | ulong* longEnd = (ulong*)(bytes - 7 + |
| | 244496 | 1352 | | (((byteEnd - bytes) >> 1 < charEnd - chars) ? |
| | 244496 | 1353 | | (byteEnd - bytes) : (charEnd - chars) << 1)); |
| | | 1354 | | |
| | | 1355 | | // Need new char* so we can check 4 at a time |
| | 244496 | 1356 | | ulong* longBytes = (ulong*)bytes; |
| | 244496 | 1357 | | ulong* longChars = (ulong*)chars; |
| | | 1358 | | |
| | 549308 | 1359 | | while (longBytes < longEnd) |
| | | 1360 | | { |
| | | 1361 | | // See if we potentially have surrogates (0x8000 bit set) |
| | | 1362 | | // (We're either big endian on a big endian machine or little endian on |
| | | 1363 | | // a little endian machine so that'll work) |
| | 389056 | 1364 | | if ((0x8000800080008000 & *longBytes) != 0) |
| | | 1365 | | { |
| | | 1366 | | // See if any of these are high or low surrogates (0xd800 - 0xdfff). If the high |
| | | 1367 | | // 5 bits looks like 11011, then its a high or low surrogate. |
| | | 1368 | | // We do the & f800 to filter the 5 bits, then ^ d800 to ensure the 0 isn't set. |
| | | 1369 | | // Note that we expect BMP characters to be more common than surrogates |
| | | 1370 | | // & each char with 11111... then ^ with 11011. Zeroes then indicate surrogates |
| | 285022 | 1371 | | ulong uTemp = (0xf800f800f800f800 & *longBytes) ^ 0xd800d800d800d800; |
| | | 1372 | | |
| | | 1373 | | // Check each of the 4 chars. 0 for those 16 bits means it was a surrogate |
| | | 1374 | | // but no clue if they're high or low. |
| | | 1375 | | // If each of the 4 characters are non-zero, then none are surrogates. |
| | 285022 | 1376 | | if ((uTemp & 0xFFFF000000000000) == 0 || |
| | 285022 | 1377 | | (uTemp & 0x0000FFFF00000000) == 0 || |
| | 285022 | 1378 | | (uTemp & 0x00000000FFFF0000) == 0 || |
| | 285022 | 1379 | | (uTemp & 0x000000000000FFFF) == 0) |
| | | 1380 | | { |
| | | 1381 | | // It has at least 1 surrogate, but we don't know if they're high or low surrogates, |
| | | 1382 | | // or if there's 1 or 4 surrogates |
| | | 1383 | | |
| | | 1384 | | // If they happen to be high/low/high/low, we may as well continue. Check the next |
| | | 1385 | | // bit to see if its set (low) or not (high) in the right pattern |
| | 84829 | 1386 | | if ((0xfc00fc00fc00fc00 & *longBytes) != |
| | 84829 | 1387 | | (BitConverter.IsLittleEndian ? (ulong)0xdc00d800dc00d800 : (ulong)0xd800dc00d800 |
| | | 1388 | | { |
| | | 1389 | | // Either there weren't 4 surrogates, or the 0x0400 bit was set when a high |
| | | 1390 | | // was hoped for or the 0x0400 bit wasn't set where a low was hoped for. |
| | | 1391 | | |
| | | 1392 | | // Drop out to the slow loop to resolve the surrogates |
| | | 1393 | | break; |
| | | 1394 | | } |
| | | 1395 | | // else they are all surrogates in High/Low/High/Low order, so we can use them. |
| | | 1396 | | } |
| | | 1397 | | // else none are surrogates, so we can use them. |
| | | 1398 | | } |
| | | 1399 | | // else all < 0x8000 so we can use them |
| | | 1400 | | |
| | | 1401 | | // We can use these 4 chars. |
| | 304812 | 1402 | | Unsafe.WriteUnaligned(longChars, *longBytes); |
| | 304812 | 1403 | | longBytes++; |
| | 304812 | 1404 | | longChars++; |
| | | 1405 | | } |
| | | 1406 | | |
| | 244496 | 1407 | | chars = (char*)longChars; |
| | 244496 | 1408 | | bytes = (byte*)longBytes; |
| | | 1409 | | |
| | 244496 | 1410 | | if (bytes >= byteEnd) |
| | | 1411 | | break; |
| | | 1412 | | } |
| | | 1413 | | #endif // FASTLOOP |
| | | 1414 | | |
| | | 1415 | | // Get 1st byte |
| | 2315976 | 1416 | | if (lastByte < 0) |
| | | 1417 | | { |
| | 1169214 | 1418 | | lastByte = *bytes++; |
| | 1169214 | 1419 | | continue; |
| | | 1420 | | } |
| | | 1421 | | |
| | | 1422 | | // Get full char |
| | | 1423 | | char ch; |
| | 1146762 | 1424 | | if (bigEndian) |
| | | 1425 | | { |
| | 0 | 1426 | | ch = (char)(lastByte << 8 | *(bytes++)); |
| | | 1427 | | } |
| | | 1428 | | else |
| | | 1429 | | { |
| | 1146762 | 1430 | | ch = (char)(*(bytes++) << 8 | lastByte); |
| | | 1431 | | } |
| | 1146762 | 1432 | | lastByte = -1; |
| | | 1433 | | |
| | | 1434 | | // See if the char's valid |
| | 1146762 | 1435 | | if (ch >= 0xd800 && ch <= 0xdfff) |
| | | 1436 | | { |
| | | 1437 | | // Was it a high surrogate? |
| | 309166 | 1438 | | if (ch <= 0xdbff) |
| | | 1439 | | { |
| | | 1440 | | // Its a high surrogate, if we had one then do fallback for previous one |
| | 176581 | 1441 | | if (lastChar > 0) |
| | | 1442 | | { |
| | | 1443 | | // Get fallback for previous high surrogate |
| | | 1444 | | // Note we have to reconstruct bytes because some may have been in decoder |
| | 65024 | 1445 | | byte[]? byteBuffer = null; |
| | 65024 | 1446 | | if (bigEndian) |
| | | 1447 | | { |
| | 0 | 1448 | | byteBuffer = [unchecked((byte)(lastChar >> 8)), unchecked((byte)lastChar)]; |
| | | 1449 | | } |
| | | 1450 | | else |
| | | 1451 | | { |
| | 65024 | 1452 | | byteBuffer = [unchecked((byte)lastChar), unchecked((byte)(lastChar >> 8))]; |
| | | 1453 | | } |
| | | 1454 | | |
| | 65024 | 1455 | | if (fallbackBuffer is null) |
| | | 1456 | | { |
| | 18107 | 1457 | | fallbackBuffer = decoder is null ? |
| | 18107 | 1458 | | this.decoderFallback.CreateFallbackBuffer() : |
| | 18107 | 1459 | | decoder.FallbackBuffer; |
| | | 1460 | | |
| | | 1461 | | // Set our internal fallback interesting things. |
| | 18107 | 1462 | | fallbackBuffer.InternalInitialize(byteStart, charEnd); |
| | | 1463 | | } |
| | | 1464 | | |
| | 65024 | 1465 | | charsForFallback = chars; // Avoid passing chars by reference to allow it to be en-registere |
| | 65024 | 1466 | | bool fallbackResult = fallbackBuffer.InternalFallback(byteBuffer, bytes, ref charsForFallbac |
| | 65024 | 1467 | | chars = charsForFallback; |
| | | 1468 | | |
| | 65024 | 1469 | | if (!fallbackResult) |
| | | 1470 | | { |
| | | 1471 | | // couldn't fall back lonely surrogate |
| | | 1472 | | // We either advanced bytes or chars should == charStart and throw below |
| | 0 | 1473 | | Debug.Assert(bytes >= byteStart + 2 || chars == charStart, |
| | 0 | 1474 | | "[UnicodeEncoding.GetChars]Expected bytes to have advanced or no output (bad surroga |
| | 0 | 1475 | | bytes -= 2; // didn't use these 2 bytes |
| | 0 | 1476 | | fallbackBuffer.InternalReset(); |
| | 0 | 1477 | | ThrowCharsOverflow(decoder, chars == charStart); // Might throw, if no chars output |
| | 0 | 1478 | | break; // couldn't fallback but didn't throw |
| | | 1479 | | } |
| | | 1480 | | } |
| | | 1481 | | |
| | | 1482 | | // Ignore the previous high surrogate which fell back already, |
| | | 1483 | | // yet remember the current high surrogate for next time. |
| | 176581 | 1484 | | lastChar = ch; |
| | 176581 | 1485 | | continue; |
| | | 1486 | | } |
| | | 1487 | | |
| | | 1488 | | // Its a low surrogate |
| | 132585 | 1489 | | if (lastChar == 0) |
| | | 1490 | | { |
| | | 1491 | | // Expected a previous high surrogate |
| | | 1492 | | // Get fallback for this low surrogate |
| | | 1493 | | // Note we have to reconstruct bytes because some may have been in decoder |
| | 105933 | 1494 | | byte[]? byteBuffer = null; |
| | 105933 | 1495 | | if (bigEndian) |
| | | 1496 | | { |
| | 0 | 1497 | | byteBuffer = [unchecked((byte)(ch >> 8)), unchecked((byte)ch)]; |
| | | 1498 | | } |
| | | 1499 | | else |
| | | 1500 | | { |
| | 105933 | 1501 | | byteBuffer = [unchecked((byte)ch), unchecked((byte)(ch >> 8))]; |
| | | 1502 | | } |
| | | 1503 | | |
| | 105933 | 1504 | | if (fallbackBuffer is null) |
| | | 1505 | | { |
| | 30942 | 1506 | | fallbackBuffer = decoder is null ? |
| | 30942 | 1507 | | this.decoderFallback.CreateFallbackBuffer() : |
| | 30942 | 1508 | | decoder.FallbackBuffer; |
| | | 1509 | | |
| | | 1510 | | // Set our internal fallback interesting things. |
| | 30942 | 1511 | | fallbackBuffer.InternalInitialize(byteStart, charEnd); |
| | | 1512 | | } |
| | | 1513 | | |
| | 105933 | 1514 | | charsForFallback = chars; // Avoid passing chars by reference to allow it to be en-registered |
| | 105933 | 1515 | | bool fallbackResult = fallbackBuffer.InternalFallback(byteBuffer, bytes, ref charsForFallback); |
| | 105933 | 1516 | | chars = charsForFallback; |
| | | 1517 | | |
| | 105933 | 1518 | | if (!fallbackResult) |
| | | 1519 | | { |
| | | 1520 | | // couldn't fall back lonely surrogate |
| | | 1521 | | // We either advanced bytes or chars should == charStart and throw below |
| | 0 | 1522 | | Debug.Assert(bytes >= byteStart + 2 || chars == charStart, |
| | 0 | 1523 | | "[UnicodeEncoding.GetChars]Expected bytes to have advanced or no output (lonely surrogat |
| | 0 | 1524 | | bytes -= 2; // didn't use these 2 bytes |
| | 0 | 1525 | | fallbackBuffer.InternalReset(); |
| | 0 | 1526 | | ThrowCharsOverflow(decoder, chars == charStart); // Might throw, if no chars output |
| | 0 | 1527 | | break; // couldn't fallback but didn't throw |
| | | 1528 | | } |
| | | 1529 | | |
| | | 1530 | | // Didn't throw, ignore this one (we already did its fallback) |
| | | 1531 | | continue; |
| | | 1532 | | } |
| | | 1533 | | |
| | | 1534 | | // Valid surrogate pair, add our lastChar (will need 2 chars) |
| | 26652 | 1535 | | if (charEnd - chars < 2) |
| | | 1536 | | { |
| | | 1537 | | // couldn't find room for this surrogate pair |
| | | 1538 | | // We either advanced bytes or chars should == charStart and throw below |
| | 0 | 1539 | | Debug.Assert(bytes >= byteStart + 2 || chars == charStart, |
| | 0 | 1540 | | "[UnicodeEncoding.GetChars]Expected bytes to have advanced or no output (surrogate pair)"); |
| | 0 | 1541 | | bytes -= 2; // didn't use these 2 bytes |
| | 0 | 1542 | | ThrowCharsOverflow(decoder, chars == charStart); // Might throw, if no chars output |
| | | 1543 | | // Leave lastChar for next call to Convert() |
| | 0 | 1544 | | break; // couldn't fallback but didn't throw |
| | | 1545 | | } |
| | | 1546 | | |
| | 26652 | 1547 | | *chars++ = lastChar; |
| | 26652 | 1548 | | lastChar = (char)0; |
| | | 1549 | | } |
| | 837596 | 1550 | | else if (lastChar > 0) |
| | | 1551 | | { |
| | | 1552 | | // Had a high surrogate, expected a low surrogate, fall back the high surrogate. |
| | 82307 | 1553 | | byte[]? byteBuffer = null; |
| | 82307 | 1554 | | if (bigEndian) |
| | | 1555 | | { |
| | 0 | 1556 | | byteBuffer = [unchecked((byte)(lastChar >> 8)), unchecked((byte)lastChar)]; |
| | | 1557 | | } |
| | | 1558 | | else |
| | | 1559 | | { |
| | 82307 | 1560 | | byteBuffer = [unchecked((byte)lastChar), unchecked((byte)(lastChar >> 8))]; |
| | | 1561 | | } |
| | | 1562 | | |
| | 82307 | 1563 | | if (fallbackBuffer is null) |
| | | 1564 | | { |
| | 26949 | 1565 | | fallbackBuffer = decoder is null ? |
| | 26949 | 1566 | | this.decoderFallback.CreateFallbackBuffer() : |
| | 26949 | 1567 | | decoder.FallbackBuffer; |
| | | 1568 | | |
| | | 1569 | | // Set our internal fallback interesting things. |
| | 26949 | 1570 | | fallbackBuffer.InternalInitialize(byteStart, charEnd); |
| | | 1571 | | } |
| | | 1572 | | |
| | 82307 | 1573 | | charsForFallback = chars; // Avoid passing chars by reference to allow it to be en-registered |
| | 82307 | 1574 | | bool fallbackResult = fallbackBuffer.InternalFallback(byteBuffer, bytes, ref charsForFallback); |
| | 82307 | 1575 | | chars = charsForFallback; |
| | | 1576 | | |
| | 82307 | 1577 | | if (!fallbackResult) |
| | | 1578 | | { |
| | | 1579 | | // couldn't fall back high surrogate, or char that would be next |
| | | 1580 | | // We either advanced bytes or chars should == charStart and throw below |
| | 0 | 1581 | | Debug.Assert(bytes >= byteStart + 2 || chars == charStart, |
| | 0 | 1582 | | "[UnicodeEncoding.GetChars]Expected bytes to have advanced or no output (no low surrogate)") |
| | 0 | 1583 | | bytes -= 2; // didn't use these 2 bytes |
| | 0 | 1584 | | fallbackBuffer.InternalReset(); |
| | 0 | 1585 | | ThrowCharsOverflow(decoder, chars == charStart); // Might throw, if no chars output |
| | 0 | 1586 | | break; // couldn't fallback but didn't throw |
| | | 1587 | | } |
| | | 1588 | | |
| | | 1589 | | // Not left over now, clear previous high surrogate and continue to add current char |
| | 82307 | 1590 | | lastChar = (char)0; |
| | | 1591 | | } |
| | | 1592 | | |
| | | 1593 | | // Valid char, room for it? |
| | 864248 | 1594 | | if (chars >= charEnd) |
| | | 1595 | | { |
| | | 1596 | | // 2 bytes couldn't fall back |
| | | 1597 | | // We either advanced bytes or chars should == charStart and throw below |
| | 0 | 1598 | | Debug.Assert(bytes >= byteStart + 2 || chars == charStart, |
| | 0 | 1599 | | "[UnicodeEncoding.GetChars]Expected bytes to have advanced or no output (normal)"); |
| | 0 | 1600 | | bytes -= 2; // didn't use these bytes |
| | 0 | 1601 | | ThrowCharsOverflow(decoder, chars == charStart); // Might throw, if no chars output |
| | 0 | 1602 | | break; // couldn't fallback but didn't throw |
| | | 1603 | | } |
| | | 1604 | | |
| | | 1605 | | // add it |
| | 864248 | 1606 | | *chars++ = ch; |
| | | 1607 | | } |
| | | 1608 | | |
| | | 1609 | | // Remember our decoder if we must |
| | 652519 | 1610 | | if (decoder is null || decoder.MustFlush) |
| | | 1611 | | { |
| | 53908 | 1612 | | if (lastChar > 0) |
| | | 1613 | | { |
| | | 1614 | | // No hanging high surrogates allowed, do fallback and remove count for it |
| | 2598 | 1615 | | byte[]? byteBuffer = null; |
| | 2598 | 1616 | | if (bigEndian) |
| | | 1617 | | { |
| | 0 | 1618 | | byteBuffer = [unchecked((byte)(lastChar >> 8)), unchecked((byte)lastChar)]; |
| | | 1619 | | } |
| | | 1620 | | else |
| | | 1621 | | { |
| | 2598 | 1622 | | byteBuffer = [unchecked((byte)lastChar), unchecked((byte)(lastChar >> 8))]; |
| | | 1623 | | } |
| | | 1624 | | |
| | 2598 | 1625 | | if (fallbackBuffer is null) |
| | | 1626 | | { |
| | 1016 | 1627 | | fallbackBuffer = decoder is null ? |
| | 1016 | 1628 | | this.decoderFallback.CreateFallbackBuffer() : |
| | 1016 | 1629 | | decoder.FallbackBuffer; |
| | | 1630 | | |
| | | 1631 | | // Set our internal fallback interesting things. |
| | 1016 | 1632 | | fallbackBuffer.InternalInitialize(byteStart, charEnd); |
| | | 1633 | | } |
| | | 1634 | | |
| | 2598 | 1635 | | charsForFallback = chars; // Avoid passing chars by reference to allow it to be en-registered |
| | 2598 | 1636 | | bool fallbackResult = fallbackBuffer.InternalFallback(byteBuffer, bytes, ref charsForFallback); |
| | 2598 | 1637 | | chars = charsForFallback; |
| | | 1638 | | |
| | 2598 | 1639 | | if (!fallbackResult) |
| | | 1640 | | { |
| | | 1641 | | // 2 bytes couldn't fall back |
| | | 1642 | | // We either advanced bytes or chars should == charStart and throw below |
| | 0 | 1643 | | Debug.Assert(bytes >= byteStart + 2 || chars == charStart, |
| | 0 | 1644 | | "[UnicodeEncoding.GetChars]Expected bytes to have advanced or no output (decoder)"); |
| | 0 | 1645 | | bytes -= 2; // didn't use these bytes |
| | 0 | 1646 | | if (lastByte >= 0) |
| | 0 | 1647 | | bytes--; // had an extra last byte hanging around |
| | 0 | 1648 | | fallbackBuffer.InternalReset(); |
| | 0 | 1649 | | ThrowCharsOverflow(decoder, chars == charStart); // Might throw, if no chars output |
| | | 1650 | | // We'll remember these in our decoder though |
| | 0 | 1651 | | bytes += 2; |
| | 0 | 1652 | | if (lastByte >= 0) |
| | 0 | 1653 | | bytes++; |
| | 0 | 1654 | | goto End; |
| | | 1655 | | } |
| | | 1656 | | |
| | | 1657 | | // done with this one |
| | 2598 | 1658 | | lastChar = (char)0; |
| | | 1659 | | } |
| | | 1660 | | |
| | 53908 | 1661 | | if (lastByte >= 0) |
| | | 1662 | | { |
| | 22452 | 1663 | | if (fallbackBuffer is null) |
| | | 1664 | | { |
| | 15809 | 1665 | | fallbackBuffer = decoder is null ? |
| | 15809 | 1666 | | this.decoderFallback.CreateFallbackBuffer() : |
| | 15809 | 1667 | | decoder.FallbackBuffer; |
| | | 1668 | | |
| | | 1669 | | // Set our internal fallback interesting things. |
| | 15809 | 1670 | | fallbackBuffer.InternalInitialize(byteStart, charEnd); |
| | | 1671 | | } |
| | | 1672 | | |
| | | 1673 | | // No hanging odd bytes allowed if must flush |
| | 22452 | 1674 | | charsForFallback = chars; // Avoid passing chars by reference to allow it to be en-registered |
| | 22452 | 1675 | | bool fallbackResult = fallbackBuffer.InternalFallback([unchecked((byte)lastByte)], bytes, ref charsF |
| | 22452 | 1676 | | chars = charsForFallback; |
| | | 1677 | | |
| | 22452 | 1678 | | if (!fallbackResult) |
| | | 1679 | | { |
| | | 1680 | | // odd byte couldn't fall back |
| | 0 | 1681 | | bytes--; // didn't use this byte |
| | 0 | 1682 | | fallbackBuffer.InternalReset(); |
| | 0 | 1683 | | ThrowCharsOverflow(decoder, chars == charStart); // Might throw, if no chars output |
| | | 1684 | | // didn't throw, but we'll remember it in the decoder |
| | 0 | 1685 | | bytes++; |
| | 0 | 1686 | | goto End; |
| | | 1687 | | } |
| | | 1688 | | |
| | | 1689 | | // Didn't fail, clear buffer |
| | 22452 | 1690 | | lastByte = -1; |
| | | 1691 | | } |
| | | 1692 | | } |
| | | 1693 | | |
| | | 1694 | | End: |
| | | 1695 | | |
| | | 1696 | | // Remember our decoder if we must |
| | 652519 | 1697 | | if (decoder is not null) |
| | | 1698 | | { |
| | 652519 | 1699 | | Debug.Assert(!decoder.MustFlush || ((lastChar == (char)0) && (lastByte == -1)), |
| | 652519 | 1700 | | "[UnicodeEncoding.GetChars] Expected no left over chars or bytes if flushing"); |
| | | 1701 | | |
| | 652519 | 1702 | | decoder._bytesUsed = (int)(bytes - byteStart); |
| | 652519 | 1703 | | decoder.lastChar = lastChar; |
| | 652519 | 1704 | | decoder.lastByte = lastByte; |
| | | 1705 | | } |
| | | 1706 | | |
| | | 1707 | | // Shouldn't have anything in fallback buffer for GetChars |
| | | 1708 | | // (don't have to check _throwOnOverflow for count or chars) |
| | 652519 | 1709 | | Debug.Assert(fallbackBuffer is null || fallbackBuffer.Remaining == 0, |
| | 652519 | 1710 | | "[UnicodeEncoding.GetChars]Expected empty fallback buffer at end"); |
| | | 1711 | | |
| | 652519 | 1712 | | return (int)(chars - charStart); |
| | | 1713 | | } |
| | | 1714 | | |
| | | 1715 | | public override Encoder GetEncoder() |
| | | 1716 | | { |
| | 4517 | 1717 | | return new EncoderNLS(this); |
| | | 1718 | | } |
| | | 1719 | | |
| | | 1720 | | public override Text.Decoder GetDecoder() |
| | | 1721 | | { |
| | 30357 | 1722 | | return new Decoder(this); |
| | | 1723 | | } |
| | | 1724 | | |
| | | 1725 | | public override byte[] GetPreamble() |
| | | 1726 | | { |
| | 0 | 1727 | | if (byteOrderMark) |
| | | 1728 | | { |
| | | 1729 | | // Note - we must allocate new byte[]'s here to prevent someone |
| | | 1730 | | // from modifying a cached byte[]. |
| | 0 | 1731 | | if (bigEndian) |
| | 0 | 1732 | | return [0xfe, 0xff]; |
| | | 1733 | | else |
| | 0 | 1734 | | return [0xff, 0xfe]; |
| | | 1735 | | } |
| | 0 | 1736 | | return []; |
| | | 1737 | | } |
| | | 1738 | | |
| | | 1739 | | public override ReadOnlySpan<byte> Preamble => |
| | 0 | 1740 | | GetType() != typeof(UnicodeEncoding) ? new ReadOnlySpan<byte>(GetPreamble()) : // in case a derived UnicodeE |
| | 0 | 1741 | | !byteOrderMark ? default : |
| | 0 | 1742 | | bigEndian ? [0xfe, 0xff] : |
| | 0 | 1743 | | [0xff, 0xfe]; |
| | | 1744 | | |
| | | 1745 | | public override int GetMaxByteCount(int charCount) |
| | | 1746 | | { |
| | 0 | 1747 | | ArgumentOutOfRangeException.ThrowIfNegative(charCount); |
| | | 1748 | | |
| | | 1749 | | // Characters would be # of characters + 1 in case left over high surrogate is ? * max fallback |
| | 0 | 1750 | | long byteCount = (long)charCount + 1; |
| | | 1751 | | |
| | 0 | 1752 | | if (EncoderFallback.MaxCharCount > 1) |
| | 0 | 1753 | | byteCount *= EncoderFallback.MaxCharCount; |
| | | 1754 | | |
| | | 1755 | | // 2 bytes per char |
| | 0 | 1756 | | byteCount <<= 1; |
| | | 1757 | | |
| | 0 | 1758 | | if (byteCount > 0x7fffffff) |
| | 0 | 1759 | | throw new ArgumentOutOfRangeException(nameof(charCount), SR.ArgumentOutOfRange_GetByteCountOverflow); |
| | | 1760 | | |
| | 0 | 1761 | | return (int)byteCount; |
| | | 1762 | | } |
| | | 1763 | | |
| | | 1764 | | public override int GetMaxCharCount(int byteCount) |
| | | 1765 | | { |
| | 0 | 1766 | | ArgumentOutOfRangeException.ThrowIfNegative(byteCount); |
| | | 1767 | | |
| | | 1768 | | // long because byteCount could be biggest int. |
| | | 1769 | | // 1 char per 2 bytes. Round up in case 1 left over in decoder. |
| | | 1770 | | // Round up using &1 in case byteCount is max size |
| | | 1771 | | // Might also need an extra 1 if there's a left over high surrogate in the decoder. |
| | 0 | 1772 | | long charCount = (long)(byteCount >> 1) + (byteCount & 1) + 1; |
| | | 1773 | | |
| | | 1774 | | // Don't forget fallback (in case they have a bunch of lonely surrogates or something bizarre like that) |
| | 0 | 1775 | | if (DecoderFallback.MaxCharCount > 1) |
| | 0 | 1776 | | charCount *= DecoderFallback.MaxCharCount; |
| | | 1777 | | |
| | 0 | 1778 | | if (charCount > 0x7fffffff) |
| | 0 | 1779 | | throw new ArgumentOutOfRangeException(nameof(byteCount), SR.ArgumentOutOfRange_GetCharCountOverflow); |
| | | 1780 | | |
| | 0 | 1781 | | return (int)charCount; |
| | | 1782 | | } |
| | | 1783 | | |
| | | 1784 | | public override bool Equals([NotNullWhen(true)] object? value) |
| | | 1785 | | { |
| | 0 | 1786 | | if (value is UnicodeEncoding that) |
| | | 1787 | | { |
| | | 1788 | | // |
| | | 1789 | | // Big Endian Unicode has different code page (1201) than small Endian one (1200), |
| | | 1790 | | // so we still have to check _codePage here. |
| | | 1791 | | // |
| | 0 | 1792 | | return (CodePage == that.CodePage) && |
| | 0 | 1793 | | byteOrderMark == that.byteOrderMark && |
| | 0 | 1794 | | // isThrowException == that.isThrowException && // Same as Encoder/Decoder being exception fall |
| | 0 | 1795 | | bigEndian == that.bigEndian && |
| | 0 | 1796 | | (EncoderFallback.Equals(that.EncoderFallback)) && |
| | 0 | 1797 | | (DecoderFallback.Equals(that.DecoderFallback)); |
| | | 1798 | | } |
| | 0 | 1799 | | return false; |
| | | 1800 | | } |
| | | 1801 | | |
| | | 1802 | | public override int GetHashCode() |
| | | 1803 | | { |
| | 0 | 1804 | | return CodePage + this.EncoderFallback.GetHashCode() + this.DecoderFallback.GetHashCode() + |
| | 0 | 1805 | | (byteOrderMark ? 4 : 0) + (bigEndian ? 8 : 0); |
| | | 1806 | | } |
| | | 1807 | | |
| | | 1808 | | private sealed class Decoder : DecoderNLS |
| | | 1809 | | { |
| | 30357 | 1810 | | internal int lastByte = -1; |
| | | 1811 | | internal char lastChar; |
| | | 1812 | | |
| | 30357 | 1813 | | public Decoder(UnicodeEncoding encoding) : base(encoding) |
| | | 1814 | | { |
| | | 1815 | | // base calls reset |
| | 30357 | 1816 | | } |
| | | 1817 | | |
| | | 1818 | | public override void Reset() |
| | | 1819 | | { |
| | 84265 | 1820 | | lastByte = -1; |
| | 84265 | 1821 | | lastChar = '\0'; |
| | 84265 | 1822 | | _fallbackBuffer?.Reset(); |
| | 35088 | 1823 | | } |
| | | 1824 | | |
| | | 1825 | | // Anything left in our decoder? |
| | 22437 | 1826 | | internal override bool HasState => this.lastByte != -1 || this.lastChar != '\0'; |
| | | 1827 | | } |
| | | 1828 | | } |
| | | 1829 | | } |
| | | 1830 | | |