| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | // |
| | | 5 | | // Don't override IsAlwaysNormalized because it is just a Unicode Transformation and could be confused. |
| | | 6 | | // |
| | | 7 | | |
| | | 8 | | using System.Diagnostics; |
| | | 9 | | using System.Diagnostics.CodeAnalysis; |
| | | 10 | | using System.Runtime.InteropServices; |
| | | 11 | | |
| | | 12 | | namespace System.Text |
| | | 13 | | { |
| | | 14 | | // Encodes text into and out of UTF-32. UTF-32 is a way of writing |
| | | 15 | | // Unicode characters with a single storage unit (32 bits) per character, |
| | | 16 | | // |
| | | 17 | | // The UTF-32 byte order mark is simply the Unicode byte order mark |
| | | 18 | | // (0x00FEFF) written in UTF-32 (0x0000FEFF or 0xFFFE0000). The byte order |
| | | 19 | | // mark is used mostly to distinguish UTF-32 text from other encodings, and doesn't |
| | | 20 | | // switch the byte orderings. |
| | | 21 | | |
| | | 22 | | public sealed class UTF32Encoding : Encoding |
| | | 23 | | { |
| | | 24 | | /* |
| | | 25 | | words bits UTF-32 representation |
| | | 26 | | ----- ---- ----------------------------------- |
| | | 27 | | 1 16 00000000 00000000 xxxxxxxx xxxxxxxx |
| | | 28 | | 2 21 00000000 000xxxxx hhhhhhll llllllll |
| | | 29 | | ----- ---- ----------------------------------- |
| | | 30 | | |
| | | 31 | | Surrogate: |
| | | 32 | | Real Unicode value = (HighSurrogate - 0xD800) * 0x400 + (LowSurrogate - 0xDC00) + 0x10000 |
| | | 33 | | */ |
| | | 34 | | |
| | | 35 | | // Used by Encoding.UTF32/BigEndianUTF32 for lazy initialization |
| | | 36 | | // The initialization code will not be run until a static member of the class is referenced |
| | 0 | 37 | | internal static readonly UTF32Encoding s_default = new UTF32Encoding(bigEndian: false, byteOrderMark: true); |
| | 0 | 38 | | internal static readonly UTF32Encoding s_bigEndianDefault = new UTF32Encoding(bigEndian: true, byteOrderMark: tr |
| | | 39 | | |
| | | 40 | | private readonly bool _emitUTF32ByteOrderMark; |
| | | 41 | | private readonly bool _isThrowException; |
| | | 42 | | private readonly bool _bigEndian; |
| | | 43 | | |
| | 0 | 44 | | public UTF32Encoding() : this(false, true) |
| | | 45 | | { |
| | 0 | 46 | | } |
| | | 47 | | |
| | | 48 | | public UTF32Encoding(bool bigEndian, bool byteOrderMark) : |
| | 0 | 49 | | base(bigEndian ? 12001 : 12000) |
| | | 50 | | { |
| | 0 | 51 | | _bigEndian = bigEndian; |
| | 0 | 52 | | _emitUTF32ByteOrderMark = byteOrderMark; |
| | 0 | 53 | | } |
| | | 54 | | |
| | | 55 | | public UTF32Encoding(bool bigEndian, bool byteOrderMark, bool throwOnInvalidCharacters) : |
| | 0 | 56 | | this(bigEndian, byteOrderMark) |
| | | 57 | | { |
| | 0 | 58 | | _isThrowException = throwOnInvalidCharacters; |
| | | 59 | | |
| | | 60 | | // Encoding constructor already did this, but it'll be wrong if we're throwing exceptions |
| | 0 | 61 | | if (_isThrowException) |
| | 0 | 62 | | SetDefaultFallbacks(); |
| | 0 | 63 | | } |
| | | 64 | | |
| | | 65 | | internal sealed override void SetDefaultFallbacks() |
| | | 66 | | { |
| | | 67 | | // For UTF-X encodings, we use a replacement fallback with an empty string |
| | 0 | 68 | | if (_isThrowException) |
| | | 69 | | { |
| | 0 | 70 | | this.encoderFallback = EncoderFallback.ExceptionFallback; |
| | 0 | 71 | | this.decoderFallback = DecoderFallback.ExceptionFallback; |
| | | 72 | | } |
| | | 73 | | else |
| | | 74 | | { |
| | 0 | 75 | | this.encoderFallback = new EncoderReplacementFallback("\xFFFD"); |
| | 0 | 76 | | this.decoderFallback = new DecoderReplacementFallback("\xFFFD"); |
| | | 77 | | } |
| | 0 | 78 | | } |
| | | 79 | | |
| | | 80 | | // The following methods are copied from EncodingNLS.cs. |
| | | 81 | | // Unfortunately EncodingNLS.cs is internal and we're public, so we have to re-implement them here. |
| | | 82 | | // These should be kept in sync for the following classes: |
| | | 83 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 84 | | |
| | | 85 | | // Returns the number of bytes required to encode a range of characters in |
| | | 86 | | // a character array. |
| | | 87 | | // |
| | | 88 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 89 | | // So if you fix this, fix the others. Currently those include: |
| | | 90 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 91 | | // parent method is safe |
| | | 92 | | |
| | | 93 | | public override unsafe int GetByteCount(char[] chars, int index, int count) |
| | | 94 | | { |
| | 0 | 95 | | ArgumentNullException.ThrowIfNull(chars); |
| | | 96 | | |
| | 0 | 97 | | ArgumentOutOfRangeException.ThrowIfNegative(index); |
| | 0 | 98 | | ArgumentOutOfRangeException.ThrowIfNegative(count); |
| | | 99 | | |
| | 0 | 100 | | if (chars.Length - index < count) |
| | 0 | 101 | | throw new ArgumentOutOfRangeException(nameof(chars), SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 102 | | |
| | | 103 | | // If no input, return 0, avoid fixed empty array problem |
| | 0 | 104 | | if (count == 0) |
| | 0 | 105 | | return 0; |
| | | 106 | | |
| | | 107 | | // Just call the pointer version |
| | 0 | 108 | | fixed (char* pChars = chars) |
| | 0 | 109 | | return GetByteCount(pChars + index, count, null); |
| | | 110 | | } |
| | | 111 | | |
| | | 112 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 113 | | // So if you fix this, fix the others. Currently those include: |
| | | 114 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 115 | | // parent method is safe |
| | | 116 | | |
| | | 117 | | public override unsafe int GetByteCount(string s) |
| | | 118 | | { |
| | 0 | 119 | | if (s is null) |
| | | 120 | | { |
| | 0 | 121 | | ThrowHelper.ThrowArgumentNullException(ExceptionArgument.s); |
| | | 122 | | } |
| | | 123 | | |
| | 0 | 124 | | fixed (char* pChars = s) |
| | 0 | 125 | | return GetByteCount(pChars, s.Length, null); |
| | | 126 | | } |
| | | 127 | | |
| | | 128 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 129 | | // So if you fix this, fix the others. Currently those include: |
| | | 130 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 131 | | |
| | | 132 | | [CLSCompliant(false)] |
| | | 133 | | public override unsafe int GetByteCount(char* chars, int count) |
| | | 134 | | { |
| | 0 | 135 | | ArgumentNullException.ThrowIfNull(chars); |
| | | 136 | | |
| | 0 | 137 | | ArgumentOutOfRangeException.ThrowIfNegative(count); |
| | | 138 | | |
| | | 139 | | // Call it with empty encoder |
| | 0 | 140 | | return GetByteCount(chars, count, null); |
| | | 141 | | } |
| | | 142 | | |
| | | 143 | | // Parent method is safe. |
| | | 144 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 145 | | // So if you fix this, fix the others. Currently those include: |
| | | 146 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 147 | | |
| | | 148 | | public override unsafe int GetBytes(string s, int charIndex, int charCount, |
| | | 149 | | byte[] bytes, int byteIndex) |
| | | 150 | | { |
| | 0 | 151 | | ArgumentNullException.ThrowIfNull(s); |
| | 0 | 152 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 153 | | |
| | 0 | 154 | | ArgumentOutOfRangeException.ThrowIfNegative(charIndex); |
| | 0 | 155 | | ArgumentOutOfRangeException.ThrowIfNegative(charCount); |
| | | 156 | | |
| | 0 | 157 | | if (s.Length - charIndex < charCount) |
| | 0 | 158 | | throw new ArgumentOutOfRangeException(nameof(s), SR.ArgumentOutOfRange_IndexCount); |
| | | 159 | | |
| | 0 | 160 | | if (byteIndex < 0 || byteIndex > bytes.Length) |
| | 0 | 161 | | throw new ArgumentOutOfRangeException(nameof(byteIndex), SR.ArgumentOutOfRange_IndexMustBeLessOrEqual); |
| | | 162 | | |
| | 0 | 163 | | int byteCount = bytes.Length - byteIndex; |
| | | 164 | | |
| | 0 | 165 | | fixed (char* pChars = s) |
| | 0 | 166 | | fixed (byte* pBytes = &MemoryMarshal.GetArrayDataReference(bytes)) |
| | | 167 | | { |
| | 0 | 168 | | return GetBytes(pChars + charIndex, charCount, pBytes + byteIndex, byteCount, null); |
| | | 169 | | } |
| | | 170 | | } |
| | | 171 | | |
| | | 172 | | // Encodes a range of characters in a character array into a range of bytes |
| | | 173 | | // in a byte array. An exception occurs if the byte array is not large |
| | | 174 | | // enough to hold the complete encoding of the characters. The |
| | | 175 | | // GetByteCount method can be used to determine the exact number of |
| | | 176 | | // bytes that will be produced for a given range of characters. |
| | | 177 | | // Alternatively, the GetMaxByteCount method can be used to |
| | | 178 | | // determine the maximum number of bytes that will be produced for a given |
| | | 179 | | // number of characters, regardless of the actual character values. |
| | | 180 | | // |
| | | 181 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 182 | | // So if you fix this, fix the others. Currently those include: |
| | | 183 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 184 | | // parent method is safe |
| | | 185 | | |
| | | 186 | | public override unsafe int GetBytes(char[] chars, int charIndex, int charCount, |
| | | 187 | | byte[] bytes, int byteIndex) |
| | | 188 | | { |
| | 0 | 189 | | ArgumentNullException.ThrowIfNull(chars); |
| | 0 | 190 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 191 | | |
| | 0 | 192 | | ArgumentOutOfRangeException.ThrowIfNegative(charIndex); |
| | 0 | 193 | | ArgumentOutOfRangeException.ThrowIfNegative(charCount); |
| | | 194 | | |
| | 0 | 195 | | if (chars.Length - charIndex < charCount) |
| | 0 | 196 | | throw new ArgumentOutOfRangeException(nameof(chars), SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 197 | | |
| | 0 | 198 | | if (byteIndex < 0 || byteIndex > bytes.Length) |
| | 0 | 199 | | throw new ArgumentOutOfRangeException(nameof(byteIndex), SR.ArgumentOutOfRange_IndexMustBeLessOrEqual); |
| | | 200 | | |
| | | 201 | | // If nothing to encode return 0, avoid fixed problem |
| | 0 | 202 | | if (charCount == 0) |
| | 0 | 203 | | return 0; |
| | | 204 | | |
| | | 205 | | // Just call pointer version |
| | 0 | 206 | | int byteCount = bytes.Length - byteIndex; |
| | | 207 | | |
| | 0 | 208 | | fixed (char* pChars = chars) |
| | 0 | 209 | | fixed (byte* pBytes = &MemoryMarshal.GetArrayDataReference(bytes)) |
| | | 210 | | { |
| | | 211 | | // Remember that byteCount is # to decode, not size of array. |
| | 0 | 212 | | return GetBytes(pChars + charIndex, charCount, pBytes + byteIndex, byteCount, null); |
| | | 213 | | } |
| | | 214 | | } |
| | | 215 | | |
| | | 216 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 217 | | // So if you fix this, fix the others. Currently those include: |
| | | 218 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 219 | | |
| | | 220 | | [CLSCompliant(false)] |
| | | 221 | | public override unsafe int GetBytes(char* chars, int charCount, byte* bytes, int byteCount) |
| | | 222 | | { |
| | 0 | 223 | | ArgumentNullException.ThrowIfNull(chars); |
| | 0 | 224 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 225 | | |
| | 0 | 226 | | ArgumentOutOfRangeException.ThrowIfNegative(charCount); |
| | 0 | 227 | | ArgumentOutOfRangeException.ThrowIfNegative(byteCount); |
| | | 228 | | |
| | 0 | 229 | | return GetBytes(chars, charCount, bytes, byteCount, null); |
| | | 230 | | } |
| | | 231 | | |
| | | 232 | | // Returns the number of characters produced by decoding a range of bytes |
| | | 233 | | // in a byte array. |
| | | 234 | | // |
| | | 235 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 236 | | // So if you fix this, fix the others. Currently those include: |
| | | 237 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 238 | | // parent method is safe |
| | | 239 | | |
| | | 240 | | public override unsafe int GetCharCount(byte[] bytes, int index, int count) |
| | | 241 | | { |
| | 0 | 242 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 243 | | |
| | 0 | 244 | | ArgumentOutOfRangeException.ThrowIfNegative(index); |
| | 0 | 245 | | ArgumentOutOfRangeException.ThrowIfNegative(count); |
| | | 246 | | |
| | 0 | 247 | | if (bytes.Length - index < count) |
| | 0 | 248 | | throw new ArgumentOutOfRangeException(nameof(bytes), SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 249 | | |
| | | 250 | | // If no input just return 0, fixed doesn't like 0 length arrays. |
| | 0 | 251 | | if (count == 0) |
| | 0 | 252 | | return 0; |
| | | 253 | | |
| | | 254 | | // Just call pointer version |
| | 0 | 255 | | fixed (byte* pBytes = bytes) |
| | 0 | 256 | | return GetCharCount(pBytes + index, count, null); |
| | | 257 | | } |
| | | 258 | | |
| | | 259 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 260 | | // So if you fix this, fix the others. Currently those include: |
| | | 261 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 262 | | |
| | | 263 | | [CLSCompliant(false)] |
| | | 264 | | public override unsafe int GetCharCount(byte* bytes, int count) |
| | | 265 | | { |
| | 0 | 266 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 267 | | |
| | 0 | 268 | | ArgumentOutOfRangeException.ThrowIfNegative(count); |
| | | 269 | | |
| | 0 | 270 | | return GetCharCount(bytes, count, null); |
| | | 271 | | } |
| | | 272 | | |
| | | 273 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 274 | | // So if you fix this, fix the others. Currently those include: |
| | | 275 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 276 | | // parent method is safe |
| | | 277 | | |
| | | 278 | | public override unsafe int GetChars(byte[] bytes, int byteIndex, int byteCount, |
| | | 279 | | char[] chars, int charIndex) |
| | | 280 | | { |
| | 0 | 281 | | ArgumentNullException.ThrowIfNull(bytes); |
| | 0 | 282 | | ArgumentNullException.ThrowIfNull(chars); |
| | | 283 | | |
| | 0 | 284 | | ArgumentOutOfRangeException.ThrowIfNegative(byteIndex); |
| | 0 | 285 | | ArgumentOutOfRangeException.ThrowIfNegative(byteCount); |
| | | 286 | | |
| | 0 | 287 | | if (bytes.Length - byteIndex < byteCount) |
| | 0 | 288 | | throw new ArgumentOutOfRangeException(nameof(bytes), SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 289 | | |
| | 0 | 290 | | if (charIndex < 0 || charIndex > chars.Length) |
| | 0 | 291 | | throw new ArgumentOutOfRangeException(nameof(charIndex), SR.ArgumentOutOfRange_IndexMustBeLessOrEqual); |
| | | 292 | | |
| | | 293 | | // If no input, return 0 & avoid fixed problem |
| | 0 | 294 | | if (byteCount == 0) |
| | 0 | 295 | | return 0; |
| | | 296 | | |
| | | 297 | | // Just call pointer version |
| | 0 | 298 | | int charCount = chars.Length - charIndex; |
| | | 299 | | |
| | 0 | 300 | | fixed (byte* pBytes = bytes) |
| | 0 | 301 | | fixed (char* pChars = &MemoryMarshal.GetArrayDataReference(chars)) |
| | | 302 | | { |
| | | 303 | | // Remember that charCount is # to decode, not size of array |
| | 0 | 304 | | return GetChars(pBytes + byteIndex, byteCount, pChars + charIndex, charCount, null); |
| | | 305 | | } |
| | | 306 | | } |
| | | 307 | | |
| | | 308 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 309 | | // So if you fix this, fix the others. Currently those include: |
| | | 310 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 311 | | |
| | | 312 | | [CLSCompliant(false)] |
| | | 313 | | public override unsafe int GetChars(byte* bytes, int byteCount, char* chars, int charCount) |
| | | 314 | | { |
| | 0 | 315 | | ArgumentNullException.ThrowIfNull(bytes); |
| | 0 | 316 | | ArgumentNullException.ThrowIfNull(chars); |
| | | 317 | | |
| | 0 | 318 | | ArgumentOutOfRangeException.ThrowIfNegative(charCount); |
| | 0 | 319 | | ArgumentOutOfRangeException.ThrowIfNegative(byteCount); |
| | | 320 | | |
| | 0 | 321 | | return GetChars(bytes, byteCount, chars, charCount, null); |
| | | 322 | | } |
| | | 323 | | |
| | | 324 | | // Returns a string containing the decoded representation of a range of |
| | | 325 | | // bytes in a byte array. |
| | | 326 | | // |
| | | 327 | | // All of our public Encodings that don't use EncodingNLS must have this (including EncodingNLS) |
| | | 328 | | // So if you fix this, fix the others. Currently those include: |
| | | 329 | | // EncodingNLS, UTF7Encoding, UTF8Encoding, UTF32Encoding, ASCIIEncoding, UnicodeEncoding |
| | | 330 | | // parent method is safe |
| | | 331 | | |
| | | 332 | | public override unsafe string GetString(byte[] bytes, int index, int count) |
| | | 333 | | { |
| | 0 | 334 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 335 | | |
| | 0 | 336 | | ArgumentOutOfRangeException.ThrowIfNegative(index); |
| | 0 | 337 | | ArgumentOutOfRangeException.ThrowIfNegative(count); |
| | | 338 | | |
| | 0 | 339 | | if (bytes.Length - index < count) |
| | 0 | 340 | | throw new ArgumentOutOfRangeException(nameof(bytes), SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 341 | | |
| | | 342 | | // Avoid problems with empty input buffer |
| | 0 | 343 | | if (count == 0) return string.Empty; |
| | | 344 | | |
| | 0 | 345 | | fixed (byte* pBytes = bytes) |
| | 0 | 346 | | return string.CreateStringFromEncoding( |
| | 0 | 347 | | pBytes + index, count, this); |
| | | 348 | | } |
| | | 349 | | |
| | | 350 | | // |
| | | 351 | | // End of standard methods copied from EncodingNLS.cs |
| | | 352 | | // |
| | | 353 | | internal override unsafe int GetByteCount(char* chars, int count, EncoderNLS? encoder) |
| | | 354 | | { |
| | 0 | 355 | | Debug.Assert(chars is not null, "[UTF32Encoding.GetByteCount]chars!=null"); |
| | 0 | 356 | | Debug.Assert(count >= 0, "[UTF32Encoding.GetByteCount]count >=0"); |
| | | 357 | | |
| | 0 | 358 | | char* end = chars + count; |
| | 0 | 359 | | char* charStart = chars; |
| | 0 | 360 | | int byteCount = 0; |
| | | 361 | | |
| | 0 | 362 | | char highSurrogate = '\0'; |
| | | 363 | | |
| | | 364 | | // For fallback we may need a fallback buffer |
| | 0 | 365 | | EncoderFallbackBuffer? fallbackBuffer = null; |
| | | 366 | | char* charsForFallback; |
| | | 367 | | |
| | 0 | 368 | | if (encoder is not null) |
| | | 369 | | { |
| | 0 | 370 | | highSurrogate = encoder._charLeftOver; |
| | 0 | 371 | | fallbackBuffer = encoder.FallbackBuffer; |
| | | 372 | | |
| | | 373 | | // We mustn't have left over fallback data when counting |
| | 0 | 374 | | if (fallbackBuffer.Remaining > 0) |
| | 0 | 375 | | throw new ArgumentException(SR.Format(SR.Argument_EncoderFallbackNotEmpty, this.EncodingName, encode |
| | | 376 | | } |
| | | 377 | | else |
| | | 378 | | { |
| | 0 | 379 | | fallbackBuffer = this.encoderFallback.CreateFallbackBuffer(); |
| | | 380 | | } |
| | | 381 | | |
| | | 382 | | // Set our internal fallback interesting things. |
| | 0 | 383 | | fallbackBuffer.InternalInitialize(charStart, end, encoder, false); |
| | | 384 | | |
| | | 385 | | char ch; |
| | | 386 | | TryAgain: |
| | | 387 | | |
| | 0 | 388 | | while (((ch = fallbackBuffer.InternalGetNextChar()) != 0) || chars < end) |
| | | 389 | | { |
| | | 390 | | // First unwind any fallback |
| | 0 | 391 | | if (ch == 0) |
| | | 392 | | { |
| | | 393 | | // No fallback, just get next char |
| | 0 | 394 | | ch = *chars; |
| | 0 | 395 | | chars++; |
| | | 396 | | } |
| | | 397 | | |
| | | 398 | | // Do we need a low surrogate? |
| | 0 | 399 | | if (highSurrogate != '\0') |
| | | 400 | | { |
| | | 401 | | // |
| | | 402 | | // In previous char, we encounter a high surrogate, so we are expecting a low surrogate here. |
| | | 403 | | // |
| | 0 | 404 | | if (char.IsLowSurrogate(ch)) |
| | | 405 | | { |
| | | 406 | | // They're all legal |
| | 0 | 407 | | highSurrogate = '\0'; |
| | | 408 | | |
| | | 409 | | // |
| | | 410 | | // One surrogate pair will be translated into 4 bytes UTF32. |
| | | 411 | | // |
| | | 412 | | |
| | 0 | 413 | | byteCount += 4; |
| | 0 | 414 | | continue; |
| | | 415 | | } |
| | | 416 | | |
| | | 417 | | // We are missing our low surrogate, decrement chars and fallback the high surrogate |
| | | 418 | | // The high surrogate may have come from the encoder, but nothing else did. |
| | 0 | 419 | | Debug.Assert(chars > charStart, |
| | 0 | 420 | | "[UTF32Encoding.GetByteCount]Expected chars to have advanced if no low surrogate"); |
| | 0 | 421 | | chars--; |
| | | 422 | | |
| | | 423 | | // Do the fallback |
| | 0 | 424 | | charsForFallback = chars; |
| | 0 | 425 | | fallbackBuffer.InternalFallback(highSurrogate, ref charsForFallback); |
| | 0 | 426 | | chars = charsForFallback; |
| | | 427 | | |
| | | 428 | | // We're going to fallback the old high surrogate. |
| | 0 | 429 | | highSurrogate = '\0'; |
| | 0 | 430 | | continue; |
| | | 431 | | } |
| | | 432 | | |
| | | 433 | | // Do we have another high surrogate? |
| | 0 | 434 | | if (char.IsHighSurrogate(ch)) |
| | | 435 | | { |
| | | 436 | | // |
| | | 437 | | // We'll have a high surrogate to check next time. |
| | | 438 | | // |
| | 0 | 439 | | highSurrogate = ch; |
| | 0 | 440 | | continue; |
| | | 441 | | } |
| | | 442 | | |
| | | 443 | | // Check for illegal characters |
| | 0 | 444 | | if (char.IsLowSurrogate(ch)) |
| | | 445 | | { |
| | | 446 | | // We have a leading low surrogate, do the fallback |
| | 0 | 447 | | charsForFallback = chars; |
| | 0 | 448 | | fallbackBuffer.InternalFallback(ch, ref charsForFallback); |
| | 0 | 449 | | chars = charsForFallback; |
| | | 450 | | |
| | | 451 | | // Try again with fallback buffer |
| | 0 | 452 | | continue; |
| | | 453 | | } |
| | | 454 | | |
| | | 455 | | // We get to add the character (4 bytes UTF32) |
| | 0 | 456 | | byteCount += 4; |
| | | 457 | | } |
| | | 458 | | |
| | | 459 | | // May have to do our last surrogate |
| | 0 | 460 | | if ((encoder is null || encoder.MustFlush) && highSurrogate > 0) |
| | | 461 | | { |
| | | 462 | | // We have to do the fallback for the lonely high surrogate |
| | 0 | 463 | | charsForFallback = chars; |
| | 0 | 464 | | fallbackBuffer.InternalFallback(highSurrogate, ref charsForFallback); |
| | 0 | 465 | | chars = charsForFallback; |
| | | 466 | | |
| | 0 | 467 | | highSurrogate = (char)0; |
| | 0 | 468 | | goto TryAgain; |
| | | 469 | | } |
| | | 470 | | |
| | | 471 | | // Check for overflows. |
| | 0 | 472 | | if (byteCount < 0) |
| | 0 | 473 | | throw new ArgumentOutOfRangeException(nameof(count), SR.ArgumentOutOfRange_GetByteCountOverflow); |
| | | 474 | | |
| | | 475 | | // Shouldn't have anything in fallback buffer for GetByteCount |
| | | 476 | | // (don't have to check _throwOnOverflow for count) |
| | 0 | 477 | | Debug.Assert(fallbackBuffer.Remaining == 0, |
| | 0 | 478 | | "[UTF32Encoding.GetByteCount]Expected empty fallback buffer at end"); |
| | | 479 | | |
| | | 480 | | // Return our count |
| | 0 | 481 | | return byteCount; |
| | | 482 | | } |
| | | 483 | | |
| | | 484 | | internal override unsafe int GetBytes(char* chars, int charCount, |
| | | 485 | | byte* bytes, int byteCount, EncoderNLS? encoder) |
| | | 486 | | { |
| | 0 | 487 | | Debug.Assert(chars is not null, "[UTF32Encoding.GetBytes]chars!=null"); |
| | 0 | 488 | | Debug.Assert(bytes is not null, "[UTF32Encoding.GetBytes]bytes!=null"); |
| | 0 | 489 | | Debug.Assert(byteCount >= 0, "[UTF32Encoding.GetBytes]byteCount >=0"); |
| | 0 | 490 | | Debug.Assert(charCount >= 0, "[UTF32Encoding.GetBytes]charCount >=0"); |
| | | 491 | | |
| | 0 | 492 | | char* charStart = chars; |
| | 0 | 493 | | char* charEnd = chars + charCount; |
| | 0 | 494 | | byte* byteStart = bytes; |
| | 0 | 495 | | byte* byteEnd = bytes + byteCount; |
| | | 496 | | |
| | 0 | 497 | | char highSurrogate = '\0'; |
| | | 498 | | |
| | | 499 | | // For fallback we may need a fallback buffer |
| | 0 | 500 | | EncoderFallbackBuffer? fallbackBuffer = null; |
| | | 501 | | char* charsForFallback; |
| | | 502 | | |
| | 0 | 503 | | if (encoder is not null) |
| | | 504 | | { |
| | 0 | 505 | | highSurrogate = encoder._charLeftOver; |
| | 0 | 506 | | fallbackBuffer = encoder.FallbackBuffer; |
| | | 507 | | |
| | | 508 | | // We mustn't have left over fallback data when not converting |
| | 0 | 509 | | if (encoder._throwOnOverflow && fallbackBuffer.Remaining > 0) |
| | 0 | 510 | | throw new ArgumentException(SR.Format(SR.Argument_EncoderFallbackNotEmpty, this.EncodingName, encode |
| | | 511 | | } |
| | | 512 | | else |
| | | 513 | | { |
| | 0 | 514 | | fallbackBuffer = this.encoderFallback.CreateFallbackBuffer(); |
| | | 515 | | } |
| | | 516 | | |
| | | 517 | | // Set our internal fallback interesting things. |
| | 0 | 518 | | fallbackBuffer.InternalInitialize(charStart, charEnd, encoder, true); |
| | | 519 | | |
| | | 520 | | char ch; |
| | | 521 | | TryAgain: |
| | | 522 | | |
| | 0 | 523 | | while (((ch = fallbackBuffer.InternalGetNextChar()) != 0) || chars < charEnd) |
| | | 524 | | { |
| | | 525 | | // First unwind any fallback |
| | 0 | 526 | | if (ch == 0) |
| | | 527 | | { |
| | | 528 | | // No fallback, just get next char |
| | 0 | 529 | | ch = *chars; |
| | 0 | 530 | | chars++; |
| | | 531 | | } |
| | | 532 | | |
| | | 533 | | // Do we need a low surrogate? |
| | 0 | 534 | | if (highSurrogate != '\0') |
| | | 535 | | { |
| | | 536 | | // |
| | | 537 | | // In previous char, we encountered a high surrogate, so we are expecting a low surrogate here. |
| | | 538 | | // |
| | 0 | 539 | | if (char.IsLowSurrogate(ch)) |
| | | 540 | | { |
| | | 541 | | // Is it a legal one? |
| | 0 | 542 | | uint iTemp = GetSurrogate(highSurrogate, ch); |
| | 0 | 543 | | highSurrogate = '\0'; |
| | | 544 | | |
| | | 545 | | // |
| | | 546 | | // One surrogate pair will be translated into 4 bytes UTF32. |
| | | 547 | | // |
| | 0 | 548 | | if (bytes + 3 >= byteEnd) |
| | | 549 | | { |
| | | 550 | | // Don't have 4 bytes |
| | 0 | 551 | | if (fallbackBuffer.bFallingBack) |
| | | 552 | | { |
| | 0 | 553 | | fallbackBuffer.MovePrevious(); // Aren't using these 2 fallback chars |
| | 0 | 554 | | fallbackBuffer.MovePrevious(); |
| | | 555 | | } |
| | | 556 | | else |
| | | 557 | | { |
| | | 558 | | // If we don't have enough room, then either we should've advanced a while |
| | | 559 | | // or we should have bytes==byteStart and throw below |
| | 0 | 560 | | Debug.Assert(chars > charStart + 1 || bytes == byteStart, |
| | 0 | 561 | | "[UnicodeEncoding.GetBytes]Expected chars to have when no room to add surrogate pair |
| | 0 | 562 | | chars -= 2; // Aren't using those 2 chars |
| | | 563 | | } |
| | 0 | 564 | | ThrowBytesOverflow(encoder, bytes == byteStart); // Throw maybe (if no bytes written) |
| | 0 | 565 | | highSurrogate = (char)0; // Nothing left over (we backed up to st |
| | 0 | 566 | | break; |
| | | 567 | | } |
| | | 568 | | |
| | 0 | 569 | | if (_bigEndian) |
| | | 570 | | { |
| | 0 | 571 | | *(bytes++) = (byte)(0x00); |
| | 0 | 572 | | *(bytes++) = (byte)(iTemp >> 16); // Implies & 0xFF, which isn't needed cause high are |
| | 0 | 573 | | *(bytes++) = (byte)(iTemp >> 8); // Implies & 0xFF |
| | 0 | 574 | | *(bytes++) = (byte)(iTemp); // Implies & 0xFF |
| | | 575 | | } |
| | | 576 | | else |
| | | 577 | | { |
| | 0 | 578 | | *(bytes++) = (byte)(iTemp); // Implies & 0xFF |
| | 0 | 579 | | *(bytes++) = (byte)(iTemp >> 8); // Implies & 0xFF |
| | 0 | 580 | | *(bytes++) = (byte)(iTemp >> 16); // Implies & 0xFF, which isn't needed cause high are |
| | 0 | 581 | | *(bytes++) = (byte)(0x00); |
| | | 582 | | } |
| | 0 | 583 | | continue; |
| | | 584 | | } |
| | | 585 | | |
| | | 586 | | // We are missing our low surrogate, decrement chars and fallback the high surrogate |
| | | 587 | | // The high surrogate may have come from the encoder, but nothing else did. |
| | 0 | 588 | | Debug.Assert(chars > charStart, |
| | 0 | 589 | | "[UTF32Encoding.GetBytes]Expected chars to have advanced if no low surrogate"); |
| | 0 | 590 | | chars--; |
| | | 591 | | |
| | | 592 | | // Do the fallback |
| | 0 | 593 | | charsForFallback = chars; |
| | 0 | 594 | | fallbackBuffer.InternalFallback(highSurrogate, ref charsForFallback); |
| | 0 | 595 | | chars = charsForFallback; |
| | | 596 | | |
| | | 597 | | // We're going to fallback the old high surrogate. |
| | 0 | 598 | | highSurrogate = '\0'; |
| | 0 | 599 | | continue; |
| | | 600 | | } |
| | | 601 | | |
| | | 602 | | // Do we have another high surrogate?, if so remember it |
| | 0 | 603 | | if (char.IsHighSurrogate(ch)) |
| | | 604 | | { |
| | | 605 | | // |
| | | 606 | | // We'll have a high surrogate to check next time. |
| | | 607 | | // |
| | 0 | 608 | | highSurrogate = ch; |
| | 0 | 609 | | continue; |
| | | 610 | | } |
| | | 611 | | |
| | | 612 | | // Check for illegal characters (low surrogate) |
| | 0 | 613 | | if (char.IsLowSurrogate(ch)) |
| | | 614 | | { |
| | | 615 | | // We have a leading low surrogate, do the fallback |
| | 0 | 616 | | charsForFallback = chars; |
| | 0 | 617 | | fallbackBuffer.InternalFallback(ch, ref charsForFallback); |
| | 0 | 618 | | chars = charsForFallback; |
| | | 619 | | |
| | | 620 | | // Try again with fallback buffer |
| | 0 | 621 | | continue; |
| | | 622 | | } |
| | | 623 | | |
| | | 624 | | // We get to add the character, yippee. |
| | 0 | 625 | | if (bytes + 3 >= byteEnd) |
| | | 626 | | { |
| | | 627 | | // Don't have 4 bytes |
| | 0 | 628 | | if (fallbackBuffer.bFallingBack) |
| | 0 | 629 | | fallbackBuffer.MovePrevious(); // Aren't using this fallback char |
| | | 630 | | else |
| | | 631 | | { |
| | | 632 | | // Must've advanced already |
| | 0 | 633 | | Debug.Assert(chars > charStart, |
| | 0 | 634 | | "[UTF32Encoding.GetBytes]Expected chars to have advanced if normal character"); |
| | 0 | 635 | | chars--; // Aren't using this char |
| | | 636 | | } |
| | 0 | 637 | | ThrowBytesOverflow(encoder, bytes == byteStart); // Throw maybe (if no bytes written) |
| | 0 | 638 | | break; // Didn't throw, stop |
| | | 639 | | } |
| | | 640 | | |
| | 0 | 641 | | if (_bigEndian) |
| | | 642 | | { |
| | 0 | 643 | | *(bytes++) = (byte)(0x00); |
| | 0 | 644 | | *(bytes++) = (byte)(0x00); |
| | 0 | 645 | | *(bytes++) = (byte)((uint)ch >> 8); // Implies & 0xFF |
| | 0 | 646 | | *(bytes++) = (byte)(ch); // Implies & 0xFF |
| | | 647 | | } |
| | | 648 | | else |
| | | 649 | | { |
| | 0 | 650 | | *(bytes++) = (byte)(ch); // Implies & 0xFF |
| | 0 | 651 | | *(bytes++) = (byte)((uint)ch >> 8); // Implies & 0xFF |
| | 0 | 652 | | *(bytes++) = (byte)(0x00); |
| | 0 | 653 | | *(bytes++) = (byte)(0x00); |
| | | 654 | | } |
| | | 655 | | } |
| | | 656 | | |
| | | 657 | | // May have to do our last surrogate |
| | 0 | 658 | | if ((encoder is null || encoder.MustFlush) && highSurrogate > 0) |
| | | 659 | | { |
| | | 660 | | // We have to do the fallback for the lonely high surrogate |
| | 0 | 661 | | charsForFallback = chars; |
| | 0 | 662 | | fallbackBuffer.InternalFallback(highSurrogate, ref charsForFallback); |
| | 0 | 663 | | chars = charsForFallback; |
| | | 664 | | |
| | 0 | 665 | | highSurrogate = (char)0; |
| | 0 | 666 | | goto TryAgain; |
| | | 667 | | } |
| | | 668 | | |
| | | 669 | | // Fix our encoder if we have one |
| | 0 | 670 | | Debug.Assert(highSurrogate == 0 || (encoder is not null && !encoder.MustFlush), |
| | 0 | 671 | | "[UTF32Encoding.GetBytes]Expected encoder to be flushed."); |
| | | 672 | | |
| | 0 | 673 | | if (encoder is not null) |
| | | 674 | | { |
| | | 675 | | // Remember our left over surrogate (or 0 if flushing) |
| | 0 | 676 | | encoder._charLeftOver = highSurrogate; |
| | | 677 | | |
| | | 678 | | // Need # chars used |
| | 0 | 679 | | encoder._charsUsed = (int)(chars - charStart); |
| | | 680 | | } |
| | | 681 | | |
| | | 682 | | // return the new length |
| | 0 | 683 | | return (int)(bytes - byteStart); |
| | | 684 | | } |
| | | 685 | | |
| | | 686 | | internal override unsafe int GetCharCount(byte* bytes, int count, DecoderNLS? baseDecoder) |
| | | 687 | | { |
| | 0 | 688 | | Debug.Assert(bytes is not null, "[UTF32Encoding.GetCharCount]bytes!=null"); |
| | 0 | 689 | | Debug.Assert(count >= 0, "[UTF32Encoding.GetCharCount]count >=0"); |
| | | 690 | | |
| | 0 | 691 | | UTF32Decoder? decoder = (UTF32Decoder?)baseDecoder; |
| | | 692 | | |
| | | 693 | | // None so far! |
| | 0 | 694 | | int charCount = 0; |
| | 0 | 695 | | byte* end = bytes + count; |
| | 0 | 696 | | byte* byteStart = bytes; |
| | | 697 | | |
| | | 698 | | // Set up decoder |
| | 0 | 699 | | int readCount = 0; |
| | 0 | 700 | | uint iChar = 0; |
| | | 701 | | |
| | | 702 | | // For fallback we may need a fallback buffer |
| | 0 | 703 | | DecoderFallbackBuffer? fallbackBuffer = null; |
| | | 704 | | |
| | | 705 | | // See if there's anything in our decoder |
| | 0 | 706 | | if (decoder is not null) |
| | | 707 | | { |
| | 0 | 708 | | readCount = decoder.readByteCount; |
| | 0 | 709 | | iChar = (uint)decoder.iChar; |
| | 0 | 710 | | fallbackBuffer = decoder.FallbackBuffer; |
| | | 711 | | |
| | | 712 | | // Shouldn't have anything in fallback buffer for GetCharCount |
| | | 713 | | // (don't have to check _throwOnOverflow for chars or count) |
| | 0 | 714 | | Debug.Assert(fallbackBuffer.Remaining == 0, |
| | 0 | 715 | | "[UTF32Encoding.GetCharCount]Expected empty fallback buffer at start"); |
| | | 716 | | } |
| | | 717 | | else |
| | | 718 | | { |
| | 0 | 719 | | fallbackBuffer = this.decoderFallback.CreateFallbackBuffer(); |
| | | 720 | | } |
| | | 721 | | |
| | | 722 | | // Set our internal fallback interesting things. |
| | 0 | 723 | | fallbackBuffer.InternalInitialize(byteStart, null); |
| | | 724 | | |
| | | 725 | | // Loop through our input, 4 characters at a time! |
| | 0 | 726 | | while (bytes < end && charCount >= 0) |
| | | 727 | | { |
| | | 728 | | // Get our next character |
| | 0 | 729 | | if (_bigEndian) |
| | | 730 | | { |
| | | 731 | | // Scoot left and add it to the bottom |
| | 0 | 732 | | iChar <<= 8; |
| | 0 | 733 | | iChar += *(bytes++); |
| | | 734 | | } |
| | | 735 | | else |
| | | 736 | | { |
| | | 737 | | // Scoot right and add it to the top |
| | 0 | 738 | | iChar >>= 8; |
| | 0 | 739 | | iChar += (uint)(*(bytes++)) << 24; |
| | | 740 | | } |
| | | 741 | | |
| | 0 | 742 | | readCount++; |
| | | 743 | | |
| | | 744 | | // See if we have all the bytes yet |
| | 0 | 745 | | if (readCount < 4) |
| | | 746 | | continue; |
| | | 747 | | |
| | | 748 | | // Have the bytes |
| | 0 | 749 | | readCount = 0; |
| | | 750 | | |
| | | 751 | | // See if its valid to encode |
| | 0 | 752 | | if (iChar > 0x10FFFF || (iChar >= 0xD800 && iChar <= 0xDFFF)) |
| | | 753 | | { |
| | | 754 | | // Need to fall back these 4 bytes |
| | | 755 | | byte[] fallbackBytes; |
| | 0 | 756 | | if (_bigEndian) |
| | | 757 | | { |
| | 0 | 758 | | fallbackBytes = [ |
| | 0 | 759 | | unchecked((byte)(iChar >> 24)), unchecked((byte)(iChar >> 16)), |
| | 0 | 760 | | unchecked((byte)(iChar >> 8)), unchecked((byte)(iChar)) ]; |
| | | 761 | | } |
| | | 762 | | else |
| | | 763 | | { |
| | 0 | 764 | | fallbackBytes = [ |
| | 0 | 765 | | unchecked((byte)(iChar)), unchecked((byte)(iChar >> 8)), |
| | 0 | 766 | | unchecked((byte)(iChar >> 16)), unchecked((byte)(iChar >> 24)) ]; |
| | | 767 | | } |
| | | 768 | | |
| | 0 | 769 | | charCount += fallbackBuffer.InternalFallback(fallbackBytes, bytes); |
| | | 770 | | |
| | | 771 | | // Ignore the illegal character |
| | 0 | 772 | | iChar = 0; |
| | 0 | 773 | | continue; |
| | | 774 | | } |
| | | 775 | | |
| | | 776 | | // Ok, we have something we can add to our output |
| | 0 | 777 | | if (iChar >= 0x10000) |
| | | 778 | | { |
| | | 779 | | // Surrogates take 2 |
| | 0 | 780 | | charCount++; |
| | | 781 | | } |
| | | 782 | | |
| | | 783 | | // Add the rest of the surrogate or our normal character |
| | 0 | 784 | | charCount++; |
| | | 785 | | |
| | | 786 | | // iChar is back to 0 |
| | 0 | 787 | | iChar = 0; |
| | | 788 | | } |
| | | 789 | | |
| | | 790 | | // See if we have something left over that has to be decoded |
| | 0 | 791 | | if (readCount > 0 && (decoder is null || decoder.MustFlush)) |
| | | 792 | | { |
| | | 793 | | // Oops, there's something left over with no place to go. |
| | 0 | 794 | | byte[] fallbackBytes = new byte[readCount]; |
| | 0 | 795 | | if (_bigEndian) |
| | | 796 | | { |
| | 0 | 797 | | while (readCount > 0) |
| | | 798 | | { |
| | 0 | 799 | | fallbackBytes[--readCount] = unchecked((byte)iChar); |
| | 0 | 800 | | iChar >>= 8; |
| | | 801 | | } |
| | | 802 | | } |
| | | 803 | | else |
| | | 804 | | { |
| | 0 | 805 | | while (readCount > 0) |
| | | 806 | | { |
| | 0 | 807 | | fallbackBytes[--readCount] = unchecked((byte)(iChar >> 24)); |
| | 0 | 808 | | iChar <<= 8; |
| | | 809 | | } |
| | | 810 | | } |
| | | 811 | | |
| | 0 | 812 | | charCount += fallbackBuffer.InternalFallback(fallbackBytes, bytes); |
| | | 813 | | } |
| | | 814 | | |
| | | 815 | | // Check for overflows. |
| | 0 | 816 | | if (charCount < 0) |
| | 0 | 817 | | throw new ArgumentOutOfRangeException(nameof(count), SR.ArgumentOutOfRange_GetByteCountOverflow); |
| | | 818 | | |
| | | 819 | | // Shouldn't have anything in fallback buffer for GetCharCount |
| | | 820 | | // (don't have to check _throwOnOverflow for chars or count) |
| | 0 | 821 | | Debug.Assert(fallbackBuffer.Remaining == 0, |
| | 0 | 822 | | "[UTF32Encoding.GetCharCount]Expected empty fallback buffer at end"); |
| | | 823 | | |
| | | 824 | | // Return our count |
| | 0 | 825 | | return charCount; |
| | | 826 | | } |
| | | 827 | | |
| | | 828 | | internal override unsafe int GetChars(byte* bytes, int byteCount, |
| | | 829 | | char* chars, int charCount, DecoderNLS? baseDecoder) |
| | | 830 | | { |
| | 0 | 831 | | Debug.Assert(chars is not null, "[UTF32Encoding.GetChars]chars!=null"); |
| | 0 | 832 | | Debug.Assert(bytes is not null, "[UTF32Encoding.GetChars]bytes!=null"); |
| | 0 | 833 | | Debug.Assert(byteCount >= 0, "[UTF32Encoding.GetChars]byteCount >=0"); |
| | 0 | 834 | | Debug.Assert(charCount >= 0, "[UTF32Encoding.GetChars]charCount >=0"); |
| | | 835 | | |
| | 0 | 836 | | UTF32Decoder? decoder = (UTF32Decoder?)baseDecoder; |
| | | 837 | | |
| | | 838 | | // None so far! |
| | 0 | 839 | | char* charStart = chars; |
| | 0 | 840 | | char* charEnd = chars + charCount; |
| | | 841 | | |
| | 0 | 842 | | byte* byteStart = bytes; |
| | 0 | 843 | | byte* byteEnd = bytes + byteCount; |
| | | 844 | | |
| | | 845 | | // See if there's anything in our decoder (but don't clear it yet) |
| | 0 | 846 | | int readCount = 0; |
| | 0 | 847 | | uint iChar = 0; |
| | | 848 | | |
| | | 849 | | // For fallback we may need a fallback buffer |
| | 0 | 850 | | DecoderFallbackBuffer? fallbackBuffer = null; |
| | | 851 | | char* charsForFallback; |
| | | 852 | | |
| | | 853 | | // See if there's anything in our decoder |
| | 0 | 854 | | if (decoder is not null) |
| | | 855 | | { |
| | 0 | 856 | | readCount = decoder.readByteCount; |
| | 0 | 857 | | iChar = (uint)decoder.iChar; |
| | 0 | 858 | | Debug.Assert(baseDecoder is not null); |
| | 0 | 859 | | fallbackBuffer = baseDecoder.FallbackBuffer; |
| | | 860 | | |
| | | 861 | | // Shouldn't have anything in fallback buffer for GetChars |
| | | 862 | | // (don't have to check _throwOnOverflow for chars) |
| | 0 | 863 | | Debug.Assert(fallbackBuffer.Remaining == 0, |
| | 0 | 864 | | "[UTF32Encoding.GetChars]Expected empty fallback buffer at start"); |
| | | 865 | | } |
| | | 866 | | else |
| | | 867 | | { |
| | 0 | 868 | | fallbackBuffer = this.decoderFallback.CreateFallbackBuffer(); |
| | | 869 | | } |
| | | 870 | | |
| | | 871 | | // Set our internal fallback interesting things. |
| | 0 | 872 | | fallbackBuffer.InternalInitialize(bytes, chars + charCount); |
| | | 873 | | |
| | | 874 | | // Loop through our input, 4 characters at a time! |
| | 0 | 875 | | while (bytes < byteEnd) |
| | | 876 | | { |
| | | 877 | | // Get our next character |
| | 0 | 878 | | if (_bigEndian) |
| | | 879 | | { |
| | | 880 | | // Scoot left and add it to the bottom |
| | 0 | 881 | | iChar <<= 8; |
| | 0 | 882 | | iChar += *(bytes++); |
| | | 883 | | } |
| | | 884 | | else |
| | | 885 | | { |
| | | 886 | | // Scoot right and add it to the top |
| | 0 | 887 | | iChar >>= 8; |
| | 0 | 888 | | iChar += (uint)(*(bytes++)) << 24; |
| | | 889 | | } |
| | | 890 | | |
| | 0 | 891 | | readCount++; |
| | | 892 | | |
| | | 893 | | // See if we have all the bytes yet |
| | 0 | 894 | | if (readCount < 4) |
| | | 895 | | continue; |
| | | 896 | | |
| | | 897 | | // Have the bytes |
| | 0 | 898 | | readCount = 0; |
| | | 899 | | |
| | | 900 | | // See if its valid to encode |
| | 0 | 901 | | if (iChar > 0x10FFFF || (iChar >= 0xD800 && iChar <= 0xDFFF)) |
| | | 902 | | { |
| | | 903 | | // Need to fall back these 4 bytes |
| | | 904 | | byte[] fallbackBytes; |
| | 0 | 905 | | if (_bigEndian) |
| | | 906 | | { |
| | 0 | 907 | | fallbackBytes = [ |
| | 0 | 908 | | unchecked((byte)(iChar >> 24)), unchecked((byte)(iChar >> 16)), |
| | 0 | 909 | | unchecked((byte)(iChar >> 8)), unchecked((byte)(iChar)) ]; |
| | | 910 | | } |
| | | 911 | | else |
| | | 912 | | { |
| | 0 | 913 | | fallbackBytes = [ |
| | 0 | 914 | | unchecked((byte)(iChar)), unchecked((byte)(iChar >> 8)), |
| | 0 | 915 | | unchecked((byte)(iChar >> 16)), unchecked((byte)(iChar >> 24)) ]; |
| | | 916 | | } |
| | | 917 | | |
| | | 918 | | // Chars won't be updated unless this works. |
| | 0 | 919 | | charsForFallback = chars; |
| | 0 | 920 | | bool fallbackResult = fallbackBuffer.InternalFallback(fallbackBytes, bytes, ref charsForFallback); |
| | 0 | 921 | | chars = charsForFallback; |
| | | 922 | | |
| | 0 | 923 | | if (!fallbackResult) |
| | | 924 | | { |
| | | 925 | | // Couldn't fallback, throw or wait til next time |
| | | 926 | | // We either read enough bytes for bytes-=4 to work, or we're |
| | | 927 | | // going to throw in ThrowCharsOverflow because chars == charStart |
| | 0 | 928 | | Debug.Assert(bytes >= byteStart + 4 || chars == charStart, |
| | 0 | 929 | | "[UTF32Encoding.GetChars]Expected to have consumed bytes or throw (bad surrogate)"); |
| | 0 | 930 | | bytes -= 4; // get back to where we were |
| | 0 | 931 | | iChar = 0; // Remembering nothing |
| | 0 | 932 | | fallbackBuffer.InternalReset(); |
| | 0 | 933 | | ThrowCharsOverflow(decoder, chars == charStart); // Might throw, if no chars output |
| | 0 | 934 | | break; // Stop here, didn't throw |
| | | 935 | | } |
| | | 936 | | |
| | | 937 | | // Ignore the illegal character |
| | 0 | 938 | | iChar = 0; |
| | 0 | 939 | | continue; |
| | | 940 | | } |
| | | 941 | | |
| | | 942 | | // Ok, we have something we can add to our output |
| | 0 | 943 | | if (iChar >= 0x10000) |
| | | 944 | | { |
| | | 945 | | // Surrogates take 2 |
| | 0 | 946 | | if (charEnd - chars < 2) |
| | | 947 | | { |
| | | 948 | | // Throwing or stopping |
| | | 949 | | // We either read enough bytes for bytes-=4 to work, or we're |
| | | 950 | | // going to throw in ThrowCharsOverflow because chars == charStart |
| | 0 | 951 | | Debug.Assert(bytes >= byteStart + 4 || chars == charStart, |
| | 0 | 952 | | "[UTF32Encoding.GetChars]Expected to have consumed bytes or throw (surrogate)"); |
| | 0 | 953 | | bytes -= 4; // get back to where we were |
| | 0 | 954 | | iChar = 0; // Remembering nothing |
| | 0 | 955 | | ThrowCharsOverflow(decoder, chars == charStart); // Might throw, if no chars output |
| | 0 | 956 | | break; // Stop here, didn't throw |
| | | 957 | | } |
| | | 958 | | |
| | 0 | 959 | | *(chars++) = GetHighSurrogate(iChar); |
| | 0 | 960 | | iChar = GetLowSurrogate(iChar); |
| | | 961 | | } |
| | | 962 | | // Bounds check for normal character |
| | 0 | 963 | | else if (chars >= charEnd) |
| | | 964 | | { |
| | | 965 | | // Throwing or stopping |
| | | 966 | | // We either read enough bytes for bytes-=4 to work, or we're |
| | | 967 | | // going to throw in ThrowCharsOverflow because chars == charStart |
| | 0 | 968 | | Debug.Assert(bytes >= byteStart + 4 || chars == charStart, |
| | 0 | 969 | | "[UTF32Encoding.GetChars]Expected to have consumed bytes or throw (normal char)"); |
| | 0 | 970 | | bytes -= 4; // get back to where we were |
| | 0 | 971 | | iChar = 0; // Remembering nothing |
| | 0 | 972 | | ThrowCharsOverflow(decoder, chars == charStart); // Might throw, if no chars output |
| | 0 | 973 | | break; // Stop here, didn't throw |
| | | 974 | | } |
| | | 975 | | |
| | | 976 | | // Add the rest of the surrogate or our normal character |
| | 0 | 977 | | *(chars++) = (char)iChar; |
| | | 978 | | |
| | | 979 | | // iChar is back to 0 |
| | 0 | 980 | | iChar = 0; |
| | | 981 | | } |
| | | 982 | | |
| | | 983 | | // See if we have something left over that has to be decoded |
| | 0 | 984 | | if (readCount > 0 && (decoder is null || decoder.MustFlush)) |
| | | 985 | | { |
| | | 986 | | // Oops, there's something left over with no place to go. |
| | 0 | 987 | | byte[] fallbackBytes = new byte[readCount]; |
| | 0 | 988 | | int tempCount = readCount; |
| | 0 | 989 | | if (_bigEndian) |
| | | 990 | | { |
| | 0 | 991 | | while (tempCount > 0) |
| | | 992 | | { |
| | 0 | 993 | | fallbackBytes[--tempCount] = unchecked((byte)iChar); |
| | 0 | 994 | | iChar >>= 8; |
| | | 995 | | } |
| | | 996 | | } |
| | | 997 | | else |
| | | 998 | | { |
| | 0 | 999 | | while (tempCount > 0) |
| | | 1000 | | { |
| | 0 | 1001 | | fallbackBytes[--tempCount] = unchecked((byte)(iChar >> 24)); |
| | 0 | 1002 | | iChar <<= 8; |
| | | 1003 | | } |
| | | 1004 | | } |
| | | 1005 | | |
| | 0 | 1006 | | charsForFallback = chars; |
| | 0 | 1007 | | bool fallbackResult = fallbackBuffer.InternalFallback(fallbackBytes, bytes, ref charsForFallback); |
| | 0 | 1008 | | chars = charsForFallback; |
| | | 1009 | | |
| | 0 | 1010 | | if (!fallbackResult) |
| | | 1011 | | { |
| | | 1012 | | // Couldn't fallback. |
| | 0 | 1013 | | fallbackBuffer.InternalReset(); |
| | 0 | 1014 | | ThrowCharsOverflow(decoder, chars == charStart); // Might throw, if no chars output |
| | | 1015 | | // Stop here, didn't throw, backed up, so still nothing in buffer |
| | | 1016 | | } |
| | | 1017 | | else |
| | | 1018 | | { |
| | | 1019 | | // Don't clear our decoder unless we could fall it back. |
| | | 1020 | | // If we caught the if above, then we're a convert() and will catch this next time. |
| | 0 | 1021 | | readCount = 0; |
| | 0 | 1022 | | iChar = 0; |
| | | 1023 | | } |
| | | 1024 | | } |
| | | 1025 | | |
| | | 1026 | | // Remember any left over stuff, clearing buffer as well for MustFlush |
| | 0 | 1027 | | if (decoder is not null) |
| | | 1028 | | { |
| | 0 | 1029 | | decoder.iChar = (int)iChar; |
| | 0 | 1030 | | decoder.readByteCount = readCount; |
| | 0 | 1031 | | decoder._bytesUsed = (int)(bytes - byteStart); |
| | | 1032 | | } |
| | | 1033 | | |
| | | 1034 | | // Shouldn't have anything in fallback buffer for GetChars |
| | | 1035 | | // (don't have to check _throwOnOverflow for chars) |
| | 0 | 1036 | | Debug.Assert(fallbackBuffer.Remaining == 0, |
| | 0 | 1037 | | "[UTF32Encoding.GetChars]Expected empty fallback buffer at end"); |
| | | 1038 | | |
| | | 1039 | | // Return our count |
| | 0 | 1040 | | return (int)(chars - charStart); |
| | | 1041 | | } |
| | | 1042 | | |
| | | 1043 | | private static uint GetSurrogate(char cHigh, char cLow) |
| | | 1044 | | { |
| | 0 | 1045 | | return (((uint)cHigh - 0xD800) * 0x400) + ((uint)cLow - 0xDC00) + 0x10000; |
| | | 1046 | | } |
| | | 1047 | | |
| | | 1048 | | private static char GetHighSurrogate(uint iChar) |
| | | 1049 | | { |
| | 0 | 1050 | | return (char)((iChar - 0x10000) / 0x400 + 0xD800); |
| | | 1051 | | } |
| | | 1052 | | |
| | | 1053 | | private static char GetLowSurrogate(uint iChar) |
| | | 1054 | | { |
| | 0 | 1055 | | return (char)((iChar - 0x10000) % 0x400 + 0xDC00); |
| | | 1056 | | } |
| | | 1057 | | |
| | | 1058 | | public override Decoder GetDecoder() |
| | | 1059 | | { |
| | 0 | 1060 | | return new UTF32Decoder(this); |
| | | 1061 | | } |
| | | 1062 | | |
| | | 1063 | | public override Encoder GetEncoder() |
| | | 1064 | | { |
| | 0 | 1065 | | return new EncoderNLS(this); |
| | | 1066 | | } |
| | | 1067 | | |
| | | 1068 | | public override int GetMaxByteCount(int charCount) |
| | | 1069 | | { |
| | 0 | 1070 | | ArgumentOutOfRangeException.ThrowIfNegative(charCount); |
| | | 1071 | | |
| | | 1072 | | // Characters would be # of characters + 1 in case left over high surrogate is ? * max fallback |
| | 0 | 1073 | | long byteCount = (long)charCount + 1; |
| | | 1074 | | |
| | 0 | 1075 | | if (EncoderFallback.MaxCharCount > 1) |
| | 0 | 1076 | | byteCount *= EncoderFallback.MaxCharCount; |
| | | 1077 | | |
| | | 1078 | | // 4 bytes per char |
| | 0 | 1079 | | byteCount *= 4; |
| | | 1080 | | |
| | 0 | 1081 | | if (byteCount > 0x7fffffff) |
| | 0 | 1082 | | throw new ArgumentOutOfRangeException(nameof(charCount), SR.ArgumentOutOfRange_GetByteCountOverflow); |
| | | 1083 | | |
| | 0 | 1084 | | return (int)byteCount; |
| | | 1085 | | } |
| | | 1086 | | |
| | | 1087 | | public override int GetMaxCharCount(int byteCount) |
| | | 1088 | | { |
| | 0 | 1089 | | ArgumentOutOfRangeException.ThrowIfNegative(byteCount); |
| | | 1090 | | |
| | | 1091 | | // A supplementary character becomes 2 surrogate characters, so 4 input bytes becomes 2 chars, |
| | | 1092 | | // plus we may have 1 surrogate char left over if the decoder has 3 bytes in it already for a non-bmp char. |
| | | 1093 | | // Have to add another one because 1/2 == 0, but 3 bytes left over could be 2 char surrogate pair |
| | 0 | 1094 | | int charCount = (byteCount / 2) + 2; |
| | | 1095 | | |
| | | 1096 | | // Also consider fallback because our input bytes could be out of range of unicode. |
| | | 1097 | | // Since fallback would fallback 4 bytes at a time, we'll only fall back 1/2 of MaxCharCount. |
| | 0 | 1098 | | if (DecoderFallback.MaxCharCount > 2) |
| | | 1099 | | { |
| | | 1100 | | // Multiply time fallback size |
| | 0 | 1101 | | charCount *= DecoderFallback.MaxCharCount; |
| | | 1102 | | |
| | | 1103 | | // We were already figuring 2 chars per 4 bytes, but fallback will be different # |
| | 0 | 1104 | | charCount /= 2; |
| | | 1105 | | } |
| | | 1106 | | |
| | 0 | 1107 | | if (charCount > 0x7fffffff) |
| | 0 | 1108 | | throw new ArgumentOutOfRangeException(nameof(byteCount), SR.ArgumentOutOfRange_GetCharCountOverflow); |
| | | 1109 | | |
| | 0 | 1110 | | return (int)charCount; |
| | | 1111 | | } |
| | | 1112 | | |
| | | 1113 | | public override byte[] GetPreamble() |
| | | 1114 | | { |
| | 0 | 1115 | | if (_emitUTF32ByteOrderMark) |
| | | 1116 | | { |
| | | 1117 | | // Allocate new array to prevent users from modifying it. |
| | 0 | 1118 | | if (_bigEndian) |
| | | 1119 | | { |
| | 0 | 1120 | | return [0x00, 0x00, 0xFE, 0xFF]; |
| | | 1121 | | } |
| | | 1122 | | else |
| | | 1123 | | { |
| | 0 | 1124 | | return [0xFF, 0xFE, 0x00, 0x00]; // 00 00 FE FF |
| | | 1125 | | } |
| | | 1126 | | } |
| | | 1127 | | else |
| | 0 | 1128 | | return []; |
| | | 1129 | | } |
| | | 1130 | | |
| | | 1131 | | public override ReadOnlySpan<byte> Preamble => |
| | 0 | 1132 | | GetType() != typeof(UTF32Encoding) ? new ReadOnlySpan<byte>(GetPreamble()) : // in case a derived UTF32Encod |
| | 0 | 1133 | | !_emitUTF32ByteOrderMark ? default : |
| | 0 | 1134 | | _bigEndian ? [0x00, 0x00, 0xFE, 0xFF] : |
| | 0 | 1135 | | [0xFF, 0xFE, 0x00, 0x00]; |
| | | 1136 | | |
| | | 1137 | | public override bool Equals([NotNullWhen(true)] object? value) |
| | | 1138 | | { |
| | 0 | 1139 | | if (value is UTF32Encoding that) |
| | | 1140 | | { |
| | 0 | 1141 | | return (_emitUTF32ByteOrderMark == that._emitUTF32ByteOrderMark) && |
| | 0 | 1142 | | (_bigEndian == that._bigEndian) && |
| | 0 | 1143 | | (EncoderFallback.Equals(that.EncoderFallback)) && |
| | 0 | 1144 | | (DecoderFallback.Equals(that.DecoderFallback)); |
| | | 1145 | | } |
| | | 1146 | | |
| | 0 | 1147 | | return false; |
| | | 1148 | | } |
| | | 1149 | | |
| | | 1150 | | public override int GetHashCode() |
| | | 1151 | | { |
| | | 1152 | | // Not great distribution, but this is relatively unlikely to be used as the key in a hashtable. |
| | 0 | 1153 | | return this.EncoderFallback.GetHashCode() + this.DecoderFallback.GetHashCode() + |
| | 0 | 1154 | | CodePage + (_emitUTF32ByteOrderMark ? 4 : 0) + (_bigEndian ? 8 : 0); |
| | | 1155 | | } |
| | | 1156 | | |
| | | 1157 | | private sealed class UTF32Decoder : DecoderNLS |
| | | 1158 | | { |
| | | 1159 | | // Need a place to store any extra bytes we may have picked up |
| | | 1160 | | internal int iChar; |
| | | 1161 | | internal int readByteCount; |
| | | 1162 | | |
| | 0 | 1163 | | public UTF32Decoder(UTF32Encoding encoding) : base(encoding) |
| | | 1164 | | { |
| | | 1165 | | // base calls reset |
| | 0 | 1166 | | } |
| | | 1167 | | |
| | | 1168 | | public override void Reset() |
| | | 1169 | | { |
| | 0 | 1170 | | this.iChar = 0; |
| | 0 | 1171 | | this.readByteCount = 0; |
| | 0 | 1172 | | _fallbackBuffer?.Reset(); |
| | 0 | 1173 | | } |
| | | 1174 | | |
| | | 1175 | | // Anything left in our decoder? |
| | | 1176 | | internal override bool HasState => |
| | | 1177 | | // ReadByteCount is our flag. (iChar==0 doesn't mean much). |
| | 0 | 1178 | | this.readByteCount != 0; |
| | | 1179 | | } |
| | | 1180 | | } |
| | | 1181 | | } |
| | | 1182 | | |