| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | using System.Buffers; |
| | | 5 | | using System.Diagnostics; |
| | | 6 | | using System.Diagnostics.CodeAnalysis; |
| | | 7 | | using System.Runtime.InteropServices; |
| | | 8 | | |
| | | 9 | | namespace System.Text |
| | | 10 | | { |
| | | 11 | | // An Encoder is used to encode a sequence of blocks of characters into |
| | | 12 | | // a sequence of blocks of bytes. Following instantiation of an encoder, |
| | | 13 | | // sequential blocks of characters are converted into blocks of bytes through |
| | | 14 | | // calls to the GetBytes method. The encoder maintains state between the |
| | | 15 | | // conversions, allowing it to correctly encode character sequences that span |
| | | 16 | | // adjacent blocks. |
| | | 17 | | // |
| | | 18 | | // Instances of specific implementations of the Encoder abstract base |
| | | 19 | | // class are typically obtained through calls to the GetEncoder method |
| | | 20 | | // of Encoding objects. |
| | | 21 | | // |
| | | 22 | | |
| | | 23 | | internal class EncoderNLS : Encoder |
| | | 24 | | { |
| | | 25 | | // Need a place for the last left over character, most of our encodings use this |
| | | 26 | | internal char _charLeftOver; |
| | | 27 | | private readonly Encoding _encoding; |
| | | 28 | | private bool _mustFlush; |
| | | 29 | | internal bool _throwOnOverflow; |
| | | 30 | | internal int _charsUsed; |
| | | 31 | | |
| | 21915 | 32 | | internal EncoderNLS(Encoding encoding) |
| | | 33 | | { |
| | 21915 | 34 | | _encoding = encoding; |
| | 21915 | 35 | | _fallback = _encoding.EncoderFallback; |
| | 21915 | 36 | | this.Reset(); |
| | 21915 | 37 | | } |
| | | 38 | | |
| | | 39 | | public override void Reset() |
| | | 40 | | { |
| | 18512 | 41 | | _charLeftOver = (char)0; |
| | 18512 | 42 | | _fallbackBuffer?.Reset(); |
| | 0 | 43 | | } |
| | | 44 | | |
| | | 45 | | public override unsafe int GetByteCount(char[] chars, int index, int count, bool flush) |
| | | 46 | | { |
| | 0 | 47 | | ArgumentNullException.ThrowIfNull(chars); |
| | | 48 | | |
| | 0 | 49 | | ArgumentOutOfRangeException.ThrowIfNegative(index); |
| | 0 | 50 | | ArgumentOutOfRangeException.ThrowIfNegative(count); |
| | | 51 | | |
| | 0 | 52 | | if (chars.Length - index < count) |
| | 0 | 53 | | throw new ArgumentOutOfRangeException(nameof(chars), |
| | 0 | 54 | | SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 55 | | |
| | | 56 | | // Just call the pointer version |
| | 0 | 57 | | int result = -1; |
| | 0 | 58 | | fixed (char* pChars = &MemoryMarshal.GetArrayDataReference(chars)) |
| | | 59 | | { |
| | 0 | 60 | | result = GetByteCount(pChars + index, count, flush); |
| | | 61 | | } |
| | 0 | 62 | | return result; |
| | | 63 | | } |
| | | 64 | | |
| | | 65 | | public override unsafe int GetByteCount(char* chars, int count, bool flush) |
| | | 66 | | { |
| | 0 | 67 | | ArgumentNullException.ThrowIfNull(chars); |
| | | 68 | | |
| | 0 | 69 | | ArgumentOutOfRangeException.ThrowIfNegative(count); |
| | | 70 | | |
| | 0 | 71 | | _mustFlush = flush; |
| | 0 | 72 | | _throwOnOverflow = true; |
| | 0 | 73 | | Debug.Assert(_encoding is not null); |
| | 0 | 74 | | return _encoding.GetByteCount(chars, count, this); |
| | | 75 | | } |
| | | 76 | | |
| | | 77 | | public override unsafe int GetBytes(char[] chars, int charIndex, int charCount, |
| | | 78 | | byte[] bytes, int byteIndex, bool flush) |
| | | 79 | | { |
| | 0 | 80 | | ArgumentNullException.ThrowIfNull(chars); |
| | 0 | 81 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 82 | | |
| | 0 | 83 | | ArgumentOutOfRangeException.ThrowIfNegative(charIndex); |
| | 0 | 84 | | ArgumentOutOfRangeException.ThrowIfNegative(charCount); |
| | | 85 | | |
| | 0 | 86 | | if (chars.Length - charIndex < charCount) |
| | 0 | 87 | | throw new ArgumentOutOfRangeException(nameof(chars), |
| | 0 | 88 | | SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 89 | | |
| | 0 | 90 | | if (byteIndex < 0 || byteIndex > bytes.Length) |
| | 0 | 91 | | throw new ArgumentOutOfRangeException(nameof(byteIndex), |
| | 0 | 92 | | SR.ArgumentOutOfRange_IndexMustBeLessOrEqual); |
| | | 93 | | |
| | 0 | 94 | | int byteCount = bytes.Length - byteIndex; |
| | | 95 | | |
| | | 96 | | // Just call pointer version |
| | 0 | 97 | | fixed (char* pChars = &MemoryMarshal.GetArrayDataReference(chars)) |
| | 0 | 98 | | fixed (byte* pBytes = &MemoryMarshal.GetArrayDataReference(bytes)) |
| | | 99 | | { |
| | | 100 | | // Remember that charCount is # to decode, not size of array. |
| | 0 | 101 | | return GetBytes(pChars + charIndex, charCount, |
| | 0 | 102 | | pBytes + byteIndex, byteCount, flush); |
| | | 103 | | } |
| | | 104 | | } |
| | | 105 | | |
| | | 106 | | public override unsafe int GetBytes(char* chars, int charCount, byte* bytes, int byteCount, bool flush) |
| | | 107 | | { |
| | 21917 | 108 | | ArgumentNullException.ThrowIfNull(chars); |
| | 21917 | 109 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 110 | | |
| | 21917 | 111 | | ArgumentOutOfRangeException.ThrowIfNegative(byteCount); |
| | 21917 | 112 | | ArgumentOutOfRangeException.ThrowIfNegative(charCount); |
| | | 113 | | |
| | 21917 | 114 | | _mustFlush = flush; |
| | 21917 | 115 | | _throwOnOverflow = true; |
| | 21917 | 116 | | Debug.Assert(_encoding is not null); |
| | 21917 | 117 | | return _encoding.GetBytes(chars, charCount, bytes, byteCount, this); |
| | | 118 | | } |
| | | 119 | | |
| | | 120 | | // This method is used when your output buffer might not be large enough for the entire result. |
| | | 121 | | // Just call the pointer version. (This gets bytes) |
| | | 122 | | public override unsafe void Convert(char[] chars, int charIndex, int charCount, |
| | | 123 | | byte[] bytes, int byteIndex, int byteCount, bool flush, |
| | | 124 | | out int charsUsed, out int bytesUsed, out bool completed) |
| | | 125 | | { |
| | 0 | 126 | | ArgumentNullException.ThrowIfNull(chars); |
| | 0 | 127 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 128 | | |
| | 0 | 129 | | ArgumentOutOfRangeException.ThrowIfNegative(charIndex); |
| | 0 | 130 | | ArgumentOutOfRangeException.ThrowIfNegative(charCount); |
| | | 131 | | |
| | 0 | 132 | | ArgumentOutOfRangeException.ThrowIfNegative(byteIndex); |
| | 0 | 133 | | ArgumentOutOfRangeException.ThrowIfNegative(byteCount); |
| | | 134 | | |
| | 0 | 135 | | if (chars.Length - charIndex < charCount) |
| | 0 | 136 | | throw new ArgumentOutOfRangeException(nameof(chars), |
| | 0 | 137 | | SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 138 | | |
| | 0 | 139 | | if (bytes.Length - byteIndex < byteCount) |
| | 0 | 140 | | throw new ArgumentOutOfRangeException(nameof(bytes), |
| | 0 | 141 | | SR.ArgumentOutOfRange_IndexCountBuffer); |
| | | 142 | | |
| | | 143 | | // Just call the pointer version (can't do this for non-msft encoders) |
| | 0 | 144 | | fixed (char* pChars = &MemoryMarshal.GetArrayDataReference(chars)) |
| | 0 | 145 | | fixed (byte* pBytes = &MemoryMarshal.GetArrayDataReference(bytes)) |
| | | 146 | | { |
| | 0 | 147 | | Convert(pChars + charIndex, charCount, pBytes + byteIndex, byteCount, flush, |
| | 0 | 148 | | out charsUsed, out bytesUsed, out completed); |
| | | 149 | | } |
| | 0 | 150 | | } |
| | | 151 | | |
| | | 152 | | // This is the version that uses pointers. We call the base encoding worker function |
| | | 153 | | // after setting our appropriate internal variables. This is getting bytes |
| | | 154 | | public override unsafe void Convert(char* chars, int charCount, |
| | | 155 | | byte* bytes, int byteCount, bool flush, |
| | | 156 | | out int charsUsed, out int bytesUsed, out bool completed) |
| | | 157 | | { |
| | 0 | 158 | | ArgumentNullException.ThrowIfNull(chars); |
| | 0 | 159 | | ArgumentNullException.ThrowIfNull(bytes); |
| | | 160 | | |
| | 0 | 161 | | ArgumentOutOfRangeException.ThrowIfNegative(charCount); |
| | 0 | 162 | | ArgumentOutOfRangeException.ThrowIfNegative(byteCount); |
| | | 163 | | |
| | | 164 | | // We don't want to throw |
| | 0 | 165 | | _mustFlush = flush; |
| | 0 | 166 | | _throwOnOverflow = false; |
| | 0 | 167 | | _charsUsed = 0; |
| | | 168 | | |
| | | 169 | | // Do conversion |
| | 0 | 170 | | Debug.Assert(_encoding is not null); |
| | 0 | 171 | | bytesUsed = _encoding.GetBytes(chars, charCount, bytes, byteCount, this); |
| | 0 | 172 | | charsUsed = _charsUsed; |
| | | 173 | | |
| | | 174 | | // If the 'completed' out parameter is set to false, it means one of two things: |
| | | 175 | | // a) this call to Convert did not consume the entire source buffer; or |
| | | 176 | | // b) this call to Convert did consume the entire source buffer, but there's |
| | | 177 | | // still pending data that needs to be written to the destination buffer. |
| | | 178 | | // |
| | | 179 | | // In either case, the caller should slice the input buffer, provide a fresh |
| | | 180 | | // destination buffer, and call Convert again in a loop until 'completed' is true. |
| | | 181 | | // |
| | | 182 | | // The caller *must* specify flush = true on the final iteration(s) of the loop |
| | | 183 | | // and iterate until 'completed' is set to true. Otherwise data loss may occur. |
| | | 184 | | // |
| | | 185 | | // Technically, the expected logic is detailed below. |
| | | 186 | | // |
| | | 187 | | // If 'flush' = false, the 'completed' parameter MUST be set to false if not all |
| | | 188 | | // elements of the source buffer have been consumed. The 'completed' parameter MUST |
| | | 189 | | // be set to true once the entire source buffer has been consumed and there is no |
| | | 190 | | // pending data for the destination buffer. (In other words, the 'completed' parameter |
| | | 191 | | // MUST be set to true if passing a zero-length source buffer and an infinite-length |
| | | 192 | | // destination buffer will make no forward progress.) The 'completed' parameter value |
| | | 193 | | // is undefined for the case where all source data has been consumed but there remains |
| | | 194 | | // pending data for the destination buffer. |
| | | 195 | | // |
| | | 196 | | // If 'flush' = true, the 'completed' parameter is set to true IF AND ONLY IF: |
| | | 197 | | // a) all elements of the source buffer have been transcoded into the destination buffer; AND |
| | | 198 | | // b) there remains no internal partial read state within this instance; AND |
| | | 199 | | // c) there remains no pending data for the destination buffer. |
| | | 200 | | // |
| | | 201 | | // In other words, if 'flush' = true, then when 'completed' is set to true it should mean |
| | | 202 | | // that all data has been converted and that this instance is indistinguishable from a |
| | | 203 | | // freshly-reset instance. |
| | | 204 | | |
| | 0 | 205 | | completed = (charsUsed == charCount) |
| | 0 | 206 | | && (!flush || !this.HasState) |
| | 0 | 207 | | && (_fallbackBuffer is null || _fallbackBuffer.Remaining == 0); |
| | 0 | 208 | | } |
| | | 209 | | |
| | | 210 | | public Encoding Encoding |
| | | 211 | | { |
| | | 212 | | get |
| | | 213 | | { |
| | 0 | 214 | | Debug.Assert(_encoding is not null); |
| | 0 | 215 | | return _encoding; |
| | | 216 | | } |
| | | 217 | | } |
| | | 218 | | |
| | 10846 | 219 | | public bool MustFlush => _mustFlush; |
| | | 220 | | |
| | | 221 | | /// <summary> |
| | | 222 | | /// States whether a call to <see cref="Encoding.GetBytes(char*, int, byte*, int, EncoderNLS)"/> must first drai |
| | | 223 | | /// </summary> |
| | 10556 | 224 | | internal bool HasLeftoverData => _charLeftOver != default || (_fallbackBuffer is not null && _fallbackBuffer.Rem |
| | | 225 | | |
| | | 226 | | // Anything left in our encoder? |
| | 0 | 227 | | internal virtual bool HasState => _charLeftOver != (char)0; |
| | | 228 | | |
| | | 229 | | // Allow encoding to clear our must flush instead of throwing (in ThrowBytesOverflow) |
| | | 230 | | internal void ClearMustFlush() |
| | | 231 | | { |
| | 0 | 232 | | _mustFlush = false; |
| | 0 | 233 | | } |
| | | 234 | | |
| | | 235 | | internal int DrainLeftoverDataForGetByteCount(ReadOnlySpan<char> chars, out int charsConsumed) |
| | | 236 | | { |
| | | 237 | | // Quick check: we _should not_ have leftover fallback data from a previous invocation, |
| | | 238 | | // as we'd end up consuming any such data and would corrupt whatever Convert call happens |
| | | 239 | | // to be in progress. |
| | | 240 | | |
| | 0 | 241 | | if (_fallbackBuffer is not null && _fallbackBuffer.Remaining > 0) |
| | | 242 | | { |
| | 0 | 243 | | throw new ArgumentException(SR.Format(SR.Argument_EncoderFallbackNotEmpty, Encoding.EncodingName, _fallb |
| | | 244 | | } |
| | | 245 | | |
| | | 246 | | // If we have a leftover high surrogate from a previous operation, consume it now. |
| | | 247 | | // We won't clear the _charLeftOver field since GetByteCount is supposed to be |
| | | 248 | | // a non-mutating operation, and we need the field to retain its value for the |
| | | 249 | | // next call to Convert. |
| | | 250 | | |
| | 0 | 251 | | charsConsumed = 0; // could be incorrect, will fix up later in the method |
| | | 252 | | |
| | 0 | 253 | | if (_charLeftOver == default) |
| | | 254 | | { |
| | 0 | 255 | | return 0; // no leftover high surrogate char - short-circuit and finish |
| | | 256 | | } |
| | | 257 | | else |
| | | 258 | | { |
| | 0 | 259 | | char secondChar = default; |
| | | 260 | | |
| | 0 | 261 | | if (chars.IsEmpty) |
| | | 262 | | { |
| | | 263 | | // If the input buffer is empty and we're not being asked to flush, no-op and return |
| | | 264 | | // success to our caller. If we're being asked to flush, the leftover high surrogate from |
| | | 265 | | // the previous operation will go through the fallback mechanism by itself. |
| | | 266 | | |
| | 0 | 267 | | if (!MustFlush) |
| | | 268 | | { |
| | 0 | 269 | | return 0; // no-op = success |
| | | 270 | | } |
| | | 271 | | } |
| | | 272 | | else |
| | | 273 | | { |
| | 0 | 274 | | secondChar = chars[0]; |
| | | 275 | | } |
| | | 276 | | |
| | | 277 | | // If we have to fallback the chars we're reading immediately below, populate the |
| | | 278 | | // fallback buffer with the invalid data. We'll just fall through to the "consume |
| | | 279 | | // fallback buffer" logic at the end of the method. |
| | | 280 | | |
| | 0 | 281 | | if (Rune.TryCreate(_charLeftOver, secondChar, out Rune rune)) |
| | | 282 | | { |
| | 0 | 283 | | charsConsumed = 1; // consumed the leftover high surrogate + the first char in the input buffer |
| | | 284 | | |
| | 0 | 285 | | Debug.Assert(_encoding is not null); |
| | 0 | 286 | | if (_encoding.TryGetByteCount(rune, out int byteCount)) |
| | | 287 | | { |
| | 0 | 288 | | Debug.Assert(byteCount >= 0, "Encoding shouldn't have returned a negative byte count."); |
| | 0 | 289 | | return byteCount; |
| | | 290 | | } |
| | | 291 | | else |
| | | 292 | | { |
| | | 293 | | // The fallback mechanism relies on a negative index to convey "the start of the invalid |
| | | 294 | | // sequence was some number of chars back before the current buffer." In this block and |
| | | 295 | | // in the block immediately thereafter, we know we have a single leftover high surrogate |
| | | 296 | | // character from a previous operation, so we provide an index of -1 to convey that the |
| | | 297 | | // char immediately before the current buffer was the start of the invalid sequence. |
| | | 298 | | |
| | 0 | 299 | | FallbackBuffer.Fallback(_charLeftOver, secondChar, index: -1); |
| | | 300 | | } |
| | | 301 | | } |
| | | 302 | | else |
| | | 303 | | { |
| | 0 | 304 | | FallbackBuffer.Fallback(_charLeftOver, index: -1); |
| | | 305 | | } |
| | | 306 | | |
| | | 307 | | // Now tally the number of bytes that would've been emitted as part of fallback. |
| | 0 | 308 | | Debug.Assert(_fallbackBuffer is not null); |
| | 0 | 309 | | return _fallbackBuffer.DrainRemainingDataForGetByteCount(); |
| | | 310 | | } |
| | | 311 | | } |
| | | 312 | | |
| | | 313 | | internal bool TryDrainLeftoverDataForGetBytes(ReadOnlySpan<char> chars, Span<byte> bytes, out int charsConsumed, |
| | | 314 | | { |
| | | 315 | | // We may have a leftover high surrogate data from a previous invocation, or we may have leftover |
| | | 316 | | // data in the fallback buffer, or we may have neither, but we will never have both. Check for these |
| | | 317 | | // conditions and handle them now. |
| | | 318 | | |
| | 0 | 319 | | charsConsumed = 0; // could be incorrect, will fix up later in the method |
| | 0 | 320 | | bytesWritten = 0; // could be incorrect, will fix up later in the method |
| | | 321 | | |
| | 0 | 322 | | if (_charLeftOver != default) |
| | | 323 | | { |
| | 0 | 324 | | char secondChar = default; |
| | | 325 | | |
| | 0 | 326 | | if (chars.IsEmpty) |
| | | 327 | | { |
| | | 328 | | // If the input buffer is empty and we're not being asked to flush, no-op and return |
| | | 329 | | // success to our caller. If we're being asked to flush, the leftover high surrogate from |
| | | 330 | | // the previous operation will go through the fallback mechanism by itself. |
| | | 331 | | |
| | 0 | 332 | | if (!MustFlush) |
| | | 333 | | { |
| | 0 | 334 | | charsConsumed = 0; |
| | 0 | 335 | | bytesWritten = 0; |
| | 0 | 336 | | return true; // no-op = success |
| | | 337 | | } |
| | | 338 | | } |
| | | 339 | | else |
| | | 340 | | { |
| | 0 | 341 | | secondChar = chars[0]; |
| | | 342 | | } |
| | | 343 | | |
| | | 344 | | // We're about to consume the leftover char. Make a local copy of it and clear |
| | | 345 | | // the backing field. We don't bother restoring its value if an exception occurs |
| | | 346 | | // because exceptional code paths corrupt instance state anyway (e.g., by |
| | | 347 | | // mutating the fallback buffer contents). |
| | | 348 | | |
| | 0 | 349 | | char charLeftOver = _charLeftOver; |
| | 0 | 350 | | _charLeftOver = default; |
| | | 351 | | |
| | | 352 | | // If we have to fallback the chars we're reading immediately below, populate the |
| | | 353 | | // fallback buffer with the invalid data. We'll just fall through to the "consume |
| | | 354 | | // fallback buffer" logic at the end of the method. |
| | | 355 | | |
| | 0 | 356 | | if (Rune.TryCreate(charLeftOver, secondChar, out Rune rune)) |
| | | 357 | | { |
| | 0 | 358 | | charsConsumed = 1; // at the very least, we consumed 1 char from the input |
| | 0 | 359 | | Debug.Assert(_encoding is not null); |
| | 0 | 360 | | switch (_encoding.EncodeRune(rune, bytes, out bytesWritten)) |
| | | 361 | | { |
| | | 362 | | case OperationStatus.Done: |
| | 0 | 363 | | return true; // that's all - we've handled the leftover data |
| | | 364 | | |
| | | 365 | | case OperationStatus.DestinationTooSmall: |
| | 0 | 366 | | _encoding.ThrowBytesOverflow(this, nothingEncoded: true); // will throw |
| | 0 | 367 | | break; |
| | | 368 | | |
| | | 369 | | case OperationStatus.InvalidData: |
| | 0 | 370 | | FallbackBuffer.Fallback(charLeftOver, secondChar, index: -1); // see comment in DrainLeftove |
| | 0 | 371 | | break; |
| | | 372 | | |
| | | 373 | | default: |
| | 0 | 374 | | Debug.Fail("Unknown return value."); |
| | | 375 | | break; |
| | | 376 | | } |
| | | 377 | | } |
| | | 378 | | else |
| | | 379 | | { |
| | 0 | 380 | | FallbackBuffer.Fallback(charLeftOver, index: -1); // see comment in DrainLeftoverDataForGetByteCount |
| | | 381 | | } |
| | | 382 | | } |
| | | 383 | | |
| | | 384 | | // Now check the fallback buffer for any remaining data. |
| | | 385 | | |
| | 0 | 386 | | if (_fallbackBuffer is not null && _fallbackBuffer.Remaining > 0) |
| | | 387 | | { |
| | 0 | 388 | | return _fallbackBuffer.TryDrainRemainingDataForGetBytes(bytes, out bytesWritten); |
| | | 389 | | } |
| | | 390 | | |
| | | 391 | | // And we're done! |
| | | 392 | | |
| | 0 | 393 | | return true; // success |
| | | 394 | | } |
| | | 395 | | } |
| | | 396 | | } |
| | | 397 | | |