< Summary

Line coverage
0%
Covered lines: 0
Uncovered lines: 355
Coverable lines: 355
Total lines: 1012
Line coverage: 0%
Branch coverage
0%
Covered branches: 0
Total branches: 270
Branch coverage: 0%
Method coverage

Feature is only available for sponsors

Upgrade to PRO version

Metrics

File(s)

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.CoreLib/src/System/Globalization/Ordinal.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Buffers;
 5using System.Diagnostics;
 6using System.Numerics;
 7using System.Runtime.CompilerServices;
 8using System.Runtime.InteropServices;
 9using System.Runtime.Intrinsics;
 10using System.Runtime.Intrinsics.X86;
 11using System.Text;
 12using System.Text.Unicode;
 13
 14namespace System.Globalization
 15{
 16    internal static partial class Ordinal
 17    {
 18        internal static int CompareStringIgnoreCase(ref char strA, int lengthA, ref char strB, int lengthB)
 19        {
 020            int length = Math.Min(lengthA, lengthB);
 021            int range = length;
 22
 023            ref char charA = ref strA;
 024            ref char charB = ref strB;
 25
 26            const char maxChar = (char)0x7F;
 27
 028            while (length != 0 && charA <= maxChar && charB <= maxChar)
 29            {
 30                // Ordinal equals or lowercase equals if the result ends up in the a-z range
 031                if (charA == charB ||
 032                    ((charA | 0x20) == (charB | 0x20) && char.IsAsciiLetter(charA)))
 33                {
 034                    length--;
 035                    charA = ref Unsafe.Add(ref charA, 1);
 036                    charB = ref Unsafe.Add(ref charB, 1);
 37                }
 38                else
 39                {
 040                    int currentA = charA;
 041                    int currentB = charB;
 42
 43                    // Uppercase both chars if needed
 044                    if (char.IsAsciiLetterLower(charA))
 45                    {
 046                        currentA -= 0x20;
 47                    }
 048                    if (char.IsAsciiLetterLower(charB))
 49                    {
 050                        currentB -= 0x20;
 51                    }
 52
 53                    // Return the (case-insensitive) difference between them.
 054                    return currentA - currentB;
 55                }
 56            }
 57
 058            if (length == 0)
 59            {
 060                return lengthA - lengthB;
 61            }
 62
 063            range -= length;
 64
 065            return CompareStringIgnoreCaseNonAscii(ref charA, lengthA - range, ref charB, lengthB - range);
 66        }
 67
 68        internal static int CompareStringIgnoreCaseNonAscii(ref char strA, int lengthA, ref char strB, int lengthB)
 69        {
 070            if (GlobalizationMode.Invariant)
 71            {
 072                return InvariantModeCasing.CompareStringIgnoreCase(ref strA, lengthA, ref strB, lengthB);
 73            }
 74
 075            if (GlobalizationMode.UseNls)
 76            {
 077                return CompareInfo.NlsCompareStringOrdinalIgnoreCase(ref strA, lengthA, ref strB, lengthB);
 78            }
 79
 080            return OrdinalCasing.CompareStringIgnoreCase(ref strA, lengthA, ref strB, lengthB);
 81        }
 82
 83        private static bool EqualsIgnoreCase_Vector<TVector>(ref char charA, ref char charB, int length)
 84            where TVector : struct, ISimdVector<TVector, ushort>
 85        {
 086            Debug.Assert(length >= TVector.ElementCount);
 87
 088            nuint lengthU = (nuint)length;
 089            nuint lengthToExamine = lengthU - (nuint)TVector.ElementCount;
 090            nuint i = 0;
 91            TVector vec1;
 92            TVector vec2;
 093            TVector loweringMask = TVector.Create(0x20);
 094            TVector vecA = TVector.Create('a');
 095            TVector vecZMinusA = TVector.Create('z' - 'a');
 96            do
 97            {
 098                vec1 = TVector.LoadUnsafe(ref Unsafe.As<char, ushort>(ref charA), i);
 099                vec2 = TVector.LoadUnsafe(ref Unsafe.As<char, ushort>(ref charB), i);
 100
 0101                if (!Utf16Utility.AllCharsInVectorAreAscii(vec1 | vec2))
 102                {
 103                    goto NON_ASCII;
 104                }
 105
 0106                TVector notEquals = ~TVector.Equals(vec1, vec2);
 0107                if (!notEquals.Equals(TVector.Zero))
 108                {
 109                    // not exact match
 110
 0111                    vec1 |= loweringMask;
 0112                    vec2 |= loweringMask;
 0113                    if (TVector.GreaterThanAny((vec1 - vecA) & notEquals, vecZMinusA) || !vec1.Equals(vec2))
 114                    {
 0115                        return false; // first input isn't in [A-Za-z], and not exact match of lowered
 116                    }
 117                }
 0118                i += (nuint)TVector.ElementCount;
 0119            } while (i <= lengthToExamine);
 120
 121            // Handle trailing elements
 0122            if (i != lengthU)
 123            {
 0124                i = lengthU - (nuint)TVector.ElementCount;
 0125                vec1 = TVector.LoadUnsafe(ref Unsafe.As<char, ushort>(ref charA), i);
 0126                vec2 = TVector.LoadUnsafe(ref Unsafe.As<char, ushort>(ref charB), i);
 127
 0128                if (!Utf16Utility.AllCharsInVectorAreAscii(vec1 | vec2))
 129                {
 130                    goto NON_ASCII;
 131                }
 132
 0133                TVector notEquals = ~TVector.Equals(vec1, vec2);
 0134                if (!notEquals.Equals(TVector.Zero))
 135                {
 136                    // not exact match
 137
 0138                    vec1 |= loweringMask;
 0139                    vec2 |= loweringMask;
 0140                    if (TVector.GreaterThanAny((vec1 - vecA) & notEquals, vecZMinusA) || !vec1.Equals(vec2))
 141                    {
 0142                        return false; // first input isn't in [A-Za-z], and not exact match of lowered
 143                    }
 144                }
 145            }
 0146            return true;
 147
 148        NON_ASCII:
 0149            if (Utf16Utility.AllCharsInVectorAreAscii(vec1) || Utf16Utility.AllCharsInVectorAreAscii(vec2))
 150            {
 151                // No need to use the fallback if one of the inputs is full-ASCII
 0152                return false;
 153            }
 154
 155            // Fallback for Non-ASCII inputs
 0156            return CompareStringIgnoreCase(
 0157                ref Unsafe.Add(ref charA, i), (int)(lengthU - i),
 0158                ref Unsafe.Add(ref charB, i), (int)(lengthU - i)) == 0;
 159        }
 160
 161        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 162        internal static bool EqualsIgnoreCase(ref char charA, ref char charB, int length)
 163        {
 0164            if (!Vector128.IsHardwareAccelerated || length < Vector128<ushort>.Count)
 165            {
 0166                return EqualsIgnoreCase_Scalar(ref charA, ref charB, length);
 167            }
 0168            if (Vector512.IsHardwareAccelerated && length >= Vector512<ushort>.Count)
 169            {
 0170                return EqualsIgnoreCase_Vector<Vector512<ushort>>(ref charA, ref charB, length);
 171            }
 0172            if (Vector256.IsHardwareAccelerated && length >= Vector256<ushort>.Count)
 173            {
 0174                return EqualsIgnoreCase_Vector<Vector256<ushort>>(ref charA, ref charB, length);
 175            }
 0176            return EqualsIgnoreCase_Vector<Vector128<ushort>>(ref charA, ref charB, length);
 177        }
 178
 179        internal static bool EqualsIgnoreCase_Scalar(ref char charA, ref char charB, int length)
 180        {
 0181            IntPtr byteOffset = IntPtr.Zero;
 182
 183#if TARGET_64BIT
 0184            ulong valueAu64 = 0;
 0185            ulong valueBu64 = 0;
 186            // Read 4 chars (64 bits) at a time from each string
 0187            while ((uint)length >= 4)
 188            {
 0189                valueAu64 = Unsafe.ReadUnaligned<ulong>(ref Unsafe.As<char, byte>(ref Unsafe.AddByteOffset(ref charA, by
 0190                valueBu64 = Unsafe.ReadUnaligned<ulong>(ref Unsafe.As<char, byte>(ref Unsafe.AddByteOffset(ref charB, by
 191
 192                // A 32-bit test - even with the bit-twiddling here - is more efficient than a 64-bit test.
 0193                ulong temp = valueAu64 | valueBu64;
 0194                if (!Utf16Utility.AllCharsInUInt32AreAscii((uint)temp | (uint)(temp >> 32)))
 195                {
 196                    goto NonAscii64; // one of the inputs contains non-ASCII data
 197                }
 198
 199                // Generally, the caller has likely performed a first-pass check that the input strings
 200                // are likely equal. Consider a dictionary which computes the hash code of its key before
 201                // performing a proper deep equality check of the string contents. We want to optimize for
 202                // the case where the equality check is likely to succeed, which means that we want to avoid
 203                // branching within this loop unless we're about to exit the loop, either due to failure or
 204                // due to us running out of input data.
 205
 0206                if (!Utf16Utility.UInt64OrdinalIgnoreCaseAscii(valueAu64, valueBu64))
 207                {
 0208                    return false;
 209                }
 210
 0211                byteOffset += 8;
 0212                length -= 4;
 213            }
 214#endif
 0215            uint valueAu32 = 0;
 0216            uint valueBu32 = 0;
 217            // Read 2 chars (32 bits) at a time from each string
 218#if TARGET_64BIT
 0219            if ((uint)length >= 2)
 220#else
 221            while ((uint)length >= 2)
 222#endif
 223            {
 0224                valueAu32 = Unsafe.ReadUnaligned<uint>(ref Unsafe.As<char, byte>(ref Unsafe.AddByteOffset(ref charA, byt
 0225                valueBu32 = Unsafe.ReadUnaligned<uint>(ref Unsafe.As<char, byte>(ref Unsafe.AddByteOffset(ref charB, byt
 226
 0227                if (!Utf16Utility.AllCharsInUInt32AreAscii(valueAu32 | valueBu32))
 228                {
 229                    goto NonAscii32; // one of the inputs contains non-ASCII data
 230                }
 231
 232                // Generally, the caller has likely performed a first-pass check that the input strings
 233                // are likely equal. Consider a dictionary which computes the hash code of its key before
 234                // performing a proper deep equality check of the string contents. We want to optimize for
 235                // the case where the equality check is likely to succeed, which means that we want to avoid
 236                // branching within this loop unless we're about to exit the loop, either due to failure or
 237                // due to us running out of input data.
 238
 0239                if (!Utf16Utility.UInt32OrdinalIgnoreCaseAscii(valueAu32, valueBu32))
 240                {
 0241                    return false;
 242                }
 243
 0244                byteOffset += 4;
 0245                length -= 2;
 246            }
 247
 0248            if (length != 0)
 249            {
 0250                Debug.Assert(length == 1);
 251
 0252                valueAu32 = Unsafe.AddByteOffset(ref charA, byteOffset);
 0253                valueBu32 = Unsafe.AddByteOffset(ref charB, byteOffset);
 254
 0255                if ((valueAu32 | valueBu32) > 0x7Fu)
 256                {
 257                    goto NonAscii32; // one of the inputs contains non-ASCII data
 258                }
 259
 0260                if (valueAu32 == valueBu32)
 261                {
 0262                    return true; // exact match
 263                }
 264
 0265                valueAu32 |= 0x20u;
 0266                if ((uint)(valueAu32 - 'a') > (uint)('z' - 'a'))
 267                {
 0268                    return false; // not exact match, and first input isn't in [A-Za-z]
 269                }
 270
 0271                return valueAu32 == (valueBu32 | 0x20u);
 272            }
 273
 0274            Debug.Assert(length == 0);
 0275            return true;
 276
 277        NonAscii32:
 278            // Both values have to be non-ASCII to use the slow fallback, in case if one of them is not we return false
 0279            if (Utf16Utility.AllCharsInUInt32AreAscii(valueAu32) || Utf16Utility.AllCharsInUInt32AreAscii(valueBu32))
 280            {
 0281                return false;
 282            }
 283            goto NonAscii;
 284
 285#if TARGET_64BIT
 286        NonAscii64:
 287            // Both values have to be non-ASCII to use the slow fallback, in case if one of them is not we return false
 0288            if (Utf16Utility.AllCharsInUInt64AreAscii(valueAu64) || Utf16Utility.AllCharsInUInt64AreAscii(valueBu64))
 289            {
 0290                return false;
 291            }
 292#endif
 293        NonAscii:
 294            // The non-ASCII case is factored out into its own helper method so that the JIT
 295            // doesn't need to emit a complex prolog for its caller (this method).
 0296            return CompareStringIgnoreCase(ref Unsafe.AddByteOffset(ref charA, byteOffset), length, ref Unsafe.AddByteOf
 297        }
 298
 299        internal static int IndexOf(string source, string value, int startIndex, int count, bool ignoreCase)
 300        {
 0301            if (source == null)
 302            {
 0303                ThrowHelper.ThrowArgumentNullException(ExceptionArgument.source);
 304            }
 305
 0306            if (value == null)
 307            {
 0308                ThrowHelper.ThrowArgumentNullException(ExceptionArgument.value);
 309            }
 310
 0311            if (!source.TryGetSpan(startIndex, count, out ReadOnlySpan<char> sourceSpan))
 312            {
 313                // Bounds check failed - figure out exactly what went wrong so that we can
 314                // surface the correct argument exception.
 315
 0316                if ((uint)startIndex > (uint)source.Length)
 317                {
 0318                    ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.startIndex, ExceptionResource.Argumen
 319                }
 320                else
 321                {
 0322                    ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.count, ExceptionResource.ArgumentOutO
 323                }
 324            }
 325
 0326            int result = ignoreCase ? IndexOfOrdinalIgnoreCase(sourceSpan, value) : sourceSpan.IndexOf(value);
 327
 0328            return result >= 0 ? result + startIndex : result;
 329        }
 330
 331        internal static int IndexOfOrdinalIgnoreCase(ReadOnlySpan<char> source, ReadOnlySpan<char> value)
 332        {
 0333            if (value.Length == 0)
 334            {
 0335                return 0;
 336            }
 337
 0338            if (value.Length > source.Length)
 339            {
 340                // A non-linguistic search compares chars directly against one another, so large
 341                // target strings can never be found inside small search spaces. This check also
 342                // handles empty 'source' spans.
 0343                return -1;
 344            }
 345
 0346            if (GlobalizationMode.Invariant)
 347            {
 0348                return InvariantModeCasing.IndexOfIgnoreCase(source, value);
 349            }
 350
 0351            if (GlobalizationMode.UseNls)
 352            {
 0353                return CompareInfo.NlsIndexOfOrdinalCore(source, value, ignoreCase: true, fromBeginning: true);
 354            }
 355
 356            // If value doesn't start with ASCII, fall back to a non-vectorized non-ASCII friendly version.
 0357            ref char valueRef = ref MemoryMarshal.GetReference(value);
 0358            char valueChar = valueRef;
 0359            if (!char.IsAscii(valueChar))
 360            {
 0361                return OrdinalCasing.IndexOf(source, value);
 362            }
 363
 364            // Hoist some expressions from the loop
 0365            int valueTailLength = value.Length - 1;
 0366            int searchSpaceMinusValueTailLength = source.Length - valueTailLength;
 0367            ref char searchSpace = ref MemoryMarshal.GetReference(source);
 0368            char valueCharU = default;
 0369            char valueCharL = default;
 0370            nint offset = 0;
 0371            bool isLetter = false;
 372
 373            // If the input is long enough and the value ends with ASCII and is at least two characters,
 374            // we can take a special vectorized path that compares both the beginning and the end at the same time.
 0375            if (Vector128.IsHardwareAccelerated &&
 0376                valueTailLength != 0 &&
 0377                searchSpaceMinusValueTailLength >= Vector128<ushort>.Count)
 378            {
 0379                valueCharU = Unsafe.Add(ref valueRef, valueTailLength);
 0380                if (char.IsAscii(valueCharU))
 381                {
 382                    goto SearchTwoChars;
 383                }
 384            }
 385
 386            // We're searching for the first character and it's known to be ASCII. If it's not a letter,
 387            // then IgnoreCase doesn't impact what it matches and we just need to do a normal search
 388            // for that single character. If it is a letter, then we need to search for both its upper
 389            // and lower-case variants.
 0390            if (char.IsAsciiLetter(valueChar))
 391            {
 0392                valueCharU = (char)(valueChar & ~0x20);
 0393                valueCharL = (char)(valueChar | 0x20);
 0394                isLetter = true;
 395            }
 396
 397            do
 398            {
 399                // Do a quick search for the first element of "value".
 0400                int relativeIndex = isLetter ?
 0401                    PackedSpanHelpers.PackedIndexOfIsSupported
 0402                        ? PackedSpanHelpers.IndexOfAnyIgnoreCase(ref Unsafe.Add(ref searchSpace, offset), valueCharL, se
 0403                        : SpanHelpers.IndexOfAnyChar(ref Unsafe.Add(ref searchSpace, offset), valueCharU, valueCharL, se
 0404                    SpanHelpers.IndexOfChar(ref Unsafe.Add(ref searchSpace, offset), valueChar, searchSpaceMinusValueTai
 0405                if (relativeIndex < 0)
 406                {
 407                    break;
 408                }
 409
 0410                searchSpaceMinusValueTailLength -= relativeIndex;
 0411                if (searchSpaceMinusValueTailLength <= 0)
 412                {
 413                    break;
 414                }
 0415                offset += relativeIndex;
 416
 417                // Found the first element of "value". See if the tail matches.
 0418                if (valueTailLength == 0 || // for single-char values we already matched first chars
 0419                    EqualsIgnoreCase(
 0420                        ref Unsafe.Add(ref searchSpace, (nuint)(offset + 1)),
 0421                        ref Unsafe.Add(ref valueRef, 1), valueTailLength))
 422                {
 0423                    return (int)offset;  // The tail matched. Return a successful find.
 424                }
 425
 0426                searchSpaceMinusValueTailLength--;
 0427                offset++;
 428            }
 0429            while (searchSpaceMinusValueTailLength > 0);
 430
 0431            return -1;
 432
 433        // Based on SpanHelpers.IndexOf(ref char, int, ref char, int), which was in turn based on
 434        // http://0x80.pl/articles/simd-strfind.html#algorithm-1-generic-simd. This version has additional
 435        // modifications to support case-insensitive searches.
 436        SearchTwoChars:
 437            // Both the first character in value (valueChar) and the last character in value (valueCharU) are ASCII. Get
 0438            valueChar = (char)(valueChar | 0x20);
 0439            valueCharU = (char)(valueCharU | 0x20);
 440
 441            // The search is more efficient if the two characters being searched for are different. As long as they are 
 442            // from the last character in the search value until we find a character that's different. Since we're deali
 443            // we compare the lowercase variants, as that's what we'll be comparing against in the main loop.
 0444            nint ch1ch2Distance = valueTailLength;
 0445            while (valueCharU == valueChar && ch1ch2Distance > 1)
 446            {
 0447                char tmp = Unsafe.Add(ref valueRef, ch1ch2Distance - 1);
 0448                if (!char.IsAscii(tmp))
 449                {
 450                    break;
 451                }
 0452                --ch1ch2Distance;
 0453                valueCharU = (char)(tmp | 0x20);
 454            }
 455
 456            // Use Vector256 if the input is long enough.
 0457            if (Vector256.IsHardwareAccelerated && searchSpaceMinusValueTailLength - Vector256<ushort>.Count >= 0)
 458            {
 459                // Create a vector for each of the lowercase ASCII characters we're searching for.
 0460                Vector256<ushort> ch1 = Vector256.Create((ushort)valueChar);
 0461                Vector256<ushort> ch2 = Vector256.Create((ushort)valueCharU);
 462
 0463                nint searchSpaceMinusValueTailLengthAndVector = searchSpaceMinusValueTailLength - (nint)Vector256<ushort
 464                do
 465                {
 466                    // Make sure we don't go out of bounds.
 0467                    Debug.Assert(offset + ch1ch2Distance + Vector256<ushort>.Count <= source.Length);
 468
 469                    // Load a vector from the current search space offset and another from the offset plus the distance 
 470                    // For each, | with 0x20 so that letters are lowercased, then & those together to get a mask. If the
 471                    // was no match.  If it wasn't, we have to do more work to check for a match.
 0472                    Vector256<ushort> cmpCh2 = Vector256.Equals(ch2, Vector256.BitwiseOr(Vector256.LoadUnsafe(ref search
 0473                    Vector256<ushort> cmpCh1 = Vector256.Equals(ch1, Vector256.BitwiseOr(Vector256.LoadUnsafe(ref search
 0474                    Vector256<byte> cmpAnd = (cmpCh1 & cmpCh2).AsByte();
 0475                    if (cmpAnd != Vector256<byte>.Zero)
 476                    {
 477                        goto CandidateFound;
 478                    }
 479
 480                LoopFooter:
 481                    // No match. Advance to the next vector.
 0482                    offset += Vector256<ushort>.Count;
 483
 484                    // If we've reached the end of the search space, bail.
 0485                    if (offset == searchSpaceMinusValueTailLength)
 486                    {
 0487                        return -1;
 488                    }
 489
 490                    // If we're within a vector's length of the end of the search space, adjust the offset
 491                    // to point to the last vector so that our next iteration will process it.
 0492                    if (offset > searchSpaceMinusValueTailLengthAndVector)
 493                    {
 0494                        offset = searchSpaceMinusValueTailLengthAndVector;
 495                    }
 496
 0497                    continue;
 498
 499                CandidateFound:
 500                    // Possible matches at the current location. Extract the bits for each element.
 501                    // For each set bits, we'll check if it's a match at that location.
 0502                    uint mask = cmpAnd.ExtractMostSignificantBits();
 503                    do
 504                    {
 505                        // Do a full IgnoreCase equality comparison. SpanHelpers.IndexOf skips comparing the two charact
 506                        // but we don't actually know that the two characters are equal, since we compared with | 0x20. 
 507                        // the full string always.
 0508                        nint charPos = (nint)(uint.TrailingZeroCount(mask) / sizeof(ushort));
 0509                        if (EqualsIgnoreCase(ref Unsafe.Add(ref searchSpace, offset + charPos), ref valueRef, value.Leng
 510                        {
 511                            // Match! Return the index.
 0512                            return (int)(offset + charPos);
 513                        }
 514
 515                        // Clear the two lowest set bits in the mask. If there are no more set bits, we're done.
 516                        // If any remain, we loop around to do the next comparison.
 0517                        mask = BitOperations.ResetLowestSetBit(BitOperations.ResetLowestSetBit(mask));
 0518                    } while (mask != 0);
 0519                    goto LoopFooter;
 520
 521                } while (true);
 522            }
 523            else // 128bit vector path (SSE2 or AdvSimd)
 524            {
 525                // Create a vector for each of the lowercase ASCII characters we're searching for.
 0526                Vector128<ushort> ch1 = Vector128.Create((ushort)valueChar);
 0527                Vector128<ushort> ch2 = Vector128.Create((ushort)valueCharU);
 528
 0529                nint searchSpaceMinusValueTailLengthAndVector = searchSpaceMinusValueTailLength - (nint)Vector128<ushort
 530                do
 531                {
 532                    // Make sure we don't go out of bounds.
 0533                    Debug.Assert(offset + ch1ch2Distance + Vector128<ushort>.Count <= source.Length);
 534
 535                    // Load a vector from the current search space offset and another from the offset plus the distance 
 536                    // For each, | with 0x20 so that letters are lowercased, then & those together to get a mask. If the
 537                    // was no match.  If it wasn't, we have to do more work to check for a match.
 0538                    Vector128<ushort> cmpCh2 = Vector128.Equals(ch2, Vector128.BitwiseOr(Vector128.LoadUnsafe(ref search
 0539                    Vector128<ushort> cmpCh1 = Vector128.Equals(ch1, Vector128.BitwiseOr(Vector128.LoadUnsafe(ref search
 0540                    Vector128<byte> cmpAnd = (cmpCh1 & cmpCh2).AsByte();
 0541                    if (cmpAnd != Vector128<byte>.Zero)
 542                    {
 543                        goto CandidateFound;
 544                    }
 545
 546                LoopFooter:
 547                    // No match. Advance to the next vector.
 0548                    offset += Vector128<ushort>.Count;
 549
 550                    // If we've reached the end of the search space, bail.
 0551                    if (offset == searchSpaceMinusValueTailLength)
 552                    {
 0553                        return -1;
 554                    }
 555
 556                    // If we're within a vector's length of the end of the search space, adjust the offset
 557                    // to point to the last vector so that our next iteration will process it.
 0558                    if (offset > searchSpaceMinusValueTailLengthAndVector)
 559                    {
 0560                        offset = searchSpaceMinusValueTailLengthAndVector;
 561                    }
 562
 0563                    continue;
 564
 565                CandidateFound:
 566                    // Possible matches at the current location. Extract the bits for each element.
 567                    // For each set bits, we'll check if it's a match at that location.
 0568                    uint mask = cmpAnd.ExtractMostSignificantBits();
 569                    do
 570                    {
 571                        // Do a full IgnoreCase equality comparison. SpanHelpers.IndexOf skips comparing the two charact
 572                        // but we don't actually know that the two characters are equal, since we compared with | 0x20. 
 573                        // the full string always.
 0574                        nint charPos = (nint)(uint.TrailingZeroCount(mask) / sizeof(ushort));
 0575                        if (EqualsIgnoreCase(ref Unsafe.Add(ref searchSpace, offset + charPos), ref valueRef, value.Leng
 576                        {
 577                            // Match! Return the index.
 0578                            return (int)(offset + charPos);
 579                        }
 580
 581                        // Clear the two lowest set bits in the mask. If there are no more set bits, we're done.
 582                        // If any remain, we loop around to do the next comparison.
 0583                        mask = BitOperations.ResetLowestSetBit(BitOperations.ResetLowestSetBit(mask));
 0584                    } while (mask != 0);
 0585                    goto LoopFooter;
 586
 587                } while (true);
 588            }
 589        }
 590
 591        internal static int LastIndexOf(string source, string value, int startIndex, int count)
 592        {
 593            int result = source.AsSpan(startIndex, count).LastIndexOf(value);
 594            if (result >= 0) { result += startIndex; } // if match found, adjust 'result' by the actual start position
 595            return result;
 596        }
 597
 598        internal static int LastIndexOf(string source, string value, int startIndex, int count, bool ignoreCase)
 599        {
 600            if (source == null)
 601            {
 602                ThrowHelper.ThrowArgumentNullException(ExceptionArgument.source);
 603            }
 604
 605            if (value == null)
 606            {
 607                ThrowHelper.ThrowArgumentNullException(ExceptionArgument.value);
 608            }
 609
 610            if (value.Length == 0)
 611            {
 612                return startIndex + 1; // startIndex is the index of the last char to include in the search space
 613            }
 614
 615            if (count == 0)
 616            {
 617                return -1;
 618            }
 619
 620            if (GlobalizationMode.Invariant)
 621            {
 622                return ignoreCase ? InvariantModeCasing.LastIndexOfIgnoreCase(source.AsSpan(startIndex, count), value) :
 623            }
 624
 625            if (GlobalizationMode.UseNls)
 626            {
 627                return CompareInfo.NlsLastIndexOfOrdinalCore(source, value, startIndex, count, ignoreCase);
 628            }
 629
 630            if (!ignoreCase)
 631            {
 632                LastIndexOf(source, value, startIndex, count);
 633            }
 634
 635            if (!source.TryGetSpan(startIndex, count, out ReadOnlySpan<char> sourceSpan))
 636            {
 637                // Bounds check failed - figure out exactly what went wrong so that we can
 638                // surface the correct argument exception.
 639
 640                if ((uint)startIndex > (uint)source.Length)
 641                {
 642                    ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.startIndex, ExceptionResource.Argumen
 643                }
 644                else
 645                {
 646                    ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.count, ExceptionResource.ArgumentOutO
 647                }
 648            }
 649
 650            int result = OrdinalCasing.LastIndexOf(sourceSpan, value);
 651
 652            if (result >= 0)
 653            {
 654                result += startIndex;
 655            }
 656            return result;
 657        }
 658
 659        internal static int LastIndexOfOrdinalIgnoreCase(ReadOnlySpan<char> source, ReadOnlySpan<char> value)
 660        {
 0661            if (value.Length == 0)
 662            {
 0663                return source.Length;
 664            }
 665
 0666            if (value.Length > source.Length)
 667            {
 668                // A non-linguistic search compares chars directly against one another, so large
 669                // target strings can never be found inside small search spaces. This check also
 670                // handles empty 'source' spans.
 671
 0672                return -1;
 673            }
 674
 0675            if (GlobalizationMode.Invariant)
 676            {
 0677                return InvariantModeCasing.LastIndexOfIgnoreCase(source, value);
 678            }
 679
 0680            if (GlobalizationMode.UseNls)
 681            {
 0682                return CompareInfo.NlsIndexOfOrdinalCore(source, value, ignoreCase: true, fromBeginning: false);
 683            }
 684
 0685            return OrdinalCasing.LastIndexOf(source, value);
 686        }
 687
 688        internal static int ToUpperOrdinal(ReadOnlySpan<char> source, Span<char> destination)
 689        {
 0690            if (source.Overlaps(destination))
 0691                ThrowHelper.ThrowInvalidOperationException(ExceptionResource.InvalidOperation_SpanOverlappedOperation);
 692
 693            // Assuming that changing case does not affect length
 0694            if (destination.Length < source.Length)
 0695                return -1;
 696
 0697            if (GlobalizationMode.Invariant)
 698            {
 0699                InvariantModeCasing.ToUpper(source, destination);
 0700                return source.Length;
 701            }
 702
 0703            if (GlobalizationMode.UseNls)
 704            {
 0705                ChangeCaseNlsOrdinal(source, destination, toUpper: true);
 0706                return source.Length;
 707            }
 708
 0709            OrdinalCasing.ToUpperOrdinal(source, destination);
 0710            return source.Length;
 711        }
 712
 713        internal static int ToLowerOrdinal(ReadOnlySpan<char> source, Span<char> destination)
 714        {
 0715            if (source.Overlaps(destination))
 0716                ThrowHelper.ThrowInvalidOperationException(ExceptionResource.InvalidOperation_SpanOverlappedOperation);
 717
 718            // Assuming that changing case does not affect length
 0719            if (destination.Length < source.Length)
 0720                return -1;
 721
 0722            if (GlobalizationMode.Invariant)
 723            {
 0724                InvariantModeCasing.ToLower(source, destination);
 0725                PreserveOrdinalLowerCasingClass(source, destination);
 0726                return source.Length;
 727            }
 728
 0729            if (GlobalizationMode.UseNls)
 730            {
 0731                ChangeCaseNlsOrdinal(source, destination, toUpper: false);
 0732                PreserveOrdinalLowerCasingClass(source, destination);
 0733                return source.Length;
 734            }
 735
 0736            OrdinalCasing.ToLowerOrdinal(source, destination);
 0737            return source.Length;
 738        }
 739
 740        private static void ChangeCaseNlsOrdinal(ReadOnlySpan<char> source, Span<char> destination, bool toUpper)
 741        {
 0742            Debug.Assert(GlobalizationMode.UseNls);
 743
 0744            OperationStatus operationStatus = toUpper
 0745                ? Ascii.ToUpper(source, destination, out int charsConsumed)
 0746                : Ascii.ToLower(source, destination, out charsConsumed);
 747
 0748            if (operationStatus != OperationStatus.InvalidData)
 749            {
 0750                Debug.Assert(operationStatus == OperationStatus.Done);
 0751                return;
 752            }
 753
 0754            source = source.Slice(charsConsumed);
 0755            destination = destination.Slice(charsConsumed);
 756
 0757            if (toUpper)
 758            {
 0759                TextInfo.Invariant.ChangeCaseToUpper(source, destination);
 760            }
 761            else
 762            {
 0763                TextInfo.Invariant.ChangeCaseToLower(source, destination);
 764            }
 765
 0766            PreserveNlsOrdinalCasingClass(source, destination);
 0767        }
 768
 769        private static void PreserveNlsOrdinalCasingClass(ReadOnlySpan<char> source, Span<char> destination)
 770        {
 0771            Debug.Assert(GlobalizationMode.UseNls);
 0772            Debug.Assert(destination.Length >= source.Length);
 773
 0774            destination = destination.Slice(0, source.Length);
 775
 0776            int i = source.IndexOfAnyInRange('\uD800', '\uDBFF');
 0777            if (i < 0)
 778            {
 0779                return;
 780            }
 781
 0782            while (i < source.Length - 1)
 783            {
 0784                if (char.IsLowSurrogate(source[i + 1]))
 785                {
 0786                    if (source[i] != destination[i] || source[i + 1] != destination[i + 1])
 787                    {
 0788                        if (CompareInfo.NlsCompareStringOrdinalIgnoreCase(
 0789                                ref MemoryMarshal.GetReference(source.Slice(i)), 2,
 0790                                ref MemoryMarshal.GetReference(destination.Slice(i)), 2) != 0)
 791                        {
 0792                            destination[i] = source[i];
 0793                            destination[i + 1] = source[i + 1];
 794                        }
 795
 0796                        i += 2;
 797                    }
 798                    else
 799                    {
 0800                        i += source.Slice(i).CommonPrefixLength(destination.Slice(i));
 801
 0802                        if (i < source.Length && char.IsLowSurrogate(source[i]) && char.IsHighSurrogate(source[i - 1]))
 803                        {
 0804                            i--;
 805                        }
 806                    }
 807                }
 808                else
 809                {
 0810                    i++;
 811                }
 812
 0813                int next = source.Slice(i).IndexOfAnyInRange('\uD800', '\uDBFF');
 0814                if (next < 0)
 815                {
 0816                    return;
 817                }
 818
 0819                i += next;
 820            }
 0821        }
 822
 823        internal static uint PreserveNlsOrdinalCasingClass(uint source, uint destination)
 824        {
 0825            Debug.Assert(GlobalizationMode.UseNls);
 0826            Debug.Assert(source > char.MaxValue);
 0827            Debug.Assert(destination > char.MaxValue);
 828
 0829            if (source == destination)
 830            {
 0831                return destination;
 832            }
 833
 0834            Span<char> sourceChars = stackalloc char[2];
 0835            Span<char> destinationChars = stackalloc char[2];
 0836            UnicodeUtility.GetUtf16SurrogatesFromSupplementaryPlaneScalar(source, out sourceChars[0], out sourceChars[1]
 0837            UnicodeUtility.GetUtf16SurrogatesFromSupplementaryPlaneScalar(destination, out destinationChars[0], out dest
 838
 0839            return CompareInfo.NlsCompareStringOrdinalIgnoreCase(
 0840                ref MemoryMarshal.GetReference(sourceChars), sourceChars.Length,
 0841                ref MemoryMarshal.GetReference(destinationChars), destinationChars.Length) == 0
 0842                    ? destination
 0843                    : source;
 844        }
 845
 846        // The only BMP scalars whose simple invariant/NLS lower mapping moves them out of their ordinal
 847        // upper-casing class. Each one is its own ordinal upper-casing (ToUpperOrdinal(c) == c) yet lowers to a
 848        // different letter, so simple lowering would break OrdinalIgnoreCase consistency for them. The ICU ordinal
 849        // table already keeps them unchanged; invariant and NLS lowering do not, so they must be restored.
 850        // U+03F4 GREEK CAPITAL THETA SYMBOL, U+1E9E LATIN CAPITAL LETTER SHARP S, U+2126 OHM SIGN,
 851        // U+212A KELVIN SIGN, U+212B ANGSTROM SIGN. The ToLowerOrdinal_EveryBmpChar tests would fail if this set
 852        // ever became incomplete.
 0853        private static readonly SearchValues<char> s_ordinalLowerCasingClassRestoreChars =
 0854            SearchValues.Create("\u03F4\u1E9E\u2126\u212A\u212B");
 855
 856        // Ordinal lower casing must never move a character out of its ordinal upper-casing class, otherwise it
 857        // would stop being consistent with OrdinalIgnoreCase (for example the Kelvin, Ohm and Angstrom signs).
 858        // Only the fixed set above is ever affected, so skip the pass entirely unless one of them is present.
 859        private static void PreserveOrdinalLowerCasingClass(ReadOnlySpan<char> source, Span<char> destination)
 860        {
 0861            int i = source.IndexOfAny(s_ordinalLowerCasingClassRestoreChars);
 0862            if (i < 0)
 863            {
 0864                return;
 865            }
 866
 867            // Restore each affected character to its original form; its ordinal lower casing is itself.
 0868            for (; i < source.Length; i++)
 869            {
 0870                if (s_ordinalLowerCasingClassRestoreChars.Contains(source[i]))
 871                {
 0872                    destination[i] = source[i];
 873                }
 874            }
 0875        }
 876    }
 877}
 878

https://raw.githubusercontent.com/dotnet/runtime/811a7eabb75c42db53440e8ba3f60c07511cfd1f/src/libraries/System.Private.CoreLib/src/System/Globalization/Ordinal.Utf8.cs

#LineLine coverage
 1// Licensed to the .NET Foundation under one or more agreements.
 2// The .NET Foundation licenses this file to you under the MIT license.
 3
 4using System.Buffers;
 5using System.Buffers.Binary;
 6using System.Runtime.CompilerServices;
 7using System.Text;
 8using System.Text.Unicode;
 9
 10namespace System.Globalization
 11{
 12    internal static partial class Ordinal
 13    {
 14        // Not optimized for large inputs: the only callers (number parsing) pass short sign/NaN/Infinity symbols.
 15        [MethodImpl(MethodImplOptions.AggressiveInlining)]
 16        internal static bool EqualsIgnoreCaseUtf8(ReadOnlySpan<byte> left, ReadOnlySpan<byte> right)
 17        {
 18            // ASCII-only loops without calls, so they stay cheap when inlined into callers
 019            while ((left.Length >= sizeof(ulong)) && (right.Length >= sizeof(ulong)))
 20            {
 021                ulong a = BinaryPrimitives.ReadUInt64LittleEndian(left);
 022                ulong b = BinaryPrimitives.ReadUInt64LittleEndian(right);
 23
 024                if (!Utf8Utility.AllBytesInUInt64AreAscii(a | b))
 25                {
 26                    break;
 27                }
 28
 029                if (!Utf8Utility.UInt64OrdinalIgnoreCaseAscii(a, b))
 30                {
 031                    return false;
 32                }
 33
 034                left = left.Slice(sizeof(ulong));
 035                right = right.Slice(sizeof(ulong));
 36            }
 37
 038            for (int i = 0; i < right.Length; i++)
 39            {
 040                if (i >= left.Length)
 41                {
 042                    return false;
 43                }
 44
 045                uint a = left[i];
 046                uint b = right[i];
 47
 048                if ((a | b) > 0x7F)
 49                {
 50                    // No non-ASCII scalar is equal to an ASCII one under ordinal casing
 051                    return ((a ^ b) <= 0x7F) && MatchIgnoreCaseUtf8(left.Slice(i), right.Slice(i), prefixOnly: false);
 52                }
 53
 54                // Ordinal equals or lowercase equals if the result ends up in the a-z range
 055                if ((a != b) && (((a | 0x20) != (b | 0x20)) || !char.IsAsciiLetter((char)a)))
 56                {
 057                    return false;
 58                }
 59            }
 60
 061            return left.Length == right.Length;
 62        }
 63
 64        internal static bool StartsWithIgnoreCaseUtf8(ReadOnlySpan<byte> source, ReadOnlySpan<byte> prefix) =>
 065            MatchIgnoreCaseUtf8(source, prefix, prefixOnly: true);
 66
 67        private static bool MatchIgnoreCaseUtf8(ReadOnlySpan<byte> source, ReadOnlySpan<byte> prefix, bool prefixOnly)
 68        {
 69            // NOTE: Two UTF-8 inputs of different length might compare as equal under
 70            // the OrdinalIgnoreCase comparer. This is distinct from UTF-16, where the
 71            // inputs being different length will mean that they can never compare as
 72            // equal under an OrdinalIgnoreCase comparer.
 73
 074            while (!prefix.IsEmpty)
 75            {
 076                if (source.IsEmpty)
 77                {
 78                    // The source ended before the prefix
 079                    return false;
 80                }
 81
 082                uint a = source[0];
 083                uint b = prefix[0];
 84
 085                if ((a | b) <= 0x7F)
 86                {
 87                    // Ordinal equals or lowercase equals if the result ends up in the a-z range
 088                    if ((a != b) && (((a | 0x20) != (b | 0x20)) || !char.IsAsciiLetter((char)a)))
 89                    {
 090                        return false;
 91                    }
 92
 093                    source = source.Slice(1);
 094                    prefix = prefix.Slice(1);
 095                    continue;
 96                }
 97
 098                if ((a ^ b) > 0x7F)
 99                {
 100                    // No non-ASCII scalar is equal to an ASCII one under ordinal casing
 0101                    return false;
 102                }
 103
 104                // NLS/ICU doesn't provide native UTF-8 support so we need to do our own corresponding ordinal compariso
 0105                OperationStatus statusA = Rune.DecodeFromUtf8(source, out Rune runeA, out int bytesConsumedA);
 0106                OperationStatus statusB = Rune.DecodeFromUtf8(prefix, out Rune runeB, out int bytesConsumedB);
 107
 0108                if (statusA != statusB)
 109                {
 0110                    return false;
 111                }
 112
 0113                if (statusA == OperationStatus.Done)
 114                {
 0115                    if ((runeA != runeB) && (Rune.ToUpperOrdinal(runeA) != Rune.ToUpperOrdinal(runeB)))
 116                    {
 0117                        return false;
 118                    }
 119                }
 0120                else if (!source.Slice(0, bytesConsumedA).SequenceEqual(prefix.Slice(0, bytesConsumedB)))
 121                {
 122                    // Invalid sequences must match exactly
 0123                    return false;
 124                }
 125
 0126                source = source.Slice(bytesConsumedA);
 0127                prefix = prefix.Slice(bytesConsumedB);
 128            }
 129
 0130            return prefixOnly || source.IsEmpty;
 131        }
 132    }
 133}
 134