| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | using System.Buffers; |
| | | 5 | | using System.Diagnostics; |
| | | 6 | | using System.Numerics; |
| | | 7 | | using System.Runtime.CompilerServices; |
| | | 8 | | using System.Runtime.InteropServices; |
| | | 9 | | using System.Runtime.Intrinsics; |
| | | 10 | | using System.Runtime.Intrinsics.X86; |
| | | 11 | | using System.Text; |
| | | 12 | | using System.Text.Unicode; |
| | | 13 | | |
| | | 14 | | namespace System.Globalization |
| | | 15 | | { |
| | | 16 | | internal static partial class Ordinal |
| | | 17 | | { |
| | | 18 | | internal static int CompareStringIgnoreCase(ref char strA, int lengthA, ref char strB, int lengthB) |
| | | 19 | | { |
| | 0 | 20 | | int length = Math.Min(lengthA, lengthB); |
| | 0 | 21 | | int range = length; |
| | | 22 | | |
| | 0 | 23 | | ref char charA = ref strA; |
| | 0 | 24 | | ref char charB = ref strB; |
| | | 25 | | |
| | | 26 | | const char maxChar = (char)0x7F; |
| | | 27 | | |
| | 0 | 28 | | while (length != 0 && charA <= maxChar && charB <= maxChar) |
| | | 29 | | { |
| | | 30 | | // Ordinal equals or lowercase equals if the result ends up in the a-z range |
| | 0 | 31 | | if (charA == charB || |
| | 0 | 32 | | ((charA | 0x20) == (charB | 0x20) && char.IsAsciiLetter(charA))) |
| | | 33 | | { |
| | 0 | 34 | | length--; |
| | 0 | 35 | | charA = ref Unsafe.Add(ref charA, 1); |
| | 0 | 36 | | charB = ref Unsafe.Add(ref charB, 1); |
| | | 37 | | } |
| | | 38 | | else |
| | | 39 | | { |
| | 0 | 40 | | int currentA = charA; |
| | 0 | 41 | | int currentB = charB; |
| | | 42 | | |
| | | 43 | | // Uppercase both chars if needed |
| | 0 | 44 | | if (char.IsAsciiLetterLower(charA)) |
| | | 45 | | { |
| | 0 | 46 | | currentA -= 0x20; |
| | | 47 | | } |
| | 0 | 48 | | if (char.IsAsciiLetterLower(charB)) |
| | | 49 | | { |
| | 0 | 50 | | currentB -= 0x20; |
| | | 51 | | } |
| | | 52 | | |
| | | 53 | | // Return the (case-insensitive) difference between them. |
| | 0 | 54 | | return currentA - currentB; |
| | | 55 | | } |
| | | 56 | | } |
| | | 57 | | |
| | 0 | 58 | | if (length == 0) |
| | | 59 | | { |
| | 0 | 60 | | return lengthA - lengthB; |
| | | 61 | | } |
| | | 62 | | |
| | 0 | 63 | | range -= length; |
| | | 64 | | |
| | 0 | 65 | | return CompareStringIgnoreCaseNonAscii(ref charA, lengthA - range, ref charB, lengthB - range); |
| | | 66 | | } |
| | | 67 | | |
| | | 68 | | internal static int CompareStringIgnoreCaseNonAscii(ref char strA, int lengthA, ref char strB, int lengthB) |
| | | 69 | | { |
| | 0 | 70 | | if (GlobalizationMode.Invariant) |
| | | 71 | | { |
| | 0 | 72 | | return InvariantModeCasing.CompareStringIgnoreCase(ref strA, lengthA, ref strB, lengthB); |
| | | 73 | | } |
| | | 74 | | |
| | 0 | 75 | | if (GlobalizationMode.UseNls) |
| | | 76 | | { |
| | 0 | 77 | | return CompareInfo.NlsCompareStringOrdinalIgnoreCase(ref strA, lengthA, ref strB, lengthB); |
| | | 78 | | } |
| | | 79 | | |
| | 0 | 80 | | return OrdinalCasing.CompareStringIgnoreCase(ref strA, lengthA, ref strB, lengthB); |
| | | 81 | | } |
| | | 82 | | |
| | | 83 | | private static bool EqualsIgnoreCase_Vector<TVector>(ref char charA, ref char charB, int length) |
| | | 84 | | where TVector : struct, ISimdVector<TVector, ushort> |
| | | 85 | | { |
| | 0 | 86 | | Debug.Assert(length >= TVector.ElementCount); |
| | | 87 | | |
| | 0 | 88 | | nuint lengthU = (nuint)length; |
| | 0 | 89 | | nuint lengthToExamine = lengthU - (nuint)TVector.ElementCount; |
| | 0 | 90 | | nuint i = 0; |
| | | 91 | | TVector vec1; |
| | | 92 | | TVector vec2; |
| | 0 | 93 | | TVector loweringMask = TVector.Create(0x20); |
| | 0 | 94 | | TVector vecA = TVector.Create('a'); |
| | 0 | 95 | | TVector vecZMinusA = TVector.Create('z' - 'a'); |
| | | 96 | | do |
| | | 97 | | { |
| | 0 | 98 | | vec1 = TVector.LoadUnsafe(ref Unsafe.As<char, ushort>(ref charA), i); |
| | 0 | 99 | | vec2 = TVector.LoadUnsafe(ref Unsafe.As<char, ushort>(ref charB), i); |
| | | 100 | | |
| | 0 | 101 | | if (!Utf16Utility.AllCharsInVectorAreAscii(vec1 | vec2)) |
| | | 102 | | { |
| | | 103 | | goto NON_ASCII; |
| | | 104 | | } |
| | | 105 | | |
| | 0 | 106 | | TVector notEquals = ~TVector.Equals(vec1, vec2); |
| | 0 | 107 | | if (!notEquals.Equals(TVector.Zero)) |
| | | 108 | | { |
| | | 109 | | // not exact match |
| | | 110 | | |
| | 0 | 111 | | vec1 |= loweringMask; |
| | 0 | 112 | | vec2 |= loweringMask; |
| | 0 | 113 | | if (TVector.GreaterThanAny((vec1 - vecA) & notEquals, vecZMinusA) || !vec1.Equals(vec2)) |
| | | 114 | | { |
| | 0 | 115 | | return false; // first input isn't in [A-Za-z], and not exact match of lowered |
| | | 116 | | } |
| | | 117 | | } |
| | 0 | 118 | | i += (nuint)TVector.ElementCount; |
| | 0 | 119 | | } while (i <= lengthToExamine); |
| | | 120 | | |
| | | 121 | | // Handle trailing elements |
| | 0 | 122 | | if (i != lengthU) |
| | | 123 | | { |
| | 0 | 124 | | i = lengthU - (nuint)TVector.ElementCount; |
| | 0 | 125 | | vec1 = TVector.LoadUnsafe(ref Unsafe.As<char, ushort>(ref charA), i); |
| | 0 | 126 | | vec2 = TVector.LoadUnsafe(ref Unsafe.As<char, ushort>(ref charB), i); |
| | | 127 | | |
| | 0 | 128 | | if (!Utf16Utility.AllCharsInVectorAreAscii(vec1 | vec2)) |
| | | 129 | | { |
| | | 130 | | goto NON_ASCII; |
| | | 131 | | } |
| | | 132 | | |
| | 0 | 133 | | TVector notEquals = ~TVector.Equals(vec1, vec2); |
| | 0 | 134 | | if (!notEquals.Equals(TVector.Zero)) |
| | | 135 | | { |
| | | 136 | | // not exact match |
| | | 137 | | |
| | 0 | 138 | | vec1 |= loweringMask; |
| | 0 | 139 | | vec2 |= loweringMask; |
| | 0 | 140 | | if (TVector.GreaterThanAny((vec1 - vecA) & notEquals, vecZMinusA) || !vec1.Equals(vec2)) |
| | | 141 | | { |
| | 0 | 142 | | return false; // first input isn't in [A-Za-z], and not exact match of lowered |
| | | 143 | | } |
| | | 144 | | } |
| | | 145 | | } |
| | 0 | 146 | | return true; |
| | | 147 | | |
| | | 148 | | NON_ASCII: |
| | 0 | 149 | | if (Utf16Utility.AllCharsInVectorAreAscii(vec1) || Utf16Utility.AllCharsInVectorAreAscii(vec2)) |
| | | 150 | | { |
| | | 151 | | // No need to use the fallback if one of the inputs is full-ASCII |
| | 0 | 152 | | return false; |
| | | 153 | | } |
| | | 154 | | |
| | | 155 | | // Fallback for Non-ASCII inputs |
| | 0 | 156 | | return CompareStringIgnoreCase( |
| | 0 | 157 | | ref Unsafe.Add(ref charA, i), (int)(lengthU - i), |
| | 0 | 158 | | ref Unsafe.Add(ref charB, i), (int)(lengthU - i)) == 0; |
| | | 159 | | } |
| | | 160 | | |
| | | 161 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 162 | | internal static bool EqualsIgnoreCase(ref char charA, ref char charB, int length) |
| | | 163 | | { |
| | 0 | 164 | | if (!Vector128.IsHardwareAccelerated || length < Vector128<ushort>.Count) |
| | | 165 | | { |
| | 0 | 166 | | return EqualsIgnoreCase_Scalar(ref charA, ref charB, length); |
| | | 167 | | } |
| | 0 | 168 | | if (Vector512.IsHardwareAccelerated && length >= Vector512<ushort>.Count) |
| | | 169 | | { |
| | 0 | 170 | | return EqualsIgnoreCase_Vector<Vector512<ushort>>(ref charA, ref charB, length); |
| | | 171 | | } |
| | 0 | 172 | | if (Vector256.IsHardwareAccelerated && length >= Vector256<ushort>.Count) |
| | | 173 | | { |
| | 0 | 174 | | return EqualsIgnoreCase_Vector<Vector256<ushort>>(ref charA, ref charB, length); |
| | | 175 | | } |
| | 0 | 176 | | return EqualsIgnoreCase_Vector<Vector128<ushort>>(ref charA, ref charB, length); |
| | | 177 | | } |
| | | 178 | | |
| | | 179 | | internal static bool EqualsIgnoreCase_Scalar(ref char charA, ref char charB, int length) |
| | | 180 | | { |
| | 0 | 181 | | IntPtr byteOffset = IntPtr.Zero; |
| | | 182 | | |
| | | 183 | | #if TARGET_64BIT |
| | 0 | 184 | | ulong valueAu64 = 0; |
| | 0 | 185 | | ulong valueBu64 = 0; |
| | | 186 | | // Read 4 chars (64 bits) at a time from each string |
| | 0 | 187 | | while ((uint)length >= 4) |
| | | 188 | | { |
| | 0 | 189 | | valueAu64 = Unsafe.ReadUnaligned<ulong>(ref Unsafe.As<char, byte>(ref Unsafe.AddByteOffset(ref charA, by |
| | 0 | 190 | | valueBu64 = Unsafe.ReadUnaligned<ulong>(ref Unsafe.As<char, byte>(ref Unsafe.AddByteOffset(ref charB, by |
| | | 191 | | |
| | | 192 | | // A 32-bit test - even with the bit-twiddling here - is more efficient than a 64-bit test. |
| | 0 | 193 | | ulong temp = valueAu64 | valueBu64; |
| | 0 | 194 | | if (!Utf16Utility.AllCharsInUInt32AreAscii((uint)temp | (uint)(temp >> 32))) |
| | | 195 | | { |
| | | 196 | | goto NonAscii64; // one of the inputs contains non-ASCII data |
| | | 197 | | } |
| | | 198 | | |
| | | 199 | | // Generally, the caller has likely performed a first-pass check that the input strings |
| | | 200 | | // are likely equal. Consider a dictionary which computes the hash code of its key before |
| | | 201 | | // performing a proper deep equality check of the string contents. We want to optimize for |
| | | 202 | | // the case where the equality check is likely to succeed, which means that we want to avoid |
| | | 203 | | // branching within this loop unless we're about to exit the loop, either due to failure or |
| | | 204 | | // due to us running out of input data. |
| | | 205 | | |
| | 0 | 206 | | if (!Utf16Utility.UInt64OrdinalIgnoreCaseAscii(valueAu64, valueBu64)) |
| | | 207 | | { |
| | 0 | 208 | | return false; |
| | | 209 | | } |
| | | 210 | | |
| | 0 | 211 | | byteOffset += 8; |
| | 0 | 212 | | length -= 4; |
| | | 213 | | } |
| | | 214 | | #endif |
| | 0 | 215 | | uint valueAu32 = 0; |
| | 0 | 216 | | uint valueBu32 = 0; |
| | | 217 | | // Read 2 chars (32 bits) at a time from each string |
| | | 218 | | #if TARGET_64BIT |
| | 0 | 219 | | if ((uint)length >= 2) |
| | | 220 | | #else |
| | | 221 | | while ((uint)length >= 2) |
| | | 222 | | #endif |
| | | 223 | | { |
| | 0 | 224 | | valueAu32 = Unsafe.ReadUnaligned<uint>(ref Unsafe.As<char, byte>(ref Unsafe.AddByteOffset(ref charA, byt |
| | 0 | 225 | | valueBu32 = Unsafe.ReadUnaligned<uint>(ref Unsafe.As<char, byte>(ref Unsafe.AddByteOffset(ref charB, byt |
| | | 226 | | |
| | 0 | 227 | | if (!Utf16Utility.AllCharsInUInt32AreAscii(valueAu32 | valueBu32)) |
| | | 228 | | { |
| | | 229 | | goto NonAscii32; // one of the inputs contains non-ASCII data |
| | | 230 | | } |
| | | 231 | | |
| | | 232 | | // Generally, the caller has likely performed a first-pass check that the input strings |
| | | 233 | | // are likely equal. Consider a dictionary which computes the hash code of its key before |
| | | 234 | | // performing a proper deep equality check of the string contents. We want to optimize for |
| | | 235 | | // the case where the equality check is likely to succeed, which means that we want to avoid |
| | | 236 | | // branching within this loop unless we're about to exit the loop, either due to failure or |
| | | 237 | | // due to us running out of input data. |
| | | 238 | | |
| | 0 | 239 | | if (!Utf16Utility.UInt32OrdinalIgnoreCaseAscii(valueAu32, valueBu32)) |
| | | 240 | | { |
| | 0 | 241 | | return false; |
| | | 242 | | } |
| | | 243 | | |
| | 0 | 244 | | byteOffset += 4; |
| | 0 | 245 | | length -= 2; |
| | | 246 | | } |
| | | 247 | | |
| | 0 | 248 | | if (length != 0) |
| | | 249 | | { |
| | 0 | 250 | | Debug.Assert(length == 1); |
| | | 251 | | |
| | 0 | 252 | | valueAu32 = Unsafe.AddByteOffset(ref charA, byteOffset); |
| | 0 | 253 | | valueBu32 = Unsafe.AddByteOffset(ref charB, byteOffset); |
| | | 254 | | |
| | 0 | 255 | | if ((valueAu32 | valueBu32) > 0x7Fu) |
| | | 256 | | { |
| | | 257 | | goto NonAscii32; // one of the inputs contains non-ASCII data |
| | | 258 | | } |
| | | 259 | | |
| | 0 | 260 | | if (valueAu32 == valueBu32) |
| | | 261 | | { |
| | 0 | 262 | | return true; // exact match |
| | | 263 | | } |
| | | 264 | | |
| | 0 | 265 | | valueAu32 |= 0x20u; |
| | 0 | 266 | | if ((uint)(valueAu32 - 'a') > (uint)('z' - 'a')) |
| | | 267 | | { |
| | 0 | 268 | | return false; // not exact match, and first input isn't in [A-Za-z] |
| | | 269 | | } |
| | | 270 | | |
| | 0 | 271 | | return valueAu32 == (valueBu32 | 0x20u); |
| | | 272 | | } |
| | | 273 | | |
| | 0 | 274 | | Debug.Assert(length == 0); |
| | 0 | 275 | | return true; |
| | | 276 | | |
| | | 277 | | NonAscii32: |
| | | 278 | | // Both values have to be non-ASCII to use the slow fallback, in case if one of them is not we return false |
| | 0 | 279 | | if (Utf16Utility.AllCharsInUInt32AreAscii(valueAu32) || Utf16Utility.AllCharsInUInt32AreAscii(valueBu32)) |
| | | 280 | | { |
| | 0 | 281 | | return false; |
| | | 282 | | } |
| | | 283 | | goto NonAscii; |
| | | 284 | | |
| | | 285 | | #if TARGET_64BIT |
| | | 286 | | NonAscii64: |
| | | 287 | | // Both values have to be non-ASCII to use the slow fallback, in case if one of them is not we return false |
| | 0 | 288 | | if (Utf16Utility.AllCharsInUInt64AreAscii(valueAu64) || Utf16Utility.AllCharsInUInt64AreAscii(valueBu64)) |
| | | 289 | | { |
| | 0 | 290 | | return false; |
| | | 291 | | } |
| | | 292 | | #endif |
| | | 293 | | NonAscii: |
| | | 294 | | // The non-ASCII case is factored out into its own helper method so that the JIT |
| | | 295 | | // doesn't need to emit a complex prolog for its caller (this method). |
| | 0 | 296 | | return CompareStringIgnoreCase(ref Unsafe.AddByteOffset(ref charA, byteOffset), length, ref Unsafe.AddByteOf |
| | | 297 | | } |
| | | 298 | | |
| | | 299 | | internal static int IndexOf(string source, string value, int startIndex, int count, bool ignoreCase) |
| | | 300 | | { |
| | 0 | 301 | | if (source == null) |
| | | 302 | | { |
| | 0 | 303 | | ThrowHelper.ThrowArgumentNullException(ExceptionArgument.source); |
| | | 304 | | } |
| | | 305 | | |
| | 0 | 306 | | if (value == null) |
| | | 307 | | { |
| | 0 | 308 | | ThrowHelper.ThrowArgumentNullException(ExceptionArgument.value); |
| | | 309 | | } |
| | | 310 | | |
| | 0 | 311 | | if (!source.TryGetSpan(startIndex, count, out ReadOnlySpan<char> sourceSpan)) |
| | | 312 | | { |
| | | 313 | | // Bounds check failed - figure out exactly what went wrong so that we can |
| | | 314 | | // surface the correct argument exception. |
| | | 315 | | |
| | 0 | 316 | | if ((uint)startIndex > (uint)source.Length) |
| | | 317 | | { |
| | 0 | 318 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.startIndex, ExceptionResource.Argumen |
| | | 319 | | } |
| | | 320 | | else |
| | | 321 | | { |
| | 0 | 322 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.count, ExceptionResource.ArgumentOutO |
| | | 323 | | } |
| | | 324 | | } |
| | | 325 | | |
| | 0 | 326 | | int result = ignoreCase ? IndexOfOrdinalIgnoreCase(sourceSpan, value) : sourceSpan.IndexOf(value); |
| | | 327 | | |
| | 0 | 328 | | return result >= 0 ? result + startIndex : result; |
| | | 329 | | } |
| | | 330 | | |
| | | 331 | | internal static int IndexOfOrdinalIgnoreCase(ReadOnlySpan<char> source, ReadOnlySpan<char> value) |
| | | 332 | | { |
| | 0 | 333 | | if (value.Length == 0) |
| | | 334 | | { |
| | 0 | 335 | | return 0; |
| | | 336 | | } |
| | | 337 | | |
| | 0 | 338 | | if (value.Length > source.Length) |
| | | 339 | | { |
| | | 340 | | // A non-linguistic search compares chars directly against one another, so large |
| | | 341 | | // target strings can never be found inside small search spaces. This check also |
| | | 342 | | // handles empty 'source' spans. |
| | 0 | 343 | | return -1; |
| | | 344 | | } |
| | | 345 | | |
| | 0 | 346 | | if (GlobalizationMode.Invariant) |
| | | 347 | | { |
| | 0 | 348 | | return InvariantModeCasing.IndexOfIgnoreCase(source, value); |
| | | 349 | | } |
| | | 350 | | |
| | 0 | 351 | | if (GlobalizationMode.UseNls) |
| | | 352 | | { |
| | 0 | 353 | | return CompareInfo.NlsIndexOfOrdinalCore(source, value, ignoreCase: true, fromBeginning: true); |
| | | 354 | | } |
| | | 355 | | |
| | | 356 | | // If value doesn't start with ASCII, fall back to a non-vectorized non-ASCII friendly version. |
| | 0 | 357 | | ref char valueRef = ref MemoryMarshal.GetReference(value); |
| | 0 | 358 | | char valueChar = valueRef; |
| | 0 | 359 | | if (!char.IsAscii(valueChar)) |
| | | 360 | | { |
| | 0 | 361 | | return OrdinalCasing.IndexOf(source, value); |
| | | 362 | | } |
| | | 363 | | |
| | | 364 | | // Hoist some expressions from the loop |
| | 0 | 365 | | int valueTailLength = value.Length - 1; |
| | 0 | 366 | | int searchSpaceMinusValueTailLength = source.Length - valueTailLength; |
| | 0 | 367 | | ref char searchSpace = ref MemoryMarshal.GetReference(source); |
| | 0 | 368 | | char valueCharU = default; |
| | 0 | 369 | | char valueCharL = default; |
| | 0 | 370 | | nint offset = 0; |
| | 0 | 371 | | bool isLetter = false; |
| | | 372 | | |
| | | 373 | | // If the input is long enough and the value ends with ASCII and is at least two characters, |
| | | 374 | | // we can take a special vectorized path that compares both the beginning and the end at the same time. |
| | 0 | 375 | | if (Vector128.IsHardwareAccelerated && |
| | 0 | 376 | | valueTailLength != 0 && |
| | 0 | 377 | | searchSpaceMinusValueTailLength >= Vector128<ushort>.Count) |
| | | 378 | | { |
| | 0 | 379 | | valueCharU = Unsafe.Add(ref valueRef, valueTailLength); |
| | 0 | 380 | | if (char.IsAscii(valueCharU)) |
| | | 381 | | { |
| | | 382 | | goto SearchTwoChars; |
| | | 383 | | } |
| | | 384 | | } |
| | | 385 | | |
| | | 386 | | // We're searching for the first character and it's known to be ASCII. If it's not a letter, |
| | | 387 | | // then IgnoreCase doesn't impact what it matches and we just need to do a normal search |
| | | 388 | | // for that single character. If it is a letter, then we need to search for both its upper |
| | | 389 | | // and lower-case variants. |
| | 0 | 390 | | if (char.IsAsciiLetter(valueChar)) |
| | | 391 | | { |
| | 0 | 392 | | valueCharU = (char)(valueChar & ~0x20); |
| | 0 | 393 | | valueCharL = (char)(valueChar | 0x20); |
| | 0 | 394 | | isLetter = true; |
| | | 395 | | } |
| | | 396 | | |
| | | 397 | | do |
| | | 398 | | { |
| | | 399 | | // Do a quick search for the first element of "value". |
| | 0 | 400 | | int relativeIndex = isLetter ? |
| | 0 | 401 | | PackedSpanHelpers.PackedIndexOfIsSupported |
| | 0 | 402 | | ? PackedSpanHelpers.IndexOfAnyIgnoreCase(ref Unsafe.Add(ref searchSpace, offset), valueCharL, se |
| | 0 | 403 | | : SpanHelpers.IndexOfAnyChar(ref Unsafe.Add(ref searchSpace, offset), valueCharU, valueCharL, se |
| | 0 | 404 | | SpanHelpers.IndexOfChar(ref Unsafe.Add(ref searchSpace, offset), valueChar, searchSpaceMinusValueTai |
| | 0 | 405 | | if (relativeIndex < 0) |
| | | 406 | | { |
| | | 407 | | break; |
| | | 408 | | } |
| | | 409 | | |
| | 0 | 410 | | searchSpaceMinusValueTailLength -= relativeIndex; |
| | 0 | 411 | | if (searchSpaceMinusValueTailLength <= 0) |
| | | 412 | | { |
| | | 413 | | break; |
| | | 414 | | } |
| | 0 | 415 | | offset += relativeIndex; |
| | | 416 | | |
| | | 417 | | // Found the first element of "value". See if the tail matches. |
| | 0 | 418 | | if (valueTailLength == 0 || // for single-char values we already matched first chars |
| | 0 | 419 | | EqualsIgnoreCase( |
| | 0 | 420 | | ref Unsafe.Add(ref searchSpace, (nuint)(offset + 1)), |
| | 0 | 421 | | ref Unsafe.Add(ref valueRef, 1), valueTailLength)) |
| | | 422 | | { |
| | 0 | 423 | | return (int)offset; // The tail matched. Return a successful find. |
| | | 424 | | } |
| | | 425 | | |
| | 0 | 426 | | searchSpaceMinusValueTailLength--; |
| | 0 | 427 | | offset++; |
| | | 428 | | } |
| | 0 | 429 | | while (searchSpaceMinusValueTailLength > 0); |
| | | 430 | | |
| | 0 | 431 | | return -1; |
| | | 432 | | |
| | | 433 | | // Based on SpanHelpers.IndexOf(ref char, int, ref char, int), which was in turn based on |
| | | 434 | | // http://0x80.pl/articles/simd-strfind.html#algorithm-1-generic-simd. This version has additional |
| | | 435 | | // modifications to support case-insensitive searches. |
| | | 436 | | SearchTwoChars: |
| | | 437 | | // Both the first character in value (valueChar) and the last character in value (valueCharU) are ASCII. Get |
| | 0 | 438 | | valueChar = (char)(valueChar | 0x20); |
| | 0 | 439 | | valueCharU = (char)(valueCharU | 0x20); |
| | | 440 | | |
| | | 441 | | // The search is more efficient if the two characters being searched for are different. As long as they are |
| | | 442 | | // from the last character in the search value until we find a character that's different. Since we're deali |
| | | 443 | | // we compare the lowercase variants, as that's what we'll be comparing against in the main loop. |
| | 0 | 444 | | nint ch1ch2Distance = valueTailLength; |
| | 0 | 445 | | while (valueCharU == valueChar && ch1ch2Distance > 1) |
| | | 446 | | { |
| | 0 | 447 | | char tmp = Unsafe.Add(ref valueRef, ch1ch2Distance - 1); |
| | 0 | 448 | | if (!char.IsAscii(tmp)) |
| | | 449 | | { |
| | | 450 | | break; |
| | | 451 | | } |
| | 0 | 452 | | --ch1ch2Distance; |
| | 0 | 453 | | valueCharU = (char)(tmp | 0x20); |
| | | 454 | | } |
| | | 455 | | |
| | | 456 | | // Use Vector256 if the input is long enough. |
| | 0 | 457 | | if (Vector256.IsHardwareAccelerated && searchSpaceMinusValueTailLength - Vector256<ushort>.Count >= 0) |
| | | 458 | | { |
| | | 459 | | // Create a vector for each of the lowercase ASCII characters we're searching for. |
| | 0 | 460 | | Vector256<ushort> ch1 = Vector256.Create((ushort)valueChar); |
| | 0 | 461 | | Vector256<ushort> ch2 = Vector256.Create((ushort)valueCharU); |
| | | 462 | | |
| | 0 | 463 | | nint searchSpaceMinusValueTailLengthAndVector = searchSpaceMinusValueTailLength - (nint)Vector256<ushort |
| | | 464 | | do |
| | | 465 | | { |
| | | 466 | | // Make sure we don't go out of bounds. |
| | 0 | 467 | | Debug.Assert(offset + ch1ch2Distance + Vector256<ushort>.Count <= source.Length); |
| | | 468 | | |
| | | 469 | | // Load a vector from the current search space offset and another from the offset plus the distance |
| | | 470 | | // For each, | with 0x20 so that letters are lowercased, then & those together to get a mask. If the |
| | | 471 | | // was no match. If it wasn't, we have to do more work to check for a match. |
| | 0 | 472 | | Vector256<ushort> cmpCh2 = Vector256.Equals(ch2, Vector256.BitwiseOr(Vector256.LoadUnsafe(ref search |
| | 0 | 473 | | Vector256<ushort> cmpCh1 = Vector256.Equals(ch1, Vector256.BitwiseOr(Vector256.LoadUnsafe(ref search |
| | 0 | 474 | | Vector256<byte> cmpAnd = (cmpCh1 & cmpCh2).AsByte(); |
| | 0 | 475 | | if (cmpAnd != Vector256<byte>.Zero) |
| | | 476 | | { |
| | | 477 | | goto CandidateFound; |
| | | 478 | | } |
| | | 479 | | |
| | | 480 | | LoopFooter: |
| | | 481 | | // No match. Advance to the next vector. |
| | 0 | 482 | | offset += Vector256<ushort>.Count; |
| | | 483 | | |
| | | 484 | | // If we've reached the end of the search space, bail. |
| | 0 | 485 | | if (offset == searchSpaceMinusValueTailLength) |
| | | 486 | | { |
| | 0 | 487 | | return -1; |
| | | 488 | | } |
| | | 489 | | |
| | | 490 | | // If we're within a vector's length of the end of the search space, adjust the offset |
| | | 491 | | // to point to the last vector so that our next iteration will process it. |
| | 0 | 492 | | if (offset > searchSpaceMinusValueTailLengthAndVector) |
| | | 493 | | { |
| | 0 | 494 | | offset = searchSpaceMinusValueTailLengthAndVector; |
| | | 495 | | } |
| | | 496 | | |
| | 0 | 497 | | continue; |
| | | 498 | | |
| | | 499 | | CandidateFound: |
| | | 500 | | // Possible matches at the current location. Extract the bits for each element. |
| | | 501 | | // For each set bits, we'll check if it's a match at that location. |
| | 0 | 502 | | uint mask = cmpAnd.ExtractMostSignificantBits(); |
| | | 503 | | do |
| | | 504 | | { |
| | | 505 | | // Do a full IgnoreCase equality comparison. SpanHelpers.IndexOf skips comparing the two charact |
| | | 506 | | // but we don't actually know that the two characters are equal, since we compared with | 0x20. |
| | | 507 | | // the full string always. |
| | 0 | 508 | | nint charPos = (nint)(uint.TrailingZeroCount(mask) / sizeof(ushort)); |
| | 0 | 509 | | if (EqualsIgnoreCase(ref Unsafe.Add(ref searchSpace, offset + charPos), ref valueRef, value.Leng |
| | | 510 | | { |
| | | 511 | | // Match! Return the index. |
| | 0 | 512 | | return (int)(offset + charPos); |
| | | 513 | | } |
| | | 514 | | |
| | | 515 | | // Clear the two lowest set bits in the mask. If there are no more set bits, we're done. |
| | | 516 | | // If any remain, we loop around to do the next comparison. |
| | 0 | 517 | | mask = BitOperations.ResetLowestSetBit(BitOperations.ResetLowestSetBit(mask)); |
| | 0 | 518 | | } while (mask != 0); |
| | 0 | 519 | | goto LoopFooter; |
| | | 520 | | |
| | | 521 | | } while (true); |
| | | 522 | | } |
| | | 523 | | else // 128bit vector path (SSE2 or AdvSimd) |
| | | 524 | | { |
| | | 525 | | // Create a vector for each of the lowercase ASCII characters we're searching for. |
| | 0 | 526 | | Vector128<ushort> ch1 = Vector128.Create((ushort)valueChar); |
| | 0 | 527 | | Vector128<ushort> ch2 = Vector128.Create((ushort)valueCharU); |
| | | 528 | | |
| | 0 | 529 | | nint searchSpaceMinusValueTailLengthAndVector = searchSpaceMinusValueTailLength - (nint)Vector128<ushort |
| | | 530 | | do |
| | | 531 | | { |
| | | 532 | | // Make sure we don't go out of bounds. |
| | 0 | 533 | | Debug.Assert(offset + ch1ch2Distance + Vector128<ushort>.Count <= source.Length); |
| | | 534 | | |
| | | 535 | | // Load a vector from the current search space offset and another from the offset plus the distance |
| | | 536 | | // For each, | with 0x20 so that letters are lowercased, then & those together to get a mask. If the |
| | | 537 | | // was no match. If it wasn't, we have to do more work to check for a match. |
| | 0 | 538 | | Vector128<ushort> cmpCh2 = Vector128.Equals(ch2, Vector128.BitwiseOr(Vector128.LoadUnsafe(ref search |
| | 0 | 539 | | Vector128<ushort> cmpCh1 = Vector128.Equals(ch1, Vector128.BitwiseOr(Vector128.LoadUnsafe(ref search |
| | 0 | 540 | | Vector128<byte> cmpAnd = (cmpCh1 & cmpCh2).AsByte(); |
| | 0 | 541 | | if (cmpAnd != Vector128<byte>.Zero) |
| | | 542 | | { |
| | | 543 | | goto CandidateFound; |
| | | 544 | | } |
| | | 545 | | |
| | | 546 | | LoopFooter: |
| | | 547 | | // No match. Advance to the next vector. |
| | 0 | 548 | | offset += Vector128<ushort>.Count; |
| | | 549 | | |
| | | 550 | | // If we've reached the end of the search space, bail. |
| | 0 | 551 | | if (offset == searchSpaceMinusValueTailLength) |
| | | 552 | | { |
| | 0 | 553 | | return -1; |
| | | 554 | | } |
| | | 555 | | |
| | | 556 | | // If we're within a vector's length of the end of the search space, adjust the offset |
| | | 557 | | // to point to the last vector so that our next iteration will process it. |
| | 0 | 558 | | if (offset > searchSpaceMinusValueTailLengthAndVector) |
| | | 559 | | { |
| | 0 | 560 | | offset = searchSpaceMinusValueTailLengthAndVector; |
| | | 561 | | } |
| | | 562 | | |
| | 0 | 563 | | continue; |
| | | 564 | | |
| | | 565 | | CandidateFound: |
| | | 566 | | // Possible matches at the current location. Extract the bits for each element. |
| | | 567 | | // For each set bits, we'll check if it's a match at that location. |
| | 0 | 568 | | uint mask = cmpAnd.ExtractMostSignificantBits(); |
| | | 569 | | do |
| | | 570 | | { |
| | | 571 | | // Do a full IgnoreCase equality comparison. SpanHelpers.IndexOf skips comparing the two charact |
| | | 572 | | // but we don't actually know that the two characters are equal, since we compared with | 0x20. |
| | | 573 | | // the full string always. |
| | 0 | 574 | | nint charPos = (nint)(uint.TrailingZeroCount(mask) / sizeof(ushort)); |
| | 0 | 575 | | if (EqualsIgnoreCase(ref Unsafe.Add(ref searchSpace, offset + charPos), ref valueRef, value.Leng |
| | | 576 | | { |
| | | 577 | | // Match! Return the index. |
| | 0 | 578 | | return (int)(offset + charPos); |
| | | 579 | | } |
| | | 580 | | |
| | | 581 | | // Clear the two lowest set bits in the mask. If there are no more set bits, we're done. |
| | | 582 | | // If any remain, we loop around to do the next comparison. |
| | 0 | 583 | | mask = BitOperations.ResetLowestSetBit(BitOperations.ResetLowestSetBit(mask)); |
| | 0 | 584 | | } while (mask != 0); |
| | 0 | 585 | | goto LoopFooter; |
| | | 586 | | |
| | | 587 | | } while (true); |
| | | 588 | | } |
| | | 589 | | } |
| | | 590 | | |
| | | 591 | | internal static int LastIndexOf(string source, string value, int startIndex, int count) |
| | | 592 | | { |
| | | 593 | | int result = source.AsSpan(startIndex, count).LastIndexOf(value); |
| | | 594 | | if (result >= 0) { result += startIndex; } // if match found, adjust 'result' by the actual start position |
| | | 595 | | return result; |
| | | 596 | | } |
| | | 597 | | |
| | | 598 | | internal static int LastIndexOf(string source, string value, int startIndex, int count, bool ignoreCase) |
| | | 599 | | { |
| | | 600 | | if (source == null) |
| | | 601 | | { |
| | | 602 | | ThrowHelper.ThrowArgumentNullException(ExceptionArgument.source); |
| | | 603 | | } |
| | | 604 | | |
| | | 605 | | if (value == null) |
| | | 606 | | { |
| | | 607 | | ThrowHelper.ThrowArgumentNullException(ExceptionArgument.value); |
| | | 608 | | } |
| | | 609 | | |
| | | 610 | | if (value.Length == 0) |
| | | 611 | | { |
| | | 612 | | return startIndex + 1; // startIndex is the index of the last char to include in the search space |
| | | 613 | | } |
| | | 614 | | |
| | | 615 | | if (count == 0) |
| | | 616 | | { |
| | | 617 | | return -1; |
| | | 618 | | } |
| | | 619 | | |
| | | 620 | | if (GlobalizationMode.Invariant) |
| | | 621 | | { |
| | | 622 | | return ignoreCase ? InvariantModeCasing.LastIndexOfIgnoreCase(source.AsSpan(startIndex, count), value) : |
| | | 623 | | } |
| | | 624 | | |
| | | 625 | | if (GlobalizationMode.UseNls) |
| | | 626 | | { |
| | | 627 | | return CompareInfo.NlsLastIndexOfOrdinalCore(source, value, startIndex, count, ignoreCase); |
| | | 628 | | } |
| | | 629 | | |
| | | 630 | | if (!ignoreCase) |
| | | 631 | | { |
| | | 632 | | LastIndexOf(source, value, startIndex, count); |
| | | 633 | | } |
| | | 634 | | |
| | | 635 | | if (!source.TryGetSpan(startIndex, count, out ReadOnlySpan<char> sourceSpan)) |
| | | 636 | | { |
| | | 637 | | // Bounds check failed - figure out exactly what went wrong so that we can |
| | | 638 | | // surface the correct argument exception. |
| | | 639 | | |
| | | 640 | | if ((uint)startIndex > (uint)source.Length) |
| | | 641 | | { |
| | | 642 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.startIndex, ExceptionResource.Argumen |
| | | 643 | | } |
| | | 644 | | else |
| | | 645 | | { |
| | | 646 | | ThrowHelper.ThrowArgumentOutOfRangeException(ExceptionArgument.count, ExceptionResource.ArgumentOutO |
| | | 647 | | } |
| | | 648 | | } |
| | | 649 | | |
| | | 650 | | int result = OrdinalCasing.LastIndexOf(sourceSpan, value); |
| | | 651 | | |
| | | 652 | | if (result >= 0) |
| | | 653 | | { |
| | | 654 | | result += startIndex; |
| | | 655 | | } |
| | | 656 | | return result; |
| | | 657 | | } |
| | | 658 | | |
| | | 659 | | internal static int LastIndexOfOrdinalIgnoreCase(ReadOnlySpan<char> source, ReadOnlySpan<char> value) |
| | | 660 | | { |
| | 0 | 661 | | if (value.Length == 0) |
| | | 662 | | { |
| | 0 | 663 | | return source.Length; |
| | | 664 | | } |
| | | 665 | | |
| | 0 | 666 | | if (value.Length > source.Length) |
| | | 667 | | { |
| | | 668 | | // A non-linguistic search compares chars directly against one another, so large |
| | | 669 | | // target strings can never be found inside small search spaces. This check also |
| | | 670 | | // handles empty 'source' spans. |
| | | 671 | | |
| | 0 | 672 | | return -1; |
| | | 673 | | } |
| | | 674 | | |
| | 0 | 675 | | if (GlobalizationMode.Invariant) |
| | | 676 | | { |
| | 0 | 677 | | return InvariantModeCasing.LastIndexOfIgnoreCase(source, value); |
| | | 678 | | } |
| | | 679 | | |
| | 0 | 680 | | if (GlobalizationMode.UseNls) |
| | | 681 | | { |
| | 0 | 682 | | return CompareInfo.NlsIndexOfOrdinalCore(source, value, ignoreCase: true, fromBeginning: false); |
| | | 683 | | } |
| | | 684 | | |
| | 0 | 685 | | return OrdinalCasing.LastIndexOf(source, value); |
| | | 686 | | } |
| | | 687 | | |
| | | 688 | | internal static int ToUpperOrdinal(ReadOnlySpan<char> source, Span<char> destination) |
| | | 689 | | { |
| | 0 | 690 | | if (source.Overlaps(destination)) |
| | 0 | 691 | | ThrowHelper.ThrowInvalidOperationException(ExceptionResource.InvalidOperation_SpanOverlappedOperation); |
| | | 692 | | |
| | | 693 | | // Assuming that changing case does not affect length |
| | 0 | 694 | | if (destination.Length < source.Length) |
| | 0 | 695 | | return -1; |
| | | 696 | | |
| | 0 | 697 | | if (GlobalizationMode.Invariant) |
| | | 698 | | { |
| | 0 | 699 | | InvariantModeCasing.ToUpper(source, destination); |
| | 0 | 700 | | return source.Length; |
| | | 701 | | } |
| | | 702 | | |
| | 0 | 703 | | if (GlobalizationMode.UseNls) |
| | | 704 | | { |
| | 0 | 705 | | ChangeCaseNlsOrdinal(source, destination, toUpper: true); |
| | 0 | 706 | | return source.Length; |
| | | 707 | | } |
| | | 708 | | |
| | 0 | 709 | | OrdinalCasing.ToUpperOrdinal(source, destination); |
| | 0 | 710 | | return source.Length; |
| | | 711 | | } |
| | | 712 | | |
| | | 713 | | internal static int ToLowerOrdinal(ReadOnlySpan<char> source, Span<char> destination) |
| | | 714 | | { |
| | 0 | 715 | | if (source.Overlaps(destination)) |
| | 0 | 716 | | ThrowHelper.ThrowInvalidOperationException(ExceptionResource.InvalidOperation_SpanOverlappedOperation); |
| | | 717 | | |
| | | 718 | | // Assuming that changing case does not affect length |
| | 0 | 719 | | if (destination.Length < source.Length) |
| | 0 | 720 | | return -1; |
| | | 721 | | |
| | 0 | 722 | | if (GlobalizationMode.Invariant) |
| | | 723 | | { |
| | 0 | 724 | | InvariantModeCasing.ToLower(source, destination); |
| | 0 | 725 | | PreserveOrdinalLowerCasingClass(source, destination); |
| | 0 | 726 | | return source.Length; |
| | | 727 | | } |
| | | 728 | | |
| | 0 | 729 | | if (GlobalizationMode.UseNls) |
| | | 730 | | { |
| | 0 | 731 | | ChangeCaseNlsOrdinal(source, destination, toUpper: false); |
| | 0 | 732 | | PreserveOrdinalLowerCasingClass(source, destination); |
| | 0 | 733 | | return source.Length; |
| | | 734 | | } |
| | | 735 | | |
| | 0 | 736 | | OrdinalCasing.ToLowerOrdinal(source, destination); |
| | 0 | 737 | | return source.Length; |
| | | 738 | | } |
| | | 739 | | |
| | | 740 | | private static void ChangeCaseNlsOrdinal(ReadOnlySpan<char> source, Span<char> destination, bool toUpper) |
| | | 741 | | { |
| | 0 | 742 | | Debug.Assert(GlobalizationMode.UseNls); |
| | | 743 | | |
| | 0 | 744 | | OperationStatus operationStatus = toUpper |
| | 0 | 745 | | ? Ascii.ToUpper(source, destination, out int charsConsumed) |
| | 0 | 746 | | : Ascii.ToLower(source, destination, out charsConsumed); |
| | | 747 | | |
| | 0 | 748 | | if (operationStatus != OperationStatus.InvalidData) |
| | | 749 | | { |
| | 0 | 750 | | Debug.Assert(operationStatus == OperationStatus.Done); |
| | 0 | 751 | | return; |
| | | 752 | | } |
| | | 753 | | |
| | 0 | 754 | | source = source.Slice(charsConsumed); |
| | 0 | 755 | | destination = destination.Slice(charsConsumed); |
| | | 756 | | |
| | 0 | 757 | | if (toUpper) |
| | | 758 | | { |
| | 0 | 759 | | TextInfo.Invariant.ChangeCaseToUpper(source, destination); |
| | | 760 | | } |
| | | 761 | | else |
| | | 762 | | { |
| | 0 | 763 | | TextInfo.Invariant.ChangeCaseToLower(source, destination); |
| | | 764 | | } |
| | | 765 | | |
| | 0 | 766 | | PreserveNlsOrdinalCasingClass(source, destination); |
| | 0 | 767 | | } |
| | | 768 | | |
| | | 769 | | private static void PreserveNlsOrdinalCasingClass(ReadOnlySpan<char> source, Span<char> destination) |
| | | 770 | | { |
| | 0 | 771 | | Debug.Assert(GlobalizationMode.UseNls); |
| | 0 | 772 | | Debug.Assert(destination.Length >= source.Length); |
| | | 773 | | |
| | 0 | 774 | | destination = destination.Slice(0, source.Length); |
| | | 775 | | |
| | 0 | 776 | | int i = source.IndexOfAnyInRange('\uD800', '\uDBFF'); |
| | 0 | 777 | | if (i < 0) |
| | | 778 | | { |
| | 0 | 779 | | return; |
| | | 780 | | } |
| | | 781 | | |
| | 0 | 782 | | while (i < source.Length - 1) |
| | | 783 | | { |
| | 0 | 784 | | if (char.IsLowSurrogate(source[i + 1])) |
| | | 785 | | { |
| | 0 | 786 | | if (source[i] != destination[i] || source[i + 1] != destination[i + 1]) |
| | | 787 | | { |
| | 0 | 788 | | if (CompareInfo.NlsCompareStringOrdinalIgnoreCase( |
| | 0 | 789 | | ref MemoryMarshal.GetReference(source.Slice(i)), 2, |
| | 0 | 790 | | ref MemoryMarshal.GetReference(destination.Slice(i)), 2) != 0) |
| | | 791 | | { |
| | 0 | 792 | | destination[i] = source[i]; |
| | 0 | 793 | | destination[i + 1] = source[i + 1]; |
| | | 794 | | } |
| | | 795 | | |
| | 0 | 796 | | i += 2; |
| | | 797 | | } |
| | | 798 | | else |
| | | 799 | | { |
| | 0 | 800 | | i += source.Slice(i).CommonPrefixLength(destination.Slice(i)); |
| | | 801 | | |
| | 0 | 802 | | if (i < source.Length && char.IsLowSurrogate(source[i]) && char.IsHighSurrogate(source[i - 1])) |
| | | 803 | | { |
| | 0 | 804 | | i--; |
| | | 805 | | } |
| | | 806 | | } |
| | | 807 | | } |
| | | 808 | | else |
| | | 809 | | { |
| | 0 | 810 | | i++; |
| | | 811 | | } |
| | | 812 | | |
| | 0 | 813 | | int next = source.Slice(i).IndexOfAnyInRange('\uD800', '\uDBFF'); |
| | 0 | 814 | | if (next < 0) |
| | | 815 | | { |
| | 0 | 816 | | return; |
| | | 817 | | } |
| | | 818 | | |
| | 0 | 819 | | i += next; |
| | | 820 | | } |
| | 0 | 821 | | } |
| | | 822 | | |
| | | 823 | | internal static uint PreserveNlsOrdinalCasingClass(uint source, uint destination) |
| | | 824 | | { |
| | 0 | 825 | | Debug.Assert(GlobalizationMode.UseNls); |
| | 0 | 826 | | Debug.Assert(source > char.MaxValue); |
| | 0 | 827 | | Debug.Assert(destination > char.MaxValue); |
| | | 828 | | |
| | 0 | 829 | | if (source == destination) |
| | | 830 | | { |
| | 0 | 831 | | return destination; |
| | | 832 | | } |
| | | 833 | | |
| | 0 | 834 | | Span<char> sourceChars = stackalloc char[2]; |
| | 0 | 835 | | Span<char> destinationChars = stackalloc char[2]; |
| | 0 | 836 | | UnicodeUtility.GetUtf16SurrogatesFromSupplementaryPlaneScalar(source, out sourceChars[0], out sourceChars[1] |
| | 0 | 837 | | UnicodeUtility.GetUtf16SurrogatesFromSupplementaryPlaneScalar(destination, out destinationChars[0], out dest |
| | | 838 | | |
| | 0 | 839 | | return CompareInfo.NlsCompareStringOrdinalIgnoreCase( |
| | 0 | 840 | | ref MemoryMarshal.GetReference(sourceChars), sourceChars.Length, |
| | 0 | 841 | | ref MemoryMarshal.GetReference(destinationChars), destinationChars.Length) == 0 |
| | 0 | 842 | | ? destination |
| | 0 | 843 | | : source; |
| | | 844 | | } |
| | | 845 | | |
| | | 846 | | // The only BMP scalars whose simple invariant/NLS lower mapping moves them out of their ordinal |
| | | 847 | | // upper-casing class. Each one is its own ordinal upper-casing (ToUpperOrdinal(c) == c) yet lowers to a |
| | | 848 | | // different letter, so simple lowering would break OrdinalIgnoreCase consistency for them. The ICU ordinal |
| | | 849 | | // table already keeps them unchanged; invariant and NLS lowering do not, so they must be restored. |
| | | 850 | | // U+03F4 GREEK CAPITAL THETA SYMBOL, U+1E9E LATIN CAPITAL LETTER SHARP S, U+2126 OHM SIGN, |
| | | 851 | | // U+212A KELVIN SIGN, U+212B ANGSTROM SIGN. The ToLowerOrdinal_EveryBmpChar tests would fail if this set |
| | | 852 | | // ever became incomplete. |
| | 0 | 853 | | private static readonly SearchValues<char> s_ordinalLowerCasingClassRestoreChars = |
| | 0 | 854 | | SearchValues.Create("\u03F4\u1E9E\u2126\u212A\u212B"); |
| | | 855 | | |
| | | 856 | | // Ordinal lower casing must never move a character out of its ordinal upper-casing class, otherwise it |
| | | 857 | | // would stop being consistent with OrdinalIgnoreCase (for example the Kelvin, Ohm and Angstrom signs). |
| | | 858 | | // Only the fixed set above is ever affected, so skip the pass entirely unless one of them is present. |
| | | 859 | | private static void PreserveOrdinalLowerCasingClass(ReadOnlySpan<char> source, Span<char> destination) |
| | | 860 | | { |
| | 0 | 861 | | int i = source.IndexOfAny(s_ordinalLowerCasingClassRestoreChars); |
| | 0 | 862 | | if (i < 0) |
| | | 863 | | { |
| | 0 | 864 | | return; |
| | | 865 | | } |
| | | 866 | | |
| | | 867 | | // Restore each affected character to its original form; its ordinal lower casing is itself. |
| | 0 | 868 | | for (; i < source.Length; i++) |
| | | 869 | | { |
| | 0 | 870 | | if (s_ordinalLowerCasingClassRestoreChars.Contains(source[i])) |
| | | 871 | | { |
| | 0 | 872 | | destination[i] = source[i]; |
| | | 873 | | } |
| | | 874 | | } |
| | 0 | 875 | | } |
| | | 876 | | } |
| | | 877 | | } |
| | | 878 | | |