| | | 1 | | // Licensed to the .NET Foundation under one or more agreements. |
| | | 2 | | // The .NET Foundation licenses this file to you under the MIT license. |
| | | 3 | | |
| | | 4 | | using System.Buffers; |
| | | 5 | | using System.Diagnostics; |
| | | 6 | | using System.Diagnostics.CodeAnalysis; |
| | | 7 | | using System.Runtime.CompilerServices; |
| | | 8 | | using System.Runtime.InteropServices; |
| | | 9 | | using System.Runtime.Serialization; |
| | | 10 | | using System.Text; |
| | | 11 | | using System.Text.Unicode; |
| | | 12 | | |
| | | 13 | | namespace System.Globalization |
| | | 14 | | { |
| | | 15 | | /// <summary> |
| | | 16 | | /// This Class defines behaviors specific to a writing system. |
| | | 17 | | /// A writing system is the collection of scripts and orthographic rules |
| | | 18 | | /// required to represent a language as text. |
| | | 19 | | /// </summary> |
| | | 20 | | public sealed partial class TextInfo : ICloneable, IDeserializationCallback |
| | | 21 | | { |
| | | 22 | | private bool _isReadOnly; |
| | | 23 | | |
| | | 24 | | private readonly string _cultureName; |
| | | 25 | | private readonly CultureData _cultureData; |
| | | 26 | | |
| | 0 | 27 | | private bool HasEmptyCultureName { get { return _cultureName.Length == 0; } } |
| | | 28 | | |
| | | 29 | | // // Name of the text info we're using (ie: _cultureData.TextInfoName) |
| | | 30 | | private readonly string _textInfoName; |
| | | 31 | | |
| | | 32 | | private NullableBool _isAsciiCasingSameAsInvariant; |
| | | 33 | | |
| | | 34 | | // Invariant text info |
| | 0 | 35 | | internal static readonly TextInfo Invariant = new TextInfo(CultureData.Invariant, readOnly: true) { _isAsciiCasi |
| | | 36 | | |
| | 0 | 37 | | internal TextInfo(CultureData cultureData) |
| | | 38 | | { |
| | | 39 | | // This is our primary data source, we don't need most of the rest of this |
| | 0 | 40 | | _cultureData = cultureData; |
| | 0 | 41 | | _cultureName = _cultureData.CultureName; |
| | 0 | 42 | | _textInfoName = _cultureData.TextInfoName; |
| | | 43 | | |
| | 0 | 44 | | if (GlobalizationMode.UseNls) |
| | | 45 | | { |
| | 0 | 46 | | _sortHandle = CompareInfo.NlsGetSortHandle(_textInfoName); |
| | | 47 | | } |
| | 0 | 48 | | } |
| | | 49 | | |
| | | 50 | | private TextInfo(CultureData cultureData, bool readOnly) |
| | 0 | 51 | | : this(cultureData) |
| | | 52 | | { |
| | 0 | 53 | | SetReadOnlyState(readOnly); |
| | 0 | 54 | | } |
| | | 55 | | |
| | | 56 | | void IDeserializationCallback.OnDeserialization(object? sender) |
| | | 57 | | { |
| | 0 | 58 | | throw new PlatformNotSupportedException(); |
| | | 59 | | } |
| | | 60 | | |
| | 0 | 61 | | public int ANSICodePage => _cultureData.ANSICodePage; |
| | | 62 | | |
| | 0 | 63 | | public int OEMCodePage => _cultureData.OEMCodePage; |
| | | 64 | | |
| | 0 | 65 | | public int MacCodePage => _cultureData.MacCodePage; |
| | | 66 | | |
| | 0 | 67 | | public int EBCDICCodePage => _cultureData.EBCDICCodePage; |
| | | 68 | | |
| | | 69 | | // Just use the LCID from our text info name |
| | 0 | 70 | | public int LCID => CultureInfo.GetCultureInfo(_textInfoName).LCID; |
| | | 71 | | |
| | 0 | 72 | | public string CultureName => _textInfoName; |
| | | 73 | | |
| | 0 | 74 | | public bool IsReadOnly => _isReadOnly; |
| | | 75 | | |
| | | 76 | | public object Clone() |
| | | 77 | | { |
| | 0 | 78 | | object o = MemberwiseClone(); |
| | 0 | 79 | | ((TextInfo)o).SetReadOnlyState(false); |
| | 0 | 80 | | return o; |
| | | 81 | | } |
| | | 82 | | |
| | | 83 | | /// <summary> |
| | | 84 | | /// Create a cloned readonly instance or return the input one if it is |
| | | 85 | | /// readonly. |
| | | 86 | | /// </summary> |
| | | 87 | | public static TextInfo ReadOnly(TextInfo textInfo) |
| | | 88 | | { |
| | 0 | 89 | | ArgumentNullException.ThrowIfNull(textInfo); |
| | | 90 | | |
| | 0 | 91 | | if (textInfo.IsReadOnly) |
| | | 92 | | { |
| | 0 | 93 | | return textInfo; |
| | | 94 | | } |
| | | 95 | | |
| | 0 | 96 | | TextInfo clonedTextInfo = (TextInfo)(textInfo.MemberwiseClone()); |
| | 0 | 97 | | clonedTextInfo.SetReadOnlyState(true); |
| | 0 | 98 | | return clonedTextInfo; |
| | | 99 | | } |
| | | 100 | | |
| | | 101 | | private void VerifyWritable() |
| | | 102 | | { |
| | 0 | 103 | | if (_isReadOnly) |
| | | 104 | | { |
| | 0 | 105 | | throw new InvalidOperationException(SR.InvalidOperation_ReadOnly); |
| | | 106 | | } |
| | 0 | 107 | | } |
| | | 108 | | |
| | | 109 | | internal void SetReadOnlyState(bool readOnly) |
| | | 110 | | { |
| | 0 | 111 | | _isReadOnly = readOnly; |
| | 0 | 112 | | } |
| | | 113 | | |
| | | 114 | | /// <summary> |
| | | 115 | | /// Returns the string used to separate items in a list. |
| | | 116 | | /// </summary> |
| | | 117 | | public string ListSeparator |
| | | 118 | | { |
| | 0 | 119 | | get => field ??= _cultureData.ListSeparator; |
| | | 120 | | set |
| | | 121 | | { |
| | 0 | 122 | | ArgumentNullException.ThrowIfNull(value); |
| | | 123 | | |
| | 0 | 124 | | VerifyWritable(); |
| | 0 | 125 | | field = value; |
| | 0 | 126 | | } |
| | | 127 | | } |
| | | 128 | | |
| | | 129 | | /// <summary> |
| | | 130 | | /// Converts the character or string to lower case. Certain locales |
| | | 131 | | /// have different casing semantics from the file systems in Win32. |
| | | 132 | | /// </summary> |
| | | 133 | | public char ToLower(char c) |
| | | 134 | | { |
| | 0 | 135 | | if (GlobalizationMode.Invariant) |
| | | 136 | | { |
| | 0 | 137 | | return InvariantModeCasing.ToLower(c); |
| | | 138 | | } |
| | | 139 | | |
| | 0 | 140 | | if (UnicodeUtility.IsAsciiCodePoint(c) && IsAsciiCasingSameAsInvariant) |
| | | 141 | | { |
| | 0 | 142 | | return ToLowerAsciiInvariant(c); |
| | | 143 | | } |
| | | 144 | | |
| | 0 | 145 | | return ChangeCase(c, toUpper: false); |
| | | 146 | | } |
| | | 147 | | |
| | | 148 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 149 | | internal static char ToLowerInvariant(char c) |
| | | 150 | | { |
| | 0 | 151 | | if (UnicodeUtility.IsAsciiCodePoint(c)) |
| | | 152 | | { |
| | 0 | 153 | | return ToLowerAsciiInvariant(c); |
| | | 154 | | } |
| | | 155 | | |
| | 0 | 156 | | if (GlobalizationMode.Invariant) |
| | | 157 | | { |
| | 0 | 158 | | return InvariantModeCasing.ToLower(c); |
| | | 159 | | } |
| | | 160 | | |
| | 0 | 161 | | return Invariant.ChangeCase(c, toUpper: false); |
| | | 162 | | } |
| | | 163 | | |
| | | 164 | | public string ToLower(string str) |
| | | 165 | | { |
| | 0 | 166 | | ArgumentNullException.ThrowIfNull(str); |
| | 0 | 167 | | return ChangeCaseCommon<ToLowerConversion>(this, str); |
| | | 168 | | } |
| | | 169 | | |
| | | 170 | | internal static string ToLowerInvariant(string str) |
| | | 171 | | { |
| | 0 | 172 | | ArgumentNullException.ThrowIfNull(str); |
| | 0 | 173 | | return ChangeCaseCommon<ToLowerConversion>(null, str); |
| | | 174 | | } |
| | | 175 | | |
| | | 176 | | internal void ToLower(ReadOnlySpan<char> source, Span<char> destination) |
| | | 177 | | { |
| | 0 | 178 | | ChangeCaseCommon<ToLowerConversion>(this, source, destination); |
| | 0 | 179 | | } |
| | | 180 | | |
| | | 181 | | private unsafe char ChangeCase(char c, bool toUpper) |
| | | 182 | | { |
| | 0 | 183 | | Debug.Assert(!GlobalizationMode.Invariant); |
| | 0 | 184 | | char dst = default; |
| | 0 | 185 | | ChangeCaseCore(&c, 1, &dst, 1, toUpper); |
| | 0 | 186 | | return dst; |
| | | 187 | | } |
| | | 188 | | |
| | | 189 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 190 | | internal static char ToUpperOrdinal(char c) |
| | | 191 | | { |
| | 0 | 192 | | if (GlobalizationMode.Invariant) |
| | | 193 | | { |
| | 0 | 194 | | return InvariantModeCasing.ToUpper(c); |
| | | 195 | | } |
| | | 196 | | |
| | 0 | 197 | | if (GlobalizationMode.UseNls) |
| | | 198 | | { |
| | 0 | 199 | | return char.IsAscii(c) |
| | 0 | 200 | | ? ToUpperAsciiInvariant(c) |
| | 0 | 201 | | : Invariant.ChangeCase(c, toUpper: true); |
| | | 202 | | } |
| | | 203 | | |
| | 0 | 204 | | return OrdinalCasing.ToUpper(c); |
| | | 205 | | } |
| | | 206 | | |
| | | 207 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 208 | | internal static char ToLowerOrdinal(char c) |
| | | 209 | | { |
| | 0 | 210 | | if (GlobalizationMode.Invariant) |
| | | 211 | | { |
| | 0 | 212 | | return char.IsAscii(c) |
| | 0 | 213 | | ? ToLowerAsciiInvariant(c) |
| | 0 | 214 | | : PreserveOrdinalLowerCasingClass(c, InvariantModeCasing.ToLower(c)); |
| | | 215 | | } |
| | | 216 | | |
| | 0 | 217 | | if (GlobalizationMode.UseNls) |
| | | 218 | | { |
| | 0 | 219 | | return char.IsAscii(c) |
| | 0 | 220 | | ? ToLowerAsciiInvariant(c) |
| | 0 | 221 | | : PreserveOrdinalLowerCasingClass(c, Invariant.ChangeCase(c, toUpper: false)); |
| | | 222 | | } |
| | | 223 | | |
| | 0 | 224 | | return OrdinalCasing.ToLower(c); |
| | | 225 | | } |
| | | 226 | | |
| | | 227 | | // Ordinal lower casing must never move a character out of its ordinal upper-casing class, otherwise it |
| | | 228 | | // would stop being consistent with OrdinalIgnoreCase (for example the Kelvin, Ohm and Angstrom signs). The |
| | | 229 | | // ICU ordinal table encodes this directly, but invariant and NLS simple lowering do not, so keep the original |
| | | 230 | | // character whenever its simple lower mapping would change its ordinal upper-casing form. |
| | | 231 | | private static char PreserveOrdinalLowerCasingClass(char c, char lower) => |
| | 0 | 232 | | lower == c || ToUpperOrdinal(lower) == ToUpperOrdinal(c) ? lower : c; |
| | | 233 | | |
| | | 234 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 235 | | internal void ChangeCaseToLower(ReadOnlySpan<char> source, Span<char> destination) |
| | | 236 | | { |
| | 0 | 237 | | Debug.Assert(destination.Length >= source.Length); |
| | 0 | 238 | | ChangeCaseCommon<ToLowerConversion>(this, source, destination); |
| | 0 | 239 | | } |
| | | 240 | | |
| | | 241 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 242 | | internal void ChangeCaseToUpper(ReadOnlySpan<char> source, Span<char> destination) |
| | | 243 | | { |
| | 0 | 244 | | Debug.Assert(destination.Length >= source.Length); |
| | 0 | 245 | | ChangeCaseCommon<ToUpperConversion>(this, source, destination); |
| | 0 | 246 | | } |
| | | 247 | | |
| | | 248 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 249 | | private static unsafe void ChangeCaseCommon<TConversion>(TextInfo? instance, ReadOnlySpan<char> source, Span<cha |
| | | 250 | | { |
| | 0 | 251 | | Debug.Assert(typeof(TConversion) == typeof(ToUpperConversion) || typeof(TConversion) == typeof(ToLowerConver |
| | | 252 | | |
| | 0 | 253 | | if (source.IsEmpty) |
| | | 254 | | { |
| | 0 | 255 | | return; |
| | | 256 | | } |
| | | 257 | | |
| | 0 | 258 | | bool toUpper = typeof(TConversion) == typeof(ToUpperConversion); // JIT will treat this as a constant in rel |
| | 0 | 259 | | int charsConsumed = 0; |
| | | 260 | | |
| | | 261 | | // instance being null indicates the invariant culture where IsAsciiCasingSameAsInvariant is always true. |
| | 0 | 262 | | if (instance == null || instance.IsAsciiCasingSameAsInvariant) |
| | | 263 | | { |
| | 0 | 264 | | OperationStatus operationStatus = toUpper |
| | 0 | 265 | | ? Ascii.ToUpper(source, destination, out charsConsumed) |
| | 0 | 266 | | : Ascii.ToLower(source, destination, out charsConsumed); |
| | | 267 | | |
| | 0 | 268 | | if (operationStatus != OperationStatus.InvalidData) |
| | | 269 | | { |
| | 0 | 270 | | Debug.Assert(operationStatus == OperationStatus.Done); |
| | 0 | 271 | | return; |
| | | 272 | | } |
| | | 273 | | } |
| | | 274 | | |
| | 0 | 275 | | if (GlobalizationMode.Invariant) |
| | | 276 | | { |
| | 0 | 277 | | if (toUpper) |
| | | 278 | | { |
| | 0 | 279 | | InvariantModeCasing.ToUpper(source, destination); |
| | | 280 | | } |
| | | 281 | | else |
| | | 282 | | { |
| | 0 | 283 | | InvariantModeCasing.ToLower(source, destination); |
| | | 284 | | } |
| | 0 | 285 | | return; |
| | | 286 | | } |
| | | 287 | | |
| | | 288 | | // instance being null means it's Invariant |
| | 0 | 289 | | instance ??= Invariant; |
| | | 290 | | |
| | 0 | 291 | | fixed (char* pSource = &MemoryMarshal.GetReference(source)) |
| | 0 | 292 | | fixed (char* pDestination = &MemoryMarshal.GetReference(destination)) |
| | | 293 | | { |
| | 0 | 294 | | instance.ChangeCaseCore(pSource + charsConsumed, source.Length - charsConsumed, |
| | 0 | 295 | | pDestination + charsConsumed, destination.Length - charsConsumed, toUpper); |
| | | 296 | | } |
| | 0 | 297 | | } |
| | | 298 | | |
| | | 299 | | private static unsafe string ChangeCaseCommon<TConversion>(TextInfo? instance, string source) where TConversion |
| | | 300 | | { |
| | 0 | 301 | | Debug.Assert(typeof(TConversion) == typeof(ToUpperConversion) || typeof(TConversion) == typeof(ToLowerConver |
| | 0 | 302 | | bool toUpper = typeof(TConversion) == typeof(ToUpperConversion); // JIT will treat this as a constant in rel |
| | | 303 | | |
| | 0 | 304 | | Debug.Assert(source != null); |
| | | 305 | | |
| | | 306 | | // If the string is empty, we're done. |
| | 0 | 307 | | if (source.Length == 0) |
| | | 308 | | { |
| | 0 | 309 | | return string.Empty; |
| | | 310 | | } |
| | | 311 | | |
| | 0 | 312 | | fixed (char* pSource = source) |
| | | 313 | | { |
| | 0 | 314 | | nuint currIdx = 0; // in chars |
| | | 315 | | |
| | | 316 | | // If this culture's casing for ASCII is the same as invariant, try to take |
| | | 317 | | // a fast path that'll work in managed code and ASCII rather than calling out |
| | | 318 | | // to the OS for culture-aware casing. |
| | | 319 | | // |
| | | 320 | | // instance being null indicates the invariant culture where IsAsciiCasingSameAsInvariant is always true |
| | 0 | 321 | | if (instance == null || instance.IsAsciiCasingSameAsInvariant) |
| | | 322 | | { |
| | | 323 | | // Read 2 chars (one 32-bit integer) at a time |
| | | 324 | | |
| | 0 | 325 | | if (source.Length >= 2) |
| | | 326 | | { |
| | 0 | 327 | | nuint lastIndexWhereCanReadTwoChars = (uint)source.Length - 2; |
| | | 328 | | do |
| | | 329 | | { |
| | | 330 | | // See the comments in ChangeCaseCommon<TConversion>(ROS<char>, Span<char>) for a full expla |
| | | 331 | | |
| | 0 | 332 | | uint tempValue = Unsafe.ReadUnaligned<uint>(pSource + currIdx); |
| | 0 | 333 | | if (!Utf16Utility.AllCharsInUInt32AreAscii(tempValue)) |
| | | 334 | | { |
| | | 335 | | goto NotAscii; |
| | | 336 | | } |
| | 0 | 337 | | if ((toUpper) ? Utf16Utility.UInt32ContainsAnyLowercaseAsciiChar(tempValue) : Utf16Utility.U |
| | | 338 | | { |
| | | 339 | | goto AsciiMustChangeCase; |
| | | 340 | | } |
| | | 341 | | |
| | 0 | 342 | | currIdx += 2; |
| | 0 | 343 | | } while (currIdx <= lastIndexWhereCanReadTwoChars); |
| | | 344 | | } |
| | | 345 | | |
| | | 346 | | // If there's a single character left to convert, do it now. |
| | 0 | 347 | | if ((source.Length & 1) != 0) |
| | | 348 | | { |
| | 0 | 349 | | uint tempValue = pSource[currIdx]; |
| | 0 | 350 | | if (tempValue > 0x7Fu) |
| | | 351 | | { |
| | | 352 | | goto NotAscii; |
| | | 353 | | } |
| | 0 | 354 | | if ((toUpper) ? ((tempValue - 'a') <= (uint)('z' - 'a')) : ((tempValue - 'A') <= (uint)('Z' - 'A |
| | | 355 | | { |
| | | 356 | | goto AsciiMustChangeCase; |
| | | 357 | | } |
| | | 358 | | } |
| | | 359 | | |
| | | 360 | | // We got through all characters without finding anything that needed to change - done! |
| | 0 | 361 | | return source; |
| | | 362 | | |
| | | 363 | | AsciiMustChangeCase: |
| | | 364 | | { |
| | | 365 | | // We reached ASCII data that requires a case change. |
| | | 366 | | // This will necessarily allocate a new string, but let's try to stay within the managed (non-lo |
| | | 367 | | // conversion code path if we can. |
| | | 368 | | |
| | 0 | 369 | | string result = string.FastAllocateString(source.Length); // changing case uses simple folding: |
| | | 370 | | |
| | | 371 | | // copy existing known-good data into the result |
| | 0 | 372 | | Span<char> resultSpan = new Span<char>(ref result.GetRawStringData(), result.Length); |
| | 0 | 373 | | source.AsSpan(0, (int)currIdx).CopyTo(resultSpan); |
| | | 374 | | |
| | | 375 | | // and re-run the fast span-based logic over the remainder of the data |
| | 0 | 376 | | ChangeCaseCommon<TConversion>(instance, source.AsSpan((int)currIdx), resultSpan.Slice((int)currI |
| | 0 | 377 | | return result; |
| | | 378 | | } |
| | | 379 | | } |
| | | 380 | | |
| | | 381 | | NotAscii: |
| | | 382 | | { |
| | 0 | 383 | | if (GlobalizationMode.Invariant) |
| | | 384 | | { |
| | 0 | 385 | | return toUpper ? InvariantModeCasing.ToUpper(source) : InvariantModeCasing.ToLower(source); |
| | | 386 | | } |
| | | 387 | | |
| | | 388 | | // We reached non-ASCII data *or* the requested culture doesn't map ASCII data the same way as the i |
| | | 389 | | // In either case we need to fall back to the localization tables. |
| | | 390 | | |
| | 0 | 391 | | string result = string.FastAllocateString(source.Length); // changing case uses simple folding: does |
| | | 392 | | |
| | 0 | 393 | | if (currIdx > 0) |
| | | 394 | | { |
| | | 395 | | // copy existing known-good data into the result |
| | 0 | 396 | | Span<char> resultSpan = new Span<char>(ref result.GetRawStringData(), result.Length); |
| | 0 | 397 | | source.AsSpan(0, (int)currIdx).CopyTo(resultSpan); |
| | | 398 | | } |
| | | 399 | | |
| | | 400 | | // instance being null means it's Invariant |
| | 0 | 401 | | instance ??= Invariant; |
| | | 402 | | |
| | | 403 | | // and run the culture-aware logic over the remainder of the data |
| | 0 | 404 | | fixed (char* pResult = result) |
| | | 405 | | { |
| | 0 | 406 | | instance.ChangeCaseCore(pSource + currIdx, source.Length - (int)currIdx, pResult + currIdx, resu |
| | | 407 | | } |
| | 0 | 408 | | return result; |
| | | 409 | | } |
| | | 410 | | } |
| | | 411 | | } |
| | | 412 | | |
| | | 413 | | internal static unsafe string ToLowerAsciiInvariant(string s) |
| | | 414 | | { |
| | 0 | 415 | | if (s.Length == 0) |
| | | 416 | | { |
| | 0 | 417 | | return string.Empty; |
| | | 418 | | } |
| | | 419 | | |
| | 0 | 420 | | int i = s.AsSpan().IndexOfAnyInRange('A', 'Z'); |
| | 0 | 421 | | if (i < 0) |
| | | 422 | | { |
| | 0 | 423 | | return s; |
| | | 424 | | } |
| | | 425 | | |
| | 0 | 426 | | fixed (char* pSource = s) |
| | | 427 | | { |
| | 0 | 428 | | string result = string.FastAllocateString(s.Length); |
| | 0 | 429 | | fixed (char* pResult = result) |
| | | 430 | | { |
| | 0 | 431 | | s.AsSpan(0, i).CopyTo(new Span<char>(pResult, result.Length)); |
| | | 432 | | |
| | 0 | 433 | | pResult[i] = (char)(pSource[i] | 0x20); |
| | 0 | 434 | | i++; |
| | | 435 | | |
| | 0 | 436 | | while (i < s.Length) |
| | | 437 | | { |
| | 0 | 438 | | pResult[i] = ToLowerAsciiInvariant(pSource[i]); |
| | 0 | 439 | | i++; |
| | | 440 | | } |
| | | 441 | | } |
| | | 442 | | |
| | 0 | 443 | | return result; |
| | | 444 | | } |
| | | 445 | | } |
| | | 446 | | |
| | | 447 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 448 | | private static char ToLowerAsciiInvariant(char c) |
| | | 449 | | { |
| | 0 | 450 | | if (char.IsAsciiLetterUpper(c)) |
| | | 451 | | { |
| | | 452 | | // on x86, extending BYTE -> DWORD is more efficient than WORD -> DWORD |
| | 0 | 453 | | c = (char)(byte)(c | 0x20); |
| | | 454 | | } |
| | 0 | 455 | | return c; |
| | | 456 | | } |
| | | 457 | | |
| | | 458 | | /// <summary> |
| | | 459 | | /// Converts the character or string to upper case. Certain locales |
| | | 460 | | /// have different casing semantics from the file systems in Win32. |
| | | 461 | | /// </summary> |
| | | 462 | | public char ToUpper(char c) |
| | | 463 | | { |
| | 0 | 464 | | if (GlobalizationMode.Invariant) |
| | | 465 | | { |
| | 0 | 466 | | return InvariantModeCasing.ToUpper(c); |
| | | 467 | | } |
| | | 468 | | |
| | 0 | 469 | | if (UnicodeUtility.IsAsciiCodePoint(c) && IsAsciiCasingSameAsInvariant) |
| | | 470 | | { |
| | 0 | 471 | | return ToUpperAsciiInvariant(c); |
| | | 472 | | } |
| | | 473 | | |
| | 0 | 474 | | return ChangeCase(c, toUpper: true); |
| | | 475 | | } |
| | | 476 | | |
| | | 477 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 478 | | internal static char ToUpperInvariant(char c) |
| | | 479 | | { |
| | 0 | 480 | | if (UnicodeUtility.IsAsciiCodePoint(c)) |
| | | 481 | | { |
| | 0 | 482 | | return ToUpperAsciiInvariant(c); |
| | | 483 | | } |
| | | 484 | | |
| | 0 | 485 | | if (GlobalizationMode.Invariant) |
| | | 486 | | { |
| | 0 | 487 | | return InvariantModeCasing.ToUpper(c); |
| | | 488 | | } |
| | | 489 | | |
| | 0 | 490 | | return Invariant.ChangeCase(c, toUpper: true); |
| | | 491 | | } |
| | | 492 | | |
| | | 493 | | public string ToUpper(string str) |
| | | 494 | | { |
| | 0 | 495 | | ArgumentNullException.ThrowIfNull(str); |
| | 0 | 496 | | return ChangeCaseCommon<ToUpperConversion>(this, str); |
| | | 497 | | } |
| | | 498 | | |
| | | 499 | | internal static string ToUpperInvariant(string str) |
| | | 500 | | { |
| | 0 | 501 | | ArgumentNullException.ThrowIfNull(str); |
| | 0 | 502 | | return ChangeCaseCommon<ToUpperConversion>(null, str); |
| | | 503 | | } |
| | | 504 | | |
| | | 505 | | internal void ToUpper(ReadOnlySpan<char> source, Span<char> destination) |
| | | 506 | | { |
| | 0 | 507 | | ChangeCaseCommon<ToUpperConversion>(this, source, destination); |
| | 0 | 508 | | } |
| | | 509 | | |
| | | 510 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 511 | | internal static char ToUpperAsciiInvariant(char c) |
| | | 512 | | { |
| | 0 | 513 | | if (char.IsAsciiLetterLower(c)) |
| | | 514 | | { |
| | 0 | 515 | | c = (char)(c & 0x5F); // = low 7 bits of ~0x20 |
| | | 516 | | } |
| | 0 | 517 | | return c; |
| | | 518 | | } |
| | | 519 | | |
| | | 520 | | /// <summary> |
| | | 521 | | /// Converts the specified rune to lowercase. |
| | | 522 | | /// </summary> |
| | | 523 | | /// <param name="value">The rune to convert to lowercase.</param> |
| | | 524 | | /// <returns>The specified rune converted to lowercase.</returns> |
| | | 525 | | public unsafe Rune ToLower(Rune value) |
| | | 526 | | { |
| | | 527 | | // Convert rune to span |
| | 0 | 528 | | ReadOnlySpan<char> valueChars = value.AsSpan(stackalloc char[Rune.MaxUtf16CharsPerRune]); |
| | | 529 | | |
| | | 530 | | // Change span to lower and convert to rune |
| | 0 | 531 | | if (valueChars.Length == 2) |
| | | 532 | | { |
| | 0 | 533 | | Span<char> lowerChars = ['\0', '\0']; |
| | 0 | 534 | | ToLower(valueChars, lowerChars); |
| | 0 | 535 | | return new Rune(lowerChars[0], lowerChars[1]); |
| | | 536 | | } |
| | | 537 | | |
| | 0 | 538 | | char lowerChar = ToLower(valueChars[0]); |
| | 0 | 539 | | return new Rune(lowerChar); |
| | | 540 | | } |
| | | 541 | | |
| | | 542 | | /// <summary> |
| | | 543 | | /// Converts the specified rune to uppercase. |
| | | 544 | | /// </summary> |
| | | 545 | | /// <param name="value">The rune to convert to uppercase.</param> |
| | | 546 | | /// <returns>The specified rune converted to uppercase.</returns> |
| | | 547 | | public unsafe Rune ToUpper(Rune value) |
| | | 548 | | { |
| | | 549 | | // Convert rune to span |
| | 0 | 550 | | ReadOnlySpan<char> valueChars = value.AsSpan(stackalloc char[Rune.MaxUtf16CharsPerRune]); |
| | | 551 | | |
| | | 552 | | // Change span to upper and convert to rune |
| | 0 | 553 | | if (valueChars.Length == 2) |
| | | 554 | | { |
| | 0 | 555 | | Span<char> upperChars = ['\0', '\0']; |
| | 0 | 556 | | ToUpper(valueChars, upperChars); |
| | 0 | 557 | | return new Rune(upperChars[0], upperChars[1]); |
| | | 558 | | } |
| | | 559 | | |
| | 0 | 560 | | char upperChar = ToUpper(valueChars[0]); |
| | 0 | 561 | | return new Rune(upperChar); |
| | | 562 | | } |
| | | 563 | | |
| | | 564 | | private bool IsAsciiCasingSameAsInvariant |
| | | 565 | | { |
| | | 566 | | [MethodImpl(MethodImplOptions.AggressiveInlining)] |
| | | 567 | | get |
| | | 568 | | { |
| | 0 | 569 | | if (_isAsciiCasingSameAsInvariant == NullableBool.Undefined) |
| | | 570 | | { |
| | 0 | 571 | | PopulateIsAsciiCasingSameAsInvariant(); |
| | | 572 | | } |
| | | 573 | | |
| | 0 | 574 | | Debug.Assert(_isAsciiCasingSameAsInvariant == NullableBool.True || _isAsciiCasingSameAsInvariant == Null |
| | 0 | 575 | | return _isAsciiCasingSameAsInvariant == NullableBool.True; |
| | | 576 | | } |
| | | 577 | | } |
| | | 578 | | |
| | | 579 | | [MethodImpl(MethodImplOptions.NoInlining)] |
| | | 580 | | private void PopulateIsAsciiCasingSameAsInvariant() |
| | | 581 | | { |
| | 0 | 582 | | bool compareResult = CultureInfo.GetCultureInfo(_textInfoName).CompareInfo.Compare("abcdefghijklmnopqrstuvwx |
| | 0 | 583 | | _isAsciiCasingSameAsInvariant = compareResult ? NullableBool.True : NullableBool.False; |
| | 0 | 584 | | } |
| | | 585 | | |
| | | 586 | | /// <summary> |
| | | 587 | | /// Returns true if the dominant direction of text and UI such as the |
| | | 588 | | /// relative position of buttons and scroll bars |
| | | 589 | | /// </summary> |
| | 0 | 590 | | public bool IsRightToLeft => _cultureData.IsRightToLeft; |
| | | 591 | | |
| | | 592 | | public override bool Equals([NotNullWhen(true)] object? obj) |
| | | 593 | | { |
| | 0 | 594 | | return obj is TextInfo otherTextInfo |
| | 0 | 595 | | && CultureName.Equals(otherTextInfo.CultureName); |
| | | 596 | | } |
| | | 597 | | |
| | 0 | 598 | | public override int GetHashCode() => CultureName.GetHashCode(); |
| | | 599 | | |
| | | 600 | | public override string ToString() |
| | | 601 | | { |
| | 0 | 602 | | return "TextInfo - " + _cultureData.CultureName; |
| | | 603 | | } |
| | | 604 | | |
| | | 605 | | /// <summary> |
| | | 606 | | /// Titlecasing refers to a casing practice wherein the first letter of a word is an uppercase letter |
| | | 607 | | /// and the rest of the letters are lowercase. The choice of which words to titlecase in headings |
| | | 608 | | /// and titles is dependent on language and local conventions. For example, "The Merry Wives of Windor" |
| | | 609 | | /// is the appropriate titlecasing of that play's name in English, with the word "of" not titlecased. |
| | | 610 | | /// In German, however, the title is "Die lustigen Weiber von Windsor," and both "lustigen" and "von" |
| | | 611 | | /// are not titlecased. In French even fewer words are titlecased: "Les joyeuses commeres de Windsor." |
| | | 612 | | /// |
| | | 613 | | /// Moreover, the determination of what actually constitutes a word is language dependent, and this can |
| | | 614 | | /// influence which letter or letters of a "word" are uppercased when titlecasing strings. For example |
| | | 615 | | /// "l'arbre" is considered two words in French, whereas "can't" is considered one word in English. |
| | | 616 | | /// </summary> |
| | | 617 | | public string ToTitleCase(string str) |
| | | 618 | | { |
| | 0 | 619 | | ArgumentNullException.ThrowIfNull(str); |
| | | 620 | | |
| | 0 | 621 | | if (str.Length == 0) |
| | | 622 | | { |
| | 0 | 623 | | return str; |
| | | 624 | | } |
| | | 625 | | |
| | 0 | 626 | | StringBuilder result = new StringBuilder(); |
| | 0 | 627 | | string? lowercaseData = null; |
| | | 628 | | // Store if the current culture is Dutch (special case). This covers both the |
| | | 629 | | // neutral culture ("nl") and any specific Dutch culture ("nl-NL", "nl-BE", etc.). |
| | 0 | 630 | | string cultureName = CultureName; |
| | 0 | 631 | | bool isDutchCulture = cultureName.StartsWith("nl", StringComparison.OrdinalIgnoreCase) && |
| | 0 | 632 | | (cultureName.Length == 2 || cultureName[2] == '-'); |
| | | 633 | | |
| | 0 | 634 | | for (int i = 0; i < str.Length; i++) |
| | | 635 | | { |
| | 0 | 636 | | UnicodeCategory charType = CharUnicodeInfo.GetUnicodeCategoryInternal(str, i, out int charLen); |
| | 0 | 637 | | if (char.CheckLetter(charType)) |
| | | 638 | | { |
| | | 639 | | // Special case to check for Dutch specific titlecasing with "IJ" characters |
| | | 640 | | // at the beginning of a word |
| | 0 | 641 | | if (isDutchCulture && i < str.Length - 1 && (str[i] == 'i' || str[i] == 'I') && (str[i + 1] == 'j' | |
| | | 642 | | { |
| | 0 | 643 | | result.Append("IJ"); |
| | 0 | 644 | | i += 2; |
| | | 645 | | } |
| | | 646 | | else |
| | | 647 | | { |
| | | 648 | | // Do the titlecasing for the first character of the word. |
| | 0 | 649 | | i = AddTitlecaseLetter(ref result, ref str, i, charLen) + 1; |
| | | 650 | | } |
| | | 651 | | |
| | | 652 | | // Convert the characters until the end of the this word |
| | | 653 | | // to lowercase. |
| | 0 | 654 | | int lowercaseStart = i; |
| | | 655 | | |
| | | 656 | | // Use hasLowerCase flag to prevent from lowercasing acronyms (like "URT", "USA", etc) |
| | | 657 | | // This is in line with Word 2000 behavior of titlecasing. |
| | 0 | 658 | | bool hasLowerCase = (charType == UnicodeCategory.LowercaseLetter); |
| | | 659 | | |
| | | 660 | | // Use a loop to find all of the other letters following this letter. |
| | 0 | 661 | | while (i < str.Length) |
| | | 662 | | { |
| | 0 | 663 | | charType = CharUnicodeInfo.GetUnicodeCategoryInternal(str, i, out charLen); |
| | 0 | 664 | | if (IsLetterCategory(charType)) |
| | | 665 | | { |
| | 0 | 666 | | if (charType == UnicodeCategory.LowercaseLetter) |
| | | 667 | | { |
| | 0 | 668 | | hasLowerCase = true; |
| | | 669 | | } |
| | 0 | 670 | | i += charLen; |
| | | 671 | | } |
| | 0 | 672 | | else if (IsApostrophe(str[i])) |
| | | 673 | | { |
| | 0 | 674 | | i++; |
| | 0 | 675 | | if (hasLowerCase) |
| | | 676 | | { |
| | 0 | 677 | | lowercaseData ??= ToLower(str); |
| | 0 | 678 | | result.Append(lowercaseData, lowercaseStart, i - lowercaseStart); |
| | | 679 | | } |
| | | 680 | | else |
| | | 681 | | { |
| | 0 | 682 | | result.Append(str, lowercaseStart, i - lowercaseStart); |
| | | 683 | | } |
| | 0 | 684 | | lowercaseStart = i; |
| | 0 | 685 | | hasLowerCase = true; |
| | | 686 | | } |
| | 0 | 687 | | else if (!IsWordSeparator(charType)) |
| | | 688 | | { |
| | | 689 | | // This category is considered to be part of the word. |
| | | 690 | | // This is any category that is marked as false in wordSeparator array. |
| | 0 | 691 | | i += charLen; |
| | | 692 | | } |
| | | 693 | | else |
| | | 694 | | { |
| | | 695 | | // A word separator. Break out of the loop. |
| | | 696 | | break; |
| | | 697 | | } |
| | | 698 | | } |
| | | 699 | | |
| | 0 | 700 | | int count = i - lowercaseStart; |
| | | 701 | | |
| | 0 | 702 | | if (count > 0) |
| | | 703 | | { |
| | 0 | 704 | | if (hasLowerCase) |
| | | 705 | | { |
| | 0 | 706 | | lowercaseData ??= ToLower(str); |
| | 0 | 707 | | result.Append(lowercaseData, lowercaseStart, count); |
| | | 708 | | } |
| | | 709 | | else |
| | | 710 | | { |
| | 0 | 711 | | result.Append(str, lowercaseStart, count); |
| | | 712 | | } |
| | | 713 | | } |
| | | 714 | | |
| | 0 | 715 | | if (i < str.Length) |
| | | 716 | | { |
| | | 717 | | // not a letter, just append it |
| | 0 | 718 | | i = AddNonLetter(ref result, ref str, i, charLen); |
| | | 719 | | } |
| | | 720 | | } |
| | | 721 | | else |
| | | 722 | | { |
| | | 723 | | // not a letter, just append it |
| | 0 | 724 | | i = AddNonLetter(ref result, ref str, i, charLen); |
| | | 725 | | } |
| | | 726 | | } |
| | 0 | 727 | | return result.ToString(); |
| | | 728 | | } |
| | | 729 | | |
| | | 730 | | private static int AddNonLetter(ref StringBuilder result, ref string input, int inputIndex, int charLen) |
| | | 731 | | { |
| | 0 | 732 | | Debug.Assert(charLen == 1 || charLen == 2, "[TextInfo.AddNonLetter] CharUnicodeInfo.InternalGetUnicodeCatego |
| | 0 | 733 | | if (charLen == 2) |
| | | 734 | | { |
| | | 735 | | // Surrogate pair |
| | 0 | 736 | | result.Append(input[inputIndex++]); |
| | 0 | 737 | | result.Append(input[inputIndex]); |
| | | 738 | | } |
| | | 739 | | else |
| | | 740 | | { |
| | 0 | 741 | | result.Append(input[inputIndex]); |
| | | 742 | | } |
| | 0 | 743 | | return inputIndex; |
| | | 744 | | } |
| | | 745 | | |
| | | 746 | | private int AddTitlecaseLetter(ref StringBuilder result, ref string input, int inputIndex, int charLen) |
| | | 747 | | { |
| | 0 | 748 | | Debug.Assert(charLen == 1 || charLen == 2, "[TextInfo.AddTitlecaseLetter] CharUnicodeInfo.InternalGetUnicode |
| | | 749 | | |
| | 0 | 750 | | if (charLen == 2) |
| | | 751 | | { |
| | | 752 | | // for surrogate pairs do a ToUpper operation on the substring |
| | 0 | 753 | | ReadOnlySpan<char> src = input.AsSpan(inputIndex, 2); |
| | 0 | 754 | | if (GlobalizationMode.Invariant) |
| | | 755 | | { |
| | 0 | 756 | | SurrogateCasing.ToUpper(src[0], src[1], out char h, out char l); |
| | 0 | 757 | | result.Append(h); |
| | 0 | 758 | | result.Append(l); |
| | | 759 | | } |
| | | 760 | | else |
| | | 761 | | { |
| | 0 | 762 | | Span<char> dst = ['\0', '\0']; |
| | 0 | 763 | | ChangeCaseToUpper(src, dst); |
| | 0 | 764 | | result.Append(dst); |
| | | 765 | | } |
| | 0 | 766 | | inputIndex++; |
| | | 767 | | } |
| | | 768 | | else |
| | | 769 | | { |
| | 0 | 770 | | switch (input[inputIndex]) |
| | | 771 | | { |
| | | 772 | | // For AppCompat, the Titlecase Case Mapping data from NDP 2.0 is used below. |
| | | 773 | | case (char)0x01C4: // DZ with Caron -> Dz with Caron |
| | | 774 | | case (char)0x01C5: // Dz with Caron -> Dz with Caron |
| | | 775 | | case (char)0x01C6: // dz with Caron -> Dz with Caron |
| | 0 | 776 | | result.Append((char)0x01C5); |
| | 0 | 777 | | break; |
| | | 778 | | case (char)0x01C7: // LJ -> Lj |
| | | 779 | | case (char)0x01C8: // Lj -> Lj |
| | | 780 | | case (char)0x01C9: // lj -> Lj |
| | 0 | 781 | | result.Append((char)0x01C8); |
| | 0 | 782 | | break; |
| | | 783 | | case (char)0x01CA: // NJ -> Nj |
| | | 784 | | case (char)0x01CB: // Nj -> Nj |
| | | 785 | | case (char)0x01CC: // nj -> Nj |
| | 0 | 786 | | result.Append((char)0x01CB); |
| | 0 | 787 | | break; |
| | | 788 | | case (char)0x01F1: // DZ -> Dz |
| | | 789 | | case (char)0x01F2: // Dz -> Dz |
| | | 790 | | case (char)0x01F3: // dz -> Dz |
| | 0 | 791 | | result.Append((char)0x01F2); |
| | 0 | 792 | | break; |
| | | 793 | | default: |
| | 0 | 794 | | result.Append(GlobalizationMode.Invariant ? InvariantModeCasing.ToUpper(input[inputIndex]) : ToU |
| | | 795 | | break; |
| | | 796 | | } |
| | | 797 | | } |
| | 0 | 798 | | return inputIndex; |
| | | 799 | | } |
| | | 800 | | |
| | | 801 | | private unsafe void ChangeCaseCore(char* src, int srcLen, char* dstBuffer, int dstBufferCapacity, bool bToUpper) |
| | | 802 | | { |
| | 0 | 803 | | if (GlobalizationMode.UseNls) |
| | | 804 | | { |
| | 0 | 805 | | NlsChangeCase(src, srcLen, dstBuffer, dstBufferCapacity, bToUpper); |
| | 0 | 806 | | return; |
| | | 807 | | } |
| | | 808 | | #if TARGET_MACCATALYST || TARGET_IOS || TARGET_TVOS |
| | | 809 | | if (GlobalizationMode.Hybrid) |
| | | 810 | | { |
| | | 811 | | ChangeCaseNative(src, srcLen, dstBuffer, dstBufferCapacity, bToUpper); |
| | | 812 | | return; |
| | | 813 | | } |
| | | 814 | | #endif |
| | 0 | 815 | | IcuChangeCase(src, srcLen, dstBuffer, dstBufferCapacity, bToUpper); |
| | 0 | 816 | | } |
| | | 817 | | |
| | | 818 | | // Used in ToTitleCase(): |
| | | 819 | | // When we find a starting letter, the following array decides if a category should be |
| | | 820 | | // considered as word separator or not. |
| | | 821 | | private const int c_wordSeparatorMask = |
| | | 822 | | /* false */ (0 << 0) | // UppercaseLetter = 0, |
| | | 823 | | /* false */ (0 << 1) | // LowercaseLetter = 1, |
| | | 824 | | /* false */ (0 << 2) | // TitlecaseLetter = 2, |
| | | 825 | | /* false */ (0 << 3) | // ModifierLetter = 3, |
| | | 826 | | /* false */ (0 << 4) | // OtherLetter = 4, |
| | | 827 | | /* false */ (0 << 5) | // NonSpacingMark = 5, |
| | | 828 | | /* false */ (0 << 6) | // SpacingCombiningMark = 6, |
| | | 829 | | /* false */ (0 << 7) | // EnclosingMark = 7, |
| | | 830 | | /* false */ (0 << 8) | // DecimalDigitNumber = 8, |
| | | 831 | | /* false */ (0 << 9) | // LetterNumber = 9, |
| | | 832 | | /* false */ (0 << 10) | // OtherNumber = 10, |
| | | 833 | | /* true */ (1 << 11) | // SpaceSeparator = 11, |
| | | 834 | | /* true */ (1 << 12) | // LineSeparator = 12, |
| | | 835 | | /* true */ (1 << 13) | // ParagraphSeparator = 13, |
| | | 836 | | /* true */ (1 << 14) | // Control = 14, |
| | | 837 | | /* true */ (1 << 15) | // Format = 15, |
| | | 838 | | /* false */ (0 << 16) | // Surrogate = 16, |
| | | 839 | | /* false */ (0 << 17) | // PrivateUse = 17, |
| | | 840 | | /* true */ (1 << 18) | // ConnectorPunctuation = 18, |
| | | 841 | | /* true */ (1 << 19) | // DashPunctuation = 19, |
| | | 842 | | /* true */ (1 << 20) | // OpenPunctuation = 20, |
| | | 843 | | /* true */ (1 << 21) | // ClosePunctuation = 21, |
| | | 844 | | /* true */ (1 << 22) | // InitialQuotePunctuation = 22, |
| | | 845 | | /* true */ (1 << 23) | // FinalQuotePunctuation = 23, |
| | | 846 | | /* true */ (1 << 24) | // OtherPunctuation = 24, |
| | | 847 | | /* true */ (1 << 25) | // MathSymbol = 25, |
| | | 848 | | /* true */ (1 << 26) | // CurrencySymbol = 26, |
| | | 849 | | /* true */ (1 << 27) | // ModifierSymbol = 27, |
| | | 850 | | /* true */ (1 << 28) | // OtherSymbol = 28, |
| | | 851 | | /* false */ (0 << 29); // OtherNotAssigned = 29; |
| | | 852 | | |
| | | 853 | | private static bool IsWordSeparator(UnicodeCategory category) |
| | | 854 | | { |
| | 0 | 855 | | return (c_wordSeparatorMask & (1 << (int)category)) != 0; |
| | | 856 | | } |
| | | 857 | | |
| | | 858 | | // Characters treated as an apostrophe within a word (e.g. contractions such as |
| | | 859 | | // "can't" or possessives such as "Grandma's"), so a following letter is not treated |
| | | 860 | | // as the start of a new word during titlecasing: |
| | | 861 | | // U+0027 APOSTROPHE |
| | | 862 | | // U+2019 RIGHT SINGLE QUOTATION MARK (the typographic curly apostrophe) |
| | | 863 | | // U+2018 LEFT SINGLE QUOTATION MARK |
| | | 864 | | // U+FF07 FULLWIDTH APOSTROPHE |
| | | 865 | | private static bool IsApostrophe(char c) |
| | | 866 | | { |
| | 0 | 867 | | return c is '\'' or '\u2019' or '\u2018' or '\uFF07'; |
| | | 868 | | } |
| | | 869 | | |
| | | 870 | | private static bool IsLetterCategory(UnicodeCategory uc) |
| | | 871 | | { |
| | 0 | 872 | | return uc == UnicodeCategory.UppercaseLetter |
| | 0 | 873 | | || uc == UnicodeCategory.LowercaseLetter |
| | 0 | 874 | | || uc == UnicodeCategory.TitlecaseLetter |
| | 0 | 875 | | || uc == UnicodeCategory.ModifierLetter |
| | 0 | 876 | | || uc == UnicodeCategory.OtherLetter; |
| | | 877 | | } |
| | | 878 | | |
| | | 879 | | // A dummy struct that is used for 'ToUpper' in generic parameters |
| | | 880 | | private readonly struct ToUpperConversion { } |
| | | 881 | | |
| | | 882 | | // A dummy struct that is used for 'ToLower' in generic parameters |
| | | 883 | | private readonly struct ToLowerConversion { } |
| | | 884 | | } |
| | | 885 | | } |
| | | 886 | | |