From 8d95f9ed0e7a437f89725180047728a90f8a5d7b Mon Sep 17 00:00:00 2001 From: Egor Bogatov Date: Thu, 1 Oct 2026 01:30:24 +0200 Subject: [PATCH 1/3] Use safe scalar code for UTF-8 Equals(OrdinalIgnoreCase) Ordinal.EqualsIgnoreCaseUtf8 is only used by UTF-8 floating-point parsing to match short NaN/Infinity symbols, so replace the vectorized/unsafe implementation with the same safe scalar loop used by StartsWithIgnoreCaseUtf8. The ASCII loop has no calls so it stays cheap when inlined into callers; runes are decoded in a separate non-ASCII fallback. Remove the now unused Utf8Utility ignore-case helpers. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: 9c5a38d5-eacd-41f0-b96d-4bcc0eecb0f4 --- .../src/System/Globalization/Ordinal.Utf8.cs | 321 ++---------------- .../MemoryExtensions.Globalization.Utf8.cs | 11 +- .../src/System/Text/Unicode/Utf8Utility.cs | 116 ------- 3 files changed, 20 insertions(+), 428 deletions(-) diff --git a/src/libraries/System.Private.CoreLib/src/System/Globalization/Ordinal.Utf8.cs b/src/libraries/System.Private.CoreLib/src/System/Globalization/Ordinal.Utf8.cs index 4b45be70ffa07f..d4f4b5728e7912 100644 --- a/src/libraries/System.Private.CoreLib/src/System/Globalization/Ordinal.Utf8.cs +++ b/src/libraries/System.Private.CoreLib/src/System/Globalization/Ordinal.Utf8.cs @@ -2,332 +2,49 @@ // The .NET Foundation licenses this file to you under the MIT license. using System.Buffers; -using System.Diagnostics; -using System.Runtime.CompilerServices; -using System.Runtime.InteropServices; -using System.Runtime.Intrinsics; using System.Text; -using System.Text.Unicode; namespace System.Globalization { internal static partial class Ordinal { - internal static bool EqualsStringIgnoreCaseUtf8(ref byte strA, int lengthA, ref byte strB, int lengthB) - { - // NOTE: Two UTF-8 inputs of different length might compare as equal under - // the OrdinalIgnoreCase comparer. This is distinct from UTF-16, where the - // inputs being different length will mean that they can never compare as - // equal under an OrdinalIgnoreCase comparer. - - int length = Math.Min(lengthA, lengthB); - int range = length; - - ref byte charA = ref strA; - ref byte charB = ref strB; - - const byte maxChar = 0x7F; - - while ((length != 0) && (charA <= maxChar) && (charB <= maxChar)) - { - // Ordinal equals or lowercase equals if the result ends up in the a-z range - if (charA == charB || - ((charA | 0x20) == (charB | 0x20) && char.IsAsciiLetter((char)charA))) - { - length--; - charA = ref Unsafe.Add(ref charA, 1); - charB = ref Unsafe.Add(ref charB, 1); - } - else - { - return false; - } - } - - if (length == 0) - { - // Success if we reached the end of both sequences - return lengthA == lengthB; - } + // Not optimized for large inputs: the only callers (number parsing) pass short sign/NaN/Infinity symbols. + internal static bool EqualsIgnoreCaseUtf8(ReadOnlySpan left, ReadOnlySpan right) => + MatchIgnoreCaseUtf8(left, right, prefixOnly: false); - range -= length; - return EqualsStringIgnoreCaseNonAsciiUtf8(ref charA, lengthA - range, ref charB, lengthB - range); - } + internal static bool StartsWithIgnoreCaseUtf8(ReadOnlySpan source, ReadOnlySpan prefix) => + MatchIgnoreCaseUtf8(source, prefix, prefixOnly: true); - internal static bool EqualsStringIgnoreCaseNonAsciiUtf8(ref byte strA, int lengthA, ref byte strB, int lengthB) + private static bool MatchIgnoreCaseUtf8(ReadOnlySpan source, ReadOnlySpan prefix, bool prefixOnly) { - // NLS/ICU doesn't provide native UTF-8 support so we need to do our own corresponding ordinal comparison - - ReadOnlySpan spanA = MemoryMarshal.CreateReadOnlySpan(ref strA, lengthA); - ReadOnlySpan spanB = MemoryMarshal.CreateReadOnlySpan(ref strB, lengthB); - - do + // ASCII-only loop without calls, so it stays cheap when inlined into callers + for (int i = 0; i < prefix.Length; i++) { - OperationStatus statusA = Rune.DecodeFromUtf8(spanA, out Rune runeA, out int bytesConsumedA); - OperationStatus statusB = Rune.DecodeFromUtf8(spanB, out Rune runeB, out int bytesConsumedB); - - if (statusA != statusB) - { - // OperationStatus don't match; fail immediately - return false; - } - - if (statusA == OperationStatus.Done) - { - if (Rune.ToUpperInvariant(runeA) != Rune.ToUpperInvariant(runeB)) - { - // Runes don't match when ignoring case; fail immediately - return false; - } - } - else if (!spanA.Slice(0, bytesConsumedA).SequenceEqual(spanB.Slice(0, bytesConsumedB))) - { - // OperationStatus match, but bytesConsumed or the sequence of bytes consumed do not; fail immediately - return false; - } - - // The current runes or invalid byte sequences matched, slice and continue. - // We'll exit the loop when the entirety of both spans have been processed. - // - // In the scenario where one buffer is empty before the other, we'll end up - // with that span returning OperationStatus.NeedsMoreData and bytesConsumed=0 - // while the other span will return a different OperationStatus or different - // bytesConsumed and thus fail the operation. - - spanA = spanA.Slice(bytesConsumedA); - spanB = spanB.Slice(bytesConsumedB); - } - while ((spanA.Length | spanB.Length) != 0); - - return true; - } - - private static bool EqualsIgnoreCaseUtf8_Vector128(ref byte charA, int lengthA, ref byte charB, int lengthB) - { - Debug.Assert(lengthA >= Vector128.Count); - Debug.Assert(lengthB >= Vector128.Count); - Debug.Assert(Vector128.IsHardwareAccelerated); - - nuint lengthU = Math.Min((uint)lengthA, (uint)lengthB); - nuint lengthToExamine = lengthU - (nuint)Vector128.Count; - - nuint i = 0; - - Vector128 vec1; - Vector128 vec2; - - do - { - vec1 = Vector128.LoadUnsafe(ref charA, i); - vec2 = Vector128.LoadUnsafe(ref charB, i); - - if (!Utf8Utility.AllBytesInVector128AreAscii(vec1 | vec2)) - { - goto NON_ASCII; - } - - if (!Utf8Utility.Vector128OrdinalIgnoreCaseAscii(vec1, vec2)) - { - return false; - } - - i += (nuint)Vector128.Count; - } - while (i <= lengthToExamine); - - if (i == lengthU) - { - // success if we reached the end of both sequences - return lengthA == lengthB; - } - - // Use scalar path for trailing elements - return EqualsIgnoreCaseUtf8_Scalar(ref Unsafe.Add(ref charA, i), (int)(lengthU - i), ref Unsafe.Add(ref charB, i), (int)(lengthU - i)); - - NON_ASCII: - if (Utf8Utility.AllBytesInVector128AreAscii(vec1) || Utf8Utility.AllBytesInVector128AreAscii(vec2)) - { - // No need to use the fallback if one of the inputs is full-ASCII - return false; - } - - // Fallback for Non-ASCII inputs - return EqualsStringIgnoreCaseUtf8( - ref Unsafe.Add(ref charA, i), lengthA - (int)i, - ref Unsafe.Add(ref charB, i), lengthB - (int)i - ); - } - - [MethodImpl(MethodImplOptions.AggressiveInlining)] - internal static bool EqualsIgnoreCaseUtf8(ref byte charA, int lengthA, ref byte charB, int lengthB) - { - if (!Vector128.IsHardwareAccelerated || (lengthA < Vector128.Count) || (lengthB < Vector128.Count)) - { - return EqualsIgnoreCaseUtf8_Scalar(ref charA, lengthA, ref charB, lengthB); - } - - return EqualsIgnoreCaseUtf8_Vector128(ref charA, lengthA, ref charB, lengthB); - } - - internal static bool EqualsIgnoreCaseUtf8_Scalar(ref byte charA, int lengthA, ref byte charB, int lengthB) - { - IntPtr byteOffset = IntPtr.Zero; - - int length = Math.Min(lengthA, lengthB); - int range = length; - -#if TARGET_64BIT - ulong valueAu64 = 0; - ulong valueBu64 = 0; - - // Read 8 chars (64 bits) at a time from each string - while ((uint)length >= 8) - { - valueAu64 = Unsafe.ReadUnaligned(ref Unsafe.AddByteOffset(ref charA, byteOffset)); - valueBu64 = Unsafe.ReadUnaligned(ref Unsafe.AddByteOffset(ref charB, byteOffset)); - - // A 32-bit test - even with the bit-twiddling here - is more efficient than a 64-bit test. - ulong temp = valueAu64 | valueBu64; - - if (!Utf8Utility.AllBytesInUInt32AreAscii((uint)temp | (uint)(temp >> 32))) - { - // one of the inputs contains non-ASCII data - goto NonAscii64; - } - - // Generally, the caller has likely performed a first-pass check that the input strings - // are likely equal. Consider a dictionary which computes the hash code of its key before - // performing a proper deep equality check of the string contents. We want to optimize for - // the case where the equality check is likely to succeed, which means that we want to avoid - // branching within this loop unless we're about to exit the loop, either due to failure or - // due to us running out of input data. - - if (!Utf8Utility.UInt64OrdinalIgnoreCaseAscii(valueAu64, valueBu64)) - { - return false; - } - - byteOffset += 8; - length -= 8; - } -#endif - - uint valueAu32 = 0; - uint valueBu32 = 0; - - // Read 4 chars (32 bits) at a time from each string -#if TARGET_64BIT - if ((uint)length >= 4) -#else - while ((uint)length >= 4) -#endif - { - valueAu32 = Unsafe.ReadUnaligned(ref Unsafe.AddByteOffset(ref charA, byteOffset)); - valueBu32 = Unsafe.ReadUnaligned(ref Unsafe.AddByteOffset(ref charB, byteOffset)); - - if (!Utf8Utility.AllBytesInUInt32AreAscii(valueAu32 | valueBu32)) - { - // one of the inputs contains non-ASCII data - goto NonAscii32; - } - - // Generally, the caller has likely performed a first-pass check that the input strings - // are likely equal. Consider a dictionary which computes the hash code of its key before - // performing a proper deep equality check of the string contents. We want to optimize for - // the case where the equality check is likely to succeed, which means that we want to avoid - // branching within this loop unless we're about to exit the loop, either due to failure or - // due to us running out of input data. - - if (!Utf8Utility.UInt32OrdinalIgnoreCaseAscii(valueAu32, valueBu32)) + if (i >= source.Length) { + // The source ended before the prefix return false; } - byteOffset += 4; - length -= 4; - } - - if (length != 0) - { - // We have 1, 2, or 3 bytes remaining. We can't do anything fancy - // like backtracking since we could have only had 1-3 bytes. So, - // instead we'll do 1 or 2 reads to get all 3 bytes. Endianness - // doesn't matter here since we only compare if all bytes are ascii - // and the ordering will be consistent between the two comparisons - - if (length == 3) - { - valueAu32 = Unsafe.ReadUnaligned(ref Unsafe.AddByteOffset(ref charA, byteOffset)); - valueBu32 = Unsafe.ReadUnaligned(ref Unsafe.AddByteOffset(ref charB, byteOffset)); - - byteOffset += 2; + uint a = source[i]; + uint b = prefix[i]; - valueAu32 |= (uint)(Unsafe.AddByteOffset(ref charA, byteOffset) << 16); - valueBu32 |= (uint)(Unsafe.AddByteOffset(ref charB, byteOffset) << 16); - } - else if (length == 2) - { - valueAu32 = Unsafe.ReadUnaligned(ref Unsafe.AddByteOffset(ref charA, byteOffset)); - valueBu32 = Unsafe.ReadUnaligned(ref Unsafe.AddByteOffset(ref charB, byteOffset)); - } - else + if ((a | b) > 0x7F) { - Debug.Assert(length == 1); - - valueAu32 = Unsafe.AddByteOffset(ref charA, byteOffset); - valueBu32 = Unsafe.AddByteOffset(ref charB, byteOffset); + return MatchIgnoreCaseNonAsciiUtf8(source.Slice(i), prefix.Slice(i), prefixOnly); } - if (!Utf8Utility.AllBytesInUInt32AreAscii(valueAu32 | valueBu32)) - { - // one of the inputs contains non-ASCII data - goto NonAscii32; - } - - if (lengthA != lengthB) + // Ordinal equals or lowercase equals if the result ends up in the a-z range + if ((a != b) && (((a | 0x20) != (b | 0x20)) || !char.IsAsciiLetter((char)a))) { - // Failure if we reached the end of one, but not both sequences return false; } - - if (valueAu32 == valueBu32) - { - // exact match - return true; - } - - return Utf8Utility.UInt32OrdinalIgnoreCaseAscii(valueAu32, valueBu32); - } - - Debug.Assert(length == 0); - return lengthA == lengthB; - - NonAscii32: - // Both values have to be non-ASCII to use the slow fallback, in case if one of them is not we return false - if (Utf8Utility.AllBytesInUInt32AreAscii(valueAu32) || Utf8Utility.AllBytesInUInt32AreAscii(valueBu32)) - { - return false; - } - goto NonAscii; - -#if TARGET_64BIT - NonAscii64: - // Both values have to be non-ASCII to use the slow fallback, in case if one of them is not we return false - if (Utf8Utility.AllBytesInUInt64AreAscii(valueAu64) || Utf8Utility.AllBytesInUInt64AreAscii(valueBu64)) - { - return false; } -#endif - NonAscii: - range -= length; - // The non-ASCII case is factored out into its own helper method so that the JIT - // doesn't need to emit a complex prolog for its caller (this method). - return EqualsStringIgnoreCaseUtf8(ref Unsafe.AddByteOffset(ref charA, byteOffset), lengthA - range, ref Unsafe.AddByteOffset(ref charB, byteOffset), lengthB - range); + return prefixOnly || (source.Length == prefix.Length); } - // Not optimized for large inputs yet: the only callers (number parsing) pass short sign symbols. - internal static bool StartsWithIgnoreCaseUtf8(ReadOnlySpan source, ReadOnlySpan prefix) + private static bool MatchIgnoreCaseNonAsciiUtf8(ReadOnlySpan source, ReadOnlySpan prefix, bool prefixOnly) { // NOTE: Two UTF-8 inputs of different length might compare as equal under // the OrdinalIgnoreCase comparer. This is distinct from UTF-16, where the @@ -390,7 +107,7 @@ internal static bool StartsWithIgnoreCaseUtf8(ReadOnlySpan source, ReadOnl prefix = prefix.Slice(bytesConsumedB); } - return true; + return prefixOnly || source.IsEmpty; } } } diff --git a/src/libraries/System.Private.CoreLib/src/System/MemoryExtensions.Globalization.Utf8.cs b/src/libraries/System.Private.CoreLib/src/System/MemoryExtensions.Globalization.Utf8.cs index 47c480dad2f6db..6166a7dee0cafb 100644 --- a/src/libraries/System.Private.CoreLib/src/System/MemoryExtensions.Globalization.Utf8.cs +++ b/src/libraries/System.Private.CoreLib/src/System/MemoryExtensions.Globalization.Utf8.cs @@ -4,7 +4,6 @@ using System.Diagnostics; using System.Globalization; using System.Runtime.CompilerServices; -using System.Runtime.InteropServices; namespace System { @@ -13,15 +12,7 @@ public static partial class MemoryExtensions [MethodImpl(MethodImplOptions.AggressiveInlining)] internal static bool EqualsOrdinalIgnoreCaseUtf8(this ReadOnlySpan span, ReadOnlySpan value) { - // For UTF-8 ist is possible for two spans of different byte length - // to compare as equal under an OrdinalIgnoreCase comparison. - - if ((span.Length | value.Length) == 0) // span.Length == value.Length == 0 - { - return true; - } - - return Ordinal.EqualsIgnoreCaseUtf8(ref MemoryMarshal.GetReference(span), span.Length, ref MemoryMarshal.GetReference(value), value.Length); + return Ordinal.EqualsIgnoreCaseUtf8(span, value); } /// diff --git a/src/libraries/System.Private.CoreLib/src/System/Text/Unicode/Utf8Utility.cs b/src/libraries/System.Private.CoreLib/src/System/Text/Unicode/Utf8Utility.cs index 17a7e7d471ded1..ef922f6ffdaadf 100644 --- a/src/libraries/System.Private.CoreLib/src/System/Text/Unicode/Utf8Utility.cs +++ b/src/libraries/System.Private.CoreLib/src/System/Text/Unicode/Utf8Utility.cs @@ -166,93 +166,6 @@ internal static ulong ConvertAllAsciiBytesInUInt64ToLowercase(ulong value) return value ^ mask; // bit flip uppercase letters [A-Z] => [a-z] } - /// - /// Given two UInt32s that represent four ASCII UTF-8 characters each, returns true iff - /// the two inputs are equal using an ordinal case-insensitive comparison. - /// - /// - /// This is a branchless implementation. - /// - [MethodImpl(MethodImplOptions.AggressiveInlining)] - internal static bool UInt32OrdinalIgnoreCaseAscii(uint valueA, uint valueB) - { - // Not currently intrinsified in mono interpreter, the UTF16 version is - // ASSUMPTION: Caller has validated that input values are ASCII. - Debug.Assert(AllBytesInUInt32AreAscii(valueA)); - Debug.Assert(AllBytesInUInt32AreAscii(valueB)); - - // The logic here is very simple and is doing SIMD Within A Register (SWAR) - // - // First we want to create a mask finding the upper-case ASCII characters - // - // To do that, we can take the above presumption that all characters are ASCII - // and therefore between 0x00 and 0x7F, inclusive. This means that `0x80 + char` - // will never overflow and will at most produce 0xFF. - // - // Given that, we can check if a byte is greater than a value by adding it to - // 0x80 and then subtracting the constant we're comparing against. So, for example, - // if we want to find all characters greater than 'A' we do `value + 0x80 - 'A'`. - // - // Given that 'A' is 0x41, we end up with `0x41 + 0x80 == 0xC1` then we subtract 'A' - // giving us `0xC1 - 0x41 == 0x80` and up to `0xBE` for 'DEL' (0x7F). This means that - // any character greater than or equal to 'A' will have the most significant bit set. - // - // This can itself be simplified down to `val + (0x80 - 'A')` or `val + 0x3F` - // - // We also want to find the characters less than or equal to 'Z' as well. This follows - // the same general principle but relies on finding the inverse instead. That is, we - // want to find all characters greater than or equal to ('Z' + 1) and then inverse it. - // - // To confirm this, lets look at 'Z' which has the value of '0x5A'. So we first do - // `0x5A + 0x80 == 0xDA`, then we subtract `[' (0x5B) giving us `0xDA - 0x5B == 0x80`. - // This means that any character greater than 'Z' will now have the most significant bit set. - // - // It then follows that taking the ones complement will give us a mask representing the bytes - // which are less than or equal to 'Z' since `!(val >= 0x5B) == (val <= 0x5A)` - // - // This then gives us that `('A' <= val) && (val <= 'Z')` is representable as - // `(val + 0x3F) & ~(val + 0x25)` - // - // However, since a `val` cannot be simultaneously less than 'A' and greater than 'Z' we - // are able to simplify this further to being just `(val + 0x3F) ^ (val + 0x25)` - // - // We then want to mask off the excess bits that aren't important to the mask and right - // shift by two. This gives us `0x20` for a byte which is an upper-case ASCII character - // and `0x00` otherwise. - // - // We now have a super efficient implementation that does a correct comparison in - // 12 instructions and with zero branching. - - uint letterMaskA = (((valueA + 0x3F3F3F3F) ^ (valueA + 0x25252525)) & 0x80808080) >> 2; - uint letterMaskB = (((valueB + 0x3F3F3F3F) ^ (valueB + 0x25252525)) & 0x80808080) >> 2; - - return (valueA | letterMaskA) == (valueB | letterMaskB); - } - - /// - /// Given two UInt64s that represent eight ASCII UTF-8 characters each, returns true iff - /// the two inputs are equal using an ordinal case-insensitive comparison. - /// - /// - /// This is a branchless implementation. - /// - [MethodImpl(MethodImplOptions.AggressiveInlining)] - internal static bool UInt64OrdinalIgnoreCaseAscii(ulong valueA, ulong valueB) - { - // Not currently intrinsified in mono interpreter, the UTF16 version is - // ASSUMPTION: Caller has validated that input values are ASCII. - Debug.Assert(AllBytesInUInt64AreAscii(valueA)); - Debug.Assert(AllBytesInUInt64AreAscii(valueB)); - - // Duplicate of logic in UInt32OrdinalIgnoreCaseAscii, but using 64-bit consts. - // See comments in that method for more info. - - ulong letterMaskA = (((valueA + 0x3F3F3F3F3F3F3F3F) ^ (valueA + 0x2525252525252525)) & 0x8080808080808080) >> 2; - ulong letterMaskB = (((valueB + 0x3F3F3F3F3F3F3F3F) ^ (valueB + 0x2525252525252525)) & 0x8080808080808080) >> 2; - - return (valueA | letterMaskA) == (valueB | letterMaskB); - } - #if NET /// /// Returns true iff the Vector128 represents 16 ASCII UTF-8 characters in machine endianness. @@ -262,35 +175,6 @@ internal static bool AllBytesInVector128AreAscii(Vector128 vec) { return (vec & Vector128.Create(unchecked((byte)(~0x7F)))) == Vector128.Zero; } - - /// - /// Given two Vector128 that represent 16 ASCII UTF-8 characters each, returns true iff - /// the two inputs are equal using an ordinal case-insensitive comparison. - /// - [MethodImpl(MethodImplOptions.AggressiveInlining)] - internal static bool Vector128OrdinalIgnoreCaseAscii(Vector128 vec1, Vector128 vec2) - { - // ASSUMPTION: Caller has validated that input values are ASCII. - - // the 0x80 bit of each word of 'lowerIndicator' will be set iff the word has value >= 'A' - Vector128 lowIndicator1 = Vector128.Create((sbyte)(0x80 - 'A')) + vec1.AsSByte(); - Vector128 lowIndicator2 = Vector128.Create((sbyte)(0x80 - 'A')) + vec2.AsSByte(); - - // the 0x80 bit of each word of 'combinedIndicator' will be set iff the word has value >= 'A' and <= 'Z' - Vector128 combIndicator1 = - Vector128.LessThan(Vector128.Create(unchecked((sbyte)(('Z' - 'A') - 0x80))), lowIndicator1); - Vector128 combIndicator2 = - Vector128.LessThan(Vector128.Create(unchecked((sbyte)(('Z' - 'A') - 0x80))), lowIndicator2); - - // Convert both vectors to lower case by adding 0x20 bit for all [A-Z][a-z] characters - Vector128 lcVec1 = - Vector128.AndNot(Vector128.Create((sbyte)0x20), combIndicator1) + vec1.AsSByte(); - Vector128 lcVec2 = - Vector128.AndNot(Vector128.Create((sbyte)0x20), combIndicator2) + vec2.AsSByte(); - - // Compare two lowercased vectors - return (lcVec1 ^ lcVec2) == Vector128.Zero; - } #endif } } From 6ec089f4c551807aaeab11ebd6af5d694c4db515 Mon Sep 17 00:00:00 2001 From: Egor Bogatov Date: Thu, 1 Oct 2026 09:00:54 +0200 Subject: [PATCH 2/3] Fix UTF-8 ignore-case parsing regressions - Compare 8 ASCII bytes at a time in EqualsIgnoreCaseUtf8 using safe BinaryPrimitives reads and the UInt64OrdinalIgnoreCaseAscii SWAR helper. - Reject ASCII vs non-ASCII bytes inline instead of calling the rune loop. - Keep StartsWithIgnoreCaseUtf8 on the single rune loop, as on main. - Mark EqualsIgnoreCaseUtf8 AggressiveInlining; its ASCII path has no calls and the rune loop stays out of line. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: 9c5a38d5-eacd-41f0-b96d-4bcc0eecb0f4 --- .../src/System/Globalization/Ordinal.Utf8.cs | 50 +++++++++++++------ .../src/System/Text/Unicode/Utf8Utility.cs | 22 ++++++++ 2 files changed, 57 insertions(+), 15 deletions(-) diff --git a/src/libraries/System.Private.CoreLib/src/System/Globalization/Ordinal.Utf8.cs b/src/libraries/System.Private.CoreLib/src/System/Globalization/Ordinal.Utf8.cs index d4f4b5728e7912..1e7527254965bf 100644 --- a/src/libraries/System.Private.CoreLib/src/System/Globalization/Ordinal.Utf8.cs +++ b/src/libraries/System.Private.CoreLib/src/System/Globalization/Ordinal.Utf8.cs @@ -2,36 +2,53 @@ // The .NET Foundation licenses this file to you under the MIT license. using System.Buffers; +using System.Buffers.Binary; +using System.Runtime.CompilerServices; using System.Text; +using System.Text.Unicode; namespace System.Globalization { internal static partial class Ordinal { // Not optimized for large inputs: the only callers (number parsing) pass short sign/NaN/Infinity symbols. - internal static bool EqualsIgnoreCaseUtf8(ReadOnlySpan left, ReadOnlySpan right) => - MatchIgnoreCaseUtf8(left, right, prefixOnly: false); + [MethodImpl(MethodImplOptions.AggressiveInlining)] + internal static bool EqualsIgnoreCaseUtf8(ReadOnlySpan left, ReadOnlySpan right) + { + // ASCII-only loops without calls, so they stay cheap when inlined into callers + while ((left.Length >= sizeof(ulong)) && (right.Length >= sizeof(ulong))) + { + ulong a = BinaryPrimitives.ReadUInt64LittleEndian(left); + ulong b = BinaryPrimitives.ReadUInt64LittleEndian(right); - internal static bool StartsWithIgnoreCaseUtf8(ReadOnlySpan source, ReadOnlySpan prefix) => - MatchIgnoreCaseUtf8(source, prefix, prefixOnly: true); + if (!Utf8Utility.AllBytesInUInt64AreAscii(a | b)) + { + break; + } - private static bool MatchIgnoreCaseUtf8(ReadOnlySpan source, ReadOnlySpan prefix, bool prefixOnly) - { - // ASCII-only loop without calls, so it stays cheap when inlined into callers - for (int i = 0; i < prefix.Length; i++) + if (!Utf8Utility.UInt64OrdinalIgnoreCaseAscii(a, b)) + { + return false; + } + + left = left.Slice(sizeof(ulong)); + right = right.Slice(sizeof(ulong)); + } + + for (int i = 0; i < right.Length; i++) { - if (i >= source.Length) + if (i >= left.Length) { - // The source ended before the prefix return false; } - uint a = source[i]; - uint b = prefix[i]; + uint a = left[i]; + uint b = right[i]; if ((a | b) > 0x7F) { - return MatchIgnoreCaseNonAsciiUtf8(source.Slice(i), prefix.Slice(i), prefixOnly); + // No non-ASCII scalar is equal to an ASCII one under ordinal casing + return ((a ^ b) <= 0x7F) && MatchIgnoreCaseUtf8(left.Slice(i), right.Slice(i), prefixOnly: false); } // Ordinal equals or lowercase equals if the result ends up in the a-z range @@ -41,10 +58,13 @@ private static bool MatchIgnoreCaseUtf8(ReadOnlySpan source, ReadOnlySpan< } } - return prefixOnly || (source.Length == prefix.Length); + return left.Length == right.Length; } - private static bool MatchIgnoreCaseNonAsciiUtf8(ReadOnlySpan source, ReadOnlySpan prefix, bool prefixOnly) + internal static bool StartsWithIgnoreCaseUtf8(ReadOnlySpan source, ReadOnlySpan prefix) => + MatchIgnoreCaseUtf8(source, prefix, prefixOnly: true); + + private static bool MatchIgnoreCaseUtf8(ReadOnlySpan source, ReadOnlySpan prefix, bool prefixOnly) { // NOTE: Two UTF-8 inputs of different length might compare as equal under // the OrdinalIgnoreCase comparer. This is distinct from UTF-16, where the diff --git a/src/libraries/System.Private.CoreLib/src/System/Text/Unicode/Utf8Utility.cs b/src/libraries/System.Private.CoreLib/src/System/Text/Unicode/Utf8Utility.cs index ef922f6ffdaadf..9d30ce2325f379 100644 --- a/src/libraries/System.Private.CoreLib/src/System/Text/Unicode/Utf8Utility.cs +++ b/src/libraries/System.Private.CoreLib/src/System/Text/Unicode/Utf8Utility.cs @@ -166,6 +166,28 @@ internal static ulong ConvertAllAsciiBytesInUInt64ToLowercase(ulong value) return value ^ mask; // bit flip uppercase letters [A-Z] => [a-z] } + /// + /// Given two UInt64s that represent eight ASCII UTF-8 characters each, returns true iff + /// the two inputs are equal using an ordinal case-insensitive comparison. + /// + /// + /// This is a branchless implementation. + /// + [MethodImpl(MethodImplOptions.AggressiveInlining)] + internal static bool UInt64OrdinalIgnoreCaseAscii(ulong valueA, ulong valueB) + { + // ASSUMPTION: Caller has validated that input values are ASCII. + Debug.Assert(AllBytesInUInt64AreAscii(valueA)); + Debug.Assert(AllBytesInUInt64AreAscii(valueB)); + + // The 0x80 bit of each byte is set iff 'A' <= byte <= 'Z'; shifting it right by 2 + // gives 0x20, which lowercases those bytes before comparing. + ulong letterMaskA = (((valueA + 0x3F3F3F3F3F3F3F3F) ^ (valueA + 0x2525252525252525)) & 0x8080808080808080) >> 2; + ulong letterMaskB = (((valueB + 0x3F3F3F3F3F3F3F3F) ^ (valueB + 0x2525252525252525)) & 0x8080808080808080) >> 2; + + return (valueA | letterMaskA) == (valueB | letterMaskB); + } + #if NET /// /// Returns true iff the Vector128 represents 16 ASCII UTF-8 characters in machine endianness. From 1c707f90c0a3bb831081872dcb7553621dd37954 Mon Sep 17 00:00:00 2001 From: Egor Bogatov Date: Thu, 1 Oct 2026 14:57:28 +0200 Subject: [PATCH 3/3] Fix Parse_CustomSigns expectation under NLS NLS doesn't case-fold supplementary characters, so U+10400 and U+10428 don't match under OrdinalIgnoreCase for both UTF-16 and UTF-8 parsing. Fixes #135014 Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: 9c5a38d5-eacd-41f0-b96d-4bcc0eecb0f4 --- .../tests/System.Runtime.Tests/System/DoubleTests.cs | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/src/libraries/System.Runtime/tests/System.Runtime.Tests/System/DoubleTests.cs b/src/libraries/System.Runtime/tests/System.Runtime.Tests/System/DoubleTests.cs index 94c98580a1be68..0c3c50c96c3ebf 100644 --- a/src/libraries/System.Runtime/tests/System.Runtime.Tests/System/DoubleTests.cs +++ b/src/libraries/System.Runtime/tests/System.Runtime.Tests/System/DoubleTests.cs @@ -1422,7 +1422,8 @@ public static IEnumerable Parse_CustomSigns_TestData() // Non-ASCII and supplementary signs are matched ignoring case yield return new object[] { "\u00C9Infinity", "\u00E9", "-", true, double.PositiveInfinity }; - yield return new object[] { "\U00010400Infinity", "\U00010428", "-", true, double.PositiveInfinity }; + // NLS doesn't case-fold supplementary characters + yield return new object[] { "\U00010400Infinity", "\U00010428", "-", !PlatformDetection.IsNlsGlobalization, PlatformDetection.IsNlsGlobalization ? 0.0 : double.PositiveInfinity }; yield return new object[] { "\u200E+\u200EInfinity", "\u200E+\u200E", "\u200E-\u200E", true, double.PositiveInfinity }; yield return new object[] { "\u200E-\u200ENaN", "\u200E+\u200E", "\u200E-\u200E", true, double.NaN };