From a9ac7d309af307bd3acc2dd3861c3d71155cefc4 Mon Sep 17 00:00:00 2001 From: Niklas Schilli Date: Tue, 29 Sep 2026 09:15:49 +0200 Subject: [PATCH 1/2] Fix OOB read in EqualsIgnoreCaseUtf8_Scalar and StartsWithIgnoreCaseUtf8_Scalar The length == 3 tail branch advanced byteOffset by 2 to compose the third byte, but the NonAscii fallback reused the now-stale byteOffset with a length computed from range (which doesn't account for the tail advance). This caused the rune-by-rune fallback to start 2 bytes past the tail with a length 2 bytes too long, reading past both buffers. Use an inline offset (byteOffset + 2) instead of mutating byteOffset. Fixes #134840 --- .../src/System/Globalization/Ordinal.Utf8.cs | 12 ++++-------- 1 file changed, 4 insertions(+), 8 deletions(-) diff --git a/src/libraries/System.Private.CoreLib/src/System/Globalization/Ordinal.Utf8.cs b/src/libraries/System.Private.CoreLib/src/System/Globalization/Ordinal.Utf8.cs index 1d715f72ace4c5..8cdcf235a2f264 100644 --- a/src/libraries/System.Private.CoreLib/src/System/Globalization/Ordinal.Utf8.cs +++ b/src/libraries/System.Private.CoreLib/src/System/Globalization/Ordinal.Utf8.cs @@ -260,10 +260,8 @@ internal static bool EqualsIgnoreCaseUtf8_Scalar(ref byte charA, int lengthA, re valueAu32 = Unsafe.ReadUnaligned(ref Unsafe.AddByteOffset(ref charA, byteOffset)); valueBu32 = Unsafe.ReadUnaligned(ref Unsafe.AddByteOffset(ref charB, byteOffset)); - byteOffset += 2; - - valueAu32 |= (uint)(Unsafe.AddByteOffset(ref charA, byteOffset) << 16); - valueBu32 |= (uint)(Unsafe.AddByteOffset(ref charB, byteOffset) << 16); + valueAu32 |= (uint)(Unsafe.AddByteOffset(ref charA, byteOffset + 2) << 16); + valueBu32 |= (uint)(Unsafe.AddByteOffset(ref charB, byteOffset + 2) << 16); } else if (length == 2) { @@ -570,10 +568,8 @@ internal static bool StartsWithIgnoreCaseUtf8_Scalar(ref byte source, int source valueAu32 = Unsafe.ReadUnaligned(ref Unsafe.AddByteOffset(ref source, byteOffset)); valueBu32 = Unsafe.ReadUnaligned(ref Unsafe.AddByteOffset(ref prefix, byteOffset)); - byteOffset += 2; - - valueAu32 |= (uint)(Unsafe.AddByteOffset(ref source, byteOffset) << 16); - valueBu32 |= (uint)(Unsafe.AddByteOffset(ref prefix, byteOffset) << 16); + valueAu32 |= (uint)(Unsafe.AddByteOffset(ref source, byteOffset + 2) << 16); + valueBu32 |= (uint)(Unsafe.AddByteOffset(ref prefix, byteOffset + 2) << 16); } else if (length == 2) { From 706f2802bd49074bab81fef0510b1751db8df549 Mon Sep 17 00:00:00 2001 From: Niklas Schilli Date: Tue, 29 Sep 2026 09:24:39 +0200 Subject: [PATCH 2/2] Add regression test for UTF-8 special-value parsing OOB read Tests that double.TryParse(ReadOnlySpan) with a custom NumberFormatInfo whose infinity symbol is a 3-byte UTF-8 sequence correctly rejects non-matching inputs regardless of trailing memory content. Covers both the direct 3-byte tail and the 4-byte-loop + 3-byte-tail path. Ref #134840 --- .../System/DoubleTests.cs | 41 +++++++++++++++++++ 1 file changed, 41 insertions(+) diff --git a/src/libraries/System.Runtime/tests/System.Runtime.Tests/System/DoubleTests.cs b/src/libraries/System.Runtime/tests/System.Runtime.Tests/System/DoubleTests.cs index 2e5a5990bf5052..ce8a3481d8a8d5 100644 --- a/src/libraries/System.Runtime/tests/System.Runtime.Tests/System/DoubleTests.cs +++ b/src/libraries/System.Runtime/tests/System.Runtime.Tests/System/DoubleTests.cs @@ -1383,6 +1383,47 @@ public static void TestSpecialValueParsingWithHyphen() Assert.True(double.IsNaN(double.Parse(Encoding.UTF8.GetBytes(value), NumberStyles.Float, format))); } + [Fact] + public static void TestUtf8SpecialValueParsingDoesNotReadPastInput() + { + // Regression test for https://github.com/dotnet/runtime/issues/134840 + // EqualsIgnoreCaseUtf8_Scalar read 2 bytes past both buffers when the + // remaining tail was 3 bytes with non-ASCII data, causing the parse + // result to depend on memory beyond the input span. + + // "\u221e" (infinity symbol used by most ICU cultures) is 3 bytes in + // UTF-8 (E2 88 9E), which triggers the affected tail path. + NumberFormatInfo nfi = new() { PositiveInfinitySymbol = "\u221e" }; + + // "\u301e" encodes as E3 80 9E \u2014 different character, same byte length. + // The char overload correctly rejects it. + Assert.False(double.TryParse("\u301e", NumberStyles.Float, nfi, out _)); + + // The UTF-8 overload must also reject it regardless of trailing memory. + byte[] withNulTrailer = [0xE3, 0x80, 0x9E, 0x00, 0x00]; + byte[] withAsciiTrailer = [0xE3, 0x80, 0x9E, 0x41, 0x41]; + + Assert.False(double.TryParse(withNulTrailer.AsSpan(0, 3), NumberStyles.Float, nfi, out _)); + Assert.False(double.TryParse(withAsciiTrailer.AsSpan(0, 3), NumberStyles.Float, nfi, out _)); + + // Also test after a 4-byte ASCII prefix (exercises the widening loop + // followed by the 3-byte tail): "ABCD\u221e" = 7 bytes UTF-8. + nfi = new() { PositiveInfinitySymbol = "ABCD\u221e" }; + + // Exact match must still succeed. + Assert.True(double.TryParse("ABCD\u221e", NumberStyles.Float, nfi, out double val)); + Assert.True(double.IsPositiveInfinity(val)); + Assert.True(double.TryParse(Encoding.UTF8.GetBytes("ABCD\u221e"), NumberStyles.Float, nfi, out val)); + Assert.True(double.IsPositiveInfinity(val)); + + // "ABCD\u301e" (41 42 43 44 E3 80 9E) must be rejected. + byte[] probe7Nul = [0x41, 0x42, 0x43, 0x44, 0xE3, 0x80, 0x9E, 0x00, 0x00]; + byte[] probe7Ascii = [0x41, 0x42, 0x43, 0x44, 0xE3, 0x80, 0x9E, 0x41, 0x41]; + + Assert.False(double.TryParse(probe7Nul.AsSpan(0, 7), NumberStyles.Float, nfi, out _)); + Assert.False(double.TryParse(probe7Ascii.AsSpan(0, 7), NumberStyles.Float, nfi, out _)); + } + [Theory] [MemberData(nameof(GenericMathTestMemberData.MaxMagnitudeNumberDouble), MemberType = typeof(GenericMathTestMemberData))] public static void MaxMagnitudeNumberTest(double x, double y, double expectedResult)