Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
39 changes: 39 additions & 0 deletions TextFilter.Test/TextFilterTests.cs
Original file line number Diff line number Diff line change
Expand Up @@ -1027,6 +1027,45 @@
Assert.IsFalse(TextFilter.IsMatch("docs/readme.md", "-*readme*", TextFilterType.Glob, TextFilterMatchOptions.ByWordAny));
}

[TestMethod]
[DataRow("src/a.cs", "src/**/*.cs")]
[DataRow("src/x/a.cs", "src/**/*.cs")]
[DataRow("src/x/y/a.cs", "src/**/*.cs")]
[DataRow("a.cs", "**/*.cs")]
[DataRow("src/a.cs", "**/*.cs")]
[DataRow("docs/readme.md", "**/readme.md")]
[DataRow(@"src\x\a.cs", @"src\**\*.cs")]
[DataRow("a/b/c/d/e/f.cs", "**/**/**/**/**/f.cs")]
public void GlobStarSlashMatchesZeroOrMoreSegments(string text, string filter)
{
Assert.IsTrue(TextFilter.IsMatch(text, filter, TextFilterType.Glob, TextFilterMatchOptions.ByWholeString));
}

[TestMethod]
[DataRow("docs/a.cs", "src/**/*.cs")]
[DataRow("src/a.md", "src/**/*.cs")]
[DataRow("docs/readme.md.bak", "**/readme.md")]
public void GlobStarSlashStillRejectsNonMatchingPaths(string text, string filter)
{
Assert.IsFalse(TextFilter.IsMatch(text, filter, TextFilterType.Glob, TextFilterMatchOptions.ByWholeString));
}

[TestMethod]
public void GlobStarSlashFiltersPathItems()
{
List<string> result = [.. TextFilter.Filter(["src/a.cs", "src/x/a.cs", "docs/readme.md"], "src/**/*.cs", TextFilterType.Glob, TextFilterMatchOptions.ByWholeString)];

Assert.AreEqual(2, result.Count);

Check warning on line 1058 in TextFilter.Test/TextFilterTests.cs

View check run for this annotation

SonarQubeCloud / SonarCloud Code Analysis

Use 'Assert.HasCount' instead of 'Assert.AreEqual'

See more on https://sonarcloud.io/project/issues?id=ktsu-dev_TextFilter&issues=AaEVSWZJj9Sw1Y3zBg0j&open=AaEVSWZJj9Sw1Y3zBg0j&pullRequest=145
Assert.AreEqual("src/a.cs", result[0]);
Assert.AreEqual("src/x/a.cs", result[1]);
}

[TestMethod]
public void GlobStarSlashMatchesCaseInsensitively()
{
Assert.IsTrue(TextFilter.IsMatch("Docs/README.md", "**/readme.md", TextFilterType.Glob, TextFilterMatchOptions.ByWholeString, TextFilterCaseSensitivity.CaseInsensitive));
}

[TestMethod]
public void GlobLiteralSlashInPatternStillMatchesSlashInText()
{
Expand Down
75 changes: 60 additions & 15 deletions TextFilter/TextFilter.cs
Original file line number Diff line number Diff line change
Expand Up @@ -83,7 +83,7 @@
private static HashSet<char> RequiredTokenPrefixes { get; } = ['+'];
private static ConcurrentDictionary<string, Regex> RegexCache { get; } = [];
// A null entry records a token that could not be parsed, so the parse is not retried on every keystroke.
private static ConcurrentDictionary<string, Glob?> GlobCache { get; } = [];
private static ConcurrentDictionary<string, Glob[]?> GlobCache { get; } = [];

// Filter patterns are caller-supplied, and in the keystroke-driven filter box this library exists
// for, every prefix of what the user types becomes its own key. Unbounded, the caches therefore
Expand Down Expand Up @@ -124,7 +124,7 @@
private static readonly TimeSpan RegexMatchTimeout = TimeSpan.FromSeconds(1);

[System.Diagnostics.CodeAnalysis.SuppressMessage("Performance", "SYSLIB1045:Convert to 'GeneratedRegexAttribute'.", Justification = "Not available in older frameworks")]
private static Regex RegexMatchAnything() => new(".*", RegexOptions.Compiled);

Check warning on line 127 in TextFilter/TextFilter.cs

View workflow job for this annotation

GitHub Actions / ci / .NET / Analyze & Release

Pass a timeout to limit the execution time.

Check warning on line 127 in TextFilter/TextFilter.cs

View workflow job for this annotation

GitHub Actions / ci / .NET / Analyze & Release

Pass a timeout to limit the execution time.

Check warning on line 127 in TextFilter/TextFilter.cs

View workflow job for this annotation

GitHub Actions / ci / .NET / Analyze & Release

Pass a timeout to limit the execution time.

Check warning on line 127 in TextFilter/TextFilter.cs

View workflow job for this annotation

GitHub Actions / ci / .NET / Analyze & Release

Pass a timeout to limit the execution time.

// Stands in for a pattern that has already run out of time once. The timeout bounds a single
// IsMatch call, so without this the same pathological pattern would pay it again for every word
Expand Down Expand Up @@ -285,7 +285,7 @@

return ExcludedTokenPrefixes.Contains(prefix)
? TextFilterTokenType.Excluded
: RequiredTokenPrefixes.Contains(prefix)

Check warning on line 288 in TextFilter/TextFilter.cs

View workflow job for this annotation

GitHub Actions / ci / .NET / Analyze & Release

Extract this nested ternary operation into an independent statement.

Check warning on line 288 in TextFilter/TextFilter.cs

View workflow job for this annotation

GitHub Actions / ci / .NET / Analyze & Release

Extract this nested ternary operation into an independent statement.

Check warning on line 288 in TextFilter/TextFilter.cs

View workflow job for this annotation

GitHub Actions / ci / .NET / Analyze & Release

Extract this nested ternary operation into an independent statement.

Check warning on line 288 in TextFilter/TextFilter.cs

View workflow job for this annotation

GitHub Actions / ci / .NET / Analyze & Release

Extract this nested ternary operation into an independent statement.

Check warning on line 288 in TextFilter/TextFilter.cs

View workflow job for this annotation

GitHub Actions / ci / .NET / Analyze & Release

Extract this nested ternary operation into an independent statement.

Check warning on line 288 in TextFilter/TextFilter.cs

View workflow job for this annotation

GitHub Actions / ci / .NET / Analyze & Release

Extract this nested ternary operation into an independent statement.
? TextFilterTokenType.Required
: TextFilterTokenType.Optional;
})
Expand Down Expand Up @@ -345,7 +345,7 @@

// An unparseable excluded token is skipped rather than treated as match-anything, which here
// would exclude every item while the user is still typing the token.
bool anyExcludedMatches = excludedTokens.Any(filterToken => ResolveGlob(filterToken, caseSensitivity) is Glob glob && textTokens.Any(token => IsGlobMatch(glob, token)));
bool anyExcludedMatches = excludedTokens.Any(filterToken => ResolveGlob(filterToken, caseSensitivity) is Glob[] globs && textTokens.Any(token => IsGlobMatch(globs, token)));

if (anyExcludedMatches)
{
Expand Down Expand Up @@ -389,9 +389,9 @@
Ensure.NotNull(filterToken);
Ensure.NotNull(textTokens);

Glob? glob = ResolveGlob(filterToken, caseSensitivity);
Glob[]? globs = ResolveGlob(filterToken, caseSensitivity);

return glob is null || textTokens.Any(token => IsGlobMatch(glob, token));
return globs is null || textTokens.Any(token => IsGlobMatch(globs, token));
}

/// <summary>
Expand All @@ -406,9 +406,9 @@
Ensure.NotNull(filterToken);
Ensure.NotNull(textTokens);

Glob? glob = ResolveGlob(filterToken, caseSensitivity);
Glob[]? globs = ResolveGlob(filterToken, caseSensitivity);

return glob is null || textTokens.All(token => IsGlobMatch(glob, token));
return globs is null || textTokens.All(token => IsGlobMatch(globs, token));
}

// DotNet.Glob is a file-path glob, so its * and ? stop at / and \. TextFilter filters arbitrary
Expand All @@ -419,35 +419,80 @@
private static string MaskPathSeparators(string value) =>
value.Replace('/', MaskedPathSeparator).Replace('\\', MaskedPathSeparator);

private static bool IsGlobMatch(Glob glob, string textToken) => glob.IsMatch(MaskPathSeparators(textToken));
private static bool IsGlobMatch(Glob[] globs, string textToken)
{
string maskedText = MaskPathSeparators(textToken);
return globs.Any(glob => glob.IsMatch(maskedText));
}

// Masking also hides the separators DotNet.Glob needs to recognise "**/" as "zero or more path
// segments", leaving "**" followed by a literal character. Since * already crosses the masked
// separator, each "**/" is expanded here instead: dropped, for the zero-segment case, or kept as
// "*/", for "anything ending at a separator". A token matches if any expansion does.
private const string MaskedGlobStarSegment = "**\uE000";

// Each "**/" doubles the expansions, so only this many are expanded both ways; any beyond it keep
// only the "*/" form and so need at least one segment.
private const int MaxExpandedGlobStarSegments = 4;

internal static IReadOnlyList<string> ExpandGlobStarSegments(string maskedToken)
{
List<string> expansions = [string.Empty];
int start = 0;
int expanded = 0;
int index;

while ((index = maskedToken.IndexOf(MaskedGlobStarSegment, start, StringComparison.Ordinal)) >= 0)
{
string literal = maskedToken[start..index];
bool expandBothWays = expanded < MaxExpandedGlobStarSegments;
List<string> next = new(expansions.Count * 2);

foreach (string prefix in expansions)
{
next.Add(prefix + literal + "*" + MaskedPathSeparator);
if (expandBothWays)
{
next.Add(prefix + literal);
}
}

expansions = next;
expanded++;
start = index + MaskedGlobStarSegment.Length;
}

string tail = maskedToken[start..];
return [.. expansions.Select(prefix => prefix + tail).Distinct(StringComparer.Ordinal)];
}

// Returns null for a token that cannot be parsed, so each caller can decide what ignoring it means.
private static Glob? ResolveGlob(string filterToken, TextFilterCaseSensitivity caseSensitivity)
private static Glob[]? ResolveGlob(string filterToken, TextFilterCaseSensitivity caseSensitivity)
{
string cacheKey = CacheKey(filterToken, caseSensitivity);

if (!GlobCache.TryGetValue(cacheKey, out Glob? glob))
if (!GlobCache.TryGetValue(cacheKey, out Glob[]? globs))
{
try
{
string maskedToken = MaskPathSeparators(filterToken);
glob = caseSensitivity is TextFilterCaseSensitivity.CaseInsensitive
? Glob.Parse(maskedToken, CaseInsensitiveGlobOptions)
: Glob.Parse(maskedToken);
globs = [.. ExpandGlobStarSegments(maskedToken).Select(pattern => caseSensitivity is TextFilterCaseSensitivity.CaseInsensitive
? Glob.Parse(pattern, CaseInsensitiveGlobOptions)
: Glob.Parse(pattern))];
}
catch (Exception ex) when (ex is not OutOfMemoryException)
{
// DotNet.Glob's tokeniser throws IndexOutOfRangeException on a range left open after the
// dash ("file[0-"), which is ordinary intermediate input while someone types a range into
// a filter box. Cache it as unparseable so the exception is not raised again on every
// keystroke. Caught broadly so the next tokeniser bug is contained too.
glob = null;
globs = null;
}

AddBounded(GlobCache, cacheKey, glob);
AddBounded(GlobCache, cacheKey, globs);
}

return glob;
return globs;
}

/// <summary>
Expand Down
Loading