Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 27 additions & 6 deletions src/glob/matcher.rs
Original file line number Diff line number Diff line change
Expand Up @@ -133,7 +133,9 @@ struct Wildcard {
/// "**"
/// Matches zero or more characters, including path separators.
/// Must match a complete path segment, i.e. followed by a path separator or
/// at the end of the pattern.
/// at the end of the pattern. A "**" that ends a brace branch qualifies when
/// the brace group itself is followed by a separator or ends the pattern, so
/// "a/{**,b}/c" means "a/**/c" or "a/b/c".
/// "[ab]"
/// Matches one of the characters contained in the brackets.
/// Character ranges (e.g. "[a-z]") are also supported.
Expand Down Expand Up @@ -220,14 +222,15 @@ fn glob_match_impl(
if is_globstar {
state.glob_index += 2;

let is_end_invalid = (state.glob_index as usize) < glob.len();
let rest = globstar_continuation(state, glob, brace_stack);
let is_end_invalid = (rest.glob_index as usize) < glob.len();

// FIXME: explain this bug fix
if is_end_invalid
&& state.path_index as usize == path.len()
&& glob.len() - state.glob_index as usize == 2
&& is_separator(glob[state.glob_index as usize])
&& glob[state.glob_index as usize + 1] == b'*'
&& glob.len() - rest.glob_index as usize == 2
&& is_separator(glob[rest.glob_index as usize])
&& glob[rest.glob_index as usize + 1] == b'*'
{
continue 'main_loop;
}
Expand All @@ -237,8 +240,10 @@ fn glob_match_impl(
// if we start at index 6 (start of **/b pattern), we don't want to index into the pattern before it
if (state.glob_index.saturating_sub(glob_start) < 3
|| glob[state.glob_index as usize - 3] == b'/')
&& (!is_end_invalid || glob[state.glob_index as usize] == b'/')
&& (!is_end_invalid || glob[rest.glob_index as usize] == b'/')
{
state.glob_index = rest.glob_index;
state.brace_depth = rest.brace_depth;
if is_end_invalid {
state.glob_index += 1;
}
Expand Down Expand Up @@ -571,6 +576,22 @@ fn skip_branch(state: &mut State, glob: &[u8], brace_stack: &BraceStack) -> bool
false
}

/// `state` has just stepped over a `**`; returns a copy of it moved to where the
/// pattern continues. When the `**` ends a brace branch, that is after the group's
/// `}` (and after the `}` of every enclosing group the `**` also ends), i.e. where
/// [`skip_branch`] would resume matching. Deciding "is this `**` a whole segment"
/// there gives `a/{**,b}/c` the meaning of its expansion `a/**/c`; decided at the
/// `,` itself, the `**` would be demoted to a plain `*`.
#[inline(always)]
fn globstar_continuation(state: &State, glob: &[u8], brace_stack: &BraceStack) -> State {
let mut rest = *state;
while rest.brace_depth > 0
&& matches!(glob.get(rest.glob_index as usize), Some(b',' | b'}'))
&& skip_branch(&mut rest, glob, brace_stack)
{}
rest
}

/// Index of the `}` matching the `{` at `open_idx`, or `glob.len()` if unterminated.
fn find_brace_end(glob: &[u8], open_idx: u32) -> u32 {
let mut i = open_idx as usize;
Expand Down
69 changes: 69 additions & 0 deletions test/js/bun/glob/match.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -413,6 +413,75 @@ describe("Glob.match", () => {
expect(new Glob("{a,b},x").match("b,x")).toBeTrue();
});

test("`**` that ends a brace branch is a globstar", () => {
// A brace group matches one of its branches, so each pattern below must
// match exactly what the brace-free patterns in its comment match. The `**`
// used to be demoted to a single-segment `*` whenever the byte after it was
// the `,` or `}` of the group instead of `/` or the end of the pattern.
const cases: [pattern: string, matches: string[], rejects: string[]][] = [
// a/** | a/b
["a/{**,b}", ["a/x", "a/x/y", "a/x/y/z", "a/b", "a/b/c", "a/"], ["a", "ab", "c/x/y"]],
// a/b | a/** (branch closed by `}` rather than `,`)
["a/{b,**}", ["a/b", "a/x/y"], ["a", "b/x"]],
// a/** | a/ (empty branch)
["a/{**,}", ["a/x/y", "a/"], ["a"]],
["a/{,**}", ["a/x/y", "a/"], ["a"]],
// a/** | b (the `**` follows a `/` inside the branch)
["{a/**,b}", ["a/", "a/x", "a/x/y", "b"], ["a", "b/x", "c/x"]],
// ** | b (the group is the whole pattern)
["{**,b}", ["x", "x/", "x/y/z", "b"], []],
// a/**/c | a/b/c
["a/{**,b}/c", ["a/c", "a/x/c", "a/x/y/c", "a/b/c", "a/b/x/c"], ["a/x/y/d", "a/x/y/c/d", "a/bc", "c"]],
// test/foo/**/baz | test/bar/baz
["test/{foo/**,bar}/baz", ["test/foo/baz", "test/foo/x/y/baz", "test/bar/baz"], ["test/bar/x/baz", "test/baz"]],
// **/c | b/c
["{**,b}/c", ["c", "x/c", "x/y/c", "b/x/c"], ["x/y/d", "cc"]],
// a/**/c | a/**/d | a/b/c | a/b/d (a later group backtracks into the globstar)
["a/{**,b}/{c,d}", ["a/c", "a/d", "a/x/y/c", "a/x/y/d", "a/b/x/d"], ["a/x/y/e", "a/x/y", "a/cd"]],
// a/**/* | a/b/* (the trailing `/*` still needs a non-empty final segment)
["a/{**,b}/*", ["a/x", "a/x/y", "a/b/z"], ["a/x/", "a/", "a"]],
// src/**/*.ts | src/lib/*.ts
["src/{**,lib}/*.ts", ["src/a.ts", "src/x/a.ts", "src/x/y/a.ts", "src/lib/a.ts"], ["src/x/y/a.js", "lib/a.ts"]],
// x/**/y/** | x/**/y/c | x/b/y/** | x/b/y/c (two such groups in one pattern)
["x/{**,b}/y/{**,c}", ["x/1/2/y/3/4", "x/y/", "x/y/3", "x/b/y/c", "x/1/y/c/2"], ["x/y", "x/1/z/3"]],
// **/**/b | **/a/b (a globstar before the group)
["**/{**,a}/b", ["b", "x/b", "x/y/b", "x/a/b"], ["x/y/c", "bb"]],
// a/c | a/** | a/d (nested: the `**` ends the inner branch and the outer one)
["a/{c,{**,d}}", ["a/c", "a/d", "a/x/y", "a/"], ["a", "b/x"]],
// a/c/e | a/**/e | a/d/e
["a/{c,{**,d}}/e", ["a/e", "a/c/e", "a/x/y/e", "a/c/x/e"], ["a/x/y/f", "a/x/y"]],
// a/**/d | a/c/d | a/b/d
["a/{{**,c},b}/d", ["a/d", "a/x/y/d", "a/b/x/d"], ["a/x/y", "a/x/y/e"]],
// !(a/** | a/b)
["!a/{**,b}", ["a", "c/x/y"], ["a/x/y", "a/b"]],

// The group continues with something other than `/`: `a/**x` is not a
// whole segment, so the `**` stays a single-segment `*` there.
// a/**x | a/bx
["a/{**,b}x", ["a/yx", "a/bx", "a/x"], ["a/y/zx"]],
// a/c | a/**x | a/dx (the inner group closes but the outer branch continues)
["a/{c,{**,d}x}", ["a/c", "a/yx", "a/dx"], ["a/y/zx"]],
// a/**c | a/**d | a/bc | a/bd
["a/{**,b}{c,d}", ["a/c", "a/yd", "a/bc"], ["a/x/yc"]],
// A `**` that does not start the segment is a `*` no matter what follows it.
// a/x** | a/b
["a/{x**,b}", ["a/xy", "a/b"], ["a/x/y"]],
// A `,` or `}` outside any brace group is a literal, not the end of a branch.
["**,x", ["a,x"], ["a/b,x"]],
["{a,b}/**,x", ["a/c,x", "b/,x"], ["a/c/d,x", "a/c"]],
];

for (const [pattern, matches, rejects] of cases) {
const glob = new Glob(pattern);
for (const path of matches) {
expect(glob.match(path), `${JSON.stringify(pattern)} should match ${JSON.stringify(path)}`).toBeTrue();
}
for (const path of rejects) {
expect(glob.match(path), `${JSON.stringify(pattern)} should not match ${JSON.stringify(path)}`).toBeFalse();
}
}
});

// Most of the potential bugs when dealing with non-ASCII patterns is when the
// pattern matching algorithm wants to deal with single chars, for example
// using the `[...]` syntax, it tries to match each char in the brackets. With
Expand Down
Loading