diff --git a/src/glob/matcher.rs b/src/glob/matcher.rs index a05a6291c1d9..0909e532c396 100644 --- a/src/glob/matcher.rs +++ b/src/glob/matcher.rs @@ -133,7 +133,9 @@ struct Wildcard { /// "**" /// Matches zero or more characters, including path separators. /// Must match a complete path segment, i.e. followed by a path separator or -/// at the end of the pattern. +/// at the end of the pattern. A "**" that ends a brace branch qualifies when +/// the brace group itself is followed by a separator or ends the pattern, so +/// "a/{**,b}/c" means "a/**/c" or "a/b/c". /// "[ab]" /// Matches one of the characters contained in the brackets. /// Character ranges (e.g. "[a-z]") are also supported. @@ -220,14 +222,15 @@ fn glob_match_impl( if is_globstar { state.glob_index += 2; - let is_end_invalid = (state.glob_index as usize) < glob.len(); + let rest = globstar_continuation(state, glob, brace_stack); + let is_end_invalid = (rest.glob_index as usize) < glob.len(); // FIXME: explain this bug fix if is_end_invalid && state.path_index as usize == path.len() - && glob.len() - state.glob_index as usize == 2 - && is_separator(glob[state.glob_index as usize]) - && glob[state.glob_index as usize + 1] == b'*' + && glob.len() - rest.glob_index as usize == 2 + && is_separator(glob[rest.glob_index as usize]) + && glob[rest.glob_index as usize + 1] == b'*' { continue 'main_loop; } @@ -237,8 +240,10 @@ fn glob_match_impl( // if we start at index 6 (start of **/b pattern), we don't want to index into the pattern before it if (state.glob_index.saturating_sub(glob_start) < 3 || glob[state.glob_index as usize - 3] == b'/') - && (!is_end_invalid || glob[state.glob_index as usize] == b'/') + && (!is_end_invalid || glob[rest.glob_index as usize] == b'/') { + state.glob_index = rest.glob_index; + state.brace_depth = rest.brace_depth; if is_end_invalid { state.glob_index += 1; } @@ -571,6 +576,22 @@ fn skip_branch(state: &mut State, glob: &[u8], brace_stack: &BraceStack) -> bool false } +/// `state` has just stepped over a `**`; returns a copy of it moved to where the +/// pattern continues. When the `**` ends a brace branch, that is after the group's +/// `}` (and after the `}` of every enclosing group the `**` also ends), i.e. where +/// [`skip_branch`] would resume matching. Deciding "is this `**` a whole segment" +/// there gives `a/{**,b}/c` the meaning of its expansion `a/**/c`; decided at the +/// `,` itself, the `**` would be demoted to a plain `*`. +#[inline(always)] +fn globstar_continuation(state: &State, glob: &[u8], brace_stack: &BraceStack) -> State { + let mut rest = *state; + while rest.brace_depth > 0 + && matches!(glob.get(rest.glob_index as usize), Some(b',' | b'}')) + && skip_branch(&mut rest, glob, brace_stack) + {} + rest +} + /// Index of the `}` matching the `{` at `open_idx`, or `glob.len()` if unterminated. fn find_brace_end(glob: &[u8], open_idx: u32) -> u32 { let mut i = open_idx as usize; diff --git a/test/js/bun/glob/match.test.ts b/test/js/bun/glob/match.test.ts index 50149b91c080..23bf3259baea 100644 --- a/test/js/bun/glob/match.test.ts +++ b/test/js/bun/glob/match.test.ts @@ -413,6 +413,75 @@ describe("Glob.match", () => { expect(new Glob("{a,b},x").match("b,x")).toBeTrue(); }); + test("`**` that ends a brace branch is a globstar", () => { + // A brace group matches one of its branches, so each pattern below must + // match exactly what the brace-free patterns in its comment match. The `**` + // used to be demoted to a single-segment `*` whenever the byte after it was + // the `,` or `}` of the group instead of `/` or the end of the pattern. + const cases: [pattern: string, matches: string[], rejects: string[]][] = [ + // a/** | a/b + ["a/{**,b}", ["a/x", "a/x/y", "a/x/y/z", "a/b", "a/b/c", "a/"], ["a", "ab", "c/x/y"]], + // a/b | a/** (branch closed by `}` rather than `,`) + ["a/{b,**}", ["a/b", "a/x/y"], ["a", "b/x"]], + // a/** | a/ (empty branch) + ["a/{**,}", ["a/x/y", "a/"], ["a"]], + ["a/{,**}", ["a/x/y", "a/"], ["a"]], + // a/** | b (the `**` follows a `/` inside the branch) + ["{a/**,b}", ["a/", "a/x", "a/x/y", "b"], ["a", "b/x", "c/x"]], + // ** | b (the group is the whole pattern) + ["{**,b}", ["x", "x/", "x/y/z", "b"], []], + // a/**/c | a/b/c + ["a/{**,b}/c", ["a/c", "a/x/c", "a/x/y/c", "a/b/c", "a/b/x/c"], ["a/x/y/d", "a/x/y/c/d", "a/bc", "c"]], + // test/foo/**/baz | test/bar/baz + ["test/{foo/**,bar}/baz", ["test/foo/baz", "test/foo/x/y/baz", "test/bar/baz"], ["test/bar/x/baz", "test/baz"]], + // **/c | b/c + ["{**,b}/c", ["c", "x/c", "x/y/c", "b/x/c"], ["x/y/d", "cc"]], + // a/**/c | a/**/d | a/b/c | a/b/d (a later group backtracks into the globstar) + ["a/{**,b}/{c,d}", ["a/c", "a/d", "a/x/y/c", "a/x/y/d", "a/b/x/d"], ["a/x/y/e", "a/x/y", "a/cd"]], + // a/**/* | a/b/* (the trailing `/*` still needs a non-empty final segment) + ["a/{**,b}/*", ["a/x", "a/x/y", "a/b/z"], ["a/x/", "a/", "a"]], + // src/**/*.ts | src/lib/*.ts + ["src/{**,lib}/*.ts", ["src/a.ts", "src/x/a.ts", "src/x/y/a.ts", "src/lib/a.ts"], ["src/x/y/a.js", "lib/a.ts"]], + // x/**/y/** | x/**/y/c | x/b/y/** | x/b/y/c (two such groups in one pattern) + ["x/{**,b}/y/{**,c}", ["x/1/2/y/3/4", "x/y/", "x/y/3", "x/b/y/c", "x/1/y/c/2"], ["x/y", "x/1/z/3"]], + // **/**/b | **/a/b (a globstar before the group) + ["**/{**,a}/b", ["b", "x/b", "x/y/b", "x/a/b"], ["x/y/c", "bb"]], + // a/c | a/** | a/d (nested: the `**` ends the inner branch and the outer one) + ["a/{c,{**,d}}", ["a/c", "a/d", "a/x/y", "a/"], ["a", "b/x"]], + // a/c/e | a/**/e | a/d/e + ["a/{c,{**,d}}/e", ["a/e", "a/c/e", "a/x/y/e", "a/c/x/e"], ["a/x/y/f", "a/x/y"]], + // a/**/d | a/c/d | a/b/d + ["a/{{**,c},b}/d", ["a/d", "a/x/y/d", "a/b/x/d"], ["a/x/y", "a/x/y/e"]], + // !(a/** | a/b) + ["!a/{**,b}", ["a", "c/x/y"], ["a/x/y", "a/b"]], + + // The group continues with something other than `/`: `a/**x` is not a + // whole segment, so the `**` stays a single-segment `*` there. + // a/**x | a/bx + ["a/{**,b}x", ["a/yx", "a/bx", "a/x"], ["a/y/zx"]], + // a/c | a/**x | a/dx (the inner group closes but the outer branch continues) + ["a/{c,{**,d}x}", ["a/c", "a/yx", "a/dx"], ["a/y/zx"]], + // a/**c | a/**d | a/bc | a/bd + ["a/{**,b}{c,d}", ["a/c", "a/yd", "a/bc"], ["a/x/yc"]], + // A `**` that does not start the segment is a `*` no matter what follows it. + // a/x** | a/b + ["a/{x**,b}", ["a/xy", "a/b"], ["a/x/y"]], + // A `,` or `}` outside any brace group is a literal, not the end of a branch. + ["**,x", ["a,x"], ["a/b,x"]], + ["{a,b}/**,x", ["a/c,x", "b/,x"], ["a/c/d,x", "a/c"]], + ]; + + for (const [pattern, matches, rejects] of cases) { + const glob = new Glob(pattern); + for (const path of matches) { + expect(glob.match(path), `${JSON.stringify(pattern)} should match ${JSON.stringify(path)}`).toBeTrue(); + } + for (const path of rejects) { + expect(glob.match(path), `${JSON.stringify(pattern)} should not match ${JSON.stringify(path)}`).toBeFalse(); + } + } + }); + // Most of the potential bugs when dealing with non-ASCII patterns is when the // pattern matching algorithm wants to deal with single chars, for example // using the `[...]` syntax, it tries to match each char in the brackets. With