diff --git a/src/runtime/shell/states/Expansion.rs b/src/runtime/shell/states/Expansion.rs index df8edbcf202d..d6ada80be68b 100644 --- a/src/runtime/shell/states/Expansion.rs +++ b/src/runtime/shell/states/Expansion.rs @@ -35,6 +35,9 @@ pub struct Expansion { /// `transition_to_glob_state`; metacharacter bytes from any other source /// (JS interpolation, quoted text, `$var`, command substitution) are data /// and must not change the expansion structure or broaden the glob. + /// Before the word reaches the glob walker, `retain_brace_syntax` removes + /// the `{`/`,`/`}` entries the brace step treated as text, so the walker + /// (whose matcher reads every `{...}` as a group) sees them as data too. pub(crate) meta_offsets: Vec, pub(crate) child_script: Option, /// Whether the in-flight command substitution was `"$(...)"` (no IFS @@ -254,6 +257,9 @@ impl Expansion { continue; } if atom.has_glob_expansion() { + // The parser set no brace hint (e.g. `{x}.*` has no comma), so + // every recorded `{`/`}`/`,` is text. + Self::retain_brace_syntax(me, &[]); return Self::transition_to_glob_state(interp, this); } Self::push_current_out(me); @@ -351,6 +357,9 @@ impl Expansion { // `current_out` (the original pattern, e.g. `src/*.{ts,tsx}`) so // the glob walker brace-expands and globs it; its matches are // appended after the literal brace variants already pushed above. + // Only the groups the lexer expanded may expand again there: a + // `{x}` it left as text must reach the walker as text too. + Self::retain_brace_syntax(me, &lexer_output.kept_as_syntax); me.state = ExpansionState::Glob; } else { me.current_out.clear(); @@ -358,12 +367,31 @@ impl Expansion { } } + /// Drop from `meta_offsets` every `{`/`,`/`}` the brace step did not keep + /// as brace syntax, so `neutralize_glob_metachars` wraps it like any other + /// data byte. `kept` is the brace lexer's `LexerOutput::kept_as_syntax`: + /// one verdict per unescaped brace byte it saw, and `do_brace_expand` + /// leaves exactly the recorded brace bytes unescaped, so the two line up + /// one to one. A word without brace expansion passes an empty `kept`: + /// every brace byte in it is data. + fn retain_brace_syntax(me: &mut Expansion, kept: &[bool]) { + let mut kept = kept.iter(); + let current_out = &me.current_out; + me.meta_offsets + .retain(|&off| match current_out[off as usize] { + b'{' | b',' | b'}' => kept.next().copied().unwrap_or(false), + _ => true, + }); + debug_assert!(kept.next().is_none()); + } + /// Build the pattern handed to the glob walker from `current_out`, - /// neutralizing every glob metacharacter byte that was *not* written by a - /// literal metacharacter atom (those positions are recorded in - /// `meta_offsets`). Metacharacters arriving via JS `${...}` interpolation, + /// neutralizing every glob metacharacter byte whose offset is not in + /// `meta_offsets`. Metacharacters arriving via JS `${...}` interpolation, /// `$var`, command substitution, or quoted text are data and must not be - /// able to broaden the match. Mirrors `do_brace_expand`'s escaping loop, + /// able to broaden the match, and neither may a template `{x}` the brace + /// step demoted to text (`retain_brace_syntax` has already dropped its + /// offsets). Mirrors `do_brace_expand`'s escaping loop, /// but the glob matcher has no general backslash-escape that survives /// `build_pattern_components` on every platform, so each byte is wrapped /// in a single-character class (`[c]`) — or a one-branch brace group for diff --git a/src/shell_parser/braces.rs b/src/shell_parser/braces.rs index 357c43e1b6a9..cf50e237802a 100644 --- a/src/shell_parser/braces.rs +++ b/src/shell_parser/braces.rs @@ -1031,6 +1031,13 @@ type Chars = ShellCharIter; pub struct LexerOutput { pub tokens: Vec, pub contains_nested: bool, + /// One entry per unescaped `{`, `,` and `}` in the input, in input order: + /// `true` when it is brace syntax in `tokens`, `false` when the lexer + /// demoted it to literal text (a group with no top-level comma, an + /// unclosed group, or a `,`/`}` outside every group). Lets a caller that + /// hands the un-expanded word to another pattern matcher escape exactly + /// the bytes this lexer treated as text. + pub kept_as_syntax: Vec, } pub(crate) type BraceLexerError = AllocError; @@ -1039,6 +1046,7 @@ pub struct NewLexer { chars: Chars, tokens: Vec, contains_nested: bool, + kept_as_syntax: Vec, } impl NewLexer { @@ -1047,6 +1055,7 @@ impl NewLexer { chars: Chars::::init(src), tokens: Vec::new(), contains_nested: false, + kept_as_syntax: Vec::new(), }; let contains_nested = this.tokenize_impl()?; @@ -1054,6 +1063,7 @@ impl NewLexer { Ok(LexerOutput { tokens: this.tokens, contains_nested, + kept_as_syntax: this.kept_as_syntax, }) } @@ -1086,6 +1096,12 @@ impl NewLexer { has_comma: bool, } let mut brace_stack: SmallVec<[OpenBrace; MAX_NESTED_BRACES]> = SmallVec::new(); + // Token index each unescaped `{`/`,`/`}` was emitted at as brace + // syntax, in input order, or `None` when it was emitted as text right + // away. The demotion below and `rollback_braces` turn syntax tokens + // back into text later, so `kept_as_syntax` is read off the tokens once + // the input is exhausted. + let mut brace_chars: Vec> = Vec::new(); loop { let Some(input) = self.eat() else { break }; @@ -1096,33 +1112,40 @@ impl NewLexer { // `char` is u32 (CodepointType unified across encodings). match char { c if c == u32::from(b'{') => { + let tok_idx = self.next_tok_idx(); brace_stack.push(OpenBrace { - tok_idx: u32::try_from(self.tokens.len()).expect("int cast"), + tok_idx, has_comma: false, }); + brace_chars.push(Some(tok_idx)); self.tokens.push(Token::Open(ExpansionVariants::default())); continue; } - c if c == u32::from(b'}') => { - if let Some(top) = brace_stack.pop() { - if top.has_comma { - self.tokens.push(Token::Close); - } else { - // A `{...}` group with no top-level comma is not a - // brace expansion (bash semantics): demote the Open - // back to a literal `{` and emit this `}` as text. - self.replace_token_with_string(top.tok_idx); - self.tokens.push(Token::Text(SmolStr::from_char(b'}'))); - } + c if c == u32::from(b'}') => match brace_stack.pop() { + Some(top) if top.has_comma => { + brace_chars.push(Some(self.next_tok_idx())); + self.tokens.push(Token::Close); continue; } - } + Some(top) => { + // A `{...}` group with no top-level comma is not a + // brace expansion (bash semantics): demote the Open + // back to a literal `{` and emit this `}` as text. + self.replace_token_with_string(top.tok_idx); + brace_chars.push(None); + self.tokens.push(Token::Text(SmolStr::from_char(b'}'))); + continue; + } + None => brace_chars.push(None), + }, c if c == u32::from(b',') => { if let Some(top) = brace_stack.last_mut() { top.has_comma = true; + brace_chars.push(Some(self.next_tok_idx())); self.tokens.push(Token::Comma); continue; } + brace_chars.push(None); } _ => {} } @@ -1139,6 +1162,12 @@ impl NewLexer { self.rollback_braces(top.tok_idx); } + let tokens = &self.tokens; + self.kept_as_syntax = brace_chars + .iter() + .map(|tok_idx| tok_idx.is_some_and(|i| !matches!(tokens[i as usize], Token::Text(_)))) + .collect(); + self.flatten_tokens()?; self.tokens.push(Token::Eof); @@ -1217,6 +1246,11 @@ impl NewLexer { } } + /// Index the next pushed token will get. + fn next_tok_idx(&self) -> u32 { + u32::try_from(self.tokens.len()).expect("int cast") + } + fn replace_token_with_string(&mut self, token_idx: u32) { let tok = &mut self.tokens[token_idx as usize]; let tok_text = tok.to_text(); @@ -1300,4 +1334,40 @@ mod tests { assert_eq!(result.tokens, expected); } } + + #[test] + fn kept_as_syntax() { + let cases: &[(&[u8], &[bool])] = &[ + (b"plain", &[]), + (b"{a,b}", &[true, true, true]), + // No top-level comma: the group is text. + (b"{foo}", &[false, false]), + // The comma sits outside every group. + (b"{a},b", &[false, false, false]), + // A literal inner group inside an expanding outer one. + (b"{a,{b}}", &[true, true, false, false, true]), + // Unclosed outer group: it and its comma roll back, the closed + // inner group keeps expanding. + (b"{a,{b,c},d", &[false, false, true, true, true, false]), + (b"{a,b", &[false, false]), + (b"}{,", &[false, false, false]), + // Escaped characters get no entry at all. + (b"\\{a\\,b\\}", &[]), + (b"\\{a,b\\}", &[false]), + (b"{a\\,b}", &[false, false]), + // Multi-byte text next to the brace characters. + ("é{a,b}😀{c}".as_bytes(), &[true, true, true, false, false]), + ]; + for &(src, expected) in cases { + let result = Lexer::tokenize(src).unwrap(); + assert_eq!(result.kept_as_syntax, expected, "{}", bstr::BStr::new(src)); + let result = NewLexer::<{ Encoding::Wtf8 }>::tokenize(src).unwrap(); + assert_eq!( + result.kept_as_syntax, + expected, + "wtf8 {}", + bstr::BStr::new(src) + ); + } + } } diff --git a/test/js/bun/shell/brace.test.ts b/test/js/bun/shell/brace.test.ts index d7702f235a48..08ef25960a41 100644 --- a/test/js/bun/shell/brace.test.ts +++ b/test/js/bun/shell/brace.test.ts @@ -332,4 +332,123 @@ describe("comma-less brace group is literal (bash 5.2)", () => { exitCode: 1, }); }); + + // The glob matcher reads any `{...}` as a brace group, comma or not. A + // template `{x}` that the brace layer left as text used to reach it as-is, + // so `{x}.*` matched `x.a` and could never match a file named `{x}.a` (bash + // matches `{x}.a`). Interpolated braces were already neutralized; only the + // template's own brace bytes leaked through. + describe.concurrent("a literal brace group globs as literal text", () => { + // Glob results are not sorted, and the walker joins path components with + // the native separator on Windows. + const words = (out: string) => out.trim().replaceAll("\\", "/").split(" ").sort(); + + test("{x}.*: no comma, so the word never enters brace expansion", async () => { + using dir = tempDir("shell-literal-brace-glob", { + "{x}.a.txt": "", + "{x}.b.txt": "", + "x.a.txt": "", + }); + const out = await $`echo {x}.*.txt`.cwd(String(dir)).text(); + expect(words(out)).toEqual(["{x}.a.txt", "{x}.b.txt"]); + }); + + test("{x}* at the start of the word", async () => { + using dir = tempDir("shell-literal-brace-glob-prefix", { + "{x}1.txt": "", + "x1.txt": "", + }); + const out = await $`echo {x}*`.cwd(String(dir)).text(); + expect(words(out)).toEqual(["{x}1.txt"]); + }); + + test("*.{x} at the end of the word", async () => { + using dir = tempDir("shell-literal-brace-glob-suffix", { + "a.{x}": "", + "a.x": "", + }); + const out = await $`echo *.{x}`.cwd(String(dir)).text(); + expect(words(out)).toEqual(["a.{x}"]); + }); + + test("{x}/* names a directory", async () => { + using dir = tempDir("shell-literal-brace-glob-dir", { + "{x}/a.txt": "", + "x/a.txt": "", + }); + const out = await $`echo {x}/*.txt`.cwd(String(dir)).text(); + expect(words(out)).toEqual(["{x}/a.txt"]); + }); + + test("{a,b*: an unclosed group with no `}` in the word", async () => { + // No `}` means no brace hint, so the `{` and the `,` are text. The + // matcher used to choke on the unclosed group and report no matches. + using dir = tempDir("shell-literal-brace-glob-unclosed", { + "{a,b1.txt": "", + "a1.txt": "", + "b1.txt": "", + }); + const out = await $`echo {a,b*`.cwd(String(dir)).text(); + expect(words(out)).toEqual(["{a,b1.txt"]); + }); + + test("{a,b}x{c*: an unclosed group after an expanding one", async () => { + // The lexer rolls the unclosed `{c` back to text while `{a,b}` still + // expands, so only `{a,b}` may expand in the walker as well. The + // literal variants `ax{c*` and `bx{c*` are emitted too, so only the + // matches are asserted. + using dir = tempDir("shell-literal-brace-glob-rollback", { + "ax{c1": "", + "bx{c2": "", + "cx{c3": "", + }); + const out = words(await $`echo {a,b}x{c*`.cwd(String(dir)).text()); + expect(out).toContain("ax{c1"); + expect(out).toContain("bx{c2"); + expect(out).not.toContain("cx{c3"); + }); + + test("{x},*: the brace step runs and demotes every brace byte", async () => { + // The brace and glob hints are both set, but no group expands. The + // literal word `{x},*.txt` is still emitted alongside the matches, so + // only the matches are asserted. + using dir = tempDir("shell-literal-brace-glob-comma", { + "{x},a.txt": "", + "x,a.txt": "", + }); + const out = words(await $`echo {x},*.txt`.cwd(String(dir)).text()); + expect(out).toContain("{x},a.txt"); + expect(out).not.toContain("x,a.txt"); + }); + + test("{a,{x}}.*: a literal group nested in an expanding one", async () => { + // `{a,...}` expands in both the brace lexer and the glob walker; the + // inner `{x}` is text in the lexer and must stay text in the walker. + // The literal variants `a.*.txt` and `{x}.*.txt` are emitted too, so + // only the matches are asserted. + using dir = tempDir("shell-literal-brace-glob-nested", { + "a.1.txt": "", + "{x}.1.txt": "", + "x.1.txt": "", + }); + const out = words(await $`echo {a,{x}}.*.txt`.cwd(String(dir)).text()); + expect(out).toContain("a.1.txt"); + expect(out).toContain("{x}.1.txt"); + expect(out).not.toContain("x.1.txt"); + }); + + test("{a,b},*: a comma outside the group is text, the group still expands", async () => { + // Guards the one-to-one pairing of the lexer's verdicts with the word's + // brace bytes: the stray comma is dropped, `{a,b}` is kept. + using dir = tempDir("shell-literal-brace-glob-stray-comma", { + "a,1.txt": "", + "b,1.txt": "", + "c,1.txt": "", + }); + const out = words(await $`echo {a,b},*.txt`.cwd(String(dir)).text()); + expect(out).toContain("a,1.txt"); + expect(out).toContain("b,1.txt"); + expect(out).not.toContain("c,1.txt"); + }); + }); });