Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
42 changes: 25 additions & 17 deletions src/runtime/shell/states/Expansion.rs
Original file line number Diff line number Diff line change
Expand Up @@ -318,24 +318,32 @@ impl Expansion {
return;
}
let count = count as usize;
let mut expanded: Vec<Vec<u8>> = (0..count).map(|_| Vec::new()).collect();

let arena = bun_alloc::Arena::new();
if let Err(e) = braces::expand(
&arena,
&mut lexer_output.tokens[..],
&mut expanded[..],
lexer_output.contains_nested,
) {
if matches!(e, braces::ParserError::TooManyBraces) {
let msg = "too many braces in brace expansion".to_string();
me.state = ExpansionState::Err(Box::new(ShellErr::Custom(msg.into_bytes().into())));
return;
let expanded: Vec<Vec<u8>> = if count == 0 {
// No `{...}` group formed an expansion (no top-level comma, or
// unbalanced); emit the word unchanged. `expand` would index
// `out[0]` of an empty slice here. Keep `current_out` for glob.
vec![me.current_out.clone()]
} else {
let mut expanded: Vec<Vec<u8>> = (0..count).map(|_| Vec::new()).collect();
let arena = bun_alloc::Arena::new();
if let Err(e) = braces::expand(
&arena,
&mut lexer_output.tokens[..],
&mut expanded[..],
lexer_output.contains_nested,
) {
if matches!(e, braces::ParserError::TooManyBraces) {
let msg = "too many braces in brace expansion".to_string();
me.state =
ExpansionState::Err(Box::new(ShellErr::Custom(msg.into_bytes().into())));
return;
}
// An unexpected token from brace expansion is a parser bug.
panic!("unexpected error from Braces.expand: {e:?}");
}
// An unexpected token from brace expansion is a parser bug.
panic!("unexpected error from Braces.expand: {e:?}");
}
drop(arena);
drop(arena);
expanded
};

// Push each variant as its own word; word boundaries are recorded
// via `bounds`.
Expand Down
106 changes: 59 additions & 47 deletions src/shell_parser/braces.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1219,7 +1219,12 @@ impl<const ENCODING: Encoding> NewLexer<ENCODING> {
// - If unclosed or encounter bad token:
// - Start at beginning of brace, replacing special tokens back with
// chars, skipping over actual closed braces
let mut brace_stack: SmallVec<[u32; MAX_NESTED_BRACES]> = SmallVec::new();
#[derive(Copy, Clone)]
struct OpenBrace {
tok_idx: u32,
has_comma: bool,
}
let mut brace_stack: SmallVec<[OpenBrace; MAX_NESTED_BRACES]> = SmallVec::new();

loop {
let Some(input) = self.eat() else { break };
Expand All @@ -1230,19 +1235,30 @@ impl<const ENCODING: Encoding> NewLexer<ENCODING> {
// `char` is u32 (CodepointType unified across encodings).
match char {
c if c == u32::from(b'{') => {
brace_stack.push(u32::try_from(self.tokens.len()).expect("int cast"));
brace_stack.push(OpenBrace {
tok_idx: u32::try_from(self.tokens.len()).expect("int cast"),
has_comma: false,
});
self.tokens.push(Token::Open(ExpansionVariants::default()));
continue;
}
c if c == u32::from(b'}') => {
if brace_stack.len() > 0 {
let _ = brace_stack.pop();
self.tokens.push(Token::Close);
if let Some(top) = brace_stack.pop() {
if top.has_comma {
self.tokens.push(Token::Close);
} else {
// A `{...}` group with no top-level comma is not a
// brace expansion (bash semantics): demote the Open
// back to a literal `{` and emit this `}` as text.
self.replace_token_with_string(top.tok_idx);
self.tokens.push(Token::Text(SmolStr::from_char(b'}')));
}
continue;
Comment thread
robobun marked this conversation as resolved.
}
}
c if c == u32::from(b',') => {
if brace_stack.len() > 0 {
if let Some(top) = brace_stack.last_mut() {
top.has_comma = true;
self.tokens.push(Token::Comma);
continue;
}
Expand All @@ -1258,9 +1274,8 @@ impl<const ENCODING: Encoding> NewLexer<ENCODING> {
}

// Unclosed braces
while brace_stack.len() > 0 {
let top_idx = brace_stack.pop().unwrap();
self.rollback_braces(top_idx);
while let Some(top) = brace_stack.pop() {
self.rollback_braces(top.tok_idx);
}

self.flatten_tokens()?;
Expand All @@ -1270,44 +1285,34 @@ impl<const ENCODING: Encoding> NewLexer<ENCODING> {
}

fn flatten_tokens(&mut self) -> Result<(), AllocError> {
if self.tokens.is_empty() {
return Ok(());
}
let mut brace_count: u32 = if matches!(self.tokens[0], Token::Open(_)) {
1
} else {
0
};
let mut i: u32 = 0;
let mut j: u32 = 1;
while (i as usize) < self.tokens.len() && (j as usize) < self.tokens.len() {
// reshaped for borrowck — branch on tags first, then borrow once.
let itok_is_text = matches!(self.tokens[i as usize], Token::Text(_));
let jtok_is_text = matches!(self.tokens[j as usize], Token::Text(_));

if itok_is_text && jtok_is_text {
let jtok_text = self.tokens[j as usize].to_text();
if let Token::Text(itxt) = &mut self.tokens[i as usize] {
itxt.append_slice(jtok_text.slice())?;
}
let _ = self.tokens.remove(j as usize);
} else {
match &self.tokens[j as usize] {
Token::Close => {
brace_count -= 1;
}
Token::Open(_) => {
brace_count += 1;
if brace_count > 1 {
self.contains_nested = true;
}
let mut brace_count: u32 = 0;
let mut write = 0usize;
for read in 0..self.tokens.len() {
match &self.tokens[read] {
Token::Open(_) => {
brace_count += 1;
if brace_count > 1 {
self.contains_nested = true;
}
_ => {}
}
i += 1;
j += 1;
Token::Close => brace_count -= 1,
_ => {}
}
if write > 0
&& matches!(self.tokens[write - 1], Token::Text(_))
&& matches!(self.tokens[read], Token::Text(_))
{
let taken = core::mem::replace(&mut self.tokens[read], Token::Eof);
if let (Token::Text(prev), Token::Text(cur)) = (&mut self.tokens[write - 1], taken)
{
prev.append_slice(cur.slice())?;
}
} else {
self.tokens.swap(write, read);
write += 1;
}
}
self.tokens.truncate(write);
Ok(())
}

Expand Down Expand Up @@ -1401,19 +1406,26 @@ mod tests {
fn lexer() {
struct TestCase(&'static [u8], Vec<Token>);
let test_cases: Vec<TestCase> = vec![
// No comma: the whole group is literal text.
TestCase(
b"{}",
vec![Token::Text(SmolStr::from_slice(b"{}").unwrap()), Token::Eof],
),
TestCase(
b"{foo}",
vec![
Token::Open(ExpansionVariants::default()),
Token::Close,
Token::Text(SmolStr::from_slice(b"{foo}").unwrap()),
Token::Eof,
],
),
// With a comma: a real brace group.
TestCase(
b"{foo}",
b"{a,b}",
vec![
Token::Open(ExpansionVariants::default()),
Token::Text(SmolStr::from_slice(b"foo").unwrap()),
Token::Text(SmolStr::from_slice(b"a").unwrap()),
Token::Comma,
Token::Text(SmolStr::from_slice(b"b").unwrap()),
Token::Close,
Token::Eof,
],
Expand Down
97 changes: 90 additions & 7 deletions test/js/bun/shell/brace.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -29,22 +29,33 @@ describe("$.braces", () => {
});

test("nested sibling product", () => {
expect($.braces(`{{d,e}{g,h}}`)).toEqual(["dg", "dh", "eg", "eh"]);
// The outer `{...}` has no comma of its own, so it is literal (bash 5.2).
expect($.braces(`{{d,e}{g,h}}`)).toEqual(["{dg}", "{dh}", "{eg}", "{eh}"]);
});

test("nested sibling product with surrounding text", () => {
expect($.braces(`pre{{a,b}{c,d}}post`)).toEqual(["preacpost", "preadpost", "prebcpost", "prebdpost"]);
expect($.braces(`pre{{a,b}{c,d}}post`)).toEqual(["pre{ac}post", "pre{ad}post", "pre{bc}post", "pre{bd}post"]);
});

test("nested sibling product mixed with variants", () => {
expect($.braces(`{a,{b,c}{d,e},f}`)).toEqual(["a", "bd", "be", "cd", "ce", "f"]);
});

test("nested sibling product triple", () => {
expect($.braces(`{{a,b}{c,d}{e,f}}`)).toEqual(["ace", "acf", "ade", "adf", "bce", "bcf", "bde", "bdf"]);
expect($.braces(`{{a,b}{c,d}{e,f}}`)).toEqual([
"{ace}",
"{acf}",
"{ade}",
"{adf}",
"{bce}",
"{bcf}",
"{bde}",
"{bdf}",
]);
});

test("very deeply nested", () => {
// The innermost `{17}` has no comma, so it is literal (bash 5.2).
const result = $.braces(`{1,{2,{3,{4,{5,{6,{7,{8,{9,{10,{11,{12,{13,{14,{15,{16,{17}}}}}}}}}}}}}}}}}`);
expect(result).toEqual([
"1",
Expand All @@ -63,7 +74,7 @@ describe("$.braces", () => {
"14",
"15",
"16",
"17",
"{17}",
]);
});

Expand Down Expand Up @@ -141,27 +152,99 @@ describe("$.braces input bounds", () => {
cmd: [
bunExe(),
"-e",
`const pattern = Buffer.alloc(50000, "{").toString() + Buffer.alloc(50000, "}").toString();
`const deep = Buffer.alloc(100000, "{,").toString() + Buffer.alloc(50000, "}").toString();
try {
Bun.$.braces(pattern);
Bun.$.braces(deep);
console.log("expanded");
} catch (e) {
console.log("rejected: " + e.message);
}
// The same shape with no commas is one literal word, not a brace expansion.
const literal = Buffer.alloc(50000, "{").toString() + Buffer.alloc(50000, "}").toString();
console.log(JSON.stringify(Bun.$.braces(literal)) === JSON.stringify([literal]));
// A reasonable pattern still expands normally.
console.log(JSON.stringify(Bun.$.braces("echo {a,b}")));`,
],
env: bunEnv,
stdout: "pipe",
stderr: "pipe",
stderr: "inherit",
});

const [stdout, exitCode] = await Promise.all([proc.stdout.text(), proc.exited]);

expect(normalizeBunSnapshot(stdout)).toMatchInlineSnapshot(`
"rejected: Too many braces in brace expansion
true
["echo a","echo b"]"
`);
expect(exitCode).toBe(0);
});
});

// A `{...}` group with no top-level comma is literal text in bash. The brace
// lexer used to tokenize it as Open/Close regardless, so a nested `{}` became
// a zero-variant expansion and the expander dropped the rest of the word.
describe("comma-less brace group is literal (bash 5.2)", () => {
const cases: [string, string[]][] = [
// Regressions: the `{}` (and the tail after it) was truncated.
["x{a,{}}y", ["xay", "x{}y"]],
["p{q{},r}s", ["pq{}s", "prs"]],
["{a,b{}}z", ["az", "b{}z"]],
["{a,{}}z", ["az", "{}z"]],
["a{{,}}b", ["a{}b", "a{}b"]],
// `{foo}` with no comma is literal at any depth.
["{a,{b}}", ["a", "{b}"]],
["{a{b,c}}", ["{ab}", "{ac}"]],
// A comma outside every `{...}` does not make one expand.
["{foo},x", ["{foo},x"]],
["{a},{b}", ["{a},{b}"]],
// Controls that were already correct.
["a{b,{c,d}}e", ["abe", "ace", "ade"]],
["{a,b}", ["a", "b"]],
];

for (const [input, expected] of cases) {
test(`$.braces(${JSON.stringify(input)})`, () => {
expect($.braces(input)).toEqual(expected);
});
}

test("shell: literal {} inside an expanding group keeps the tail", async () => {
// Subprocess so the pre-fix `}{,` panic is observed as a non-zero exit;
// `echo` is a builtin so argv is observed exactly on every platform.
const script = `
const { $ } = require("bun");
$.nothrow();
const cases = ${JSON.stringify([...cases, ["}{,", ["}{,"]]])};
for (const [input] of cases) {
const { stdout } = await $\`echo \${{ raw: input }}\`.quiet();
console.log(JSON.stringify([input, stdout.toString().slice(0, -1)]));
}
`;
Comment thread
robobun marked this conversation as resolved.
await using proc = Bun.spawn({
cmd: [bunExe(), "-e", script],
env: bunEnv,
stdout: "pipe",
stderr: "inherit",
});
const [stdout, exitCode] = await Promise.all([proc.stdout.text(), proc.exited]);
const lines = stdout
.trim()
.split("\n")
.map(l => JSON.parse(l));
expect(lines).toEqual([...cases.map(([input, expected]) => [input, expected.join(" ")]), ["}{,", "}{,"]]);
expect(exitCode).toBe(0);
});

test("a word with a comma-less brace group and a glob keeps its pattern", async () => {
// `{x},*.txt` sets both the brace and glob hints; after the lexer demotes
// `{x}` to text the brace-expand count is 0. The original pattern must
// still reach the glob walker rather than being taken as the literal word.
using dir = tempDir("shell-brace-literal-glob", { "a.txt": "" });
const { stderr, exitCode } = await $`echo {x},*.txt`.cwd(String(dir)).nothrow().quiet();
expect({ stderr: stderr.toString(), exitCode }).toEqual({
stderr: "bun: no matches found: {x},*.txt\n",
exitCode: 1,
});
});
});
2 changes: 1 addition & 1 deletion test/js/bun/shell/bunshell.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -963,7 +963,7 @@ booga"

doTest(
`{1,{2,{3,{4,{5,{6,{7,{8,{9,{10,{11,{12,{13,{14,{15,{16,{17}}}}}}}}}}}}}}}}}`,
"1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17",
"1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 {17}",
);
});

Expand Down
Loading