From ba1b441d09bf3fc338a460893125e2752efd7702 Mon Sep 17 00:00:00 2001 From: robobun <117481402+robobun@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:29:59 +0000 Subject: [PATCH] shell: break words on a tab in the lexer The lexer's word-break arm matched only a space. An unquoted tab fell through to the default arm and joined the current word, so a tab-indented script failed with "command not found: \techo" and `echo ab` passed one argument. The arm now matches a tab too. A tab inside quotes, or after a backslash, stays a literal character. --- src/shell_parser/parse.rs | 3 ++- test/js/bun/shell/bunshell.test.ts | 38 ++++++++++++++++++++++++++++++ test/js/bun/shell/lex.test.ts | 31 ++++++++++++++++++++++++ 3 files changed, 71 insertions(+), 1 deletion(-) diff --git a/src/shell_parser/parse.rs b/src/shell_parser/parse.rs index abba829b5596..ce47517ddcd1 100644 --- a/src/shell_parser/parse.rs +++ b/src/shell_parser/parse.rs @@ -2937,8 +2937,9 @@ impl<'bump, const ENCODING: StringEncoding> Lexer<'bump, ENCODING> { fell_through = true; } // 3. Word breakers - c if c == u32::from(b' ') => { + c if c == u32::from(b' ') || c == u32::from(b'\t') => { const _: () = assert!(SPECIAL_CHARS_TABLE.is_set(b' ' as usize)); + const _: () = assert!(SPECIAL_CHARS_TABLE.is_set(b'\t' as usize)); if self.chars.state == CharState::Normal { self.break_word(AddDelimiter::AfterWord)?; fell_through = true; diff --git a/test/js/bun/shell/bunshell.test.ts b/test/js/bun/shell/bunshell.test.ts index 387e7e7f1ec8..a31f30adae83 100644 --- a/test/js/bun/shell/bunshell.test.ts +++ b/test/js/bun/shell/bunshell.test.ts @@ -888,6 +888,44 @@ bar\n`, .runAsTest("Double Quote Dollar Paren Single Quote"); }); + // A tab outside quotes separates words, as a space does. + describe("tab", () => { + const script = (source: string) => TestBuilder.command`${{ raw: source }}`; + + script("if true; then\n\techo tabbed\nfi").stdout("tabbed\n").runAsTest("indents the body of an if clause"); + + script("echo a\tb\t\tc").stdout("a b c\n").runAsTest("separates arguments"); + + script("echo a\t|\tcat\t&&\techo b").stdout("a\nb\n").runAsTest("surrounds operators"); + + script("FOO=bar\tBAZ=qux\necho $FOO $BAZ").stdout("bar qux\n").runAsTest("separates assignments"); + + script("if [[\t-n a\t]]; then echo yes; fi").stdout("yes\n").runAsTest("separates the operands of [[ ]]"); + + script(`echo "a\tb" 'c\td' e\\\tf`) + .stdout("a\tb c\td e\tf\n") + .runAsTest("is a literal character in quotes and after a backslash"); + + TestBuilder.command`${BUN} script.sh` + .ensureTempDir() + .file("script.sh", "if true; then\n\techo tabbed\nfi\n") + .stdout("tabbed\n") + .runAsTest("indents a .sh file"); + + // A tab-indented line is a command now, so the body of an unsupported + // construct must not run. The parse fails on the construct. + const reserved = (w: string) => + `"${w}" is a reserved word that Bun Shell does not support yet. To run a command named "${w}", quote it.`; + test.each([ + ["for", "for\tf in a b; do\n\techo BODY\ndone", reserved("for")], + ["while", "while\tfalse; do\n\techo BODY\ndone", reserved("while")], + ["!", "if\t!\tfalse; then\n\techo THEN\nelse\n\techo ELSE\nfi", reserved("!")], + ["a brace group", "false &&\t{\techo BODY;\t}", reserved("{")], + ])("does not run the body of an unsupported construct: %s", (_name, source, message) => { + expect(shellParseError(source)).toBe(message); + }); + }); + describe("escaped_newline", () => { const printArgs = /* ts */ `console.log(JSON.stringify(process.argv))`; diff --git a/test/js/bun/shell/lex.test.ts b/test/js/bun/shell/lex.test.ts index aebe12bc91a2..a4c0ad3e0fdb 100644 --- a/test/js/bun/shell/lex.test.ts +++ b/test/js/bun/shell/lex.test.ts @@ -706,6 +706,37 @@ describe("lex shell", () => { expect(JSON.parse(result)).toEqual(expected); }); + // A tab outside quotes separates words, as a space does. + test.each([ + ["between words", "echo\tfoo"], + ["at the start of a line", "if true; then\n\techo foo\nfi"], + ["at the end of a line", "echo foo\t\nls"], + ["in a run of blanks", "echo \t \tfoo"], + ["around an operator", "echo foo\t|\tcat"], + ["after a variable", "echo $FOO\tbar"], + ["after a closing quote", '"a"\tb'], + ["after a brace group", "{a,b}\tc"], + ["around a redirect", "echo foo\t>\tout.txt"], + ["before a comment", "echo foo\t# comment"], + ["before a comment that ends a line", "echo a\t# c\n\techo b"], + ["in a command substitution", "echo $(echo\tfoo)"], + ["in a script with a non-ascii character", "echo\tfoo é"], + ])("tab: %s lexes like a space", (_name, source) => { + expect(JSON.parse(lex({ raw: [source] }))).toEqual(JSON.parse(lex({ raw: [source.replaceAll("\t", " ")] }))); + }); + + test.each([ + ["in double quotes", 'echo "a\tb"', [{ Text: "echo" }, { Delimit: {} }, { DoubleQuotedText: "a\tb" }, { Eof: {} }]], + ["in single quotes", "echo 'a\tb'", [{ Text: "echo" }, { Delimit: {} }, { SingleQuotedText: "a\tb" }, { Eof: {} }]], + [ + "after a backslash", + "echo a\\\tb", + [{ Text: "echo" }, { Delimit: {} }, { Text: "a\tb" }, { Delimit: {} }, { Eof: {} }], + ], + ])("tab: %s is a literal character", (_name, source, expected) => { + expect(JSON.parse(lex({ raw: [source] }))).toEqual(expected); + }); + // Where the lexer puts `Delimit` when a word's last part is not plain text // (a variable, a closing quote, a brace group), or when the word is split // into several tokens. Whitespace and operators delimit such a word; `;`