Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion src/runtime/shell/states/Assigns.rs
Original file line number Diff line number Diff line change
Expand Up @@ -86,7 +86,7 @@ impl Assigns {
io,
ExpansionOpts {
for_spawn: false,
single: false,
single: true,
},
);
return Expansion::start(interp, child);
Expand Down
37 changes: 24 additions & 13 deletions src/runtime/shell/states/Cmd.rs
Original file line number Diff line number Diff line change
Expand Up @@ -262,9 +262,12 @@ impl Cmd {
atom,
this,
io,
// Redirect targets are field-split (bash errors
// on >1 field; we keep the pre-existing glue),
// so surrounding IFS whitespace is stripped.
ExpansionOpts {
for_spawn: false,
single: true,
single: false,
Comment thread
claude[bot] marked this conversation as resolved.
},
);
return Expansion::start(interp, child);
Expand Down Expand Up @@ -395,10 +398,11 @@ impl Cmd {
let me = interp.as_cmd_mut(this);
if out.bounds.is_empty() {
// An empty
// expansion that did *not* see a `""` literal pushes
// no arg at all — `$unset` vanishes, only `""` yields
// an empty argv word.
if !out.buf.is_empty() || out.has_quoted_empty {
// expansion that did *not* see a `""` literal or a
// field from IFS splitting pushes no arg at all —
// `$unset` vanishes, while `""` and a sole empty IFS
// field each yield one empty argv word.
if !out.buf.is_empty() || out.has_quoted_empty || out.has_empty_field {
me.args.push(out.buf);
}
} else {
Expand All @@ -412,14 +416,21 @@ impl Cmd {
}
CmdState::ExpandingRedirect { ref mut idx } => {
*idx += 1;
// NUL-terminate a
// non-empty result; leave an empty expansion empty so the
// ambiguous-redirect check in `Builtin::init_redirections`
// still fires.
let mut buf = out.buf;
if !buf.is_empty() && buf.last() != Some(&0) {
buf.push(0);
}
// A target that field-split into more than one word is an
// ambiguous redirect; leave `redirection_file` empty so the
// check in `init_redirections` / `init_subproc_redirections`
// fires (matching bash). Otherwise NUL-terminate the single
// word; an empty expansion also stays empty and is caught by
// the same check.
let buf = if out.bounds.is_empty() {
let mut buf = out.buf;
if !buf.is_empty() && buf.last() != Some(&0) {
buf.push(0);
}
buf
} else {
Vec::new()
};
interp.as_cmd_mut(this).redirection_file = buf;
}
_ => {}
Expand Down
198 changes: 152 additions & 46 deletions src/runtime/shell/states/Expansion.rs
Original file line number Diff line number Diff line change
Expand Up @@ -50,6 +50,10 @@ pub struct Expansion {
/// Exit code of a sole-command-substitution arg — propagated to `Cmd`
/// so `$(false)` as argv0 fails.
pub out_exit_code: ExitCode,
/// Expand to a single word: suppress field splitting of command
/// substitutions. Set for assignment values and `[[ ]]` operands, which
/// are exempt from field splitting (redirect targets are still split).
pub single: bool,
}

#[derive(Default, strum::IntoStaticStr)]
Expand Down Expand Up @@ -80,6 +84,16 @@ pub struct ExpansionOut {
/// empty, this distinguishes `""` (push one empty arg) from `$unset`
/// (push no arg). See [`Expansion::has_quoted_empty`].
pub has_quoted_empty: bool,
/// Whether at least one word has been committed to `buf`. Drives boundary
/// recording instead of `buf`/`bounds` emptiness, which cannot tell "no
/// word yet" from "one empty word committed" (a leading empty field from
/// non-whitespace IFS splitting).
pub committed: bool,
/// Set when IFS field splitting yielded at least one field, so a result
/// that splits down to a single empty field (e.g. `$(echo ,)` with
/// `IFS=,`) still produces one empty argv word. Distinct from
/// `has_quoted_empty` so it does not affect leading-tilde handling.
pub has_empty_field: bool,
}

#[derive(Clone, Copy, Default)]
Expand All @@ -95,7 +109,7 @@ impl Expansion {
node: *const ast::Atom,
parent: NodeId,
io: IO,
_opts: ExpansionOpts,
opts: ExpansionOpts,
) -> NodeId {
interp.alloc_node(Node::Expansion(Expansion {
base: Base::new(StateKind::Expansion, parent, shell),
Expand All @@ -115,6 +129,7 @@ impl Expansion {
cmd_subst_quoted: false,
has_quoted_empty: false,
out_exit_code: 0,
single: opts.single,
}))
}

Expand Down Expand Up @@ -266,7 +281,12 @@ impl Expansion {
if atom.has_glob_expansion() {
return Self::transition_to_glob_state(interp, this);
}
Self::push_current_out(me);
// Flush the in-progress word. Skip an empty `current_out` once a
// word was already committed: a trailing IFS separator already
// committed the last field, so there is no final word to add.
if !me.current_out.is_empty() || !me.out.committed {
Self::push_current_out(me);
}
me.state = ExpansionState::Done;
}
let parent = interp.as_expansion(this).base.parent;
Expand Down Expand Up @@ -337,13 +357,15 @@ impl Expansion {
}
drop(arena);

// Push each variant as its own word; word boundaries are recorded
// via `bounds`.
for s in expanded {
if !me.out.buf.is_empty() {
// Push each non-empty variant as its own word; word boundaries are
// recorded via `bounds`. Empty variants are dropped as unquoted null
// words (`{,a}` -> `a`, `{,}` -> nothing), matching bash.
for s in expanded.iter().filter(|s| !s.is_empty()) {
if me.out.committed {
me.out.bounds.push(me.out.buf.len() as u32);
}
me.out.buf.extend_from_slice(&s);
me.out.buf.extend_from_slice(s);
me.out.committed = true;
}

let node = me.node;
Expand Down Expand Up @@ -533,59 +555,142 @@ impl Expansion {
/// end-offset so the consumer's `[prev..bound]` slicing reconstructs each
/// word and the trailing `[prev..]` slice yields the final one.
fn push_current_out(me: &mut Expansion) {
if !me.out.buf.is_empty() {
// A boundary precedes every word after the first, so record one once a
// word has already been committed — even an empty one leaves `buf`
// empty, so `committed` (not `buf` emptiness) is the right test.
if me.out.committed {
me.out.bounds.push(me.out.buf.len() as u32);
}
me.out.buf.append(&mut me.current_out);
me.out.committed = true;
me.meta_offsets.clear();
}

/// Newlines→spaces, trim, then split on whitespace runs into separate
/// argv words.
/// Reads the script-assigned value of `IFS`. Returns `None` when `IFS` is
/// unset so the caller applies the default separators. Only `shell_env` is
/// consulted, never `export_env`: like every POSIX shell, Bun Shell must
/// not let an inherited environment `IFS` (here, `process.env.IFS`) control
/// field splitting, and `export_env` holds the inherited process
/// environment. A bare `IFS=...` assignment lands in `shell_env`, so the
/// common idiom works. The `export IFS=...` spelling (which writes only
/// `export_env`) is not yet honored for splitting; distinguishing
/// script-exported vars from inherited ones needs a separate change.
fn get_ifs(shell: &ShellExecEnv) -> Option<Vec<u8>> {
use crate::shell::env_str::EnvStr;
let entry = shell.shell_env.get(EnvStr::init_slice(b"IFS"))?;
let bytes = entry.slice().to_vec();
entry.deref();
Some(bytes)
}
Comment thread
claude[bot] marked this conversation as resolved.
Comment thread
claude[bot] marked this conversation as resolved.

/// POSIX field splitting of a command-substitution result. `ifs` is the
/// active separator set (default `" \t\n"`). Leading/trailing IFS
/// whitespace is ignored, runs of IFS whitespace collapse to one
/// delimiter, and each non-whitespace IFS byte (with adjacent IFS
/// whitespace) delimits one field, so consecutive ones yield empty fields.
///
/// Also returns whether a separator touched each edge of the input: leading
/// IFS whitespace (`.1`) and a trailing delimiter (`.2`). A leading
/// non-whitespace delimiter instead surfaces as an empty first field, so it
/// is not reported here. The caller uses these to break the word against an
/// adjacent literal (`a$(echo " b")` -> `a`, `b`).
fn ifs_split_fields<'a>(s: &'a [u8], ifs: &[u8]) -> (Vec<&'a [u8]>, bool, bool) {
let is_ifs = |b: u8| ifs.contains(&b);
let is_ifs_ws = |b: u8| matches!(b, b' ' | b'\t' | b'\n') && ifs.contains(&b);

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🟡 nit: is_ifs_ws hardcodes {space, tab, newline}, but bash 5.2 (the PR's stated reference) treats \r/\f/\v as IFS whitespace when they appear in IFS — so IFS=$'\r'; set -- $(printf 'a\r\rb') gives 2 words in bash but 3 (with a spurious empty middle) here. Obscure trigger and matches dash, so not blocking; the fix is one line: add b'\r' | b'\x0c' | b'\x0b' to the matches!.

Extended reasoning...

What the bug is

The IFS-whitespace predicate in ifs_split_fields is:

let is_ifs_ws = |b: u8| matches!(b, b' ' | b'\t' | b'\n') && ifs.contains(&b);

POSIX §2.6.5 defines IFS white space as any character that is both in IFS and in the current locale's [:space:] class — in the C/POSIX locale that is {space, tab, newline, carriage-return, form-feed, vertical-tab}. Bash 5.0+ implements exactly this (its CHANGES file documents the switch to the locale space class), and the PR description explicitly says output "matches bash 5.2 across the matrix." With this predicate, an \r, \f, or \v in IFS is classified as a non-whitespace IFS byte — a hard delimiter — instead of collapsing whitespace.

How it manifests / step-by-step proof

Take IFS = "\r" and input "a\r\rb" (after trailing-newline stripping). Tracing ifs_split_fields:

  1. is_ifs_ws('\r') = false (not in the matches! set), so the leading-whitespace skip does nothing; i = 0.
  2. Field scan: i = 0..1 ('a' is not in IFS). Push "a". i = 1.
  3. Delimiter consumption: the first while is_ifs_ws loop is a no-op. is_ifs(s[1]) is true → consume one byte, i = 2. The trailing while is_ifs_ws loop is again a no-op.
  4. Field scan at i = 2: s[2] = '\r' is in IFS, so the loop body doesn't advance. Push "" (empty slice [2..2]).
  5. Consume one delimiter → i = 3. Field scan pushes "b".

Result: ["a", "", "b"] — 3 fields. Bash 5.2 gives 2:

$ bash -c 'IFS=$(printf "\r"); set -- $(printf "a\r\rb"); echo $#'
2
$ bash -c 'IFS=$(printf " \r"); set -- $(printf "\ra\r"); echo $#'
1

The same divergence applies to \f (0x0c) and \v (0x0b), also verified against bash 5.2.21.

Why nothing else prevents it

The predicate is new in this PR; nothing else classifies IFS whitespace. The old post_subshell_expansion never consulted IFS at all (and its trim step happened to include \r), but that path is gone.

Impact and why this is a nit

The trigger requires deliberately putting \r/\f/\v into IFS, which is awkward in Bun Shell — there's no $'...' quoting, so it requires an external printf via command substitution. The realistic case is processing CRLF output with IFS containing \r, where a blank line (\r\n\r\n) yields a spurious empty argv word. Note that dash uses the same space/tab/newline-only interpretation and gives 3 for the test above, so POSIX shells genuinely disagree here — Bun matches dash but not bash. Given the obscurity and the split among reference shells, this shouldn't block merge.

Fix

One line — extend the whitespace set to the full C-locale [:space:] class:

let is_ifs_ws = |b: u8| matches!(b, b' ' | b'\t' | b'\n' | b'\r' | b'\x0c' | b'\x0b') && ifs.contains(&b);

Comment on lines +598 to +599

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🟡 nit: is_ifs tests membership per-byte via ifs.contains(&b), so a multibyte IFS character is treated as a set of independent delimiter bytes. IFS=£ (0xC2 0xA3) makes each byte its own non-whitespace delimiter, so $(echo a£b) yields ["a","","b"] (bash: ["a","b"]), and output containing a different character sharing a byte (e.g. ¢ = 0xC2 0xA2) is torn mid-codepoint into invalid UTF-8. Unlike the deferred \r/\f/\v case this is directly reachable via a literal IFS=£ in source (no $'...' needed) and testable — matches dash but not bash; fine as a follow-up alongside the other deferred edges, worth documenting the byte-oriented semantics.

Extended reasoning...

What the bug is

ifs_split_fields tests IFS membership per-byte:

let is_ifs = |b: u8| ifs.contains(&b);

so a multibyte UTF-8 character in IFS is treated as a set of independent delimiter bytes rather than one delimiter character. Each byte then acts as its own non-whitespace IFS delimiter, producing spurious empty fields between them and — worse — matching unrelated characters that share a byte.

Distinct from the deferred \r/\f/\v comment

That comment was deferred on the grounds that the trigger requires those bytes to be in IFS, which Bun Shell can't currently express (no $'...' quoting, and a literal CR is eaten by the lexer). That rationale does not apply here: a non-ASCII IFS is trivially expressible as a literal — IFS=£ in the script source (or await $`IFS=£; ...` from JS) stores the UTF-8 bytes [0xC2, 0xA3] in shell_env with no special quoting. So this path is reachable through the shell surface today and is testable.

Step-by-step: IFS=£; cmd $(printf 'a£b')

ifs = [0xC2, 0xA3], s = [0x61, 0xC2, 0xA3, 0x62].

  1. Leading-ws skip: no-op (0x61 not IFS-ws). i=0.
  2. Field scan: !is_ifs(0x61) → i=1. is_ifs(0xC2) (contained in ifs) → stop. Push "a".
  3. Delimiter: is_ifs_ws(0xC2)=false; is_ifs(0xC2) → consume one byte, i=2; trailing-ws no-op.
  4. Field scan at i=2: is_ifs(0xA3) → stop immediately. Push "".
  5. Delimiter: consume 0xA3, i=3.
  6. Field scan: push "b".

Result: ["a", "", "b"] — 3 argv words with a spurious empty middle. bash 5.2 in a UTF-8 locale gives ["a", "b"] (verified: bash -c 'IFS=£; set -- $(printf "a£b"); echo $#' → 2). For a££b bash gives 3; this gives 5.

The worse case: mid-character splitting

If the substitution output contains a different character sharing a byte with the IFS character, that byte is stripped and the remaining continuation byte becomes an argv word of invalid UTF-8. E.g. IFS=£ (0xC2 0xA3) with output "a¢b" (¢ = 0xC2 0xA2): is_ifs(0xC2) matches, so field 1 = "a", delimiter consumes 0xC2, field 2 = [0xA2, 0x62] — a lone continuation byte 0xA2 glued to b. The ¢ is silently corrupted. bash leaves "a¢b" as one word (no £ present).

Why nothing prevents it

get_ifs returns raw bytes and ifs_split_fields never groups them into characters. There is no ASCII fast-path / WTF-8 slow-path split as do_brace_expand has (Expansion.rs:319-324). Before this PR the path did not exist (IFS was ignored), so it is new to this rewrite in the sense that the arity changes from 1 (old, IFS ignored) to 3 (new) where bash gives 2.

Impact / severity

Nit: requires deliberately setting IFS to a non-ASCII delimiter, which is uncommon. dash has the same byte-wise behavior, so Bun matches dash but not bash — same disclaimer as the \r case. But unlike that case it is directly reachable and testable, and the mid-character-split variant silently produces mojibake in argv with exit 0. Shouldn't block merge given dash parity and the exotic trigger.

Fix

Either (a) document that IFS is byte-oriented (like dash / POSIX C-locale), or (b) when ifs contains a byte ≥ 0x80, decode both ifs and s as WTF-8 code points and match character-wise, mirroring the ASCII/WTF-8 branching in do_brace_expand. A test:

TestBuilder.command`IFS=£; BUN_TEST_VAR=1 ${BUN} -e ${ARGV} $(echo a£b)`
  .stdout(`["a","b"]\n`)
  .runAsTest("multibyte IFS char is one delimiter");

let n = s.len();
let mut fields: Vec<&[u8]> = Vec::new();
let mut i = 0usize;
let ws_start = i;
while i < n && is_ifs_ws(s[i]) {
i += 1;
}
Comment thread
robobun marked this conversation as resolved.
Comment thread
coderabbitai[bot] marked this conversation as resolved.
let leading_ws_sep = i > ws_start;
let mut trailing_sep = false;
while i < n {
let start = i;
while i < n && !is_ifs(s[i]) {
i += 1;
}
fields.push(&s[start..i]);
if i >= n {
break;
}
// Consume one delimiter: IFS whitespace, then at most one
// non-whitespace IFS byte, then trailing IFS whitespace.
while i < n && is_ifs_ws(s[i]) {
i += 1;
}
if i < n && is_ifs(s[i]) {
i += 1;
while i < n && is_ifs_ws(s[i]) {
i += 1;
Comment thread
robobun marked this conversation as resolved.
}
}
// A delimiter that consumed the rest of the input is a trailing
// separator: no further field follows it.
if i >= n {
trailing_sep = true;
}
}
(fields, leading_ws_sep, trailing_sep)
}

/// Split an unquoted command-substitution result into argv words using
/// POSIX field splitting driven by `IFS`.
fn post_subshell_expansion(me: &mut Expansion, mut stdout: Vec<u8>) {
// Strip a single trailing newline, then convert remaining newlines
// to spaces.
if stdout.last() == Some(&b'\n') {
// Command substitution deletes trailing newlines before field splitting.
while stdout.last() == Some(&b'\n') {
stdout.pop();
}
Comment on lines 640 to 644

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🟡 Nit: this rewrite now strips only trailing \n from unquoted $(...) (previously [ \t\r\n]), while the quoted path in child_done still trims trailing \r/space/tab — so "$(tool)" strips a CRLF-terminating \r but $(tool) keeps it, the reverse of what quoting normally implies. The new unquoted behavior is bash-correct so it shouldn't block, but it's worth deciding: either also strip trailing \r here (while matches!(stdout.last(), Some(b'\n' | b'\r'))) as a documented cross-platform deviation for Windows tools that emit CRLF (where.exe, MSVCRT-linked CLIs), or align the quoted path to POSIX (strip only trailing \n) so the two are consistent.

Extended reasoning...

What changed

The old post_subshell_expansion trimmed leading/trailing b' ' | b'\n' | b'\r' | b'\t' before splitting. The rewrite now strips only trailing newlines:

while stdout.last() == Some(&b'\n') {
    stdout.pop();
}

then either appends verbatim (single / empty-IFS) or field-splits on IFS (default b" \t\n", which does not contain \r). So an unquoted $(cmd) whose output ends in CRLF now yields an argv word with a trailing \r where it previously did not. This is POSIX-correct — verified: bash -c 'set -- $(printf "a\r\n"); printf "%s" "$1" | od -c' retains the \r. And bash-compat is the PR's stated goal, so on its own this would just be a "note the behavior change" item.

The internal inconsistency

The stronger point is that the quoted path in child_done was not touched and still does:

while hi > 0 && matches!(stdout[hi - 1], b' ' | b'\n' | b'\r' | b'\t') {
    hi -= 1;
}

So after this PR, "$(tool)" strips a trailing \r while $(tool) keeps it — the opposite of what a user would expect from adding quotes (quoting normally reduces trimming/mangling, never increases it). The two code paths now disagree about what "remove trailing newlines from a command substitution" means.

Step-by-step trace

Take cmd $(tool) where tool prints "path\r\n" (any Windows-native CLI: where.exe, PowerShell, cmd /c echo, or an MSVCRT-linked binary), with default IFS:

  1. stdout = b"path\r\n".
  2. while stdout.last() == Some(&b'\n') → pop once → b"path\r". \r != \n, loop stops.
  3. Not empty, not single, IFS unset → ifs_bytes = b" \t\n".
  4. ifs_split_fields(b"path\r", b" \t\n"): \r is not in the IFS set, so the field scan runs to end. Returns [b"path\r"].
  5. current_out = b"path\r". cmd receives argv ["path\r"].

Before this PR: after stripping the \n and converting newlines→spaces, the trim loop matched \r and dropped it → cmd received ["path"].

Compare the quoted spelling cmd "$(tool)" after this PR: child_done's while hi > 0 && matches!(stdout[hi-1], b' ' | b'\n' | b'\r' | b'\t') decrements hi twice → cmd receives ["path"]. Quoted strips more than unquoted.

The single: true path (assignments, redirect targets, [[ ]] operands — new in 747b149) hits the same shape: it appends verbatim after \n-strip only, so VAR=$(tool) and > $(tool) also gain a trailing \r they did not have before.

Why it matters (and why it's still a nit)

Bun Shell explicitly targets cross-platform scripting. On Windows, $(where.exe node), $(cmd /c echo %CD%), and any MSVCRT-linked tool print CRLF. Before this PR the \r was silently stripped and the result was directly usable as a path/arg; after, every such substitution carries an invisible \r that becomes "file not found" / "command not found" downstream, exit ≠ 0, no hint that a stray CR is the cause.

That said: (1) the new unquoted behavior is exactly bash-correct, which is what the PR set out to do; (2) the old \r-trimming was itself the deviation; (3) the quoted-path over-trim is pre-existing code the PR didn't touch. So this shouldn't block merge — it just deserves a conscious decision rather than an accident.

Fix options

Either:

  • (a) Also strip trailing \r here as a deliberate cross-platform deviation from POSIX, matching what the quoted path already does: while matches!(stdout.last(), Some(b'\n' | b'\r')) { stdout.pop(); }. This keeps $(where.exe node) working on Windows.
  • (b) Accept the bash-correct behavior, and align the quoted path in child_done to POSIX (strip only trailing \n) so at least the two are consistent — and call out the CRLF implication in release notes.

Either way the two paths should agree.

for b in stdout.iter_mut() {
if *b == b'\n' {
*b = b' ';
}
if stdout.is_empty() {
return;
}
// Trim leading/trailing whitespace.
let s: &[u8] = {
let mut lo = 0usize;
let mut hi = stdout.len();
while lo < hi && matches!(stdout[lo], b' ' | b'\n' | b'\r' | b'\t') {
lo += 1;
}
while hi > lo && matches!(stdout[hi - 1], b' ' | b'\n' | b'\r' | b'\t') {
hi -= 1;
// Assignment values and `[[ ]]` operands expand to a single word:
// they are exempt from field splitting.
if me.single {
me.current_out.extend_from_slice(&stdout);
return;
}
let ifs = Self::get_ifs(me.base.shell());
let ifs_bytes: &[u8] = match &ifs {
// `IFS=` (set to empty) disables field splitting entirely.
Some(v) if v.is_empty() => {
me.current_out.extend_from_slice(&stdout);
return;
}
&stdout[lo..hi]
Some(v) => v,
None => b" \t\n",
};
if s.is_empty() {
let (fields, leading_ws_sep, trailing_sep) = Self::ifs_split_fields(&stdout, ifs_bytes);
// Leading IFS whitespace separates a preceding literal (the prefix held
// in `current_out`) from the first field, so flush the prefix as its
// own word before appending. A leading non-whitespace delimiter needs
// no special handling: it surfaces as an empty first field below.
if leading_ws_sep && !me.current_out.is_empty() {
Self::push_current_out(me);
}
// Commit every field but the last as its own word; the last stays in
// `current_out` so following atoms can concatenate onto it (it is
// flushed later by `push_current_out`). `push_current_out` tracks word
// commits via `out.committed`, so leading empty fields are preserved.
let Some((last, rest)) = fields.split_last() else {
return;
};
// At least one field exists, so the expansion yields at least one word
// even if it is a single empty field (`$(echo ,)` with `IFS=,`), which
// leaves `buf`/`bounds` empty.
me.out.has_empty_field = true;
for field in rest {
me.current_out.extend_from_slice(field);
Self::push_current_out(me);
}
// Split on runs of spaces — each run is a word boundary.
let mut prev_ws = false;
let mut a = 0usize;
for (i, &c) in s.iter().enumerate() {
if prev_ws {
if c != b' ' {
a = i;
prev_ws = false;
}
continue;
}
if c == b' ' {
prev_ws = true;
me.current_out.extend_from_slice(&s[a..i]);
Self::push_current_out(me);
}
me.current_out.extend_from_slice(last);
Comment thread
claude[bot] marked this conversation as resolved.
Comment on lines +683 to +687

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🟣 Pre-existing (not introduced by this PR), noting for the deferred edge-delimiter follow-up: in a compound word ~$(cmd) where the substitution field-splits into ≥2 fields, the leading ~ lands on the last field instead of the first — bash gives ~$(printf "a b") → ["~a","b"], Bun gives ["a","~b"]. post_subshell_expansion flushes all-but-last fields via push_current_out mid-walk, leaving only the last field in current_out; the deferred leading-tilde post-processing in next() then operates solely on current_out. When the follow-up restructures how post_subshell_expansion reports boundaries, it should also route the tilde onto the first committed field rather than current_out.

Extended reasoning...

What the bug is

In a compound word ~$(cmd), when the unquoted $(cmd) field-splits into two or more fields, the leading ~ is attached to the last field instead of the first. bash 5.2:

$ bash -c 'set -- ~$(printf "a b"); printf "[%s]" "$@"'
[~a][b]

Bun (both before and after this PR) produces ["a","~b"]. The rewrite widens reachability slightly — with the new custom-IFS support, IFS=,; ~$(printf ",a") gives ["","~a"] where bash gives ["~","a"] — but the underlying defect is the same.

Code path

next() (Expansion.rs:186-189) detects leading_tilde for a compound atom starting with SimpleAtom::Tilde and sets word_idx = 1, deferring tilde handling until after the atom walk. The walk then hits the CmdSubst atom, spawns a Script, and on child_done calls post_subshell_expansion. For a multi-field result, the for field in rest loop (Expansion.rs:656-659) calls Self::push_current_out(me) once per non-last field — flushing each into out.buf and clearing current_out — and leaves only the last field in current_out. Back in next() with word_idx >= atoms_len, the leading-tilde post-processing block (Expansion.rs:246-273) runs match me.current_out.first() and does me.current_out.insert(0, b'~') (or the HOME splice) — on current_out, which now holds only the last field. The already-committed first field in out.buf never sees the tilde.

Why nothing prevents it

The tilde post-processing was written under the assumption that current_out still holds the entire expanded word when it runs. That assumption breaks whenever post_subshell_expansion flushes mid-walk. leading_tilde is not consulted inside post_subshell_expansion, and there is no mechanism to prefix the tilde onto the first committed field rather than the residual current_out.

Step-by-step: ~$(printf "a b")

  1. Compound atom = [Tilde, CmdSubst]. leading_tilde = true → word_idx = 1.
  2. CmdSubst runs, stdout = "a b". Not quoted → post_subshell_expansion.
  3. ifs_split_fields("a b", " \t\n") → ["a","b"]. split_last() → last="b", rest=["a"].
  4. rest loop: current_out = "a"; push_current_out(me) → out.buf = "a", out.bounds = [] (first commit), out.committed = true, current_out cleared.
  5. current_out.extend("b") → current_out = "b". Return.
  6. next() re-enters, word_idx = 2 >= atoms_len. Leading-tilde block: current_out.first() = Some(b'b') → Some(_) arm → current_out.insert(0, b'~') → current_out = "~b".
  7. push_current_out → out.buf = "a~b", out.bounds = [1].
  8. Cmd::child_done reconstructs ["a","~b"]. Bash: ["~a","b"].

Why this is pre-existing

The old post_subshell_expansion (removed in this diff) also called Self::push_current_out(me) on each interior space and left only the trailing segment in current_out via the final extend_from_slice(&s[a..]). So ~$(echo "a b") already produced ["a","~b"] on main via default-IFS whitespace splitting. The rewrite preserves the same flush-then-postprocess ordering; it does not introduce the defect.

Impact and fix

Narrow trigger: a compound word starting with a literal ~ immediately followed by an unquoted $(...) that produces multiple fields. Not blocking for this PR.

Flagging so the deferred edge-delimiter follow-up (which will restructure how post_subshell_expansion reports boundaries to its caller) can also route the deferred tilde onto the first committed field. One option: when leading_tilde is set, prepend the ~/HOME to current_out before the first push_current_out call inside post_subshell_expansion (i.e., handle it at the point the first field is about to be committed), rather than after the walk completes.

// A trailing delimiter separates the last field from a following
// literal, so commit it now; the walk-end flush skips the now-empty
// `current_out` (see `next`).
if trailing_sep {
Self::push_current_out(me);
}
me.current_out.extend_from_slice(&s[a..]);
}

pub fn child_done(
Expand Down Expand Up @@ -694,10 +799,11 @@ impl Expansion {
// Push each match as its own argv word. The
// walker arena owns the strings, so they were `to_vec`'d already.
for entry in result {
if !me.out.buf.is_empty() {
if me.out.committed {
me.out.bounds.push(me.out.buf.len() as u32);
}
me.out.buf.extend_from_slice(&entry);
me.out.committed = true;
}
me.state = ExpansionState::Done;
}
Expand Down
Loading
Loading