Skip to content
Closed
7 changes: 6 additions & 1 deletion src/ast/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1177,7 +1177,12 @@ impl BabyString {
if text.is_empty() {
return BabyString::new(0, 0);
}
let off = bun_core::strings::index_of(parent, text).expect("unreachable");
// `parent` is a formatted message the caller promises embeds `text`
// verbatim. A broken promise (e.g. a formatter rendering `text`
// lossily) degrades to "no specifier" instead of panicking.
let Some(off) = bun_core::strings::index_of(parent, text) else {
return BabyString::new(0, 0);
};
BabyString::new(off as u16, text.len() as u16) // @truncate
}

Expand Down
11 changes: 6 additions & 5 deletions src/bun_core/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1230,11 +1230,12 @@ pub use crate::string::immutable::{
has_suffix_comptime, index_of, index_of_scalar, index_of_t, is_all_whitespace, is_ip_address,
is_npm_package_name, is_npm_package_name_ignore_length, is_on_char_boundary,
is_utf8_char_boundary, is_valid_utf8, join, last_index_of, last_index_of_t,
length_of_leading_whitespace_ascii, memmem, order, order_t, percent_encode_write, sort_asc,
sort_desc, split, starts_with_case_insensitive_ascii, starts_with_char, str_utf8,
to_ascii_hex_value, to_utf16_alloc, trim_leading_char, trim_prefix, trim_prefix_comptime,
trim_spaces, trim_suffix, trim_suffix_comptime, utf8_byte_sequence_length, utf16_eql_string,
without_prefix, without_prefix_comptime, without_suffix_comptime, without_utf8_bom,
length_of_leading_whitespace_ascii, memmem, order, order_t, percent_encode_write,
replace_invalid_utf8, sort_asc, sort_desc, split, starts_with_case_insensitive_ascii,
starts_with_char, str_utf8, to_ascii_hex_value, to_utf16_alloc, trim_leading_char, trim_prefix,
trim_prefix_comptime, trim_spaces, trim_suffix, trim_suffix_comptime,
utf8_byte_sequence_length, utf16_eql_string, without_prefix, without_prefix_comptime,
without_suffix_comptime, without_utf8_bom,
};

#[allow(deprecated)]
Expand Down
29 changes: 28 additions & 1 deletion src/bun_core/string/immutable.rs
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ use core::ffi::c_int;

use crate::BoundedArray;
use crate::Error;
use bun_alloc::AllocError;
use bun_alloc::{AllocError, ArenaVec, MimallocArena};
use bun_highway as highway;
use bun_simdutf_sys::simdutf;

Expand Down Expand Up @@ -1556,6 +1556,33 @@ pub fn str_utf8(bytes: &[u8]) -> Option<&str> {
}
}

/// Well-formed UTF-8 view of `bytes`: returns `bytes` unchanged when it is
/// already valid (the SIMD-validated common case), else an `arena` copy with
/// each maximal ill-formed subsequence replaced by U+FFFD.
pub fn replace_invalid_utf8<'a>(bytes: &'a [u8], arena: &'a MimallocArena) -> &'a [u8] {
if is_valid_utf8(bytes) {
return bytes;
}
// Size exactly: each maximal ill-formed subsequence (1-3 bytes) becomes
// one 3-byte U+FFFD.
let out_len = bytes.utf8_chunks().fold(0usize, |len, chunk| {
let replacement = if chunk.invalid().is_empty() {
0
} else {
UNICODE_REPLACEMENT_STR.len()
};
len + chunk.valid().len() + replacement
});
let mut out = ArenaVec::with_capacity_in(out_len, arena);
for chunk in bytes.utf8_chunks() {
out.extend_from_slice(chunk.valid().as_bytes());
if !chunk.invalid().is_empty() {
out.extend_from_slice(&UNICODE_REPLACEMENT_STR);
}
}
out.leak()
}

pub use index_of_newline_or_non_ascii as index_of_newline_or_non_ascii_or_ansi;

/// Checks if slice[offset..] has any < 0x20 or > 127 characters
Expand Down
9 changes: 8 additions & 1 deletion src/bundler/ParseTask.rs
Original file line number Diff line number Diff line change
Expand Up @@ -2300,7 +2300,14 @@ pub mod parse_worker {
}
*step = Step::Parse;

let entry_contents: &[u8] = entry.contents.as_slice();
// The CSS tokenizer requires well-formed UTF-8 (token payloads are raw
// sub-slices). Decode before `source` is built so its contents, token
// positions, error line text, and source maps index the same buffer.
let entry_contents: &[u8] = if loader == Loader::Css {
strings::replace_invalid_utf8(entry.contents.as_slice(), bump)
} else {
entry.contents.as_slice()
};
let is_empty = strings::is_all_whitespace(entry_contents);

// SAFETY: `transpiler` derived from a live `&mut` above. Reborrow only the
Expand Down
5 changes: 4 additions & 1 deletion src/bundler/transpiler.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3159,9 +3159,12 @@ impl<'a> Transpiler<'a> {
// `'bump`-threading note).
let alloc: &'static Arena = unsafe { bun_ptr::detach_lifetime_ref::<Arena>(self.arena) };

// The CSS tokenizer requires well-formed UTF-8 (token payloads are raw
// sub-slices of the source).
let code = strings::replace_invalid_utf8(entry.contents(), alloc);
let (mut sheet, extra) = match bun_css::StyleSheet::<bun_css::DefaultAtRule>::parse(
alloc,
entry.contents(),
code,
opts,
None,
bun_ast::Index::INVALID,
Expand Down
157 changes: 157 additions & 0 deletions test/bundler/css/invalid-utf8.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,157 @@
import { describe, expect, test } from "bun:test";
import { bunEnv, bunExe, tempDir } from "harness";
import { writeFileSync } from "node:fs";
import { join } from "node:path";

// CSS sources whose bytes are not valid UTF-8. The bundler must decode them
// (invalid sequences become U+FFFD) instead of tokenizing the raw bytes,
// which used to crash with `panic: unreachable` once an unresolvable
// `@import` specifier containing the raw byte reached the error formatter.
//
// 0xE2 is a three-byte UTF-8 lead with no continuation bytes after it.
const importWithInvalidByte = Buffer.concat([
Buffer.from('@import url("./x'),
Buffer.from([0xe2]),
Buffer.from('y.css");\n'),
]);

describe("css with invalid utf-8", () => {
test.concurrent("unresolvable @import reports a resolve error", async () => {
using dir = tempDir("css-invalid-utf8-import", {});
writeFileSync(join(String(dir), "in.css"), importWithInvalidByte);

await using proc = Bun.spawn({
cmd: [bunExe(), "build", "./in.css", "--outdir=out"],
env: bunEnv,
cwd: String(dir),
stdout: "pipe",
stderr: "pipe",
});
const [stdout, stderr, exitCode] = await Promise.all([proc.stdout.text(), proc.stderr.text(), proc.exited]);

// The invalid byte is replaced with U+FFFD, exactly as a browser decodes
// the stylesheet, so the import fails to resolve like any other typo.
expect(stderr).toContain('Could not resolve: "./x\uFFFDy.css"');
expect(stdout).not.toContain("Bundled");
expect(exitCode).toBe(1);
});

test.concurrent("Bun.build reports the failure on a ResolveMessage", async () => {
using dir = tempDir("css-invalid-utf8-api", {
"build.js": `
const result = await Bun.build({ entrypoints: ["./in.css"], throw: false });
const log = result.logs[0];
console.log(JSON.stringify({ success: result.success, message: log.message, specifier: log.specifier }));
`,
});
writeFileSync(join(String(dir), "in.css"), importWithInvalidByte);

await using proc = Bun.spawn({
cmd: [bunExe(), "run", "./build.js"],
env: bunEnv,
cwd: String(dir),
stdout: "pipe",
stderr: "pipe",
});
const [stdout, stderr, exitCode] = await Promise.all([proc.stdout.text(), proc.stderr.text(), proc.exited]);

// Checked before JSON.parse so a crashed child reports its stderr
// instead of a JSON syntax error on the empty stdout.
expect({ stdout, stderr, exitCode }).toMatchObject({ exitCode: 0 });

// Only the ASCII affixes of the specifier are asserted here: the
// ResolveMessage getters mis-decode non-ASCII text today (reproducible
// on its own with `import "./café.js"`), which is a separate issue.
Comment thread
claude[bot] marked this conversation as resolved.
const log = JSON.parse(stdout);
expect(log.success).toBe(false);
expect(log.message).toStartWith('Could not resolve: "./x');
expect(log.specifier).toStartWith("./x");
expect(log.specifier).toEndWith("y.css");
});

test.concurrent("url() token in an at-rule prelude reports a resolve error", async () => {
using dir = tempDir("css-invalid-utf8-url", {});
writeFileSync(
join(String(dir), "in.css"),
Buffer.concat([Buffer.from("@-x url(a"), Buffer.from([0xe2]), Buffer.from("b) tok;\n")]),
);

await using proc = Bun.spawn({
cmd: [bunExe(), "build", "./in.css", "--outdir=out"],
env: bunEnv,
cwd: String(dir),
stdout: "pipe",
stderr: "pipe",
});
const [stdout, stderr, exitCode] = await Promise.all([proc.stdout.text(), proc.stderr.text(), proc.exited]);

expect(stderr).toContain('Could not resolve: "a\uFFFDb"');
expect(exitCode).toBe(1);
});

test.concurrent("invalid bytes outside an import become U+FFFD in the output", async () => {
using dir = tempDir("css-invalid-utf8-content", {});
// `content: "caf<0xE9>"` (latin-1 "café"): the build succeeds and the
// emitted stylesheet is well-formed UTF-8, matching how a browser would
// have decoded the input.
writeFileSync(
join(String(dir), "in.css"),
Buffer.concat([Buffer.from('a { content: "caf'), Buffer.from([0xe9]), Buffer.from('"; }\n')]),
);

await using proc = Bun.spawn({
cmd: [bunExe(), "build", "./in.css", "--outdir=out"],
env: bunEnv,
cwd: String(dir),
stdout: "pipe",
stderr: "pipe",
});
const [stdout, stderr, exitCode] = await Promise.all([proc.stdout.text(), proc.stderr.text(), proc.exited]);
expect(stderr).not.toContain("error:");

const out = new Uint8Array(await Bun.file(join(String(dir), "out", "in.css")).arrayBuffer());
expect(Buffer.from(out).includes(Buffer.from('content: "caf\uFFFD"'))).toBe(true);
expect(Buffer.from(out).includes(0xe9)).toBe(false);
Comment thread
coderabbitai[bot] marked this conversation as resolved.
expect(exitCode).toBe(0);
});

test.concurrent("escaped invalid byte does not swallow the bytes after it", async () => {
using dir = tempDir("css-invalid-utf8-escape", {});
// `\` followed by a lone 0xC3 lead byte. The escape must decode to U+FFFD
// and consume exactly that byte; it used to consume three (the UTF-8
// length of U+FFFD), eating ident characters, string content, or the
// closing quote.
writeFileSync(
join(String(dir), "in.css"),
Buffer.concat([
Buffer.from(".x\\"),
Buffer.from([0xc3]),
Buffer.from('yz { content: "\\'),
Buffer.from([0xc3]),
Buffer.from('x AFTER"; }\n.a { content: "\\'),
Buffer.from([0xc3]),
Buffer.from('"; color: green; }\n'),
]),
);

await using proc = Bun.spawn({
cmd: [bunExe(), "build", "./in.css", "--outdir=out"],
env: bunEnv,
cwd: String(dir),
stdout: "pipe",
stderr: "pipe",
});
const [stdout, stderr, exitCode] = await Promise.all([proc.stdout.text(), proc.stderr.text(), proc.exited]);
expect(stderr).not.toContain("error:");

const out = await Bun.file(join(String(dir), "out", "in.css")).text();
// Selector: `yz` used to be eaten out of the class name.
expect(out).toContain(".x\uFFFDyz");
// String content: the `x ` after the escape used to be eaten.
expect(out).toContain('content: "\uFFFDx AFTER"');
// The closing quote used to be eaten, absorbing the rest of the rule.
expect(out).toContain('content: "\uFFFD";');
expect(out).toContain("color: green;");
expect(exitCode).toBe(0);
});
});
Comment thread
robobun marked this conversation as resolved.
Loading