Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion src/bun_core/string/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1833,7 +1833,7 @@ pub mod printer {

const MALFORMED: i32 = -1;

/// Same algorithm as `bun_js_printer::write_pre_quoted_string`, except malformed UTF-8 becomes U+FFFD.
/// Same algorithm as `bun_js_printer::write_pre_quoted_string`.
/// PERF: (quote_char, ascii_only, json, encoding) are runtime params —
/// profile if it shows up on a hot path.
pub fn write_pre_quoted_string<W: PrinterWriter + ?Sized>(
Expand Down
38 changes: 30 additions & 8 deletions src/js_printer/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1031,6 +1031,8 @@ where
write_pre_quoted_string_inner::<W, ENCODING>(text_in, writer, QUOTE_CHAR, ASCII_ONLY, JSON)
}

const MALFORMED: i32 = -1;

/// `quote_char` / `ascii_only` / `json` are runtime args (were `const`): the
/// branches on them are cheap and well-predicted, and collapsing the
/// monomorphizations keeps the hot transpile pages dense (see the facade above).
Expand Down Expand Up @@ -1083,14 +1085,23 @@ where
let clamped_width = (width as usize).min(n.saturating_sub(i));
let c: i32 = match ENCODING {
Encoding::Utf8 => {
let bytes: [u8; 4] = match clamped_width {
1 => [text[i], 0, 0, 0],
2 => [text[i], text[i + 1], 0, 0],
3 => [text[i], text[i + 1], text[i + 2], 0],
4 => [text[i], text[i + 1], text[i + 2], text[i + 3]],
_ => unreachable!(),
};
strings::decode_wtf8_rune_t::<i32>(bytes, width, 0)
if width == 1 {
// width 1 with a byte >= 0x80 is a stray continuation byte or an invalid lead.
if text[i] >= 0x80 {
MALFORMED
} else {
text[i] as i32
}
} else {
let bytes: [u8; 4] = match clamped_width {
1 => [text[i], 0, 0, 0],
2 => [text[i], text[i + 1], 0, 0],
3 => [text[i], text[i + 1], text[i + 2], 0],
4 => [text[i], text[i + 1], text[i + 2], text[i + 3]],
_ => unreachable!(),
};
strings::decode_wtf8_rune_t::<i32>(bytes, width, MALFORMED)
}
}
Encoding::Ascii => {
debug_assert!(text[i] <= 0x7F);
Expand All @@ -1105,6 +1116,17 @@ where
}
};

if c == MALFORMED {
if ascii_only {
writer.write_all(&bmp_escape(0xFFFD))?;
} else {
writer.write_all("\u{FFFD}".as_bytes())?;
}
// One byte, not `width`, so the bytes after a truncated sequence survive.
i += 1;
continue;
}

if can_print_without_escape(c, ascii_only) {
match ENCODING {
Encoding::Ascii | Encoding::Utf8 => {
Expand Down
40 changes: 40 additions & 0 deletions test/bundler/bundler_loader.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -432,6 +432,46 @@ describe("bundler", async () => {
},
});

// A byte that is not valid UTF-8 in a text import or in a string of a data
// file prints as U+FFFD. The bytes around it are kept and the output stays
// valid UTF-8. "bun" prints the literal with ASCII escapes, "browser" writes
// the characters as UTF-8.
{
// "A" E9 "B" C3 "Z" 80 FF "[" C3 A9, E2 82 AC, F0 9F 98 80 "]" E2
// E9 and C3 start a sequence that the next byte does not continue, 80 is a
// continuation byte on its own, FF is never valid, the bracket holds valid
// 2, 3 and 4 byte sequences, and the E2 at the end is cut off by the end
// of the file. Every bad sequence here breaks at its second byte, so one
// U+FFFD per bad byte (the printer) and one per maximal subpart
// (TextDecoder) give the same string.
const body = Buffer.from("41e942c35a80ff5bc3a9e282acf09f98805d", "hex");
const text = Buffer.concat([body, Buffer.from([0xe2])]);
const json = Buffer.concat([Buffer.from('{"key":"'), body, Buffer.from('"}')]);
const expected = "41 fffd 42 fffd 5a fffd fffd 5b e9 20ac 1f600 5d";
for (const target of ["bun", "browser"] as const) {
itBundled(`${target}/loader-ill-formed-utf8`, {
target,
files: {
"/entry.ts": /* js */ `
import text from "./chars.txt" with { type: "text" };
import json from "./chars.json";
const codePoints = (s) => [...s].map(c => c.codePointAt(0).toString(16)).join(" ");
console.log(codePoints(text));
console.log(codePoints(json.key));
`,
"/chars.txt": text,
"/chars.json": json,
},
onAfterBundle(api) {
const output = fs.readFileSync(api.outfile);
expect(() => new TextDecoder("utf-8", { fatal: true }).decode(output)).not.toThrow();
Comment thread
coderabbitai[bot] marked this conversation as resolved.
expect(output.includes(target === "bun" ? "\\uFFFD" : "\uFFFD")).toBe(true);
},
run: { stdout: `${expected} fffd\n${expected}` },
});
}
}

const loaders: Loader[] = ["wasm", "json", "file" /* "napi" */, "text"];
const exts = ["wasm", "json", "lmao" /* ".node" */, "txt"];
for (let i = 0; i < loaders.length; i++) {
Expand Down
27 changes: 24 additions & 3 deletions test/js/bun/import-attributes/import-attributes.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -99,7 +99,7 @@ type Tests = Record<
const default_tests = Object.fromEntries(
loaders.map(loader => [loader, { loader, filename: "no_extension" }]),
) as Tests;
async function compileAndTest(code: string, tests: Tests = default_tests): Promise<Record<string, unknown>> {
async function compileAndTest(code: string | Buffer, tests: Tests = default_tests): Promise<Record<string, unknown>> {
const [v1, v2, v3] = await Promise.all([
compileAndTest_inner(code, tests, testBunRun),
compileAndTest_inner(code, tests, testBunRunAwaitImport),
Expand All @@ -114,7 +114,7 @@ async function compileAndTest(code: string, tests: Tests = default_tests): Promi
return v1;
}
async function compileAndTest_inner(
code: string,
code: string | Buffer,
tests: Tests,
cb: (dir: string, loader: string | null, filename: string) => Promise<unknown>,
): Promise<Record<string, unknown>> {
Expand All @@ -131,7 +131,7 @@ async function compileAndTest_inner(
);
let res: Record<string, unknown> = Object.fromEntries(results);
if (Object.hasOwn(res, "text")) {
expect(res.text).toEqual({ default: code });
expect(res.text).toEqual({ default: typeof code === "string" ? code : code.toString("utf8") });
delete res.text;
}
if (Object.hasOwn(res, "yaml")) {
Expand Down Expand Up @@ -292,6 +292,27 @@ test("yaml", async () => {
`);
});

// A byte that is not valid UTF-8 becomes U+FFFD and the bytes around it are
// kept, both in a text import and in a string value of a data file.
test("ill-formed UTF-8", async () => {
// {"key":"A E9 B C3 Z 80 FF [é€😀]"}: E9 and C3 start a sequence that the
// next byte does not continue, 80 is a continuation byte on its own, FF is
// never valid, and the bracket holds valid 2, 3 and 4 byte sequences. Every
// bad sequence breaks at its second byte, so one U+FFFD per bad byte (the
// printer) and one per maximal subpart (TextDecoder, the runtime JSON
// loader) give the same string and all import paths agree.
const code = Buffer.concat([
Buffer.from('{"key":"'),
Buffer.from("41e942c35a80ff5bc3a9e282acf09f98805d", "hex"),
Buffer.from('"}'),
]);
const key = "A\uFFFDB\uFFFDZ\uFFFD\uFFFD[\u00E9\u20AC\u{1F600}]";
expect(await compileAndTest(code)).toEqual({
"js,jsx,ts,tsx,toml": "error",
"json,jsonc,yaml": { default: { key }, key },
});
});

test("tsconfig.json is assumed jsonc", async () => {
const tests: Tests = {
"tsconfig.json": { loader: null, filename: "tsconfig.json" },
Expand Down
Loading