Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion src/bundler/transpiler.rs
Original file line number Diff line number Diff line change
Expand Up @@ -2265,7 +2265,7 @@ fn parse_unsupported_loader(loader: options::Loader, path: &bun_paths::fs::Path<
// `format` is a runtime arg rather than a const generic —
// `bun_js_printer::Format` doesn't derive `ConstParamTy` (and can't be added
// from this crate). All callers pass a literal anyway; the inner
// `print_ast::<_, ASCII_ONLY, ENABLE_SOURCE_MAP>` keeps both const-generic
// `print_ast::<_, IS_BUN_PLATFORM, ENABLE_SOURCE_MAP>` keeps both const-generic
// bools, so codegen monomorphizes the printer body identically.
// PERF: outer `match format` is one extra branch — profile if hot.
// ══════════════════════════════════════════════════════════════════════════
Expand Down
60 changes: 38 additions & 22 deletions src/js_printer/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1009,12 +1009,12 @@ where
}

while i < n {
let width: u8 = match ENCODING {
let mut width: u8 = match ENCODING {
Encoding::Latin1 | Encoding::Ascii => 1,
Encoding::Utf8 => strings::wtf8_byte_sequence_length_with_invalid(text[i]),
Encoding::Utf16 => 1,
};
let clamped_width = (width as usize).min(n.saturating_sub(i));
let mut clamped_width = (width as usize).min(n.saturating_sub(i));
let c: i32 = match ENCODING {
Encoding::Utf8 => {
let bytes: [u8; 4] = match clamped_width {
Expand All @@ -1032,10 +1032,19 @@ where
}
Encoding::Latin1 => text[i] as i32,
Encoding::Utf16 => {
// TODO: if this is a part of a surrogate pair, we could parse the whole codepoint in order
// to emit it as a single \u{result} rather than two paired \uLOW\uHIGH.
// eg: "\u{10334}" will convert to "𐌴" without this.
code_unit_at!(i)
let unit = code_unit_at!(i);
if strings::u16_is_lead(unit as u16) && i + 1 < n {
let trail = code_unit_at!(i + 1);
if strings::u16_is_trail(trail as u16) {
width = 2;
clamped_width = 2;
strings::u16_get_supplementary(unit as u16, trail as u16) as i32
} else {
unit
}
} else {
unit
}
}
};

Expand Down Expand Up @@ -4598,7 +4607,7 @@ pub mod __gated_printer {
self.print(b" ");
}

if IS_BUN_PLATFORM {
if ASCII_ONLY {
// Translate any non-ASCII to unicode escape sequences
let mut ascii_start: usize = 0;
let mut is_ascii = false;
Expand Down Expand Up @@ -7794,11 +7803,10 @@ pub type BufferPrinter = Writer<BufferWriter>;
pub enum Format {
Esm,
Cjs,
// bun.js must escape non-latin1 identifiers in the output This is because
// we load JavaScript as a UTF-8 buffer instead of a UTF-16 buffer
// JavaScriptCore does not support UTF-8 identifiers when the source code
// string is loaded as const char* We don't want to double the size of code
// in memory...
// Runtime-transpiler path. The printer now emits UTF-8 for this variant and
// consumers call `String::clone_utf8`, so non-ASCII source text is preserved
// verbatim (observable via `Function.prototype.toString`, `RegExp#source`,
// tagged-template `.raw`). The name is kept for churn avoidance.
EsmAscii,
CjsAscii,
}
Expand Down Expand Up @@ -7903,7 +7911,12 @@ pub fn get_source_map_builder<const IS_BUN_PLATFORM: bool>(
// Top-level print entry points
// ───────────────────────────────────────────────────────────────────────────

pub fn print_ast<'a, W: WriterTrait, const ASCII_ONLY: bool, const GENERATE_SOURCE_MAP: bool>(
pub fn print_ast<
'a,
W: WriterTrait,
const IS_BUN_PLATFORM: bool,
const GENERATE_SOURCE_MAP: bool,
>(
Comment thread
robobun marked this conversation as resolved.
_writer: W,
bump: &'a bun_alloc::Arena,
tree: &'a Ast,
Expand Down Expand Up @@ -8015,9 +8028,11 @@ pub fn print_ast<'a, W: WriterTrait, const ASCII_ONLY: bool, const GENERATE_SOUR

// defer: if minify_identifiers { renamer.deinit() } — Drop handles.

// `is_bun_platform = ascii_only` for printAst.
type PrinterType<'a, W, const A: bool, const G: bool> =
Printer<'a, W, A, false, /*IS_BUN_PLATFORM=*/ A, false, G>;
// Runtime-transpiler path: emit UTF-8 so `Function.prototype.toString`,
// `RegExp#source` and tagged-template `.raw` reflect the author's source
// text. Consumers pass the output through `String::clone_utf8`.
type PrinterType<'a, W, const B: bool, const G: bool> =
Printer<'a, W, /*ASCII_ONLY=*/ false, false, /*IS_BUN_PLATFORM=*/ B, false, G>;
let mut writer = _writer;
// Pre-size the output buffer ~proportional to the source. Transpiled output
// is almost always within a small factor of the input, so reserving up front
Expand All @@ -8026,13 +8041,13 @@ pub fn print_ast<'a, W: WriterTrait, const ASCII_ONLY: bool, const GENERATE_SOUR
let _ = writer.reserve(source.contents().len() as u64);

let mut opts = opts;
let source_map_builder = get_source_map_builder::<ASCII_ONLY>(
let source_map_builder = get_source_map_builder::<IS_BUN_PLATFORM>(
GenerateSourceMap::lazy_if(GENERATE_SOURCE_MAP),
&mut opts,
source,
tree,
);
let mut printer = PrinterType::<W, ASCII_ONLY, GENERATE_SOURCE_MAP>::init(
let mut printer = PrinterType::<W, IS_BUN_PLATFORM, GENERATE_SOURCE_MAP>::init(
writer,
bump,
tree.import_records.as_slice(),
Expand All @@ -8050,7 +8065,7 @@ pub fn print_ast<'a, W: WriterTrait, const ASCII_ONLY: bool, const GENERATE_SOUR
// Borrowck: `opts` was moved into `Printer::init`; populate
// `printer.module_info` by taking it back out of `printer.options`
// (see `print_with_writer_and_platform`).
if PrinterType::<W, ASCII_ONLY, GENERATE_SOURCE_MAP>::MAY_HAVE_MODULE_INFO {
if PrinterType::<W, IS_BUN_PLATFORM, GENERATE_SOURCE_MAP>::MAY_HAVE_MODULE_INFO {
printer.module_info = printer.options.module_info.take();
}
printer.binary_expression_stack = Vec::new();
Expand All @@ -8070,7 +8085,7 @@ pub fn print_ast<'a, W: WriterTrait, const ASCII_ONLY: bool, const GENERATE_SOUR
// `require` must be an unbound variable.
printer.print(b"var {require}=import.meta;");

if PrinterType::<W, ASCII_ONLY, GENERATE_SOURCE_MAP>::MAY_HAVE_MODULE_INFO {
if PrinterType::<W, IS_BUN_PLATFORM, GENERATE_SOURCE_MAP>::MAY_HAVE_MODULE_INFO {
if let Some(mi) = printer.module_info.as_deref_mut() {
mi.flags.contains_import_meta = true;
let s = mi.str(b"require");
Expand All @@ -8088,8 +8103,9 @@ pub fn print_ast<'a, W: WriterTrait, const ASCII_ONLY: bool, const GENERATE_SOUR
}
printer.check_stack_overflow()?;

let have_module_info = PrinterType::<W, ASCII_ONLY, GENERATE_SOURCE_MAP>::MAY_HAVE_MODULE_INFO
&& printer.module_info.is_some();
let have_module_info =
PrinterType::<W, IS_BUN_PLATFORM, GENERATE_SOURCE_MAP>::MAY_HAVE_MODULE_INFO
&& printer.module_info.is_some();
if have_module_info {
printer
.module_info
Expand Down
2 changes: 1 addition & 1 deletion src/jsc/AsyncModule.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1375,7 +1375,7 @@ impl AsyncModule {
}

Ok(ResolvedSource {
source_code: BunString::clone_latin1(printer.ctx.get_written()),
source_code: BunString::clone_utf8(printer.ctx.get_written()),
specifier: BunString::init(specifier),
source_url: BunString::init(path.text),
is_commonjs_module,
Expand Down
14 changes: 7 additions & 7 deletions src/jsc/RuntimeTranspilerCache.rs
Original file line number Diff line number Diff line change
Expand Up @@ -41,7 +41,8 @@ bun_core::declare_scope!(cache, visible);
/// Version 22: Serialize `has_tla` in the cached ESM record flags byte. Entries
/// written before #30888 carried `has_tla=false` for every module; the cache-HIT
/// path reinstates the bug for any previously-cached TLA module (#30887).
const EXPECTED_VERSION: u32 = 22;
/// Version 23: Runtime transpiler emits UTF-8 instead of ASCII-escaped output.
const EXPECTED_VERSION: u32 = 23;

/// Source files smaller than this are not written to / read from the on-disk
/// transpiler cache. Originally 50 KiB, which excluded almost every file in a
Expand Down Expand Up @@ -1014,7 +1015,7 @@ impl RuntimeTranspilerCache {
return;
}
debug_assert!(self.entry.is_none());
let output_code = BunString::clone_latin1(output_code_bytes);
let output_code = BunString::clone_utf8(output_code_bytes);
// Refcount stays at 1, sole owner.
// BunString is Copy with no Drop, so an extra dupe_ref here would leak.
self.output_code = Some(output_code);
Expand All @@ -1033,7 +1034,7 @@ impl RuntimeTranspilerCache {
}
#[cfg(debug_assertions)]
{
bun_core::scoped_log!(cache, "put() = {} bytes", output_code.latin1().len());
bun_core::scoped_log!(cache, "put() = {} bytes", output_code_bytes.len());
}
}
}
Expand Down Expand Up @@ -1083,10 +1084,9 @@ bun_ast::link_impl_TranspilerCacheImpl! {
}
debug_assert!(this.entry.is_none());

// Borrowed Latin-1 view: `to_file` only reads `byte_slice()` + the encoding
// tag (unmarked 8-bit ZigString -> Encoding::LATIN1, same as clone_latin1),
// and `output_code_bytes` outlives the synchronous `to_file` call.
let output_code = BunString::ascii(output_code_bytes);
// UTF-8-tagged so `to_file` records Encoding::UTF8 (it then
// Box-copies into `OutputCode::Utf8`; see that fn's PERF note).
let output_code = BunString::borrow_utf8(output_code_bytes);
let result = RuntimeTranspilerCache::to_file(
this.input_byte_length.unwrap(),
this.input_hash.unwrap(),
Expand Down
4 changes: 2 additions & 2 deletions src/jsc/RuntimeTranspilerStore.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1034,7 +1034,7 @@ impl TranspilerJob {
_ => (ptr::null_mut(), 0),
};
self.resolved_source = OwnedResolvedSource::from(ResolvedSource {
source_code: String::clone_latin1(&parse_result.source.contents),
source_code: String::clone_utf8(&parse_result.source.contents),
already_bundled: true,
bytecode_cache,
bytecode_cache_size,
Expand Down Expand Up @@ -1181,7 +1181,7 @@ impl TranspilerJob {
// `cache.output_code` (only the `r#impl == None` fallback does,
// and `r#impl` is `Some(Jsc)` here), so it is always `None`.
debug_assert!(cache.output_code.is_none());
let result = String::clone_latin1(written);
let result = String::clone_utf8(written);

// SAFETY: leaf scalar field read on `*vm`; see `vm` note above.
if written.len() > 1024 * 1024 * 2 || unsafe { (*vm).smol } {
Expand Down
12 changes: 12 additions & 0 deletions src/jsc/VirtualMachine.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3785,6 +3785,18 @@ impl VirtualMachine {
..Default::default()
};
}
// `ref_counted_string` wraps the bytes as a Latin-1 external string.
// The runtime transpiler now emits UTF-8, so non-ASCII output would be
// mojibaked; transcode it instead of interning on that (rare) path.
if !bun_core::strings::is_all_ascii(code) {
return ResolvedSource {
source_code: bun_core::String::clone_utf8(code),
specifier,
source_url: create_if_different(&specifier, source_url),
source_code_needs_deref: true,
..Default::default()
};
}
// Const-generic bool can't be `!ADD_DOUBLE_REF`, so branch.
let source = if ADD_DOUBLE_REF {
self.ref_counted_string::<false>(code, hash_)
Expand Down
2 changes: 1 addition & 1 deletion src/runtime/cli/repl.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1862,7 +1862,7 @@ impl<'a> Repl<'a> {

if bun_js_printer::print_ast::<
_,
/* ASCII_ONLY */ true,
/* IS_BUN_PLATFORM */ true,
/* GENERATE_SOURCE_MAP */ false,
>(
&mut buffer_printer,
Expand Down
4 changes: 2 additions & 2 deletions src/runtime/jsc_hooks.rs
Original file line number Diff line number Diff line change
Expand Up @@ -2645,7 +2645,7 @@ fn transpile_source_code_inner(
_ => (core::ptr::null_mut(), 0),
};
return Ok(OwnedResolvedSource::from(ResolvedSource {
source_code: bun_core::String::clone_latin1(&source.contents),
source_code: bun_core::String::clone_utf8(&source.contents),
specifier: input_specifier.dupe_ref(),
source_url: create_if_different(input_specifier, path.text),
already_bundled: true,
Expand Down Expand Up @@ -3059,7 +3059,7 @@ fn transpile_source_code_inner(
// `None`.
debug_assert!(cache.output_code.is_none());
let written_len = written.len();
let source_code = bun_core::String::clone_latin1(written);
let source_code = bun_core::String::clone_utf8(written);
Comment thread
robobun marked this conversation as resolved.
// `printer.ctx.buffer.deinit()`: release the
// large/--smol print buffer now instead of holding it until the
// next transpile. Replacing the printer drops the old buffer
Expand Down
11 changes: 7 additions & 4 deletions test/bundler/transpiler/transpiler.test.js
Original file line number Diff line number Diff line change
Expand Up @@ -2308,13 +2308,16 @@ console.log(<div {...obj} key="after" />);`),
});

it("unicode surrogates", () => {
expectPrinted_(`console.log("𐌴")`, 'console.log("\\uD800\\uDF34")');
expectPrinted_(`console.log("\\u{10334}")`, 'console.log("\\uD800\\uDF34")');
expectPrinted_(`console.log("\\uD800\\uDF34")`, 'console.log("\\uD800\\uDF34")');
expectPrinted_(`console.log("𐌴")`, 'console.log("𐌴")');
expectPrinted_(`console.log("\\u{10334}")`, 'console.log("𐌴")');
expectPrinted_(`console.log("\\uD800\\uDF34")`, 'console.log("𐌴")');
expectPrinted_(`console.log("\\u{10334}" === "\\uD800\\uDF34")`, "console.log(true)");
expectPrinted_(`console.log("\\u{10334}" === "\\uDF34\\uD800")`, "console.log(false)");
expectPrintedMin_(`console.log("abc" + "def")`, 'console.log("abcdef")');
// Lone surrogates stay escaped; concatenation must not combine across operands.
expectPrintedMin_(`console.log("\\uD800" + "\\uDF34")`, 'console.log("\\uD800" + "\\uDF34")');
expectPrinted_(`console.log("\\uD800")`, 'console.log("\\uD800")');
expectPrinted_(`console.log("\\uDF34")`, 'console.log("\\uDF34")');
});

it("fold string addition", () => {
Expand Down Expand Up @@ -3945,7 +3948,7 @@ console.log(foo, array);

const input = `let list = ["•", "-", "◦", "▪", "▫"];\n`;
const result = await transpiler.transform(input);
expect(result).toBe(`let list = [\"\\u2022\", \"-\", \"\\u25E6\", \"\\u25AA\", \"\\u25AB\"];\n`);
expect(result).toBe(`let list = ["•", "-", "◦", "▪", "▫"];\n`);
});
});

Expand Down
5 changes: 1 addition & 4 deletions test/js/bun/shell/bunshell.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -342,10 +342,7 @@ describe("bunshell", () => {

test("escape unicode", async () => {
const { stdout } = await $`echo \\弟\\気`;
// TODO: Uncomment and replace after unicode in template tags is supported
// expect(stdout.toString("utf8")).toEqual(`\弟\気\n`);
// Set this here for now, because unicode in template tags while using .raw is broken, but should be fixed
expect(stdout.toString("utf8")).toEqual("\\u5F1F\\u6C17\n");
expect(stdout.toString("utf8")).toEqual("\\弟\\気\n");
});

/**
Expand Down
Loading
Loading