Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
131 changes: 95 additions & 36 deletions src/bun_core/string/immutable.rs
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@ use core::cmp::Ordering;

use crate::BoundedArray;
use crate::CrateError as Error;
use bun_alloc::AllocError;
use bun_alloc::{AllocError, ArenaVec, ArenaVecExt, MimallocArena};
use bun_highway as highway;
use bun_simdutf_sys::simdutf;

Expand Down Expand Up @@ -198,18 +198,55 @@ pub mod lexer_step {
wtf8_byte_sequence_length_with_invalid,
};

/// Non-ASCII tail of [`next_codepoint`]. Kept out-of-line so the hot
/// ASCII path stays small enough to inline into every `step()` site.
///
/// `#[cold]` is required: with fat LTO + `codegen-units = 1`, LLVM's
/// single-caller heuristic merges an `#[inline(never)]`-only callee back
/// into its sole caller, which then makes `next_codepoint` too large to
/// inline into `next()` (perf showed it as a separate ~2.6% symbol with
/// the multibyte decode folded in). `cold` parks this in `.text.unlikely`
/// and survives LTO's IPO inliner.
/// The lexer's non-ASCII step. `#[inline]`: `Lexer::next_codepoint_multibyte` is the out-of-line copy.
#[inline]
pub fn next_codepoint_multibyte(
contents: &[u8],
current: &mut usize,
first: u8,
ill_formed: &mut bool,
) -> CodePoint {
let at = *current;
// Well-formed UTF-8 with four readable bytes: one load and no call.
if let Some(chunk) = contents.get(at..).and_then(|tail| tail.first_chunk::<4>()) {
let word = u32::from_le_bytes(*chunk);
if first < 0xE0 {
if first >= 0xC2 && word & 0xC000 == 0x8000 {
*current = at + 2;
return (((word & 0x1F) << 6) | ((word >> 8) & 0x3F)) as CodePoint;
}
} else if first < 0xF0 {
if word & 0x00C0_C000 == 0x0080_8000 {
let cp = ((word & 0x0F) << 12) | ((word >> 2) & 0x0FC0) | ((word >> 16) & 0x3F);
// Neither overlong (below U+0800) nor a surrogate (U+D800..=U+DFFF).
if !matches!(cp >> 11, 0 | 0x1B) {
*current = at + 3;
return cp as CodePoint;
}
}
} else if first <= 0xF4 && word & 0xC0C0_C000 == 0x8080_8000 {
let cp = ((word & 0x07) << 18)
| ((word << 4) & 0x3_F000)
| ((word >> 10) & 0x0FC0)
| ((word >> 24) & 0x3F);
if cp.wrapping_sub(0x1_0000) < 0x10_0000 {
*current = at + 4;
return cp as CodePoint;
}
}
}
next_codepoint_ill_formed_or_at_end(contents, current, first, ill_formed)
}

/// Sets `*ill_formed` for bytes that are not UTF-8. The lexer learns that nowhere else.
#[cold]
#[inline(never)]
pub fn next_codepoint_multibyte(contents: &[u8], current: &mut usize, first: u8) -> CodePoint {
fn next_codepoint_ill_formed_or_at_end(
contents: &[u8],
current: &mut usize,
first: u8,
ill_formed: &mut bool,
) -> CodePoint {
let len = contents.len();
let cp_len = wtf8_byte_sequence_length_with_invalid(first) as usize;
let avail = len - *current;
Expand All @@ -218,33 +255,31 @@ pub mod lexer_step {
// may still be 1 for invalid lead bytes (0x80-0xBF, 0xF8-0xFF) — those must yield the
// raw byte, NOT the EOF sentinel, so the main lex loop falls through to its syntax-error
// arm instead of silently emitting TEndOfFile mid-stream.
let code_point: CodePoint = if cp_len == 1 {
first as CodePoint
} else if avail < cp_len {
if cp_len == 1 {
*ill_formed = true;
*current += 1;
return first as CodePoint;
}
if avail < cp_len {
// truncated multibyte at EOF
-1
} else {
let mut quad = [0u8; 4];
// SAFETY: `*current < len` (checked by caller), `cp_len ∈ 2..=4`, and
// `avail >= cp_len`, so `contents[current..current + cp_len]` is in-bounds.
// `decode_wtf8_rune_t_multibyte` only dereferences `p[0..len]`; pad bytes are
// never read.
unsafe {
core::ptr::copy_nonoverlapping(
contents.as_ptr().add(*current),
quad.as_mut_ptr(),
cp_len,
);
}
decode_wtf8_rune_t_multibyte(quad, cp_len as u8, UNICODE_REPLACEMENT as CodePoint)
};

*current += if code_point != UNICODE_REPLACEMENT as CodePoint {
cp_len
} else {
1
};
*ill_formed = true;
*current += cp_len;
return -1;
}

let mut quad = [0u8; 4];
quad[..cp_len].copy_from_slice(&contents[*current..*current + cp_len]);
let code_point: CodePoint = decode_wtf8_rune_t_multibyte(quad, cp_len as u8, -1);
if code_point < 0 {
*ill_formed = true;
*current += 1;
return UNICODE_REPLACEMENT as CodePoint;
}
// An encoded surrogate is WTF-8, not UTF-8.
if (0xD800..=0xDFFF).contains(&code_point) {
*ill_formed = true;
}
*current += cp_len;
code_point
}
}
Expand Down Expand Up @@ -1356,6 +1391,30 @@ pub fn str_utf8(bytes: &[u8]) -> Option<&str> {
}
}

/// `bytes` if already valid UTF-8, else an `arena` copy with each ill-formed sequence replaced by U+FFFD.
pub fn replace_invalid_utf8<'a>(bytes: &'a [u8], arena: &'a MimallocArena) -> &'a [u8] {
if is_valid_utf8(bytes) {
return bytes;
}
const REPLACEMENT: &[u8] = "\u{FFFD}".as_bytes();
let mut out_len = 0;
for chunk in bytes.utf8_chunks() {
out_len += chunk.valid().len();
if !chunk.invalid().is_empty() {
out_len += REPLACEMENT.len();
}
}
let mut out = ArenaVec::<u8>::with_capacity_in(out_len, arena);
for chunk in bytes.utf8_chunks() {
out.extend_from_slice(chunk.valid().as_bytes());
if !chunk.invalid().is_empty() {
out.extend_from_slice(REPLACEMENT);
}
}
debug_assert_eq!(out.len(), out_len);
out.into_bump_slice()
}

pub use index_of_newline_or_non_ascii as index_of_newline_or_non_ascii_or_ansi;

/// Checks if slice[offset..] has any < 0x20 or > 127 characters
Expand Down
25 changes: 17 additions & 8 deletions src/bundler/ParseTask.rs
Original file line number Diff line number Diff line change
Expand Up @@ -816,13 +816,15 @@ pub mod parse_worker {
opts: ParserOptions<'static>,
bump: &'static Bump,
resolver: *mut Resolver,
source: &'static Source,
// The JavaScript arm replaces it when it had to decode the file (see `cache::JavaScript::parse`).
source_slot: &mut &'static Source,
loader: Loader,
unique_key_prefix: u64,
unique_key_for_additional_file: &mut FileLoaderHash,
has_any_css_locals: &AtomicU32,
) -> core::result::Result<JSAst<'static>, AnyError> {
use core::fmt::Write as _;
let source: &'static Source = *source_slot;

// SAFETY: `transpiler` is a live worker-owned `*mut Transpiler`.
// `options` and `resolver` are disjoint fields of `Transpiler`; reborrowing
Expand All @@ -840,16 +842,23 @@ pub mod parse_worker {
// field drift is a hard error) before the move.
let fallback_opts = opts.clone_for_lazy_export();
let module_type = opts.module_type;
return if let Some(res) =
(crate::cache::JavaScript {}).parse(bump, opts, &topts.define, log, source)?
{
let parsed = (crate::cache::JavaScript {}).parse(
bump,
opts,
&topts.define,
log,
source_slot,
)?;
let source: &'static Source = *source_slot;
return if let Some(res) = parsed {
// `Cached`/`AlreadyBundled` are runtime-loader
// states that never reach the bundler's `getAST`, so unwrap.
match res {
bun_js_parser::Result::Ast(ast) => Ok(JSAst::init(*ast)),
bun_js_parser::Result::Cached
| bun_js_parser::Result::AlreadyBundled(_) => {
unreachable!("bundler parse never yields Cached/AlreadyBundled")
| bun_js_parser::Result::AlreadyBundled(_)
| bun_js_parser::Result::NotUtf8(_) => {
unreachable!("bundler parse never yields Cached/AlreadyBundled/NotUtf8")
}
}
} else if module_type == options::ModuleType::Esm {
Expand Down Expand Up @@ -2440,7 +2449,7 @@ pub mod parse_worker {

// Allocated in the worker arena so `js_parser::new_lazy_export_ast`'s
// `&'bump Source` parameter is satisfied (`bump` is the same arena).
let source: &'static Source = bump.alloc(Source {
let mut source: &'static Source = bump.alloc(Source {
// `Source.path` is `bun_paths::fs::Path<'static>`, distinct from
// `bun_resolver::fs::Path` (TYPE_ONLY mirror). Construct
// field-by-field across the type boundary.
Expand Down Expand Up @@ -2692,7 +2701,7 @@ pub mod parse_worker {
opts,
bump,
resolver,
source,
&mut source,
loader,
task_ctx.unique_key,
&mut unique_key_for_additional_file,
Expand Down
40 changes: 39 additions & 1 deletion src/bundler/cache.rs
Original file line number Diff line number Diff line change
Expand Up @@ -74,8 +74,42 @@ impl JavaScript {
opts: js_parser::ParserOptions<'a>,
defines: &'a Define,
log: &mut bun_ast::Log,
source: &'a bun_ast::Source,
source: &mut &'a bun_ast::Source,
) -> Result<Option<js_parser::Result<'a>>, crate::Error> {
self.parse_impl(bump, opts, defines, log, source, true)
}

/// Only reached when the lexer flagged bytes that are not UTF-8.
#[cold]
fn parse_decoded<'a>(
&self,
bump: &'a Bump,
opts: js_parser::ParserOptions<'a>,
defines: &'a Define,
log: &mut bun_ast::Log,
source: &mut &'a bun_ast::Source,
) -> Result<Option<js_parser::Result<'a>>, crate::Error> {
let decoded = strings::replace_invalid_utf8(source.contents(), bump);
// The caller prints and maps with `*source`, so it has to be the text the AST was parsed from.
*source = bump.alloc(bun_ast::Source {
contents: std::borrow::Cow::Borrowed(bun_ast::StoreStr::new(decoded).slice()),
..(**source).clone()
});
// `false`: parse whatever `decoded` holds, so this cannot recurse again.
self.parse_impl(bump, opts, defines, log, source, false)
Comment thread
robobun marked this conversation as resolved.
}

fn parse_impl<'a>(
&self,
bump: &'a Bump,
mut opts: js_parser::ParserOptions<'a>,
defines: &'a Define,
log: &mut bun_ast::Log,
source_slot: &mut &'a bun_ast::Source,
stop_on_ill_formed_utf8: bool,
) -> Result<Option<js_parser::Result<'a>>, crate::Error> {
let source: &'a bun_ast::Source = *source_slot;
opts.features.stop_on_ill_formed_utf8 = stop_on_ill_formed_utf8;
let mut temp_log = bun_ast::Log::init();
temp_log.level = log.level;
let parser = match js_parser::Parser::init(opts, &mut temp_log, source, defines, bump) {
Expand All @@ -87,6 +121,10 @@ impl JavaScript {
};

let result = match parser.parse() {
// `temp_log` holds messages about text that was read wrong; drop it.
Ok(js_parser::Result::NotUtf8(opts)) => {
return self.parse_decoded(bump, *opts, defines, log, source_slot);
}
Ok(r) => {
// The parser halts on every logged error.
debug_assert_eq!(temp_log.errors, 0);
Expand Down
15 changes: 11 additions & 4 deletions src/bundler/transpiler.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1405,7 +1405,7 @@ impl<'a> Transpiler<'a> {
// (`Drop` is a no-op).
let mut source_backing: resolver::cache::Contents = resolver::cache::Contents::Empty;

let source: &'a bun_ast::Source = arena.alloc('brk: {
let mut source: &'a bun_ast::Source = arena.alloc('brk: {
if let Some(virtual_source) = this_parse.virtual_source {
break 'brk virtual_source.clone();
}
Expand Down Expand Up @@ -1711,9 +1711,13 @@ impl<'a> Transpiler<'a> {
// the real `parse` lives on `crate::cache::JavaScript`. Both
// are stateless unit structs, so calling the bundler-crate one
// directly is equivalent.
let parsed = match crate::cache::JavaScript::init()
.parse(arena, opts, define, log, source)
{
let parsed = match crate::cache::JavaScript::init().parse(
arena,
opts,
define,
log,
&mut source,
) {
Ok(Some(r)) => r,
Ok(None) | Err(_) => return None,
};
Expand Down Expand Up @@ -1808,6 +1812,9 @@ impl<'a> Transpiler<'a> {
empty: false,
source_contents_backing: source_backing,
},
js_ast::Result::NotUtf8(_) => {
unreachable!("`cache::JavaScript::parse` parses the decoded text itself")
}
});
}
// TODO: use lazy export AST
Expand Down
Loading
Loading