Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion src/ast/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1105,7 +1105,10 @@ impl BabyString {
if substring.is_empty() {
return BabyString::new(0, 0);
}
let off = bun_core::strings::index_of(container, substring).expect("unreachable");
// A lossily formatted `container` may not embed `substring`; report no specifier.
let Some(off) = bun_core::strings::index_of(container, substring) else {
return BabyString::new(0, 0);
};
BabyString::new(off as u16, substring.len() as u16) // @truncate
}

Expand Down
36 changes: 35 additions & 1 deletion src/bun_core/string/immutable.rs
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@ use core::cmp::Ordering;

use crate::BoundedArray;
use crate::CrateError as Error;
use bun_alloc::AllocError;
use bun_alloc::{AllocError, MimallocArena};
use bun_highway as highway;
use bun_simdutf_sys::simdutf;

Expand Down Expand Up @@ -1356,6 +1356,40 @@ pub fn str_utf8(bytes: &[u8]) -> Option<&str> {
}
}

/// `(bytes, None)` if already valid UTF-8, else an `arena` copy with each ill-formed sequence replaced by U+FFFD, and the offset of the first one.
pub fn replace_invalid_utf8<'a>(
bytes: &'a [u8],
arena: &'a MimallocArena,
) -> (&'a [u8], Option<usize>) {
if is_valid_utf8(bytes) {
return (bytes, None);
}
const REPLACEMENT: &[u8] = "\u{FFFD}".as_bytes();
let mut first_invalid = None;
let mut out_len = 0;
for chunk in bytes.utf8_chunks() {
out_len += chunk.valid().len();
if !chunk.invalid().is_empty() {
first_invalid.get_or_insert(out_len);
out_len += REPLACEMENT.len();
}
}
// Not `ArenaVec<u8>`: an instance of its shared generics in this crate replaces the one `bun_ast` inlines.
let out = arena.alloc_slice_fill_copy(out_len, 0u8);
let mut at = 0;
for chunk in bytes.utf8_chunks() {
let valid = chunk.valid().as_bytes();
out[at..at + valid.len()].copy_from_slice(valid);
at += valid.len();
if !chunk.invalid().is_empty() {
out[at..at + REPLACEMENT.len()].copy_from_slice(REPLACEMENT);
at += REPLACEMENT.len();
}
}
debug_assert_eq!(at, out_len);
(out, first_invalid)
}

pub use index_of_newline_or_non_ascii as index_of_newline_or_non_ascii_or_ansi;

/// Checks if slice[offset..] has any < 0x20 or > 127 characters
Expand Down
10 changes: 9 additions & 1 deletion src/bundler/ParseTask.rs
Original file line number Diff line number Diff line change
Expand Up @@ -2395,7 +2395,12 @@ pub mod parse_worker {
}
*step = Step::Parse;

let entry_contents: &[u8] = entry.contents.as_slice();
// The CSS tokenizer needs valid UTF-8; decode before `source` is built so offsets match.
let (entry_contents, first_invalid_utf8) = if loader == Loader::Css {
strings::replace_invalid_utf8(entry.contents.as_slice(), bump)
} else {
(entry.contents.as_slice(), None)
};
let is_empty = strings::is_all_whitespace(entry_contents);

// SAFETY: `transpiler` derived from a live `&mut` above. Reborrow only the
Comment thread
robobun marked this conversation as resolved.
Expand Down Expand Up @@ -2462,6 +2467,9 @@ pub mod parse_worker {
contents_is_recycled: false,
..Default::default()
});
if let Some(first_invalid) = first_invalid_utf8 {
bun_css::warn_invalid_utf8(log, source, first_invalid);
}

let target = (if task.source_index.get() == 1 {
target_from_hashbang(entry_contents)
Expand Down
8 changes: 7 additions & 1 deletion src/bundler/transpiler.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3180,9 +3180,15 @@ impl<'a> Transpiler<'a> {
// `'bump`-threading note).
let alloc: &'static Arena = unsafe { bun_ptr::detach_lifetime_ref::<Arena>(self.arena) };

// The CSS tokenizer requires well-formed UTF-8 (see `ParseTask`).
let (code, first_invalid_utf8) = strings::replace_invalid_utf8(entry.contents(), alloc);
if let Some(first_invalid) = first_invalid_utf8 {
let source = bun_ast::Source::init_path_string(file_path_text, code);
bun_css::warn_invalid_utf8(self.log_mut(), &source, first_invalid);
}
let (mut sheet, extra) = match bun_css::StyleSheet::<bun_css::DefaultAtRule>::parse(
alloc,
entry.contents(),
code,
opts,
None,
bun_ast::Index::source(0u32),
Expand Down
55 changes: 55 additions & 0 deletions src/css/css_parser.rs
Original file line number Diff line number Diff line change
Expand Up @@ -4098,6 +4098,10 @@ pub(crate) unsafe fn src_str(s: &[u8]) -> &'static [u8] {

impl<'a> Tokenizer<'a> {
pub(crate) fn init_with_arena(src: &'a [u8], arena: &'a Bump) -> Tokenizer<'a> {
debug_assert!(
strings::is_valid_utf8(src),
"CSS tokenizer input must be valid UTF-8 (see strings::replace_invalid_utf8)"
);
Comment thread
coderabbitai[bot] marked this conversation as resolved.
Tokenizer {
src,
position: 0,
Expand Down Expand Up @@ -5161,6 +5165,57 @@ pub(crate) fn split_source_map(contents: &[u8]) -> Option<&[u8]> {
None
}

/// Warns that `strings::replace_invalid_utf8` changed `source`, at the first replaced sequence.
pub fn warn_invalid_utf8(log: &mut Log, source: &bun_ast::Source, first_invalid: usize) {
let range = bun_ast::Range {
loc: bun_ast::Loc {
start: i32::try_from(first_invalid).unwrap_or(i32::MAX),
},
len: REPLACEMENT_CHAR_UTF8.len() as i32,
};
match ignored_charset(&source.contents) {
Some(charset) => log.add_range_warning_fmt(
Some(source),
range,
format_args!(
"@charset \"{}\" was ignored, this file was read as UTF-8 and each invalid byte sequence was replaced with U+FFFD",
bstr::BStr::new(charset)
),
),
None => log.add_range_warning_fmt(
Comment thread
robobun marked this conversation as resolved.
Some(source),
range,
format_args!(
"This file is not valid UTF-8, each invalid byte sequence was replaced with U+FFFD"
),
),
}
}

/// The label of a leading `@charset "…";` (the byte pattern of css-syntax-3 §3.2), unless it is empty or names UTF-8.
fn ignored_charset(contents: &[u8]) -> Option<&[u8]> {
const UTF8_LABELS: [&[u8]; 6] = [
b"utf-8",
b"utf8",
b"unicode-1-1-utf-8",
b"unicode11utf8",
b"unicode20utf8",
b"x-unicode20utf8",
];
let rest = contents[..contents.len().min(1024)].strip_prefix(b"@charset \"")?;
let end = strings::index_of_char_usize(rest, b'"')?;
if rest.get(end + 1) != Some(&b';')
|| !rest[..end]
.iter()
.all(|b| matches!(b, 0x16..=0x21 | 0x23..=0x7F))
{
return None;
}
let label = strings::trim(&rest[..end], b" ");
(!label.is_empty() && !strings::eql_any_case_insensitive_ascii(label, &UTF8_LABELS))
.then_some(label)
}

// ───────────────────────────── Token ─────────────────────────────

#[derive(Clone, Copy, PartialEq, Eq, strum::IntoStaticStr)]
Expand Down
2 changes: 1 addition & 1 deletion src/css/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -185,7 +185,7 @@ pub use dependencies::Dependency;
// for css_jsc / bundler.
pub use css_parser::{
DefaultAtRule, LocalsResultsMap, MinifyOptions, Parser, ParserFlags, ParserInput,
ParserOptions, StyleAttribute, StyleSheet, StylesheetExtra, ToCssResult,
ParserOptions, StyleAttribute, StyleSheet, StylesheetExtra, ToCssResult, warn_invalid_utf8,
};
pub use printer::{ImportInfo, Printer, PrinterOptions, PseudoClasses};
/// Dependent crates name this `ImportRecordHandler`; the surviving type is
Expand Down
5 changes: 4 additions & 1 deletion src/runtime/cli/pm_diff_normalize.rs
Original file line number Diff line number Diff line change
Expand Up @@ -738,7 +738,10 @@ fn normalize_css(path: &[u8], bytes: &[u8]) -> Option<Normalized> {
static ARENA: std::sync::LazyLock<Arena> = std::sync::LazyLock::new(Arena::new);
static FED: core::sync::atomic::AtomicUsize = core::sync::atomic::AtomicUsize::new(0);
const CSS_BUDGET: usize = 64 * 1024 * 1024;
if FED.fetch_add(bytes.len(), core::sync::atomic::Ordering::Relaxed) + bytes.len() > CSS_BUDGET
// The CSS tokenizer requires well-formed UTF-8; anything else diffs as text.
if !bun_core::strings::is_valid_utf8(bytes)
Comment thread
robobun marked this conversation as resolved.
|| FED.fetch_add(bytes.len(), core::sync::atomic::Ordering::Relaxed) + bytes.len()
> CSS_BUDGET
{
Comment thread
robobun marked this conversation as resolved.
return None;
}
Expand Down
11 changes: 11 additions & 0 deletions test/cli/install/bun-pm-diff.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1139,6 +1139,17 @@ describe.concurrent("bun pm diff (engine invariants)", () => {
expect(exitCode).toBe(0);
});

test("CSS that is not valid UTF-8 is diffed as text, so a change to one ill-formed byte still shows", async () => {
// The CSS parser only takes valid UTF-8. A lossy decode maps 0xE9 and 0xE8 to the same U+FFFD, so the two
// re-prints would be equal and the change would read as "formatting only".
const sheet = (byte: number) =>
Buffer.concat([Buffer.from('.a::before { content: "caf'), Buffer.from([byte]), Buffer.from('"; }\n')]);
const { text, exitCode } = await pretty({ "a/latin1.css": sheet(0xe9), "b/latin1.css": sheet(0xe8) });
expect(text).toMatch(/\nlatin1\.css ─+ not parsed \+1 -1\n/);
expect(text).not.toContain("formatting only");
expect(exitCode).toBe(0);
});

// A debug build byte-scans the 64 MB line slowly (~30 s); release is well under a second.
test.skipIf(isDebug)(
"a file over the normalization size limit is diffed as text and says so",
Expand Down
72 changes: 0 additions & 72 deletions test/js/bun/css/invalid-utf8-column.test.ts

This file was deleted.

Loading
Loading