Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
195 changes: 188 additions & 7 deletions src/string/immutable/grapheme.zig
Original file line number Diff line number Diff line change
Expand Up @@ -6,21 +6,202 @@ pub fn graphemeBreak(cp1: u21, cp2: u21, state: *BreakState) bool {
(Precompute.Key{
.gbc1 = table.get(cp1).grapheme_boundary_class,
.gbc2 = table.get(cp2).grapheme_boundary_class,
.state = state.*,
.state = state.precomputed,
}).index()
];
state.* = value.state;
state.precomputed = value.state;

// GB9c: \p{InCB=Consonant} [\p{InCB=Extend}\p{InCB=Linker}]* \p{InCB=Linker} [\p{InCB=Extend}\p{InCB=Linker}]* × \p{InCB=Consonant}
// If the precomputed rules say "break", check if GB9c prevents it.
const incb2 = indicConjunctBreakProperty(cp2);
if (value.result) {
if (state.incb == .after_linker and incb2 == .consonant) {
// GB9c applies: don't break between the linker sequence and the consonant
state.incb = .after_consonant;
return false;
}
}

// Update InCB state for next iteration.
// The state tracks the InCB sequence ending at cp2.
// We also check cp1's InCB property for the case where cp1 is the first consonant
// in a new sequence (state.incb was .none from initialization or a previous break).
const incb1 = indicConjunctBreakProperty(cp1);
switch (incb2) {
.consonant => state.incb = .after_consonant,
.linker => {
// A linker extends the sequence if preceded by a consonant (directly or via state)
if (state.incb == .after_consonant or state.incb == .after_linker or incb1 == .consonant) {
state.incb = .after_linker;
} else {
state.incb = .none;
}
},
.extend => {
// Extend characters maintain InCB state.
// If state was .none but cp1 is a consonant, start tracking.
if (state.incb == .none and incb1 == .consonant) {
state.incb = .after_consonant;
}
},
.none => state.incb = .none,
}

return value.result;
}

pub const BreakState = packed struct(u2) {
pub const BreakState = struct {
precomputed: PrecomputedState = .{},
incb: InCBState = .none,
};

const PrecomputedState = packed struct(u2) {
extended_pictographic: bool = false,
regional_indicator: bool = false,
};
Comment on lines +53 to 61

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🛠️ Refactor suggestion | 🟠 Major

🧩 Analysis chain

🏁 Script executed:

# First, find the grapheme.zig file to understand its structure
find . -name "grapheme.zig" -type f

Repository: oven-sh/bun

Length of output: 90


🏁 Script executed:

# Search for external usage of .precomputed and .incb fields
rg -n '\.precomputed\b' --type=zig
rg -n '\.incb\b' --type=zig

Repository: oven-sh/bun

Length of output: 1157


🏁 Script executed:

# Search for BreakState instantiations and usage
rg -n '\bBreakState\b' --type=zig -A 2 -B 2

Repository: oven-sh/bun

Length of output: 1540


🏁 Script executed:

# Search for PrecomputedState usage
rg -n '\bPrecomputedState\b' --type=zig -A 2 -B 2

Repository: oven-sh/bun

Length of output: 2064


Make BreakState and PrecomputedState fields private using # prefix.

These fields are accessed only within grapheme.zig. External code in visible.zig merely instantiates BreakState and passes it to functions; it never directly accesses the fields. Mark them private to match the repo's Zig guidelines and avoid unnecessary API coupling.

♻️ Proposed refactor
 pub const BreakState = struct {
-    precomputed: PrecomputedState = .{},
-    incb: InCBState = .none,
+    `#precomputed`: PrecomputedState = .{},
+    `#incb`: InCBState = .none,
 };
 
 const PrecomputedState = packed struct(u2) {
-    extended_pictographic: bool = false,
-    regional_indicator: bool = false,
+    `#extended_pictographic`: bool = false,
+    `#regional_indicator`: bool = false,
 };
🤖 Prompt for AI Agents
In `@src/string/immutable/grapheme.zig` around lines 53 - 61, Make the struct
fields private by prefixing their names with `#`: change BreakState fields
`precomputed` and `incb` to `#precomputed` and `#incb`, and change
PrecomputedState fields `extended_pictographic` and `regional_indicator` to
`#extended_pictographic` and `#regional_indicator`; update all internal
references in grapheme.zig that read/write these fields (look for uses of
BreakState and PrecomputedState within this file) so they use the new private
names, leaving external callers (which only construct/passthrough BreakState)
unchanged.


/// Indic Conjunct Break state for GB9c
const InCBState = enum(u2) {
none = 0,
after_consonant = 1, // Seen Consonant + [Extend|Linker]*
after_linker = 2, // Seen Consonant + [Extend|Linker]* + Linker + [Extend|Linker]*
_unused = 3,
};

/// Indic Conjunct Break property (Unicode 15.1, UAX #29)
const InCBProperty = enum(u2) {
none = 0,
consonant = 1,
linker = 2,
extend = 3,
};

/// Returns the Indic_Conjunct_Break property for a codepoint.
/// Based on Unicode 15.1 IndicConjunctBreak.txt
fn indicConjunctBreakProperty(cp: u21) InCBProperty {
// InCB=Linker: Virama/Halant characters
if (isInCBLinker(cp)) return .linker;
// InCB=Consonant: Indic script consonants
if (isInCBConsonant(cp)) return .consonant;
// InCB=Extend: Characters with GBC=Extend or GBC=ZWJ
const gbc = table.get(cp).grapheme_boundary_class;
if (gbc == .extend or gbc == .zwj) return .extend;
return .none;
}

/// InCB=Linker: Virama/Halant characters from various Indic scripts
pub fn isInCBLinker(cp: u21) bool {
return switch (cp) {
0x094D, // DEVANAGARI SIGN VIRAMA
0x09CD, // BENGALI SIGN VIRAMA
0x0A4D, // GURMUKHI SIGN VIRAMA
0x0ACD, // GUJARATI SIGN VIRAMA
0x0B4D, // ORIYA SIGN VIRAMA
0x0BCD, // TAMIL SIGN VIRAMA
0x0C4D, // TELUGU SIGN VIRAMA
0x0CCD, // KANNADA SIGN VIRAMA
0x0D4D, // MALAYALAM SIGN VIRAMA
0x0DCA, // SINHALA SIGN AL-LAKUNA
0x1B44, // BALINESE ADEG ADEG
0xA806, // SYLOTI NAGRI SIGN HASANTA
0xA8C4, // SAURASHTRA SIGN VIRAMA
0xA953, // REJANG VIRAMA
0xA9C0, // JAVANESE PANGKON
0x11046, // BRAHMI VIRAMA
0x1107F, // BRAHMI NUMBER JOINER
0x110B9, // KAITHI SIGN VIRAMA
0x11133,
0x11134, // CHAKMA VIRAMA..CHAKMA MAAYYAA
0x111C0, // SHARADA SIGN VIRAMA
0x11235, // KHOJKI SIGN VIRAMA
0x112EA, // KHUDAWADI SIGN VIRAMA
0x1134D, // GRANTHA SIGN VIRAMA
0x11442, // NEWA SIGN VIRAMA
0x114C2, // TIRHUTA SIGN VIRAMA
0x115BF, // SIDDHAM SIGN VIRAMA
0x1163F, // MODI SIGN VIRAMA
0x116B6, // TAKRI SIGN VIRAMA
0x1172B, // AHOM SIGN KILLER
0x11839, // DOGRA SIGN VIRAMA
0x1193D,
0x1193E, // DIVES AKURU SIGN HALANTA..VIRAMA
0x119E0, // NANDINAGARI SIGN VIRAMA
0x11A34, // ZANABAZAR SQUARE SIGN VIRAMA
0x11A47, // ZANABAZAR SQUARE SUBJOINER
0x11A99, // SOYOMBO SUBJOINER
0x11C3F, // BHAIKSUKI SIGN VIRAMA
0x11D44,
0x11D45, // MASARAM GONDI SIGN HALANTA..VIRAMA
0x11D97, // GUNJALA GONDI VIRAMA
=> true,
else => false,
};
}

/// InCB=Consonant: Indic script consonants (base letters that form conjuncts)
fn isInCBConsonant(cp: u21) bool {
return switch (cp) {
// Devanagari consonants
0x0915...0x0939,
0x0958...0x095F,
0x0979...0x097F,
// Bengali consonants
0x0995...0x09A8,
0x09AA...0x09B0,
0x09B2,
0x09B6...0x09B9,
0x09DC...0x09DD,
0x09DF,
// Gurmukhi consonants
0x0A15...0x0A28,
0x0A2A...0x0A30,
0x0A32...0x0A33,
0x0A35...0x0A36,
0x0A38...0x0A39,
0x0A59...0x0A5C,
0x0A5E,
// Gujarati consonants
0x0A95...0x0AA8,
0x0AAA...0x0AB0,
0x0AB2...0x0AB3,
0x0AB5...0x0AB9,
// Oriya consonants
0x0B15...0x0B28,
0x0B2A...0x0B30,
0x0B32...0x0B33,
0x0B35...0x0B39,
0x0B5C...0x0B5D,
0x0B5F,
// Tamil consonants
0x0B95,
0x0B99...0x0B9A,
0x0B9C,
0x0B9E...0x0B9F,
0x0BA3...0x0BA4,
0x0BA8...0x0BAA,
0x0BAE...0x0BB9,
// Telugu consonants
0x0C15...0x0C28,
0x0C2A...0x0C39,
0x0C58...0x0C5A,
// Kannada consonants
0x0C95...0x0CA8,
0x0CAA...0x0CB3,
0x0CB5...0x0CB9,
0x0CDE,
// Malayalam consonants
0x0D15...0x0D3A,
0x0D4E,
// Sinhala consonants
0x0D9A...0x0DC6,
=> true,
else => false,
};
Comment on lines +141 to +199

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue | 🟠 Major

🧩 Analysis chain

🌐 Web query:

Unicode 15.1 IndicConjunctBreak.txt InCB=Consonant character ranges

💡 Result:

The InCB=Consonant ranges for Unicode 15.1 are listed in the Unicode UCD file IndicConjunctBreak.txt (Indic_Conjunct_Break). See the full list in the UCD; example excerpt (Unicode 15.1 / DerivedCoreProperties build showing InCB=Consonant ranges): 0915..0939; 0958..095F; 0978..097F; 0995..09A8; 09AA..09B0; 09B2; 09B6..09B9; 09DC..09DD; 09DF; 09F0..09F1; 0A95..0AA8; 0AAA..0AB0; 0AB2..0AB3; … (complete canonical list is in IndicConjunctBreak.txt). [1][2]

References

  • Derived/core-data excerpt showing InCB=Consonant ranges (Unicode 15.1). [1]
  • PropertyValueAliases / Indic_Conjunct_Break (InCB) listing (Unicode 15.1). [2]

If you want, I can fetch and paste the complete, verbatim InCB=Consonant range list from IndicConjunctBreak.txt.


🌐 Web query:

complete list Unicode 15.1 IndicConjunctBreak.txt InCB=Consonant all ranges

💡 Result:

Do you want me to fetch the official Unicode 15.1 extracted file (IndicConjunctBreak.txt) and return the complete InCB=Consonant ranges (raw file or formatted list)?


🌐 Web query:

fetch Unicode 15.1 official IndicConjunctBreak.txt file complete InCB=Consonant ranges formatted list

💡 Result:

I can fetch that and return the full list of InCB=Consonant ranges from the official Unicode 15.1 IndicConjunctBreak.txt. Do you want them (a) as hexadecimal ranges (e.g. 0A95..0A96), (b) as code points with names, or (c) both?


🌐 Web query:

Unicode 15.1 IndicConjunctBreak.txt InCB=Consonant ranges hexadecimal format complete list

💡 Result:

I can fetch and return the complete hexadecimal ranges for InCB=Consonant from Unicode 15.1 — do you want the full list now? (I’ll pull the authoritative UCD file and paste the ranges.) (unicode.org)

Citations:


🌐 Web query:

site:unicode.org IndicConjunctBreak.txt 15.1 Consonant

💡 Result:

The Unicode UCD includes an Indic_Conjunct_Break (InCB) data file named IndicConjunctBreak.txt (e.g. in the Public/15.1.0/ucd/auxiliary/ location). One of the InCB property values is "Consonant". [1][2]

Sources:

  • Property value aliases showing InCB; Consonant. [1]
  • Unicode UAX/common references listing the auxiliary file IndicConjunctBreak.txt (version 15.1). [2]

🌐 Web query:

unicode.org/Public/15.1.0/ucd/auxiliary/IndicConjunctBreak.txt

💡 Result:

The specific file URL you requested (https://www.unicode.org/Public/15.1.0/ucd/auxiliary/IndicConjunctBreak.txt) returns 404 — it is not available at that path. ()

For the Unicode definition and derived data for Indic_Conjunct_Break, see the Unicode Character Database / TR44 description of the property and the Grapheme Cluster (segmentation) test material (these point to the current UCD auxiliary data in the Public/UCD/… area). (unicode.org)

If you want, I can:

Citations:


🌐 Web query:

Unicode 15.1 Indic Conjunct Break Consonant code points comprehensive list hexadecimal

💡 Result:

The complete set of code points with Indic_Conjunct_Break=Consonant for Unicode 15.1 is published in the Unicode Character Database (PropList / Derived property files). Download the UCD for 15.1 and inspect the Indic_Conjunct_Break (InCB) section (PropList/DerivedCoreProperties). [1][2]

If you want the property as ranges (hex) ready to use, see the extracted list (Indic_Conjunct_Break=Consonant) published in platform updates that mirror the UCD; example excerpt (ranges are hexadecimal):

0915..0939
0958..095F
0978..097F
0995..09A8
09AA..09B0
09B2
09B6..09B9
09DC..09DD
09DF
09F0..09F1
0A95..0AA8
0AAA..0AB0
0AB2..0AB3
0AB5..0AB9
0ABD..0AC2
0AC9
0B95..0BB9
0C15..0C39
0C58..0C59
0C95..0CB9
0D15..0D39
0D60..0D61
0D66..0D6F
0D7A..0D7F
0D85..0D96
0D9A..0DB1
0DC0..0DC6
0E01..0E30
0E31
0E32..0E33
0E40..0E46
0E4F..0E5B
0E81..0E82
0E84
0E87..0E8A
0EA1..0EA3
0EA5
0EA7..0EA8
0EAA..0EAB
0EAD..0EB0
0EB2..0EB3
0EBD
0EC0..0EC4
0EC6
0EDC..0EDF
0F40..0F47
0F49..0F6C
0F88..0F8C
10300..10323
1032D
10350..10375
A8F2..A8F7
1B05..1B33
1B45..1B4B
1B83..1BA0
1BAE..1BBA
AA60..AA76
AAB0..AAB0
AAB2..AAB4
AAB7..AAB8
AABE..AABF
AAC0..AAC0
A9E0..A9E4

(For the authoritative, full, and canonical listing including exact singletons and all ranges, download the Unicode 15.1 UCD files — PropList / DerivedCoreProperties — or inspect the Indic_Conjunct_Break section in the UCD zip.) [1][2]

Sources:

  • Unicode 15.1 UCD (UCD.zip / PropList / Derived files). [1]
  • Platform extraction showing Indic_Conjunct_Break=Consonant ranges (example: GCC / OpenJDK updates that mirror UCD). [2]

Add missing InCB=Consonant ranges from Unicode 15.1.

The function is incomplete. Unicode 15.1 IndicConjunctBreak.txt includes consonant ranges for Thai, Lao, Tibetan, Old Italic, Gothic, Balinese, Sundanese, Myanmar Extended-B, Tai Viet, Javanese, and others that are not covered. Additionally, within the Indic scripts already listed, several ranges are missing or incorrect:

  • Devanagari: missing 0x0978
  • Bengali: missing 0x09F0..0x09F1
  • Gujarati: missing 0x0ABD..0x0AC2, 0x0AC9
  • Malayalam: missing 0x0D60..0x0D61, 0x0D66..0x0D6F, 0x0D7A..0x0D7F
  • Sinhala: missing 0x0D85..0x0D96

Review the complete Unicode 15.1 IndicConjunctBreak.txt file to add all missing ranges and correct existing ones.

🤖 Prompt for AI Agents
In `@src/string/immutable/grapheme.zig` around lines 138 - 163, The
isInCBConsonant function is missing and has incorrect Unicode ranges per Unicode
15.1; update the function (isInCBConsonant in src/string/immutable/grapheme.zig)
to match IndicConjunctBreak.txt by adding all missing script ranges (e.g., add
Devanagari 0x0978; Bengali 0x09F0..0x09F1; Gujarati 0x0ABD..0x0AC2 and 0x0AC9;
Malayalam 0x0D60..0x0D61, 0x0D66..0x0D6F, 0x0D7A..0x0D7F; Sinhala
0x0D85..0x0D96) and include the additional scripts listed in Unicode 15.1 (Thai,
Lao, Tibetan, Old Italic, Gothic, Balinese, Sundanese, Myanmar Extended-B, Tai
Viet, Javanese, etc.) by adding their ranges to the switch arms in
isInCBConsonant so the switch exactly mirrors IndicConjunctBreak.txt.

}

const Precompute = struct {
const Key = packed struct(u10) {
state: BreakState,
state: PrecomputedState,

gbc1: GraphemeBoundaryClass,
gbc2: GraphemeBoundaryClass,
Expand All @@ -32,7 +213,7 @@ const Precompute = struct {

const Value = packed struct(u3) {
result: bool,
state: BreakState,
state: PrecomputedState,
};

const data = precompute: {
Expand All @@ -43,7 +224,7 @@ const Precompute = struct {
for (0..std.math.maxInt(u2) + 1) |state_init| {
for (info.fields) |field1| {
for (info.fields) |field2| {
var state: BreakState = @bitCast(@as(u2, @intCast(state_init)));
var state: PrecomputedState = @bitCast(@as(u2, @intCast(state_init)));
const key: Key = .{
.gbc1 = @field(GraphemeBoundaryClass, field1.name),
.gbc2 = @field(GraphemeBoundaryClass, field2.name),
Expand All @@ -62,7 +243,7 @@ const Precompute = struct {
fn graphemeBreakClass(
gbc1: GraphemeBoundaryClass,
gbc2: GraphemeBoundaryClass,
state: *BreakState,
state: *PrecomputedState,
) bool {
// GB11: Emoji Extend* ZWJ x Emoji
if (!state.extended_pictographic and gbc1 == .extended_pictographic) {
Expand Down
9 changes: 8 additions & 1 deletion src/string/immutable/visible.zig
Original file line number Diff line number Diff line change
Expand Up @@ -792,7 +792,8 @@ pub const visible = struct {
zwj: bool = false,
vs15: bool = false,
vs16: bool = false,
_pad: u5 = 0,
has_incb_linker: bool = false, // Indic conjunct with virama
_pad: u4 = 0,
};

const GraphemeState = struct {
Expand Down Expand Up @@ -825,6 +826,7 @@ pub const visible = struct {
.regional_indicator = isRegionalIndicator(cp),
.skin_tone = isSkinToneModifier(cp),
.zwj = cp == 0x200D,
.has_incb_linker = grapheme.isInCBLinker(@truncate(cp)),
};
}

Expand All @@ -837,6 +839,7 @@ pub const visible = struct {
self.s.zwj = self.s.zwj or (cp == 0x200D);
self.s.vs15 = self.s.vs15 or (cp == 0xFE0E);
self.s.vs16 = self.s.vs16 or (cp == 0xFE0F);
self.s.has_incb_linker = self.s.has_incb_linker or grapheme.isInCBLinker(@truncate(cp));

if (!isZeroWidthCodepointType(u32, cp)) {
self.s.non_emoji_width +|= visibleCodepointWidthType(u32, cp, ambiguousAsWide);
Expand Down Expand Up @@ -868,6 +871,10 @@ pub const visible = struct {
return 1;
}

// Indic conjuncts (virama+consonant sequences) render as a single glyph
// Use the width of the first base character only
if (s.has_incb_linker) return s.base_width;

return s.non_emoji_width;
}

Expand Down
11 changes: 11 additions & 0 deletions test/js/bun/util/stringWidth.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -480,6 +480,17 @@ describe("stringWidth extended", () => {
expect(Bun.stringWidth("कि")).toBe(1); // Ka + vowel sign i (combining)
});

test("Devanagari conjuncts (GB9c)", () => {
// Conjuncts formed with virama+ZWJ should render as a single glyph (width 1)
// क्‍ष = Ka + Virama + ZWJ + Ssa (kṣa conjunct)
expect(Bun.stringWidth("क्\u200Dष")).toBe(1);
// Conjuncts without ZWJ (virama alone joins consonants)
// क्ष = Ka + Virama + Ssa
expect(Bun.stringWidth("क्ष")).toBe(1);
// Multiple conjuncts
expect(Bun.stringWidth("क्\u200Dष क्\u200Dष")).toBe(3); // conjunct + space + conjunct
});

test("Thai with combining marks", () => {
expect(Bun.stringWidth("ก")).toBe(1); // Ko kai
expect(Bun.stringWidth("ก็")).toBe(1); // With maitaikhu
Expand Down
Loading