diff --git a/docs/runtime/utils.mdx b/docs/runtime/utils.mdx index c268b11418bc..4e570d4b37f0 100644 --- a/docs/runtime/utils.mdx +++ b/docs/runtime/utils.mdx @@ -322,12 +322,16 @@ Example usage: Bun.stringWidth("hello"); // => 5 Bun.stringWidth("\u001b[31mhello\u001b[0m"); // => 5 Bun.stringWidth("\u001b[31mhello\u001b[0m", { countAnsiEscapeCodes: true }); // => 12 +Bun.stringWidth("๐Ÿ‘จโ€๐Ÿ‘ฉโ€๐Ÿ‘งโ€๐Ÿ‘ฆ"); // => 2 +Bun.stringWidth("๐Ÿ‘จโ€๐Ÿ‘ฉโ€๐Ÿ‘งโ€๐Ÿ‘ฆ", { perCodePoint: true }); // => 8 ``` Use it to align text in a terminal or to check whether a string contains ANSI escape codes. The API matches the "string-width" npm package, so existing code can be ported to Bun and vice versa. +By default, an emoji sequence (or any other grapheme cluster) is measured once, like `string-width` does. Pass `perCodePoint: true` to measure every code point individually instead, which is how Node.js measures strings when aligning `console.table` and `util.inspect` output. In that mode each emoji in a ZWJ sequence is counted, so the family emoji above measures 8 instead of 2. Use it when your output has to line up with output produced by Node.js. This option is specific to Bun: `string-width` has no equivalent, so code that uses it does not port back unchanged. + [In this benchmark](https://github.com/oven-sh/bun/blob/5147c0ba7379d85d4d1ed0714b84d6544af917eb/bench/snippets/string-width.mjs#L13), `Bun.stringWidth` is ~6,756x faster than the `string-width` npm package for input larger than about 500 characters. Big thanks to [sindresorhus](https://github.com/sindresorhus) for their work on `string-width`. ```ts @@ -440,6 +444,17 @@ namespace Bun { * @default true */ ambiguousIsNarrow?: boolean; + /** + * If `true`, measure every Unicode code point individually (East Asian + * Width plus emoji presentation, the algorithm Node.js uses for + * `console.table` and `util.inspect` alignment), so each member of an + * emoji ZWJ sequence is counted: `"๐Ÿ‘จโ€๐Ÿ‘ฉโ€๐Ÿ‘งโ€๐Ÿ‘ฆ"` measures 8. If `false`, + * emoji sequences and other grapheme clusters count once: `"๐Ÿ‘จโ€๐Ÿ‘ฉโ€๐Ÿ‘งโ€๐Ÿ‘ฆ"` + * measures 2. + * + * @default false + */ + perCodePoint?: boolean; }, ): number; } diff --git a/test/js/bun/util/stringWidth.test.ts b/test/js/bun/util/stringWidth.test.ts index bde985ac6c31..d2114e6a9150 100644 --- a/test/js/bun/util/stringWidth.test.ts +++ b/test/js/bun/util/stringWidth.test.ts @@ -1285,8 +1285,10 @@ test("options lookup ignores Object.prototype pollution", () => { try { (Object.prototype as any).countAnsiEscapeCodes = true; (Object.prototype as any).ambiguousIsNarrow = false; + (Object.prototype as any).perCodePoint = true; expect(Bun.stringWidth("\x1b[31mhello\x1b[39m", {})).toBe(5); expect(Bun.stringWidth("โ˜…", {})).toBe(1); + expect(Bun.stringWidth("๐Ÿ‘ฉโ€๐Ÿ‘ฉโ€๐Ÿ‘งโ€๐Ÿ‘ฆ", {})).toBe(2); // An explicit own property still wins. expect(Bun.stringWidth("\x1b[31mhello\x1b[39m", { countAnsiEscapeCodes: true })).toBe(13); // Inherited properties from a non-Object.prototype prototype are honored, @@ -1295,9 +1297,71 @@ test("options lookup ignores Object.prototype pollution", () => { } finally { delete (Object.prototype as any).countAnsiEscapeCodes; delete (Object.prototype as any).ambiguousIsNarrow; + delete (Object.prototype as any).perCodePoint; } }); +// perCodePoint: true measures each code point on its own (node's GetColumnWidth in +// src/node_i18n.cc, which node's console.table and util.inspect align with) instead +// of once per grapheme cluster. These are the values documented in +// docs/runtime/utils.mdx and bun.d.ts; every perCodePoint value below equals what +// node v26's internalBinding("icu").getStringWidth returns for the same string. +describe("perCodePoint", () => { + const family = "๐Ÿ‘จโ€๐Ÿ‘ฉโ€๐Ÿ‘งโ€๐Ÿ‘ฆ"; // four emoji joined by three ZWJs + const widths = (input: string) => ({ + default: Bun.stringWidth(input), + perCodePoint: Bun.stringWidth(input, { perCodePoint: true }), + }); + + test("is off by default", () => { + expect(Bun.stringWidth(family)).toBe(2); + expect(Bun.stringWidth(family, {})).toBe(2); + expect(Bun.stringWidth(family, { perCodePoint: false })).toBe(2); + }); + + test("counts every member of an emoji sequence", () => { + expect(widths(family)).toEqual({ default: 2, perCodePoint: 8 }); + expect(widths("๐Ÿ‡ฏ๐Ÿ‡ต")).toEqual({ default: 2, perCodePoint: 4 }); // two regional indicators + expect(widths("๐Ÿ‘๐Ÿฝ")).toEqual({ default: 2, perCodePoint: 4 }); // skin tone modifier is East Asian Wide + expect(widths("๐Ÿณ๏ธโ€๐ŸŒˆ")).toEqual({ default: 2, perCodePoint: 3 }); // ๐Ÿณ (1) + VS16 (0) + ZWJ (0) + ๐ŸŒˆ (2) + expect(widths("1\uFE0F\u20E3")).toEqual({ default: 2, perCodePoint: 1 }); // keycap: digit + two marks + }); + + test("wide, combining and control characters keep their widths", () => { + expect(widths("hello")).toEqual({ default: 5, perCodePoint: 5 }); + expect(widths("ๆ—ฅๆœฌ")).toEqual({ default: 4, perCodePoint: 4 }); + expect(widths("x\u0300")).toEqual({ default: 1, perCodePoint: 1 }); + expect(widths("a\u0007b\u0085c")).toEqual({ default: 3, perCodePoint: 3 }); // Latin-1 C0 and C1 controls + expect(widths("")).toEqual({ default: 0, perCodePoint: 0 }); + }); + + test("soft hyphen takes a column, as in node", () => { + expect(widths("a\u00ADb")).toEqual({ default: 2, perCodePoint: 3 }); + }); + + test("composes with countAnsiEscapeCodes", () => { + const latin1 = "\x1b[31mhello\x1b[0m"; + expect(Bun.stringWidth(latin1, { perCodePoint: true })).toBe(5); + expect(Bun.stringWidth(latin1, { perCodePoint: true, countAnsiEscapeCodes: true })).toBe(12); + + const c1 = "\x9b31mhello\x9b0m"; + expect(Bun.stringWidth(c1, { perCodePoint: true })).toBe(5); + expect(Bun.stringWidth(c1, { perCodePoint: true, countAnsiEscapeCodes: true })).toBe(10); + + const utf16 = `\x1b[31m${family}\x1b[0m`; + expect(Bun.stringWidth(utf16, { perCodePoint: true })).toBe(8); + expect(Bun.stringWidth(utf16, { perCodePoint: true, countAnsiEscapeCodes: true })).toBe(15); + }); + + test("composes with ambiguousIsNarrow", () => { + expect(Bun.stringWidth("โ˜…โ˜†", { perCodePoint: true })).toBe(2); + expect(Bun.stringWidth("โ˜…โ˜†", { perCodePoint: true, ambiguousIsNarrow: true })).toBe(2); + expect(Bun.stringWidth("โ˜…โ˜†", { perCodePoint: true, ambiguousIsNarrow: false })).toBe(4); + // U+00AD is East Asian Ambiguous, so this also covers the Latin-1 path. + expect(Bun.stringWidth("a\u00ADb", { perCodePoint: true, ambiguousIsNarrow: false })).toBe(4); + }); +}); + // The Latin-1 ANSI-excluding width walks 16-64 byte SIMD chunks counting // printable bytes until the next escape introducer, then hands it to the // shared recognizer (ANSI::consumeANSI). These tests pin its behavior at and