From 6dbca39af1eb52fbfd664f56fa37d4d4f00d21d7 Mon Sep 17 00:00:00 2001 From: Peter Steinberger Date: Thu, 1 Oct 2026 09:24:26 -0700 Subject: [PATCH] fix(text): keep grapheme lookups on cluster boundaries under Bun (#162749) Fix Bun message chunking and terminal truncation selecting the preceding grapheme when JSC containing() probes an emoji high surrogate. Route all seven lookups through normalization-core, normalize numeric indexes before the offset, and retain the new eager dependency in the native PR wrapper inventory. Upstream engine fix: oven-sh/WebKit#753. Node 24 and fork Bun each pass 586 tests (one skipped) across the requested 39 files. Node matches native containing() for all 1,236 corpus lookups. Changed-file checks, import-cycle validation, wrapper closure checks, and independent P2 reviews pass. The remaining CI cron copy-fault and Discord skills-watcher teardown failures reproduce on clean main and use the authorized native pre-existing-failure exception. The original wrapper inventory regression was fixed. --- docs/install/bun-compatibility.md | 1 + .../normalization-core/src/grapheme.test.ts | 74 +++++++++++++++++++ packages/normalization-core/src/grapheme.ts | 30 ++++++-- packages/terminal-core/src/ansi.ts | 3 +- scripts/pr-lib/wrapper-components.txt | 1 + src/commands/text-format.test.ts | 5 ++ src/commands/text-format.ts | 3 +- 7 files changed, 110 insertions(+), 7 deletions(-) create mode 100644 packages/normalization-core/src/grapheme.test.ts diff --git a/docs/install/bun-compatibility.md b/docs/install/bun-compatibility.md index 919363853e66..a81a5a7712ea 100644 --- a/docs/install/bun-compatibility.md +++ b/docs/install/bun-compatibility.md @@ -177,6 +177,7 @@ config or state. Gateway startup logs include the decision and its reason. ## Known limitations +- **Text boundaries:** OpenClaw works around a [JSC segment lookup bug](https://github.com/oven-sh/WebKit/pull/753) that can include the preceding cluster when a lookup starts on an emoji's high surrogate. Message chunking and terminal cells preserve the intended grapheme boundaries on Bun without runtime configuration changes. - **Desktop WebSockets:** OpenClaw uses the installed `ws` transport for desktop observers and paired-node desktop/portal streams. Bun 1.4.2's built-in `ws` server adapter lacks pause/resume and the Duplex stream bridge; the installed transport preserves backpressure, payload limits, and cleanup when a desktop disconnects. - **Lifecycle scripts:** Bun blocks dependency lifecycle scripts unless explicitly trusted with `bun pm trust`. - **Package scripts:** Some scripts hardcode pnpm, so `bun run` still invokes pnpm internally. diff --git a/packages/normalization-core/src/grapheme.test.ts b/packages/normalization-core/src/grapheme.test.ts new file mode 100644 index 000000000000..1b7e2d62b05e --- /dev/null +++ b/packages/normalization-core/src/grapheme.test.ts @@ -0,0 +1,74 @@ +import { describe, expect, it } from "vitest"; +import { containingSegment, findGraphemeChunkEnd } from "./grapheme.js"; + +const inputs = [ + "", + "abc", + "๐Ÿ˜€".repeat(50), + "a๐Ÿ˜€b", + "a๐Ÿ‡ฏ๐Ÿ‡ต๐Ÿ‡ฆ๐Ÿ‡นb", + "a๐Ÿ‘จโ€๐Ÿ‘ฉโ€๐Ÿ‘งโ€๐Ÿ‘ฆb", + "a๐Ÿ‘๐Ÿฝ๐Ÿ‘๐Ÿปb", + "Hi. ๐Ÿ‘ Bye.", + "\ud83d", + "\ude00", + "\ud83dabc\ud83d", + "\ude00abc\ude00", + "\ud83d๐Ÿ˜€\ude00", + "a\ud83db\ude00c", + "\ud83d\ud83d\ude00\ude00", +]; + +describe("containingSegment", () => { + it.each(["grapheme", "word", "sentence"] as const)( + "matches %s iteration at every UTF-16 index in either query order", + (granularity) => { + for (const text of inputs) { + const segments = new Intl.Segmenter("en", { granularity }).segment(text); + const expected = Array.from(segments); + const indices = Array.from({ length: text.length + 2 }, (_, index) => index - 1); + for (const order of [indices, indices.toReversed()]) { + for (const index of order) { + const actual = containingSegment(segments, text, index); + expect(actual).toEqual( + expected.find( + (segment) => + segment.index <= index && index < segment.index + segment.segment.length, + ), + ); + if (!process.versions.bun) { + expect(actual).toEqual(segments.containing(index)); + } + } + } + } + }, + ); + + it.each([ + Number.NEGATIVE_INFINITY, + -1.5, + -0.5, + 0.5, + 0.9999999999999999, + 1.5, + 2.5, + Number.NaN, + Number.POSITIVE_INFINITY, + ])("preserves numeric index coercion for %s", (index) => { + const text = "๐Ÿ˜€๐Ÿ˜€"; + const segments = new Intl.Segmenter("en", { granularity: "grapheme" }).segment(text); + const offset = Number.isNaN(index) ? 0 : Math.trunc(index); + expect(containingSegment(segments, text, index)).toEqual( + Array.from(segments).find( + (segment) => segment.index <= offset && offset < segment.index + segment.segment.length, + ), + ); + }); +}); + +describe("findGraphemeChunkEnd", () => { + it.each(["๐Ÿ˜€", "๐Ÿ‡ฏ๐Ÿ‡ต", "๐Ÿ‘จโ€๐Ÿ‘ฉโ€๐Ÿ‘งโ€๐Ÿ‘ฆ", "๐Ÿ‘๐Ÿฝ"])("uses the full budget before another %s cluster", (cluster) => { + expect(findGraphemeChunkEnd(cluster.repeat(3), 0, cluster.length * 2)).toBe(cluster.length * 2); + }); +}); diff --git a/packages/normalization-core/src/grapheme.ts b/packages/normalization-core/src/grapheme.ts index 6295936599b3..0dda39947bd7 100644 --- a/packages/normalization-core/src/grapheme.ts +++ b/packages/normalization-core/src/grapheme.ts @@ -8,6 +8,23 @@ function getGraphemeSegmenter(): Intl.Segmenter { return graphemeSegmenter; } +/** Uses the same source text that created segments. */ +export function containingSegment( + segments: Intl.Segments, + text: string, + index: number, +): Intl.SegmentData | undefined { + // JSC can include the preceding segment for high-surrogate queries in grapheme, + // word, and sentence mode. Query the low half of a valid pair instead. + // https://github.com/oven-sh/WebKit/pull/753 + // Integer-normalize first, as containing() does, so the offset cannot round past a boundary. + const position = Number.isNaN(index) ? 0 : Math.trunc(index); + const codePoint = text.codePointAt(position); + return segments.containing( + codePoint !== undefined && codePoint > 0xffff ? position + 1 : position, + ); +} + /** * Chooses a whole-grapheme cut within the hard budget, honoring a usable preference. * If no whole grapheme fits, allowPartial permits a surrogate-safe progress cut; @@ -33,9 +50,12 @@ export function findGraphemeChunkEnd( } const segments = getGraphemeSegmenter().segment(text); - let end = segments.containing(preferred)?.index ?? preferred; + let end = containingSegment(segments, text, preferred)?.index ?? preferred; if (end <= start && preferred < hardEnd) { - end = hardEnd === text.length ? hardEnd : (segments.containing(hardEnd)?.index ?? hardEnd); + end = + hardEnd === text.length + ? hardEnd + : (containingSegment(segments, text, hardEnd)?.index ?? hardEnd); } return end > start ? end @@ -49,7 +69,7 @@ export function firstGraphemeClusterLength(text: string): number { if (!text) { return 0; } - return getGraphemeSegmenter().segment(text).containing(0)?.segment.length ?? 0; + return containingSegment(getGraphemeSegmenter().segment(text), text, 0)?.segment.length ?? 0; } const WHITESPACE_GRAPHEME_RE = /^\s+$/u; @@ -66,7 +86,7 @@ export function skipWhitespaceGraphemes( const segments = getGraphemeSegmenter().segment(text); let cursor = start; for (let count = 0; count < maxGraphemes && cursor < text.length; count += 1) { - const cluster = segments.containing(cursor); + const cluster = containingSegment(segments, text, cursor); if (!cluster || cluster.index !== cursor || !WHITESPACE_GRAPHEME_RE.test(cluster.segment)) { break; } @@ -83,7 +103,7 @@ export function trimEndWhitespaceGraphemes(text: string, end = text.length): str const segments = getGraphemeSegmenter().segment(text); let cursor = end; while (cursor > 0) { - const cluster = segments.containing(cursor - 1); + const cluster = containingSegment(segments, text, cursor - 1); if ( !cluster || cluster.index + cluster.segment.length > cursor || diff --git a/packages/terminal-core/src/ansi.ts b/packages/terminal-core/src/ansi.ts index b441dbe78d1c..55bab06ab7ed 100644 --- a/packages/terminal-core/src/ansi.ts +++ b/packages/terminal-core/src/ansi.ts @@ -1,3 +1,4 @@ +import { containingSegment } from "@openclaw/normalization-core/grapheme"; import stringWidth from "string-width"; import { ANSI_COMPAT_CONTROL_SEQUENCE_PATTERN, @@ -232,7 +233,7 @@ export function truncateToVisibleWidth(input: string, maxWidth: number): string position >= current.index + current.segment.length ) { // SAFETY: the end sentinel returns above; other probes resolve inside this segment. - current = segments.containing(position) as Intl.SegmentData; + current = containingSegment(segments, segment, position) as Intl.SegmentData; candidateWidth = current.index === 0 ? 0 diff --git a/scripts/pr-lib/wrapper-components.txt b/scripts/pr-lib/wrapper-components.txt index 6184d770e67d..3d80204275df 100644 --- a/scripts/pr-lib/wrapper-components.txt +++ b/scripts/pr-lib/wrapper-components.txt @@ -81,6 +81,7 @@ packages/normalization-core/src/code-points.ts packages/normalization-core/src/error-coercion.ts packages/normalization-core/src/expect.ts packages/normalization-core/src/format.ts +packages/normalization-core/src/grapheme.ts packages/normalization-core/src/home-dir.ts packages/normalization-core/src/index.ts packages/normalization-core/src/json-coercion.ts diff --git a/src/commands/text-format.test.ts b/src/commands/text-format.test.ts index 83f1563ca7c4..6ea0c7e35c33 100644 --- a/src/commands/text-format.test.ts +++ b/src/commands/text-format.test.ts @@ -30,6 +30,11 @@ describe("formatTextCell raw output bound", () => { it.each([ ["zero-width overflow", "\u200b".repeat(32), "\u200b".repeat(14) + "โ€ฆ "], ["exact zero-width raw boundary", "\u200b".repeat(14), "\u200b".repeat(14) + " "], + [ + "astral character at the raw boundary", + "\u200b".repeat(14) + "๐Ÿ˜€", + "\u200b".repeat(14) + "โ€ฆ ", + ], ["one oversized combining cluster", "e" + "\u0301".repeat(16), "โ€ฆ "], ["one oversized ZWJ cluster", "๐Ÿ‘ฉ" + "\u200d๐Ÿ‘ฉ".repeat(8), "โ€ฆ "], ["prefix before an oversized cluster", "Ae" + "\u0301".repeat(16), "Aโ€ฆ"], diff --git a/src/commands/text-format.ts b/src/commands/text-format.ts index 8155b363d810..f5844fb4ae90 100644 --- a/src/commands/text-format.ts +++ b/src/commands/text-format.ts @@ -1,3 +1,4 @@ +import { containingSegment } from "@openclaw/normalization-core/grapheme"; import * as terminalAnsi from "../../packages/terminal-core/src/ansi.js"; const graphemeSegmenter = new Intl.Segmenter(undefined, { granularity: "grapheme" }); @@ -19,7 +20,7 @@ export const shortenText = (value: string, maxLen: number) => { export function formatTextCell(text: string, width: number): string { // Eight UTF-16 units per column allow ordinary accents/emoji; reserve width for padding. // Whole-cluster raw bounds also catch invisible runs and oversized single graphemes. - const overflow = graphemeSegmenter.segment(text).containing(width * 7); + const overflow = containingSegment(graphemeSegmenter.segment(text), text, width * 7); const bounded = overflow ? `${text.slice(0, overflow.index)}โ€ฆ` : text; const boundedWidth = terminalAnsi.visibleWidth(bounded); const fitted =