fix(text): keep grapheme lookups on cluster boundaries under Bun (#162749)

Fix Bun message chunking and terminal truncation selecting the preceding grapheme when JSC containing() probes an emoji high surrogate. Route all seven lookups through normalization-core, normalize numeric indexes before the offset, and retain the new eager dependency in the native PR wrapper inventory. Upstream engine fix: oven-sh/WebKit#753.

Node 24 and fork Bun each pass 586 tests (one skipped) across the requested 39 files. Node matches native containing() for all 1,236 corpus lookups. Changed-file checks, import-cycle validation, wrapper closure checks, and independent P2 reviews pass.

The remaining CI cron copy-fault and Discord skills-watcher teardown failures reproduce on clean main and use the authorized native pre-existing-failure exception. The original wrapper inventory regression was fixed.
This commit is contained in:
Peter Steinberger 2026-10-01 09:24:26 -07:00 • committed by GitHub
parent a12736320c
commit 6dbca39af1
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
7 changed files with 110 additions and 7 deletions

View file

@ -177,6 +177,7 @@ config or state. Gateway startup logs include the decision and its reason.
## Known limitations
- **Text boundaries:** OpenClaw works around a [JSC segment lookup bug](https://github.com/oven-sh/WebKit/pull/753) that can include the preceding cluster when a lookup starts on an emoji's high surrogate. Message chunking and terminal cells preserve the intended grapheme boundaries on Bun without runtime configuration changes.
- **Desktop WebSockets:** OpenClaw uses the installed `ws` transport for desktop observers and paired-node desktop/portal streams. Bun 1.4.2's built-in `ws` server adapter lacks pause/resume and the Duplex stream bridge; the installed transport preserves backpressure, payload limits, and cleanup when a desktop disconnects.
- **Lifecycle scripts:** Bun blocks dependency lifecycle scripts unless explicitly trusted with `bun pm trust`.
- **Package scripts:** Some scripts hardcode pnpm, so `bun run` still invokes pnpm internally.

View file

@ -0,0 +1,74 @@
import { describe, expect, it } from "vitest";
import { containingSegment, findGraphemeChunkEnd } from "./grapheme.js";
const inputs = [
"",
"abc",
"😀".repeat(50),
"a😀b",
"a🇯🇵🇦🇹b",
"a👨‍👩‍👧‍👦b",
"a👍🏽👍🏻b",
"Hi. 👍 Bye.",
"\ud83d",
"\ude00",
"\ud83dabc\ud83d",
"\ude00abc\ude00",
"\ud83d😀\ude00",
"a\ud83db\ude00c",
"\ud83d\ud83d\ude00\ude00",
];
describe("containingSegment", () => {
it.each(["grapheme", "word", "sentence"] as const)(
"matches %s iteration at every UTF-16 index in either query order",
(granularity) => {
for (const text of inputs) {
const segments = new Intl.Segmenter("en", { granularity }).segment(text);
const expected = Array.from(segments);
const indices = Array.from({ length: text.length + 2 }, (_, index) => index - 1);
for (const order of [indices, indices.toReversed()]) {
for (const index of order) {
const actual = containingSegment(segments, text, index);
expect(actual).toEqual(
expected.find(
(segment) =>
segment.index <= index && index < segment.index + segment.segment.length,
),
);
if (!process.versions.bun) {
expect(actual).toEqual(segments.containing(index));
}
}
}
}
},
);
it.each([
Number.NEGATIVE_INFINITY,
-1.5,
-0.5,
0.5,
0.9999999999999999,
1.5,
2.5,
Number.NaN,
Number.POSITIVE_INFINITY,
])("preserves numeric index coercion for %s", (index) => {
const text = "😀😀";
const segments = new Intl.Segmenter("en", { granularity: "grapheme" }).segment(text);
const offset = Number.isNaN(index) ? 0 : Math.trunc(index);
expect(containingSegment(segments, text, index)).toEqual(
Array.from(segments).find(
(segment) => segment.index <= offset && offset < segment.index + segment.segment.length,
),
);
});
});
describe("findGraphemeChunkEnd", () => {
it.each(["😀", "🇯🇵", "👨‍👩‍👧‍👦", "👍🏽"])("uses the full budget before another %s cluster", (cluster) => {
expect(findGraphemeChunkEnd(cluster.repeat(3), 0, cluster.length * 2)).toBe(cluster.length * 2);
});
});

View file

@ -8,6 +8,23 @@ function getGraphemeSegmenter(): Intl.Segmenter {
return graphemeSegmenter;
}
/** Uses the same source text that created segments. */
export function containingSegment(
segments: Intl.Segments,
text: string,
index: number,
): Intl.SegmentData | undefined {
// JSC can include the preceding segment for high-surrogate queries in grapheme,
// word, and sentence mode. Query the low half of a valid pair instead.
// https://github.com/oven-sh/WebKit/pull/753
// Integer-normalize first, as containing() does, so the offset cannot round past a boundary.
const position = Number.isNaN(index) ? 0 : Math.trunc(index);
const codePoint = text.codePointAt(position);
return segments.containing(
codePoint !== undefined && codePoint > 0xffff ? position + 1 : position,
);
}
/**
* Chooses a whole-grapheme cut within the hard budget, honoring a usable preference.
* If no whole grapheme fits, allowPartial permits a surrogate-safe progress cut;
@ -33,9 +50,12 @@ export function findGraphemeChunkEnd(
}
const segments = getGraphemeSegmenter().segment(text);
let end = segments.containing(preferred)?.index ?? preferred;
let end = containingSegment(segments, text, preferred)?.index ?? preferred;
if (end <= start && preferred < hardEnd) {
end = hardEnd === text.length ? hardEnd : (segments.containing(hardEnd)?.index ?? hardEnd);
end =
hardEnd === text.length
? hardEnd
: (containingSegment(segments, text, hardEnd)?.index ?? hardEnd);
}
return end > start
? end
@ -49,7 +69,7 @@ export function firstGraphemeClusterLength(text: string): number {
if (!text) {
return 0;
}
return getGraphemeSegmenter().segment(text).containing(0)?.segment.length ?? 0;
return containingSegment(getGraphemeSegmenter().segment(text), text, 0)?.segment.length ?? 0;
}
const WHITESPACE_GRAPHEME_RE = /^\s+$/u;
@ -66,7 +86,7 @@ export function skipWhitespaceGraphemes(
const segments = getGraphemeSegmenter().segment(text);
let cursor = start;
for (let count = 0; count < maxGraphemes && cursor < text.length; count += 1) {
const cluster = segments.containing(cursor);
const cluster = containingSegment(segments, text, cursor);
if (!cluster || cluster.index !== cursor || !WHITESPACE_GRAPHEME_RE.test(cluster.segment)) {
break;
}
@ -83,7 +103,7 @@ export function trimEndWhitespaceGraphemes(text: string, end = text.length): str
const segments = getGraphemeSegmenter().segment(text);
let cursor = end;
while (cursor > 0) {
const cluster = segments.containing(cursor - 1);
const cluster = containingSegment(segments, text, cursor - 1);
if (
!cluster ||
cluster.index + cluster.segment.length > cursor ||

View file

@ -1,3 +1,4 @@
import { containingSegment } from "@openclaw/normalization-core/grapheme";
import stringWidth from "string-width";
import {
ANSI_COMPAT_CONTROL_SEQUENCE_PATTERN,
@ -232,7 +233,7 @@ export function truncateToVisibleWidth(input: string, maxWidth: number): string
position >= current.index + current.segment.length
) {
// SAFETY: the end sentinel returns above; other probes resolve inside this segment.
current = segments.containing(position) as Intl.SegmentData;
current = containingSegment(segments, segment, position) as Intl.SegmentData;
candidateWidth =
current.index === 0
? 0

View file

@ -81,6 +81,7 @@ packages/normalization-core/src/code-points.ts
packages/normalization-core/src/error-coercion.ts
packages/normalization-core/src/expect.ts
packages/normalization-core/src/format.ts
packages/normalization-core/src/grapheme.ts
packages/normalization-core/src/home-dir.ts
packages/normalization-core/src/index.ts
packages/normalization-core/src/json-coercion.ts

View file

@ -30,6 +30,11 @@ describe("formatTextCell raw output bound", () => {
it.each([
["zero-width overflow", "\u200b".repeat(32), "\u200b".repeat(14) + "… "],
["exact zero-width raw boundary", "\u200b".repeat(14), "\u200b".repeat(14) + " "],
[
"astral character at the raw boundary",
"\u200b".repeat(14) + "😀",
"\u200b".repeat(14) + "… ",
],
["one oversized combining cluster", "e" + "\u0301".repeat(16), "… "],
["one oversized ZWJ cluster", "👩" + "\u200d👩".repeat(8), "… "],
["prefix before an oversized cluster", "Ae" + "\u0301".repeat(16), "A…"],

View file

@ -1,3 +1,4 @@
import { containingSegment } from "@openclaw/normalization-core/grapheme";
import * as terminalAnsi from "../../packages/terminal-core/src/ansi.js";
const graphemeSegmenter = new Intl.Segmenter(undefined, { granularity: "grapheme" });
@ -19,7 +20,7 @@ export const shortenText = (value: string, maxLen: number) => {
export function formatTextCell(text: string, width: number): string {
// Eight UTF-16 units per column allow ordinary accents/emoji; reserve width for padding.
// Whole-cluster raw bounds also catch invisible runs and oversized single graphemes.
const overflow = graphemeSegmenter.segment(text).containing(width * 7);
const overflow = containingSegment(graphemeSegmenter.segment(text), text, width * 7);
const bounded = overflow ? `${text.slice(0, overflow.index)}…` : text;
const boundedWidth = terminalAnsi.visibleWidth(bounded);
const fitted =