mirror of
https://github.com/openclaw/openclaw.git
synced 2026-10-03 01:29:56 +00:00
fix(text): keep grapheme lookups on cluster boundaries under Bun (#162749)
Fix Bun message chunking and terminal truncation selecting the preceding grapheme when JSC containing() probes an emoji high surrogate. Route all seven lookups through normalization-core, normalize numeric indexes before the offset, and retain the new eager dependency in the native PR wrapper inventory. Upstream engine fix: oven-sh/WebKit#753. Node 24 and fork Bun each pass 586 tests (one skipped) across the requested 39 files. Node matches native containing() for all 1,236 corpus lookups. Changed-file checks, import-cycle validation, wrapper closure checks, and independent P2 reviews pass. The remaining CI cron copy-fault and Discord skills-watcher teardown failures reproduce on clean main and use the authorized native pre-existing-failure exception. The original wrapper inventory regression was fixed.
This commit is contained in:
parent
a12736320c
commit
6dbca39af1
7 changed files with 110 additions and 7 deletions
|
|
@ -177,6 +177,7 @@ config or state. Gateway startup logs include the decision and its reason.
|
|||
|
||||
## Known limitations
|
||||
|
||||
- **Text boundaries:** OpenClaw works around a [JSC segment lookup bug](https://github.com/oven-sh/WebKit/pull/753) that can include the preceding cluster when a lookup starts on an emoji's high surrogate. Message chunking and terminal cells preserve the intended grapheme boundaries on Bun without runtime configuration changes.
|
||||
- **Desktop WebSockets:** OpenClaw uses the installed `ws` transport for desktop observers and paired-node desktop/portal streams. Bun 1.4.2's built-in `ws` server adapter lacks pause/resume and the Duplex stream bridge; the installed transport preserves backpressure, payload limits, and cleanup when a desktop disconnects.
|
||||
- **Lifecycle scripts:** Bun blocks dependency lifecycle scripts unless explicitly trusted with `bun pm trust`.
|
||||
- **Package scripts:** Some scripts hardcode pnpm, so `bun run` still invokes pnpm internally.
|
||||
|
|
|
|||
74
packages/normalization-core/src/grapheme.test.ts
Normal file
74
packages/normalization-core/src/grapheme.test.ts
Normal file
|
|
@ -0,0 +1,74 @@
|
|||
import { describe, expect, it } from "vitest";
|
||||
import { containingSegment, findGraphemeChunkEnd } from "./grapheme.js";
|
||||
|
||||
const inputs = [
|
||||
"",
|
||||
"abc",
|
||||
"😀".repeat(50),
|
||||
"a😀b",
|
||||
"a🇯🇵🇦🇹b",
|
||||
"a👨👩👧👦b",
|
||||
"a👍🏽👍🏻b",
|
||||
"Hi. 👍 Bye.",
|
||||
"\ud83d",
|
||||
"\ude00",
|
||||
"\ud83dabc\ud83d",
|
||||
"\ude00abc\ude00",
|
||||
"\ud83d😀\ude00",
|
||||
"a\ud83db\ude00c",
|
||||
"\ud83d\ud83d\ude00\ude00",
|
||||
];
|
||||
|
||||
describe("containingSegment", () => {
|
||||
it.each(["grapheme", "word", "sentence"] as const)(
|
||||
"matches %s iteration at every UTF-16 index in either query order",
|
||||
(granularity) => {
|
||||
for (const text of inputs) {
|
||||
const segments = new Intl.Segmenter("en", { granularity }).segment(text);
|
||||
const expected = Array.from(segments);
|
||||
const indices = Array.from({ length: text.length + 2 }, (_, index) => index - 1);
|
||||
for (const order of [indices, indices.toReversed()]) {
|
||||
for (const index of order) {
|
||||
const actual = containingSegment(segments, text, index);
|
||||
expect(actual).toEqual(
|
||||
expected.find(
|
||||
(segment) =>
|
||||
segment.index <= index && index < segment.index + segment.segment.length,
|
||||
),
|
||||
);
|
||||
if (!process.versions.bun) {
|
||||
expect(actual).toEqual(segments.containing(index));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
);
|
||||
|
||||
it.each([
|
||||
Number.NEGATIVE_INFINITY,
|
||||
-1.5,
|
||||
-0.5,
|
||||
0.5,
|
||||
0.9999999999999999,
|
||||
1.5,
|
||||
2.5,
|
||||
Number.NaN,
|
||||
Number.POSITIVE_INFINITY,
|
||||
])("preserves numeric index coercion for %s", (index) => {
|
||||
const text = "😀😀";
|
||||
const segments = new Intl.Segmenter("en", { granularity: "grapheme" }).segment(text);
|
||||
const offset = Number.isNaN(index) ? 0 : Math.trunc(index);
|
||||
expect(containingSegment(segments, text, index)).toEqual(
|
||||
Array.from(segments).find(
|
||||
(segment) => segment.index <= offset && offset < segment.index + segment.segment.length,
|
||||
),
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
describe("findGraphemeChunkEnd", () => {
|
||||
it.each(["😀", "🇯🇵", "👨👩👧👦", "👍🏽"])("uses the full budget before another %s cluster", (cluster) => {
|
||||
expect(findGraphemeChunkEnd(cluster.repeat(3), 0, cluster.length * 2)).toBe(cluster.length * 2);
|
||||
});
|
||||
});
|
||||
|
|
@ -8,6 +8,23 @@ function getGraphemeSegmenter(): Intl.Segmenter {
|
|||
return graphemeSegmenter;
|
||||
}
|
||||
|
||||
/** Uses the same source text that created segments. */
|
||||
export function containingSegment(
|
||||
segments: Intl.Segments,
|
||||
text: string,
|
||||
index: number,
|
||||
): Intl.SegmentData | undefined {
|
||||
// JSC can include the preceding segment for high-surrogate queries in grapheme,
|
||||
// word, and sentence mode. Query the low half of a valid pair instead.
|
||||
// https://github.com/oven-sh/WebKit/pull/753
|
||||
// Integer-normalize first, as containing() does, so the offset cannot round past a boundary.
|
||||
const position = Number.isNaN(index) ? 0 : Math.trunc(index);
|
||||
const codePoint = text.codePointAt(position);
|
||||
return segments.containing(
|
||||
codePoint !== undefined && codePoint > 0xffff ? position + 1 : position,
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Chooses a whole-grapheme cut within the hard budget, honoring a usable preference.
|
||||
* If no whole grapheme fits, allowPartial permits a surrogate-safe progress cut;
|
||||
|
|
@ -33,9 +50,12 @@ export function findGraphemeChunkEnd(
|
|||
}
|
||||
|
||||
const segments = getGraphemeSegmenter().segment(text);
|
||||
let end = segments.containing(preferred)?.index ?? preferred;
|
||||
let end = containingSegment(segments, text, preferred)?.index ?? preferred;
|
||||
if (end <= start && preferred < hardEnd) {
|
||||
end = hardEnd === text.length ? hardEnd : (segments.containing(hardEnd)?.index ?? hardEnd);
|
||||
end =
|
||||
hardEnd === text.length
|
||||
? hardEnd
|
||||
: (containingSegment(segments, text, hardEnd)?.index ?? hardEnd);
|
||||
}
|
||||
return end > start
|
||||
? end
|
||||
|
|
@ -49,7 +69,7 @@ export function firstGraphemeClusterLength(text: string): number {
|
|||
if (!text) {
|
||||
return 0;
|
||||
}
|
||||
return getGraphemeSegmenter().segment(text).containing(0)?.segment.length ?? 0;
|
||||
return containingSegment(getGraphemeSegmenter().segment(text), text, 0)?.segment.length ?? 0;
|
||||
}
|
||||
|
||||
const WHITESPACE_GRAPHEME_RE = /^\s+$/u;
|
||||
|
|
@ -66,7 +86,7 @@ export function skipWhitespaceGraphemes(
|
|||
const segments = getGraphemeSegmenter().segment(text);
|
||||
let cursor = start;
|
||||
for (let count = 0; count < maxGraphemes && cursor < text.length; count += 1) {
|
||||
const cluster = segments.containing(cursor);
|
||||
const cluster = containingSegment(segments, text, cursor);
|
||||
if (!cluster || cluster.index !== cursor || !WHITESPACE_GRAPHEME_RE.test(cluster.segment)) {
|
||||
break;
|
||||
}
|
||||
|
|
@ -83,7 +103,7 @@ export function trimEndWhitespaceGraphemes(text: string, end = text.length): str
|
|||
const segments = getGraphemeSegmenter().segment(text);
|
||||
let cursor = end;
|
||||
while (cursor > 0) {
|
||||
const cluster = segments.containing(cursor - 1);
|
||||
const cluster = containingSegment(segments, text, cursor - 1);
|
||||
if (
|
||||
!cluster ||
|
||||
cluster.index + cluster.segment.length > cursor ||
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
import { containingSegment } from "@openclaw/normalization-core/grapheme";
|
||||
import stringWidth from "string-width";
|
||||
import {
|
||||
ANSI_COMPAT_CONTROL_SEQUENCE_PATTERN,
|
||||
|
|
@ -232,7 +233,7 @@ export function truncateToVisibleWidth(input: string, maxWidth: number): string
|
|||
position >= current.index + current.segment.length
|
||||
) {
|
||||
// SAFETY: the end sentinel returns above; other probes resolve inside this segment.
|
||||
current = segments.containing(position) as Intl.SegmentData;
|
||||
current = containingSegment(segments, segment, position) as Intl.SegmentData;
|
||||
candidateWidth =
|
||||
current.index === 0
|
||||
? 0
|
||||
|
|
|
|||
|
|
@ -81,6 +81,7 @@ packages/normalization-core/src/code-points.ts
|
|||
packages/normalization-core/src/error-coercion.ts
|
||||
packages/normalization-core/src/expect.ts
|
||||
packages/normalization-core/src/format.ts
|
||||
packages/normalization-core/src/grapheme.ts
|
||||
packages/normalization-core/src/home-dir.ts
|
||||
packages/normalization-core/src/index.ts
|
||||
packages/normalization-core/src/json-coercion.ts
|
||||
|
|
|
|||
|
|
@ -30,6 +30,11 @@ describe("formatTextCell raw output bound", () => {
|
|||
it.each([
|
||||
["zero-width overflow", "\u200b".repeat(32), "\u200b".repeat(14) + "… "],
|
||||
["exact zero-width raw boundary", "\u200b".repeat(14), "\u200b".repeat(14) + " "],
|
||||
[
|
||||
"astral character at the raw boundary",
|
||||
"\u200b".repeat(14) + "😀",
|
||||
"\u200b".repeat(14) + "… ",
|
||||
],
|
||||
["one oversized combining cluster", "e" + "\u0301".repeat(16), "… "],
|
||||
["one oversized ZWJ cluster", "👩" + "\u200d👩".repeat(8), "… "],
|
||||
["prefix before an oversized cluster", "Ae" + "\u0301".repeat(16), "A…"],
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
import { containingSegment } from "@openclaw/normalization-core/grapheme";
|
||||
import * as terminalAnsi from "../../packages/terminal-core/src/ansi.js";
|
||||
|
||||
const graphemeSegmenter = new Intl.Segmenter(undefined, { granularity: "grapheme" });
|
||||
|
|
@ -19,7 +20,7 @@ export const shortenText = (value: string, maxLen: number) => {
|
|||
export function formatTextCell(text: string, width: number): string {
|
||||
// Eight UTF-16 units per column allow ordinary accents/emoji; reserve width for padding.
|
||||
// Whole-cluster raw bounds also catch invisible runs and oversized single graphemes.
|
||||
const overflow = graphemeSegmenter.segment(text).containing(width * 7);
|
||||
const overflow = containingSegment(graphemeSegmenter.segment(text), text, width * 7);
|
||||
const bounded = overflow ? `${text.slice(0, overflow.index)}…` : text;
|
||||
const boundedWidth = terminalAnsi.visibleWidth(bounded);
|
||||
const fitted =
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue