openclaw/scripts/lib/test-source-term-matcher.mts
Peter Steinberger 181405f477
fix(ci): planner ownership census times out on loaded runners (#160586)
* perf(ci): scan wide import-graph frontiers in one pass

The PR-exempt ownership census plans a PR that edits all 906 PR-exempt
test files. Its consumer check sends ~2,700 terms through the wide native
reader path over ~42.5k tracked files. #160531 and #160517 each added a
match-before-parse filter, so that path spawned one reader child to match
every file and a second to re-read and parse the 2,136 matches.
readImportGraphEdges already requests matchingOnly, so drop the extra
pass: one child matches, parses only uncached matches, and still leaves
nonmatches uncached.

Build the Aho-Corasick automaton as a dense Int32Array over the UTF-16
code units the terms use, completing failure rows breadth-first, so the
scan takes one table load per code unit instead of Map lookups and
failure walks. Files with no match skip projecting every requested term.
Matching semantics, reference checks, and ordering are unchanged.

For the real 906-file request, reader output is byte-identical. Reader
CPU fell from ~12.1s over two passes to ~5.9s. Inside the Vitest worker
the census case's reader cost fell from 16.5s to 8.8s, and the case from
39.1s to 30.4s on a comparably loaded host. Timeout, assertions, and file
coverage are unchanged.

* perf(ci): bound term-matcher rows to the ASCII alphabet

Search terms come from tracked paths, which have no character-diversity
bound. One dense column per distinct UTF-16 unit could therefore grow the
transition table by nodes x alphabet on Unicode-heavy frontiers.

Keep dense Int32Array rows only for the ASCII units the terms use, at most
129 columns per node, and route other term units through sparse per-node
edges with the classic failure walk. Size the table once to its node upper
bound so it never reallocates; only rows of real nodes are written. Fold
isReferenceCharacter into an equivalent character-class test, which is
exhaustively equal for every UTF-16 unit and NaN.

The matcher contract fixture gains terms whose failure links cross dense
and sparse edges. A 24,000-case randomized differential against main's
matcher and a String.includes oracle shows no differences, and reader
output for the 906-file census request stays byte-identical.
2026-09-28 15:45:40 -07:00

133 lines
4.6 KiB
TypeScript

type SourceTerm = { text: string; index: number; reference: boolean };
// Out-of-range reads are NaN, which String.fromCharCode maps to NUL.
const isReferenceCharacter = (code: number) => /[\w.@+/-]/u.test(String.fromCharCode(code));
/** Match literal terms and complete source tokens without rescanning once per term. */
export function createSourceTermMatcher(terms: readonly string[]) {
// String.includes and the source-token contract operate on UTF-16 code units.
// Dense rows cover only the ASCII units terms use, so each node costs at most
// 129 columns however diverse the paths are. Other units take sparse edges.
const symbols = new Uint8Array(128);
let width = 1;
let maxNodes = 1;
for (const text of terms) {
maxNodes += text.length;
for (let index = 0; index < text.length; index++) {
const code = text.charCodeAt(index);
if (code < 128) {
symbols[code] ||= width++;
}
}
}
// Each code unit adds at most one node; rows past the last node stay untouched.
const transitions = new Int32Array(maxNodes * width);
const wide: (Map<number, number> | undefined)[] = [];
const outputs: SourceTerm[][] = [[]];
const unique = new Map<string, SourceTerm>();
let referenceCount = 0;
const requested = terms.map((text) => {
const existing = unique.get(text);
if (existing) {
return existing;
}
const term = {
text,
index: unique.size,
reference: /^[A-Za-z0-9_.@+/-]{4,}$/u.test(text),
};
unique.set(text, term);
referenceCount += Number(term.reference);
let node = 0;
for (let index = 0; index < text.length; index++) {
const code = text.charCodeAt(index);
if (code >= 128) {
const edges = (wide[node] ??= new Map());
const next = edges.get(code) ?? outputs.push([]) - 1;
edges.set(code, next);
node = next;
continue;
}
const slot = node * width + symbols[code]!;
transitions[slot] ||= outputs.push([]) - 1;
node = transitions[slot]!;
}
if (text.length > 0) {
outputs[node]!.push(term);
}
return term;
});
const failures = new Int32Array(outputs.length);
const wideStep = (from: number, code: number): number => {
for (let node = from; ; node = failures[node]!) {
const next = wide[node]?.get(code);
if (next !== undefined || node === 0) {
return next ?? 0;
}
}
};
// Breadth-first order completes each failure row before its children use it,
// so the scan below takes one table load per ASCII code unit.
const queue = [0];
for (const node of queue) {
for (let symbol = 1; symbol < width; symbol++) {
const slot = node * width + symbol;
const fallback = node === 0 ? 0 : transitions[failures[node]! * width + symbol]!;
const child = transitions[slot]!;
if (child === 0) {
transitions[slot] = fallback;
continue;
}
failures[child] = fallback;
outputs[child]!.push(...outputs[fallback]!);
queue.push(child);
}
for (const [code, child] of wide[node] ?? []) {
failures[child] = node === 0 ? 0 : wideStep(failures[node]!, code);
outputs[child]!.push(...outputs[failures[child]!]!);
queue.push(child);
}
}
const empty = unique.get("");
return (source: string) => {
const matches = new Uint8Array(unique.size);
const references = new Uint8Array(unique.size);
let matched = 0;
let referenced = 0;
if (empty) {
matches[empty.index] = 1;
matched++;
}
let node = 0;
for (let index = 0; index < source.length; index++) {
if (matched === unique.size && referenced === referenceCount) {
break;
}
const code = source.charCodeAt(index);
node = code < 128 ? transitions[node * width + symbols[code]!]! : wideStep(node, code);
for (const term of outputs[node]!) {
if (!matches[term.index]) {
matches[term.index] = 1;
matched++;
}
if (
term.reference &&
!references[term.index] &&
!isReferenceCharacter(source.charCodeAt(index - term.text.length)) &&
!isReferenceCharacter(source.charCodeAt(index + 1))
) {
references[term.index] = 1;
referenced++;
}
}
}
// Most scanned files match nothing; skip projecting every requested term.
if (matched === 0) {
return { matches: [], references: [] };
}
return {
matches: requested.filter((term) => matches[term.index]).map((term) => term.text),
references: requested.filter((term) => references[term.index]).map((term) => term.text),
};
};
}