mirror of
https://github.com/openclaw/openclaw.git
synced 2026-10-04 10:10:01 +00:00
* perf(ci): scan wide import-graph frontiers in one pass The PR-exempt ownership census plans a PR that edits all 906 PR-exempt test files. Its consumer check sends ~2,700 terms through the wide native reader path over ~42.5k tracked files. #160531 and #160517 each added a match-before-parse filter, so that path spawned one reader child to match every file and a second to re-read and parse the 2,136 matches. readImportGraphEdges already requests matchingOnly, so drop the extra pass: one child matches, parses only uncached matches, and still leaves nonmatches uncached. Build the Aho-Corasick automaton as a dense Int32Array over the UTF-16 code units the terms use, completing failure rows breadth-first, so the scan takes one table load per code unit instead of Map lookups and failure walks. Files with no match skip projecting every requested term. Matching semantics, reference checks, and ordering are unchanged. For the real 906-file request, reader output is byte-identical. Reader CPU fell from ~12.1s over two passes to ~5.9s. Inside the Vitest worker the census case's reader cost fell from 16.5s to 8.8s, and the case from 39.1s to 30.4s on a comparably loaded host. Timeout, assertions, and file coverage are unchanged. * perf(ci): bound term-matcher rows to the ASCII alphabet Search terms come from tracked paths, which have no character-diversity bound. One dense column per distinct UTF-16 unit could therefore grow the transition table by nodes x alphabet on Unicode-heavy frontiers. Keep dense Int32Array rows only for the ASCII units the terms use, at most 129 columns per node, and route other term units through sparse per-node edges with the classic failure walk. Size the table once to its node upper bound so it never reallocates; only rows of real nodes are written. Fold isReferenceCharacter into an equivalent character-class test, which is exhaustively equal for every UTF-16 unit and NaN. The matcher contract fixture gains terms whose failure links cross dense and sparse edges. A 24,000-case randomized differential against main's matcher and a String.includes oracle shows no differences, and reader output for the 906-file census request stays byte-identical.
133 lines
4.6 KiB
TypeScript
133 lines
4.6 KiB
TypeScript
type SourceTerm = { text: string; index: number; reference: boolean };
|
|
|
|
// Out-of-range reads are NaN, which String.fromCharCode maps to NUL.
|
|
const isReferenceCharacter = (code: number) => /[\w.@+/-]/u.test(String.fromCharCode(code));
|
|
|
|
/** Match literal terms and complete source tokens without rescanning once per term. */
|
|
export function createSourceTermMatcher(terms: readonly string[]) {
|
|
// String.includes and the source-token contract operate on UTF-16 code units.
|
|
// Dense rows cover only the ASCII units terms use, so each node costs at most
|
|
// 129 columns however diverse the paths are. Other units take sparse edges.
|
|
const symbols = new Uint8Array(128);
|
|
let width = 1;
|
|
let maxNodes = 1;
|
|
for (const text of terms) {
|
|
maxNodes += text.length;
|
|
for (let index = 0; index < text.length; index++) {
|
|
const code = text.charCodeAt(index);
|
|
if (code < 128) {
|
|
symbols[code] ||= width++;
|
|
}
|
|
}
|
|
}
|
|
// Each code unit adds at most one node; rows past the last node stay untouched.
|
|
const transitions = new Int32Array(maxNodes * width);
|
|
const wide: (Map<number, number> | undefined)[] = [];
|
|
const outputs: SourceTerm[][] = [[]];
|
|
const unique = new Map<string, SourceTerm>();
|
|
let referenceCount = 0;
|
|
const requested = terms.map((text) => {
|
|
const existing = unique.get(text);
|
|
if (existing) {
|
|
return existing;
|
|
}
|
|
const term = {
|
|
text,
|
|
index: unique.size,
|
|
reference: /^[A-Za-z0-9_.@+/-]{4,}$/u.test(text),
|
|
};
|
|
unique.set(text, term);
|
|
referenceCount += Number(term.reference);
|
|
let node = 0;
|
|
for (let index = 0; index < text.length; index++) {
|
|
const code = text.charCodeAt(index);
|
|
if (code >= 128) {
|
|
const edges = (wide[node] ??= new Map());
|
|
const next = edges.get(code) ?? outputs.push([]) - 1;
|
|
edges.set(code, next);
|
|
node = next;
|
|
continue;
|
|
}
|
|
const slot = node * width + symbols[code]!;
|
|
transitions[slot] ||= outputs.push([]) - 1;
|
|
node = transitions[slot]!;
|
|
}
|
|
if (text.length > 0) {
|
|
outputs[node]!.push(term);
|
|
}
|
|
return term;
|
|
});
|
|
const failures = new Int32Array(outputs.length);
|
|
const wideStep = (from: number, code: number): number => {
|
|
for (let node = from; ; node = failures[node]!) {
|
|
const next = wide[node]?.get(code);
|
|
if (next !== undefined || node === 0) {
|
|
return next ?? 0;
|
|
}
|
|
}
|
|
};
|
|
// Breadth-first order completes each failure row before its children use it,
|
|
// so the scan below takes one table load per ASCII code unit.
|
|
const queue = [0];
|
|
for (const node of queue) {
|
|
for (let symbol = 1; symbol < width; symbol++) {
|
|
const slot = node * width + symbol;
|
|
const fallback = node === 0 ? 0 : transitions[failures[node]! * width + symbol]!;
|
|
const child = transitions[slot]!;
|
|
if (child === 0) {
|
|
transitions[slot] = fallback;
|
|
continue;
|
|
}
|
|
failures[child] = fallback;
|
|
outputs[child]!.push(...outputs[fallback]!);
|
|
queue.push(child);
|
|
}
|
|
for (const [code, child] of wide[node] ?? []) {
|
|
failures[child] = node === 0 ? 0 : wideStep(failures[node]!, code);
|
|
outputs[child]!.push(...outputs[failures[child]!]!);
|
|
queue.push(child);
|
|
}
|
|
}
|
|
const empty = unique.get("");
|
|
return (source: string) => {
|
|
const matches = new Uint8Array(unique.size);
|
|
const references = new Uint8Array(unique.size);
|
|
let matched = 0;
|
|
let referenced = 0;
|
|
if (empty) {
|
|
matches[empty.index] = 1;
|
|
matched++;
|
|
}
|
|
let node = 0;
|
|
for (let index = 0; index < source.length; index++) {
|
|
if (matched === unique.size && referenced === referenceCount) {
|
|
break;
|
|
}
|
|
const code = source.charCodeAt(index);
|
|
node = code < 128 ? transitions[node * width + symbols[code]!]! : wideStep(node, code);
|
|
for (const term of outputs[node]!) {
|
|
if (!matches[term.index]) {
|
|
matches[term.index] = 1;
|
|
matched++;
|
|
}
|
|
if (
|
|
term.reference &&
|
|
!references[term.index] &&
|
|
!isReferenceCharacter(source.charCodeAt(index - term.text.length)) &&
|
|
!isReferenceCharacter(source.charCodeAt(index + 1))
|
|
) {
|
|
references[term.index] = 1;
|
|
referenced++;
|
|
}
|
|
}
|
|
}
|
|
// Most scanned files match nothing; skip projecting every requested term.
|
|
if (matched === 0) {
|
|
return { matches: [], references: [] };
|
|
}
|
|
return {
|
|
matches: requested.filter((term) => matches[term.index]).map((term) => term.text),
|
|
references: requested.filter((term) => references[term.index]).map((term) => term.text),
|
|
};
|
|
};
|
|
}
|