openclaw/scripts/mantis/telegram-proof-plan.ts
Martin Cleary a960844a9c
feat(qa): run isolated behavioral proof for inline reviews (#138953)
* feat(telegram): add isolated Test Server proof workflow

Add maintainer-only exact-head admission, durable at-most-once QA lease consumption, isolated candidate execution, and normalized trusted Telegram Test Server observations.

Co-authored-by: brokemac79 <255583030+brokemac79@users.noreply.github.com>

* feat(proof): bind named Web UI and canonical Telegram QA evidence

Reuse the existing formatting QA recipe, preserve isolated exact-candidate execution and produce the consumer request-bound receipt. Keep named smoke scenarios distinct and protect stationary harness ancestry.

Co-authored-by: brokemac79 <255583030+brokemac79@users.noreply.github.com>

* fix(proof): complete isolated QA execution and bounded failure capture

Reuse canonical ephemeral device pairing and QA RPC scopes, preserve strict startup probes and recorder locks, and retain bounded wrong-text attempts without Telegram delivery.

Co-authored-by: brokemac79 <255583030+brokemac79@users.noreply.github.com>

* fix(mantis): revoke proof forwarding and enforce lease roles

* fix(mantis): close proof producer CI gates

* test(mantis): consolidate related proof suites within CI budget

* test(mantis): preserve fast QA ownership when grouping integration cases

* fix(mantis): accept exact branch-qualified workflow paths

* fix(mantis): align live admission workflow path checks

* feat(mantis): collect selected proof inside the originating review

* fix(mantis): bound proof storage and preserve failure evidence

Reuse verified storage across request-bound candidates, reserve backing capacity, retain sanitized rejection evidence, and repair cleanup and observation finalization. Scoped checks and dirty review pass; full Gateway and sandboxed browser runtime proof remain required before publication.

* fix(mantis): retain bridge identity before startup

* fix(mantis): keep candidate config readable under private umask

* fix(qa): use verified rootless networking and join candidate shutdown

* fix(qa): repair proof tooling checks and deterministic recorder fixture

---------

Co-authored-by: brokemac79 <255583030+brokemac79@users.noreply.github.com>
2026-09-07 14:09:37 +01:00

74 lines
2.8 KiB
TypeScript

import { createHash } from "node:crypto";
import { z } from "zod";
const text = z.string().min(1).max(4096);
const offset = z.number().int().min(0).max(60_000);
// These are recorder data, never the broader userbot command/config DSL.
export const telegramProofPlanSchema = z
.strictObject({
claim: z.string().min(1).max(1024),
actions: z
.array(
z.discriminatedUnion("type", [
z.strictObject({ type: z.literal("send"), atMs: offset, text }),
z.strictObject({
type: z.literal("click"),
atMs: offset,
messageText: text,
buttonText: z.string().min(1).max(256),
timeoutMs: z.number().int().min(1).max(10_000),
}),
]),
)
.min(1)
.max(8),
modelReplies: z.array(text).max(8),
settings: z.strictObject({
streaming: z.enum(["off", "partial", "block"]),
nativeCommands: z.boolean(),
}),
maxDurationMs: z.number().int().min(1000).max(90_000),
expectations: z.array(z.string().min(1).max(1024)).min(1).max(8),
})
.superRefine((plan, context) => {
if (plan.actions[0]?.type !== "send") {
context.addIssue({ code: "custom", message: "A proof begins with a tester send." });
}
for (const [index, action] of plan.actions.entries()) {
const previous = plan.actions[index - 1];
const completeBy = action.atMs + (action.type === "click" ? action.timeoutMs : 0);
if (completeBy >= plan.maxDurationMs || (previous && action.atMs < previous.atMs)) {
context.addIssue({
code: "custom",
message: "Actions must be ordered inside the recording budget.",
});
}
}
});
export type TelegramProofPlan = z.infer<typeof telegramProofPlanSchema>;
// Shared wire canonicalization: recursive lexicographic object keys, array order
// unchanged, JSON.stringify string/number encoding, UTF-8, no trailing newline.
function canonicalPlan(value: unknown): string {
if (Array.isArray(value)) {
return `[${value.map(canonicalPlan).join(",")}]`;
}
if (value !== null && typeof value === "object") {
return `{${Object.entries(value)
.toSorted(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0))
.map(([key, item]) => `${JSON.stringify(key)}:${canonicalPlan(item)}`)
.join(",")}}`;
}
return JSON.stringify(value);
}
export function parseTelegramProofPlan(encoded: string, expectedDigest: string) {
if (Buffer.byteLength(encoded, "utf8") > 48 * 1024 || !/^[a-f0-9]{64}$/.test(expectedDigest)) {
throw new Error("Invalid proof plan envelope.");
}
const plan = telegramProofPlanSchema.parse(JSON.parse(encoded));
const digest = createHash("sha256").update(canonicalPlan(plan), "utf8").digest("hex");
if (digest !== expectedDigest) {
throw new Error("Proof plan digest mismatch.");
}
return plan;
}