mirror of
https://github.com/openclaw/openclaw.git
synced 2026-10-03 09:39:25 +00:00
fix(qa): use bounded instruction injection evidence (#158634)
This commit is contained in:
parent
a8272f099f
commit
ca5f05c87e
3 changed files with 191 additions and 122 deletions
180
extensions/qa-lab/src/scenario-catalog.prompt-evidence.test.ts
Normal file
180
extensions/qa-lab/src/scenario-catalog.prompt-evidence.test.ts
Normal file
|
|
@ -0,0 +1,180 @@
|
|||
import path from "node:path";
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { readQaScenarioById, type QaScenarioFlow } from "./scenario-catalog.js";
|
||||
import { runLoadedScenarioFlow } from "./scenario-flow-runner.test-support.js";
|
||||
|
||||
const scenarioId = "instruction-profile-artifact-followthrough-live";
|
||||
const sessionKey = "agent:qa:instruction-profile-artifact:test";
|
||||
const currentObservation = {
|
||||
egress: "responses-sdk",
|
||||
payloadVariant: "initial",
|
||||
promptSource: "input.developer",
|
||||
expectedChars: 4096,
|
||||
observedChars: 4096,
|
||||
matchesAssembledPrompt: true,
|
||||
};
|
||||
const currentEvent = {
|
||||
type: "provider.prompt.observed",
|
||||
runId: "current-run",
|
||||
data: currentObservation,
|
||||
};
|
||||
|
||||
async function runPromptEvidence(
|
||||
params: {
|
||||
events?: unknown[];
|
||||
report?: Record<string, unknown>;
|
||||
reportSessionKey?: string;
|
||||
} = {},
|
||||
) {
|
||||
const scenario = readQaScenarioById(scenarioId);
|
||||
const actions = scenario.execution.flow?.steps[0]?.actions;
|
||||
if (!actions) {
|
||||
throw new Error("instruction profile scenario has no actions");
|
||||
}
|
||||
const exportIndex = actions.findIndex(
|
||||
(action) =>
|
||||
typeof action === "object" &&
|
||||
action !== null &&
|
||||
"call" in action &&
|
||||
action.call === "runQaCli",
|
||||
);
|
||||
const assertionIndex = actions.findIndex(
|
||||
(action, index) =>
|
||||
index > exportIndex &&
|
||||
typeof action === "object" &&
|
||||
action !== null &&
|
||||
"assert" in action &&
|
||||
JSON.stringify(action).includes("current-run provider prompt evidence mismatch"),
|
||||
);
|
||||
if (exportIndex < 0 || assertionIndex < 0) {
|
||||
throw new Error("instruction profile scenario has no provider prompt evidence assertion");
|
||||
}
|
||||
const instructionContents = scenario.execution.config?.instructionContents;
|
||||
if (typeof instructionContents !== "string") {
|
||||
throw new Error("instruction profile scenario has no instruction contents");
|
||||
}
|
||||
const flow: QaScenarioFlow = {
|
||||
steps: [
|
||||
{
|
||||
name: "acquires bounded current-run prompt evidence",
|
||||
actions: [
|
||||
{ set: "sessionKey", value: sessionKey },
|
||||
{ set: "turn", value: { started: { runId: "current-run" } } },
|
||||
...actions.slice(exportIndex, assertionIndex + 1),
|
||||
],
|
||||
},
|
||||
],
|
||||
};
|
||||
return await runLoadedScenarioFlow(scenarioId, {
|
||||
flow,
|
||||
api: {
|
||||
path,
|
||||
env: {
|
||||
gateway: {
|
||||
call: async (method: string, input: Record<string, unknown>) => {
|
||||
expect(method).toBe("sessions.usage");
|
||||
expect(input).toEqual({
|
||||
key: sessionKey,
|
||||
agentId: "qa",
|
||||
range: "all",
|
||||
limit: 1,
|
||||
includeContextWeight: true,
|
||||
});
|
||||
return {
|
||||
sessions: [
|
||||
{
|
||||
key: params.reportSessionKey ?? sessionKey,
|
||||
contextWeight: {
|
||||
injectedWorkspaceFiles: [
|
||||
{
|
||||
path: "/qa/AGENTS.md",
|
||||
missing: false,
|
||||
truncated: false,
|
||||
rawChars: instructionContents.trimEnd().length,
|
||||
injectedChars: instructionContents.trimEnd().length,
|
||||
...params.report,
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
],
|
||||
};
|
||||
},
|
||||
},
|
||||
},
|
||||
runQaCli: async () => ({ outputDir: "/qa/trajectory" }),
|
||||
fs: {
|
||||
readFile: async (file: string) => {
|
||||
if (file === path.join("/qa/trajectory", "prompts.json")) {
|
||||
return JSON.stringify({ captured: true });
|
||||
}
|
||||
if (file === path.join("/qa/trajectory", "events.jsonl")) {
|
||||
return [
|
||||
{ type: "trace.metadata", data: { prompting: "[Truncated]" } },
|
||||
...(params.events ?? [currentEvent]),
|
||||
]
|
||||
.map((event) => JSON.stringify(event))
|
||||
.join("\n");
|
||||
}
|
||||
throw new Error(`unexpected evidence file: ${file}`);
|
||||
},
|
||||
rm: async () => undefined,
|
||||
},
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
describe("instruction profile prompt evidence", () => {
|
||||
it("acquires full injection evidence despite truncated metadata and stale provider mismatches", async () => {
|
||||
const result = await runPromptEvidence({
|
||||
events: [
|
||||
{
|
||||
...currentEvent,
|
||||
runId: "stale-run",
|
||||
data: { ...currentObservation, observedChars: 0, matchesAssembledPrompt: false },
|
||||
},
|
||||
currentEvent,
|
||||
],
|
||||
});
|
||||
expect(result.status).toBe("pass");
|
||||
});
|
||||
|
||||
it("excludes marker-bearing diagnostic context from bounded no-leak evidence", async () => {
|
||||
const marker = "INSTRUCTION-PROFILE-CONTEXT-MARKER-A6E29D4B";
|
||||
const result = await runPromptEvidence({
|
||||
events: [
|
||||
{
|
||||
type: "context.compiled",
|
||||
runId: "current-run",
|
||||
data: { systemPrompt: `diagnostic support context ${marker}` },
|
||||
},
|
||||
{
|
||||
...currentEvent,
|
||||
data: {
|
||||
...currentObservation,
|
||||
egress: "native-codex-websocket",
|
||||
promptSource: "instructions",
|
||||
},
|
||||
},
|
||||
],
|
||||
});
|
||||
expect(result.status).toBe("pass");
|
||||
});
|
||||
|
||||
it.each([
|
||||
{ name: "missing file", report: { missing: true } },
|
||||
{ name: "truncated injection", report: { truncated: true } },
|
||||
{ name: "incomplete source", report: { rawChars: 1 } },
|
||||
{ name: "incomplete injection", report: { injectedChars: 1 } },
|
||||
{ name: "another session's report", reportSessionKey: "agent:qa:other" },
|
||||
{ name: "missing current-run dispatch", events: [{ ...currentEvent, runId: "stale-run" }] },
|
||||
{
|
||||
name: "mismatched dispatch",
|
||||
events: [{ ...currentEvent, data: { ...currentObservation, matchesAssembledPrompt: false } }],
|
||||
},
|
||||
])("rejects $name", async (params) => {
|
||||
await expect(runPromptEvidence(params)).rejects.toThrow(
|
||||
"current-run provider prompt evidence mismatch",
|
||||
);
|
||||
});
|
||||
});
|
||||
|
|
@ -74,66 +74,6 @@ async function runWebchatTranscriptWait(
|
|||
});
|
||||
}
|
||||
|
||||
function readCurrentRunProviderPromptEvidenceFlow(trajectoryEvents: unknown[]): QaScenarioFlow {
|
||||
const scenario = readQaScenarioById("instruction-profile-artifact-followthrough-live");
|
||||
const actions = scenario.execution.flow?.steps[0]?.actions;
|
||||
if (!actions) {
|
||||
throw new Error("instruction profile scenario has no actions");
|
||||
}
|
||||
const evidenceIndex = actions.findIndex(
|
||||
(action) =>
|
||||
typeof action === "object" &&
|
||||
action !== null &&
|
||||
"set" in action &&
|
||||
action.set === "providerPromptEvidence",
|
||||
);
|
||||
const assertionIndex = actions.findIndex(
|
||||
(action, index) =>
|
||||
index > evidenceIndex &&
|
||||
typeof action === "object" &&
|
||||
action !== null &&
|
||||
"assert" in action &&
|
||||
JSON.stringify(action).includes("current-run provider prompt evidence mismatch"),
|
||||
);
|
||||
if (evidenceIndex < 0 || assertionIndex < 0) {
|
||||
throw new Error("instruction profile scenario has no provider prompt evidence assertion");
|
||||
}
|
||||
const instructionContents = scenario.execution.config?.instructionContents;
|
||||
const instructionChars =
|
||||
typeof instructionContents === "string" ? instructionContents.trimEnd().length : 0;
|
||||
return {
|
||||
steps: [
|
||||
{
|
||||
name: "proves current-run provider prompt evidence",
|
||||
actions: [
|
||||
{ set: "turn", value: { started: { runId: "current-run" } } },
|
||||
{
|
||||
set: "instructionProfileReport",
|
||||
value: {
|
||||
missing: false,
|
||||
truncated: false,
|
||||
rawChars: instructionChars,
|
||||
injectedChars: instructionChars,
|
||||
},
|
||||
},
|
||||
{ set: "trajectoryEvents", value: trajectoryEvents },
|
||||
...actions
|
||||
.slice(evidenceIndex, assertionIndex + 1)
|
||||
.filter(
|
||||
(action) =>
|
||||
!(
|
||||
typeof action === "object" &&
|
||||
action !== null &&
|
||||
"call" in action &&
|
||||
action.call === "fs.rm"
|
||||
),
|
||||
),
|
||||
],
|
||||
},
|
||||
],
|
||||
};
|
||||
}
|
||||
|
||||
const planningEvidenceCoverageIds = new Set([
|
||||
"agent-runtime.external-harness-selection-planning",
|
||||
"openai.codex-harness-no-meta-leak",
|
||||
|
|
@ -308,64 +248,6 @@ describe("scenario-flow-runner", () => {
|
|||
assertTelegramRichObservationFlow,
|
||||
);
|
||||
|
||||
it("ignores stale provider prompt mismatches when the current run matches", async () => {
|
||||
const currentObservation = {
|
||||
egress: "responses-sdk",
|
||||
payloadVariant: "initial",
|
||||
promptSource: "input.developer",
|
||||
expectedChars: 4096,
|
||||
observedChars: 4096,
|
||||
matchesAssembledPrompt: true,
|
||||
};
|
||||
const result = await runLoadedScenarioFlow("instruction-profile-artifact-followthrough-live", {
|
||||
flow: readCurrentRunProviderPromptEvidenceFlow([
|
||||
{
|
||||
type: "provider.prompt.observed",
|
||||
runId: "stale-run",
|
||||
data: {
|
||||
...currentObservation,
|
||||
promptSource: "missing",
|
||||
observedChars: 0,
|
||||
matchesAssembledPrompt: false,
|
||||
},
|
||||
},
|
||||
{ type: "provider.prompt.observed", runId: "current-run", data: currentObservation },
|
||||
]),
|
||||
});
|
||||
|
||||
expect(result.status).toBe("pass");
|
||||
});
|
||||
|
||||
it("excludes marker-bearing diagnostic trajectory context from bounded no-leak evidence", async () => {
|
||||
const marker = "INSTRUCTION-PROFILE-CONTEXT-MARKER-A6E29D4B";
|
||||
const trajectoryEvents = [
|
||||
{
|
||||
type: "context.compiled",
|
||||
runId: "current-run",
|
||||
data: { systemPrompt: `diagnostic support context ${marker}` },
|
||||
},
|
||||
{
|
||||
type: "provider.prompt.observed",
|
||||
runId: "current-run",
|
||||
data: {
|
||||
egress: "native-codex-websocket",
|
||||
payloadVariant: "initial",
|
||||
promptSource: "instructions",
|
||||
expectedChars: 4096,
|
||||
observedChars: 4096,
|
||||
matchesAssembledPrompt: true,
|
||||
},
|
||||
},
|
||||
];
|
||||
|
||||
expect(JSON.stringify(trajectoryEvents)).toContain(marker);
|
||||
const result = await runLoadedScenarioFlow("instruction-profile-artifact-followthrough-live", {
|
||||
flow: readCurrentRunProviderPromptEvidenceFlow(trajectoryEvents),
|
||||
});
|
||||
|
||||
expect(result.status).toBe("pass");
|
||||
});
|
||||
|
||||
it("keeps live goal followthrough inside the active-goal context limit", async () => {
|
||||
const state = createQaBusState();
|
||||
const artifactFile = "goal-continuance-live-00000000.txt";
|
||||
|
|
|
|||
|
|
@ -173,12 +173,19 @@ flow:
|
|||
- --json
|
||||
- json: true
|
||||
timeoutMs: 30000
|
||||
- set: promptsCapture
|
||||
value:
|
||||
expr: "JSON.parse(await fs.readFile(path.join(trajectoryExport.outputDir, 'prompts.json'), 'utf8'))"
|
||||
- call: env.gateway.call
|
||||
saveAs: sessionUsage
|
||||
args:
|
||||
- sessions.usage
|
||||
- key:
|
||||
ref: sessionKey
|
||||
agentId: qa
|
||||
range: all
|
||||
limit: 1
|
||||
includeContextWeight: true
|
||||
- set: instructionProfileReport
|
||||
value:
|
||||
expr: "promptsCapture.systemPromptReport?.injectedWorkspaceFiles?.find((entry) => path.basename(String(entry.path ?? '')) === config.instructionFile)"
|
||||
expr: "sessionUsage.sessions?.find((session) => session.key === sessionKey)?.contextWeight?.injectedWorkspaceFiles?.find((entry) => path.basename(String(entry.path ?? '')) === config.instructionFile)"
|
||||
- set: trajectoryEvents
|
||||
value:
|
||||
expr: "(await fs.readFile(path.join(trajectoryExport.outputDir, 'events.jsonl'), 'utf8')).split(/\\r?\\n/u).filter((line) => line.trim()).map((line) => JSON.parse(line))"
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue