fix(qa): use bounded instruction injection evidence (#158634)

This commit is contained in:
Josh Avant 2026-09-26 03:55:03 -05:00 • committed by GitHub
parent a8272f099f
commit ca5f05c87e
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 191 additions and 122 deletions

View file

@ -0,0 +1,180 @@
import path from "node:path";
import { describe, expect, it } from "vitest";
import { readQaScenarioById, type QaScenarioFlow } from "./scenario-catalog.js";
import { runLoadedScenarioFlow } from "./scenario-flow-runner.test-support.js";
const scenarioId = "instruction-profile-artifact-followthrough-live";
const sessionKey = "agent:qa:instruction-profile-artifact:test";
const currentObservation = {
egress: "responses-sdk",
payloadVariant: "initial",
promptSource: "input.developer",
expectedChars: 4096,
observedChars: 4096,
matchesAssembledPrompt: true,
};
const currentEvent = {
type: "provider.prompt.observed",
runId: "current-run",
data: currentObservation,
};
async function runPromptEvidence(
params: {
events?: unknown[];
report?: Record<string, unknown>;
reportSessionKey?: string;
} = {},
) {
const scenario = readQaScenarioById(scenarioId);
const actions = scenario.execution.flow?.steps[0]?.actions;
if (!actions) {
throw new Error("instruction profile scenario has no actions");
}
const exportIndex = actions.findIndex(
(action) =>
typeof action === "object" &&
action !== null &&
"call" in action &&
action.call === "runQaCli",
);
const assertionIndex = actions.findIndex(
(action, index) =>
index > exportIndex &&
typeof action === "object" &&
action !== null &&
"assert" in action &&
JSON.stringify(action).includes("current-run provider prompt evidence mismatch"),
);
if (exportIndex < 0 || assertionIndex < 0) {
throw new Error("instruction profile scenario has no provider prompt evidence assertion");
}
const instructionContents = scenario.execution.config?.instructionContents;
if (typeof instructionContents !== "string") {
throw new Error("instruction profile scenario has no instruction contents");
}
const flow: QaScenarioFlow = {
steps: [
{
name: "acquires bounded current-run prompt evidence",
actions: [
{ set: "sessionKey", value: sessionKey },
{ set: "turn", value: { started: { runId: "current-run" } } },
...actions.slice(exportIndex, assertionIndex + 1),
],
},
],
};
return await runLoadedScenarioFlow(scenarioId, {
flow,
api: {
path,
env: {
gateway: {
call: async (method: string, input: Record<string, unknown>) => {
expect(method).toBe("sessions.usage");
expect(input).toEqual({
key: sessionKey,
agentId: "qa",
range: "all",
limit: 1,
includeContextWeight: true,
});
return {
sessions: [
{
key: params.reportSessionKey ?? sessionKey,
contextWeight: {
injectedWorkspaceFiles: [
{
path: "/qa/AGENTS.md",
missing: false,
truncated: false,
rawChars: instructionContents.trimEnd().length,
injectedChars: instructionContents.trimEnd().length,
...params.report,
},
],
},
},
],
};
},
},
},
runQaCli: async () => ({ outputDir: "/qa/trajectory" }),
fs: {
readFile: async (file: string) => {
if (file === path.join("/qa/trajectory", "prompts.json")) {
return JSON.stringify({ captured: true });
}
if (file === path.join("/qa/trajectory", "events.jsonl")) {
return [
{ type: "trace.metadata", data: { prompting: "[Truncated]" } },
...(params.events ?? [currentEvent]),
]
.map((event) => JSON.stringify(event))
.join("\n");
}
throw new Error(`unexpected evidence file: ${file}`);
},
rm: async () => undefined,
},
},
});
}
describe("instruction profile prompt evidence", () => {
it("acquires full injection evidence despite truncated metadata and stale provider mismatches", async () => {
const result = await runPromptEvidence({
events: [
{
...currentEvent,
runId: "stale-run",
data: { ...currentObservation, observedChars: 0, matchesAssembledPrompt: false },
},
currentEvent,
],
});
expect(result.status).toBe("pass");
});
it("excludes marker-bearing diagnostic context from bounded no-leak evidence", async () => {
const marker = "INSTRUCTION-PROFILE-CONTEXT-MARKER-A6E29D4B";
const result = await runPromptEvidence({
events: [
{
type: "context.compiled",
runId: "current-run",
data: { systemPrompt: `diagnostic support context ${marker}` },
},
{
...currentEvent,
data: {
...currentObservation,
egress: "native-codex-websocket",
promptSource: "instructions",
},
},
],
});
expect(result.status).toBe("pass");
});
it.each([
{ name: "missing file", report: { missing: true } },
{ name: "truncated injection", report: { truncated: true } },
{ name: "incomplete source", report: { rawChars: 1 } },
{ name: "incomplete injection", report: { injectedChars: 1 } },
{ name: "another session's report", reportSessionKey: "agent:qa:other" },
{ name: "missing current-run dispatch", events: [{ ...currentEvent, runId: "stale-run" }] },
{
name: "mismatched dispatch",
events: [{ ...currentEvent, data: { ...currentObservation, matchesAssembledPrompt: false } }],
},
])("rejects $name", async (params) => {
await expect(runPromptEvidence(params)).rejects.toThrow(
"current-run provider prompt evidence mismatch",
);
});
});

View file

@ -74,66 +74,6 @@ async function runWebchatTranscriptWait(
});
}
function readCurrentRunProviderPromptEvidenceFlow(trajectoryEvents: unknown[]): QaScenarioFlow {
const scenario = readQaScenarioById("instruction-profile-artifact-followthrough-live");
const actions = scenario.execution.flow?.steps[0]?.actions;
if (!actions) {
throw new Error("instruction profile scenario has no actions");
}
const evidenceIndex = actions.findIndex(
(action) =>
typeof action === "object" &&
action !== null &&
"set" in action &&
action.set === "providerPromptEvidence",
);
const assertionIndex = actions.findIndex(
(action, index) =>
index > evidenceIndex &&
typeof action === "object" &&
action !== null &&
"assert" in action &&
JSON.stringify(action).includes("current-run provider prompt evidence mismatch"),
);
if (evidenceIndex < 0 || assertionIndex < 0) {
throw new Error("instruction profile scenario has no provider prompt evidence assertion");
}
const instructionContents = scenario.execution.config?.instructionContents;
const instructionChars =
typeof instructionContents === "string" ? instructionContents.trimEnd().length : 0;
return {
steps: [
{
name: "proves current-run provider prompt evidence",
actions: [
{ set: "turn", value: { started: { runId: "current-run" } } },
{
set: "instructionProfileReport",
value: {
missing: false,
truncated: false,
rawChars: instructionChars,
injectedChars: instructionChars,
},
},
{ set: "trajectoryEvents", value: trajectoryEvents },
...actions
.slice(evidenceIndex, assertionIndex + 1)
.filter(
(action) =>
!(
typeof action === "object" &&
action !== null &&
"call" in action &&
action.call === "fs.rm"
),
),
],
},
],
};
}
const planningEvidenceCoverageIds = new Set([
"agent-runtime.external-harness-selection-planning",
"openai.codex-harness-no-meta-leak",
@ -308,64 +248,6 @@ describe("scenario-flow-runner", () => {
assertTelegramRichObservationFlow,
);
it("ignores stale provider prompt mismatches when the current run matches", async () => {
const currentObservation = {
egress: "responses-sdk",
payloadVariant: "initial",
promptSource: "input.developer",
expectedChars: 4096,
observedChars: 4096,
matchesAssembledPrompt: true,
};
const result = await runLoadedScenarioFlow("instruction-profile-artifact-followthrough-live", {
flow: readCurrentRunProviderPromptEvidenceFlow([
{
type: "provider.prompt.observed",
runId: "stale-run",
data: {
...currentObservation,
promptSource: "missing",
observedChars: 0,
matchesAssembledPrompt: false,
},
},
{ type: "provider.prompt.observed", runId: "current-run", data: currentObservation },
]),
});
expect(result.status).toBe("pass");
});
it("excludes marker-bearing diagnostic trajectory context from bounded no-leak evidence", async () => {
const marker = "INSTRUCTION-PROFILE-CONTEXT-MARKER-A6E29D4B";
const trajectoryEvents = [
{
type: "context.compiled",
runId: "current-run",
data: { systemPrompt: `diagnostic support context ${marker}` },
},
{
type: "provider.prompt.observed",
runId: "current-run",
data: {
egress: "native-codex-websocket",
payloadVariant: "initial",
promptSource: "instructions",
expectedChars: 4096,
observedChars: 4096,
matchesAssembledPrompt: true,
},
},
];
expect(JSON.stringify(trajectoryEvents)).toContain(marker);
const result = await runLoadedScenarioFlow("instruction-profile-artifact-followthrough-live", {
flow: readCurrentRunProviderPromptEvidenceFlow(trajectoryEvents),
});
expect(result.status).toBe("pass");
});
it("keeps live goal followthrough inside the active-goal context limit", async () => {
const state = createQaBusState();
const artifactFile = "goal-continuance-live-00000000.txt";

View file

@ -173,12 +173,19 @@ flow:
- --json
- json: true
timeoutMs: 30000
- set: promptsCapture
value:
expr: "JSON.parse(await fs.readFile(path.join(trajectoryExport.outputDir, 'prompts.json'), 'utf8'))"
- call: env.gateway.call
saveAs: sessionUsage
args:
- sessions.usage
- key:
ref: sessionKey
agentId: qa
range: all
limit: 1
includeContextWeight: true
- set: instructionProfileReport
value:
expr: "promptsCapture.systemPromptReport?.injectedWorkspaceFiles?.find((entry) => path.basename(String(entry.path ?? '')) === config.instructionFile)"
expr: "sessionUsage.sessions?.find((session) => session.key === sessionKey)?.contextWeight?.injectedWorkspaceFiles?.find((entry) => path.basename(String(entry.path ?? '')) === config.instructionFile)"
- set: trajectoryEvents
value:
expr: "(await fs.readFile(path.join(trajectoryExport.outputDir, 'events.jsonl'), 'utf8')).split(/\\r?\\n/u).filter((line) => line.trim()).map((line) => JSON.parse(line))"