From ca5f05c87ee2111b830ecd04da5dfa068dcbe5c4 Mon Sep 17 00:00:00 2001 From: Josh Avant <830519+joshavant@users.noreply.github.com> Date: Sat, 26 Sep 2026 03:55:03 -0500 Subject: [PATCH] fix(qa): use bounded instruction injection evidence (#158634) --- .../scenario-catalog.prompt-evidence.test.ts | 180 ++++++++++++++++++ .../qa-lab/src/scenario-flow-runner.test.ts | 118 ------------ ...n-profile-artifact-followthrough-live.yaml | 15 +- 3 files changed, 191 insertions(+), 122 deletions(-) create mode 100644 extensions/qa-lab/src/scenario-catalog.prompt-evidence.test.ts diff --git a/extensions/qa-lab/src/scenario-catalog.prompt-evidence.test.ts b/extensions/qa-lab/src/scenario-catalog.prompt-evidence.test.ts new file mode 100644 index 000000000000..1ccdc249a429 --- /dev/null +++ b/extensions/qa-lab/src/scenario-catalog.prompt-evidence.test.ts @@ -0,0 +1,180 @@ +import path from "node:path"; +import { describe, expect, it } from "vitest"; +import { readQaScenarioById, type QaScenarioFlow } from "./scenario-catalog.js"; +import { runLoadedScenarioFlow } from "./scenario-flow-runner.test-support.js"; + +const scenarioId = "instruction-profile-artifact-followthrough-live"; +const sessionKey = "agent:qa:instruction-profile-artifact:test"; +const currentObservation = { + egress: "responses-sdk", + payloadVariant: "initial", + promptSource: "input.developer", + expectedChars: 4096, + observedChars: 4096, + matchesAssembledPrompt: true, +}; +const currentEvent = { + type: "provider.prompt.observed", + runId: "current-run", + data: currentObservation, +}; + +async function runPromptEvidence( + params: { + events?: unknown[]; + report?: Record; + reportSessionKey?: string; + } = {}, +) { + const scenario = readQaScenarioById(scenarioId); + const actions = scenario.execution.flow?.steps[0]?.actions; + if (!actions) { + throw new Error("instruction profile scenario has no actions"); + } + const exportIndex = actions.findIndex( + (action) => + typeof action === "object" && + action !== null && + "call" in action && + action.call === "runQaCli", + ); + const assertionIndex = actions.findIndex( + (action, index) => + index > exportIndex && + typeof action === "object" && + action !== null && + "assert" in action && + JSON.stringify(action).includes("current-run provider prompt evidence mismatch"), + ); + if (exportIndex < 0 || assertionIndex < 0) { + throw new Error("instruction profile scenario has no provider prompt evidence assertion"); + } + const instructionContents = scenario.execution.config?.instructionContents; + if (typeof instructionContents !== "string") { + throw new Error("instruction profile scenario has no instruction contents"); + } + const flow: QaScenarioFlow = { + steps: [ + { + name: "acquires bounded current-run prompt evidence", + actions: [ + { set: "sessionKey", value: sessionKey }, + { set: "turn", value: { started: { runId: "current-run" } } }, + ...actions.slice(exportIndex, assertionIndex + 1), + ], + }, + ], + }; + return await runLoadedScenarioFlow(scenarioId, { + flow, + api: { + path, + env: { + gateway: { + call: async (method: string, input: Record) => { + expect(method).toBe("sessions.usage"); + expect(input).toEqual({ + key: sessionKey, + agentId: "qa", + range: "all", + limit: 1, + includeContextWeight: true, + }); + return { + sessions: [ + { + key: params.reportSessionKey ?? sessionKey, + contextWeight: { + injectedWorkspaceFiles: [ + { + path: "/qa/AGENTS.md", + missing: false, + truncated: false, + rawChars: instructionContents.trimEnd().length, + injectedChars: instructionContents.trimEnd().length, + ...params.report, + }, + ], + }, + }, + ], + }; + }, + }, + }, + runQaCli: async () => ({ outputDir: "/qa/trajectory" }), + fs: { + readFile: async (file: string) => { + if (file === path.join("/qa/trajectory", "prompts.json")) { + return JSON.stringify({ captured: true }); + } + if (file === path.join("/qa/trajectory", "events.jsonl")) { + return [ + { type: "trace.metadata", data: { prompting: "[Truncated]" } }, + ...(params.events ?? [currentEvent]), + ] + .map((event) => JSON.stringify(event)) + .join("\n"); + } + throw new Error(`unexpected evidence file: ${file}`); + }, + rm: async () => undefined, + }, + }, + }); +} + +describe("instruction profile prompt evidence", () => { + it("acquires full injection evidence despite truncated metadata and stale provider mismatches", async () => { + const result = await runPromptEvidence({ + events: [ + { + ...currentEvent, + runId: "stale-run", + data: { ...currentObservation, observedChars: 0, matchesAssembledPrompt: false }, + }, + currentEvent, + ], + }); + expect(result.status).toBe("pass"); + }); + + it("excludes marker-bearing diagnostic context from bounded no-leak evidence", async () => { + const marker = "INSTRUCTION-PROFILE-CONTEXT-MARKER-A6E29D4B"; + const result = await runPromptEvidence({ + events: [ + { + type: "context.compiled", + runId: "current-run", + data: { systemPrompt: `diagnostic support context ${marker}` }, + }, + { + ...currentEvent, + data: { + ...currentObservation, + egress: "native-codex-websocket", + promptSource: "instructions", + }, + }, + ], + }); + expect(result.status).toBe("pass"); + }); + + it.each([ + { name: "missing file", report: { missing: true } }, + { name: "truncated injection", report: { truncated: true } }, + { name: "incomplete source", report: { rawChars: 1 } }, + { name: "incomplete injection", report: { injectedChars: 1 } }, + { name: "another session's report", reportSessionKey: "agent:qa:other" }, + { name: "missing current-run dispatch", events: [{ ...currentEvent, runId: "stale-run" }] }, + { + name: "mismatched dispatch", + events: [{ ...currentEvent, data: { ...currentObservation, matchesAssembledPrompt: false } }], + }, + ])("rejects $name", async (params) => { + await expect(runPromptEvidence(params)).rejects.toThrow( + "current-run provider prompt evidence mismatch", + ); + }); +}); diff --git a/extensions/qa-lab/src/scenario-flow-runner.test.ts b/extensions/qa-lab/src/scenario-flow-runner.test.ts index 25d2e773bff8..f6f8def04385 100644 --- a/extensions/qa-lab/src/scenario-flow-runner.test.ts +++ b/extensions/qa-lab/src/scenario-flow-runner.test.ts @@ -74,66 +74,6 @@ async function runWebchatTranscriptWait( }); } -function readCurrentRunProviderPromptEvidenceFlow(trajectoryEvents: unknown[]): QaScenarioFlow { - const scenario = readQaScenarioById("instruction-profile-artifact-followthrough-live"); - const actions = scenario.execution.flow?.steps[0]?.actions; - if (!actions) { - throw new Error("instruction profile scenario has no actions"); - } - const evidenceIndex = actions.findIndex( - (action) => - typeof action === "object" && - action !== null && - "set" in action && - action.set === "providerPromptEvidence", - ); - const assertionIndex = actions.findIndex( - (action, index) => - index > evidenceIndex && - typeof action === "object" && - action !== null && - "assert" in action && - JSON.stringify(action).includes("current-run provider prompt evidence mismatch"), - ); - if (evidenceIndex < 0 || assertionIndex < 0) { - throw new Error("instruction profile scenario has no provider prompt evidence assertion"); - } - const instructionContents = scenario.execution.config?.instructionContents; - const instructionChars = - typeof instructionContents === "string" ? instructionContents.trimEnd().length : 0; - return { - steps: [ - { - name: "proves current-run provider prompt evidence", - actions: [ - { set: "turn", value: { started: { runId: "current-run" } } }, - { - set: "instructionProfileReport", - value: { - missing: false, - truncated: false, - rawChars: instructionChars, - injectedChars: instructionChars, - }, - }, - { set: "trajectoryEvents", value: trajectoryEvents }, - ...actions - .slice(evidenceIndex, assertionIndex + 1) - .filter( - (action) => - !( - typeof action === "object" && - action !== null && - "call" in action && - action.call === "fs.rm" - ), - ), - ], - }, - ], - }; -} - const planningEvidenceCoverageIds = new Set([ "agent-runtime.external-harness-selection-planning", "openai.codex-harness-no-meta-leak", @@ -308,64 +248,6 @@ describe("scenario-flow-runner", () => { assertTelegramRichObservationFlow, ); - it("ignores stale provider prompt mismatches when the current run matches", async () => { - const currentObservation = { - egress: "responses-sdk", - payloadVariant: "initial", - promptSource: "input.developer", - expectedChars: 4096, - observedChars: 4096, - matchesAssembledPrompt: true, - }; - const result = await runLoadedScenarioFlow("instruction-profile-artifact-followthrough-live", { - flow: readCurrentRunProviderPromptEvidenceFlow([ - { - type: "provider.prompt.observed", - runId: "stale-run", - data: { - ...currentObservation, - promptSource: "missing", - observedChars: 0, - matchesAssembledPrompt: false, - }, - }, - { type: "provider.prompt.observed", runId: "current-run", data: currentObservation }, - ]), - }); - - expect(result.status).toBe("pass"); - }); - - it("excludes marker-bearing diagnostic trajectory context from bounded no-leak evidence", async () => { - const marker = "INSTRUCTION-PROFILE-CONTEXT-MARKER-A6E29D4B"; - const trajectoryEvents = [ - { - type: "context.compiled", - runId: "current-run", - data: { systemPrompt: `diagnostic support context ${marker}` }, - }, - { - type: "provider.prompt.observed", - runId: "current-run", - data: { - egress: "native-codex-websocket", - payloadVariant: "initial", - promptSource: "instructions", - expectedChars: 4096, - observedChars: 4096, - matchesAssembledPrompt: true, - }, - }, - ]; - - expect(JSON.stringify(trajectoryEvents)).toContain(marker); - const result = await runLoadedScenarioFlow("instruction-profile-artifact-followthrough-live", { - flow: readCurrentRunProviderPromptEvidenceFlow(trajectoryEvents), - }); - - expect(result.status).toBe("pass"); - }); - it("keeps live goal followthrough inside the active-goal context limit", async () => { const state = createQaBusState(); const artifactFile = "goal-continuance-live-00000000.txt"; diff --git a/qa/scenarios/agents/instruction-profile-artifact-followthrough-live.yaml b/qa/scenarios/agents/instruction-profile-artifact-followthrough-live.yaml index 0cc0622735a5..8308cb9e6204 100644 --- a/qa/scenarios/agents/instruction-profile-artifact-followthrough-live.yaml +++ b/qa/scenarios/agents/instruction-profile-artifact-followthrough-live.yaml @@ -173,12 +173,19 @@ flow: - --json - json: true timeoutMs: 30000 - - set: promptsCapture - value: - expr: "JSON.parse(await fs.readFile(path.join(trajectoryExport.outputDir, 'prompts.json'), 'utf8'))" + - call: env.gateway.call + saveAs: sessionUsage + args: + - sessions.usage + - key: + ref: sessionKey + agentId: qa + range: all + limit: 1 + includeContextWeight: true - set: instructionProfileReport value: - expr: "promptsCapture.systemPromptReport?.injectedWorkspaceFiles?.find((entry) => path.basename(String(entry.path ?? '')) === config.instructionFile)" + expr: "sessionUsage.sessions?.find((session) => session.key === sessionKey)?.contextWeight?.injectedWorkspaceFiles?.find((entry) => path.basename(String(entry.path ?? '')) === config.instructionFile)" - set: trajectoryEvents value: expr: "(await fs.readFile(path.join(trajectoryExport.outputDir, 'events.jsonl'), 'utf8')).split(/\\r?\\n/u).filter((line) => line.trim()).map((line) => JSON.parse(line))"