From fd81db011042850cf985d5ffaf1989e14ae1a5dd Mon Sep 17 00:00:00 2001 From: Josh Avant <830519+joshavant@users.noreply.github.com> Date: Wed, 30 Sep 2026 23:54:25 -0500 Subject: [PATCH] fix(qa): accept valid Code Mode completion proof (#162397) --- .../qa-lab/src/live-scenario-timeouts.test.ts | 150 ++++++++++++------ .../issue-109025-completion-policy-live.yaml | 4 +- 2 files changed, 105 insertions(+), 49 deletions(-) diff --git a/extensions/qa-lab/src/live-scenario-timeouts.test.ts b/extensions/qa-lab/src/live-scenario-timeouts.test.ts index 1868e1628c0f..89918a7fb586 100644 --- a/extensions/qa-lab/src/live-scenario-timeouts.test.ts +++ b/extensions/qa-lab/src/live-scenario-timeouts.test.ts @@ -1,4 +1,5 @@ import { describe, expect, it } from "vitest"; +import { nestedToolHistoryFixture } from "../test/nested-tool-activity-fixture.js"; import { createQaBusState } from "./bus-state.js"; import type { QaNativeSubagentRun } from "./execution-identity-storage-inspection.js"; import { resolveQaLiveTurnTimeoutMs } from "./live-timeout.js"; @@ -61,6 +62,8 @@ function runCompletionPolicyFlow( delayedParentReplyHistoryReads?: number; parentExecCompletedAt?: number; parentExecCommand?: string; + parentExecMode?: "direct" | "code-mode"; + parentExecCustomType?: string; parentExecToolCallId?: string; parentOutbound?: CompletionParentOutboundFixture | null; parentReply?: CompletionParentReplyFixture | null; @@ -170,12 +173,36 @@ function runCompletionPolicyFlow( }; const parentReplyIsVisible = parentReply && gatewayCalls.length > (params.delayedParentReplyHistoryReads ?? 0); + const execMessages = + params.parentExecMode === "code-mode" + ? [ + { + role: "assistant", + content: [ + { + type: "toolCall", + id: "exec-qa", + name: "exec", + arguments: { + code: `text(await tools.exec(${JSON.stringify(execToolCall.arguments)}));`, + }, + }, + ], + }, + { + ...nestedToolHistoryFixture({ + toolName: "exec", + toolCallId: execToolCall.id, + input: execToolCall.arguments, + text: completionProofMarker, + }), + customType: params.parentExecCustomType ?? "openclaw.nested-tool.v1", + }, + ] + : [{ role: "assistant", content: [execToolCall] }]; return { messages: [ - { - role: "assistant", - content: [execToolCall], - }, + ...execMessages, ...(parentReplyIsVisible ? [ { @@ -671,11 +698,15 @@ describe("live subagent scenario timeouts", () => { ).rejects.toThrow("parent successful tool timeline was incomplete"); }); - it("rejects a visible exec tool call that does not match the successful result identity", async () => { - await expect( - runCompletionPolicyFlow({ parentExecToolCallId: "another-exec-call" }).result, - ).rejects.toThrow("parent exec did not use the exact delivered-completion command"); - }); + it.each(["direct", "code-mode"] as const)( + "rejects a %s exec tool call that does not match the successful result identity", + async (parentExecMode) => { + await expect( + runCompletionPolicyFlow({ parentExecMode, parentExecToolCallId: "another-exec-call" }) + .result, + ).rejects.toThrow("parent exec did not use the exact delivered-completion command"); + }, + ); it("rejects premature parent exec with a guessed completion token", async () => { await expect( @@ -683,11 +714,33 @@ describe("live subagent scenario timeouts", () => { ).rejects.toThrow("completion proof did not contain the delivered child marker"); }); - it("rejects inline filesystem discovery even when the exec proof forges the child marker", async () => { + it.each(["direct", "code-mode"] as const)( + "rejects %s inline filesystem discovery even when the exec proof forges the child marker", + async (parentExecMode) => { + await expect( + runCompletionPolicyFlow({ + parentExecMode, + parentExecCommand: `node -e 'require("node:fs").readFileSync("guessed-chain")'`, + proofCompletionText: "__DELIVERED_CHILD_TOKEN__", + }).result, + ).rejects.toThrow("parent exec did not use the exact delivered-completion command"); + }, + ); + + it("rejects a Code Mode nested exec with the wrong command despite matching output", async () => { await expect( runCompletionPolicyFlow({ - parentExecCommand: `node -e 'require("node:fs").readFileSync("guessed-chain")'`, - proofCompletionText: "__DELIVERED_CHILD_TOKEN__", + parentExecMode: "code-mode", + parentExecCommand: "printf wrong-command", + }).result, + ).rejects.toThrow("parent exec did not use the exact delivered-completion command"); + }); + + it("rejects an unrelated custom row with the exact exec ID and command", async () => { + await expect( + runCompletionPolicyFlow({ + parentExecMode: "code-mode", + parentExecCustomType: "unrelated", }).result, ).rejects.toThrow("parent exec did not use the exact delivered-completion command"); }); @@ -733,42 +786,45 @@ describe("live subagent scenario timeouts", () => { await expect(runCompletionPolicyFlow(fixture).result).rejects.toThrow(fixture.failure); }); - it("accepts a current requester-owned delivered child with complete transcript proof", async () => { - const { - gatewayCalls, - result, - state, - runRequesterSessionKeys, - transcriptReadOptions, - transcriptSessionKeys, - } = runCompletionPolicyFlow(); + it.each(["direct", "code-mode"] as const)( + "accepts a current requester-owned delivered child with complete %s transcript proof", + async (parentExecMode) => { + const { + gatewayCalls, + result, + state, + runRequesterSessionKeys, + transcriptReadOptions, + transcriptSessionKeys, + } = runCompletionPolicyFlow({ parentExecMode }); - await expect(result).resolves.toMatchObject({ status: "pass" }); - expect(gatewayCalls).toEqual([ - { - method: "chat.history", - request: { sessionKey: completionParentSessionKey, limit: 100, maxChars: 131_072 }, - }, - ]); - expect(runRequesterSessionKeys).toEqual([completionParentSessionKey]); - expect(transcriptSessionKeys).toEqual([ - completionParentSessionKey, - completionParentSessionKey, - completionChildSessionKey, - ]); - expect(transcriptReadOptions).toEqual([ - { allowEmpty: true }, - { afterEventCursor: completionParentTranscriptCursor }, - undefined, - ]); - expect(state.getSnapshot().messages[0]?.text).toContain("CHILD_DONE"); - expect(state.getSnapshot().messages[1]).toMatchObject({ - accountId: "qa-channel", - conversation: { id: "issue-109025-completion", kind: "direct" }, - direction: "outbound", - text: completionProofMarker, - }); - }); + await expect(result).resolves.toMatchObject({ status: "pass" }); + expect(gatewayCalls).toEqual([ + { + method: "chat.history", + request: { sessionKey: completionParentSessionKey, limit: 100, maxChars: 131_072 }, + }, + ]); + expect(runRequesterSessionKeys).toEqual([completionParentSessionKey]); + expect(transcriptSessionKeys).toEqual([ + completionParentSessionKey, + completionParentSessionKey, + completionChildSessionKey, + ]); + expect(transcriptReadOptions).toEqual([ + { allowEmpty: true }, + { afterEventCursor: completionParentTranscriptCursor }, + undefined, + ]); + expect(state.getSnapshot().messages[0]?.text).toContain("CHILD_DONE"); + expect(state.getSnapshot().messages[1]).toMatchObject({ + accountId: "qa-channel", + conversation: { id: "issue-109025-completion", kind: "direct" }, + direction: "outbound", + text: completionProofMarker, + }); + }, + ); it("waits until the exact final requester reply is visible after the exec result", async () => { const { gatewayCalls, result } = runCompletionPolicyFlow({ delayedParentReplyHistoryReads: 2 }); diff --git a/qa/scenarios/agents/issue-109025-completion-policy-live.yaml b/qa/scenarios/agents/issue-109025-completion-policy-live.yaml index a4db1ecd10d5..58d71043521a 100644 --- a/qa/scenarios/agents/issue-109025-completion-policy-live.yaml +++ b/qa/scenarios/agents/issue-109025-completion-policy-live.yaml @@ -234,7 +234,7 @@ flow: expr: |- (async () => { const history = await env.gateway.call('chat.history', { sessionKey: parentSessionKey, limit: 100, maxChars: 131072 }, { timeoutMs: liveTurnTimeoutMs(env, 15000) }); - const messages = history.messages ?? []; + const messages = projectQaToolMessages(history.messages ?? []); const execIndex = messages.findLastIndex((message) => message?.role === 'assistant' && Array.isArray(message.content) && message.content.some((item) => (item?.type === 'toolCall' || item?.type === 'toolUse') && (item?.name ?? item?.toolName) === 'exec')); if (execIndex < 0) return undefined; const finalReply = messages.slice(execIndex + 1).findLast((message) => message?.role === 'assistant' && message.phase !== 'commentary' && !message.openclawMessageToolMirror && !message.openclawDeliveryMirror && !(message.provider === 'openclaw' && ['delivery-mirror', 'gateway-injected'].includes(message.model)) && (!Array.isArray(message.content) || !message.content.some((item) => ['toolCall', 'toolUse', 'tool_use', 'toolResult', 'tool_result'].includes(item?.type)))); @@ -245,7 +245,7 @@ flow: - expr: liveTurnTimeoutMs(env, 60000) - 250 - assert: - expr: "(parentHistory.messages ?? []).some((message) => message?.role === 'assistant' && Array.isArray(message.content) && message.content.some((item) => (item?.type === 'toolCall' || item?.type === 'toolUse') && (item?.name ?? item?.toolName) === 'exec' && item?.id === parentSuccessfulToolEvents[2].toolCallId && (item?.arguments ?? item?.input)?.command === execCommand.replace('__CHILD_COMPLETION_TOKEN__', childTerminalMarker)))" + expr: "projectQaToolMessages(parentHistory.messages ?? []).some((message) => message?.role === 'assistant' && Array.isArray(message.content) && message.content.some((item) => (item?.type === 'toolCall' || item?.type === 'toolUse') && (item?.name ?? item?.toolName) === 'exec' && item?.id === parentSuccessfulToolEvents[2].toolCallId && (item?.arguments ?? item?.input)?.command === execCommand.replace('__CHILD_COMPLETION_TOKEN__', childTerminalMarker)))" message: parent exec did not use the exact delivered-completion command - call: waitForOutboundMessage saveAs: parentOutbound