fix(qa): accept valid Code Mode completion proof (#162397)

This commit is contained in:
Josh Avant 2026-09-30 23:54:25 -05:00 • committed by GitHub
parent d76d5af33d
commit fd81db0110
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 105 additions and 49 deletions

View file

@ -1,4 +1,5 @@
import { describe, expect, it } from "vitest";
import { nestedToolHistoryFixture } from "../test/nested-tool-activity-fixture.js";
import { createQaBusState } from "./bus-state.js";
import type { QaNativeSubagentRun } from "./execution-identity-storage-inspection.js";
import { resolveQaLiveTurnTimeoutMs } from "./live-timeout.js";
@ -61,6 +62,8 @@ function runCompletionPolicyFlow(
delayedParentReplyHistoryReads?: number;
parentExecCompletedAt?: number;
parentExecCommand?: string;
parentExecMode?: "direct" | "code-mode";
parentExecCustomType?: string;
parentExecToolCallId?: string;
parentOutbound?: CompletionParentOutboundFixture | null;
parentReply?: CompletionParentReplyFixture | null;
@ -170,12 +173,36 @@ function runCompletionPolicyFlow(
};
const parentReplyIsVisible =
parentReply && gatewayCalls.length > (params.delayedParentReplyHistoryReads ?? 0);
const execMessages =
params.parentExecMode === "code-mode"
? [
{
role: "assistant",
content: [
{
type: "toolCall",
id: "exec-qa",
name: "exec",
arguments: {
code: `text(await tools.exec(${JSON.stringify(execToolCall.arguments)}));`,
},
},
],
},
{
...nestedToolHistoryFixture({
toolName: "exec",
toolCallId: execToolCall.id,
input: execToolCall.arguments,
text: completionProofMarker,
}),
customType: params.parentExecCustomType ?? "openclaw.nested-tool.v1",
},
]
: [{ role: "assistant", content: [execToolCall] }];
return {
messages: [
{
role: "assistant",
content: [execToolCall],
},
...execMessages,
...(parentReplyIsVisible
? [
{
@ -671,11 +698,15 @@ describe("live subagent scenario timeouts", () => {
).rejects.toThrow("parent successful tool timeline was incomplete");
});
it("rejects a visible exec tool call that does not match the successful result identity", async () => {
await expect(
runCompletionPolicyFlow({ parentExecToolCallId: "another-exec-call" }).result,
).rejects.toThrow("parent exec did not use the exact delivered-completion command");
});
it.each(["direct", "code-mode"] as const)(
"rejects a %s exec tool call that does not match the successful result identity",
async (parentExecMode) => {
await expect(
runCompletionPolicyFlow({ parentExecMode, parentExecToolCallId: "another-exec-call" })
.result,
).rejects.toThrow("parent exec did not use the exact delivered-completion command");
},
);
it("rejects premature parent exec with a guessed completion token", async () => {
await expect(
@ -683,11 +714,33 @@ describe("live subagent scenario timeouts", () => {
).rejects.toThrow("completion proof did not contain the delivered child marker");
});
it("rejects inline filesystem discovery even when the exec proof forges the child marker", async () => {
it.each(["direct", "code-mode"] as const)(
"rejects %s inline filesystem discovery even when the exec proof forges the child marker",
async (parentExecMode) => {
await expect(
runCompletionPolicyFlow({
parentExecMode,
parentExecCommand: `node -e 'require("node:fs").readFileSync("guessed-chain")'`,
proofCompletionText: "__DELIVERED_CHILD_TOKEN__",
}).result,
).rejects.toThrow("parent exec did not use the exact delivered-completion command");
},
);
it("rejects a Code Mode nested exec with the wrong command despite matching output", async () => {
await expect(
runCompletionPolicyFlow({
parentExecCommand: `node -e 'require("node:fs").readFileSync("guessed-chain")'`,
proofCompletionText: "__DELIVERED_CHILD_TOKEN__",
parentExecMode: "code-mode",
parentExecCommand: "printf wrong-command",
}).result,
).rejects.toThrow("parent exec did not use the exact delivered-completion command");
});
it("rejects an unrelated custom row with the exact exec ID and command", async () => {
await expect(
runCompletionPolicyFlow({
parentExecMode: "code-mode",
parentExecCustomType: "unrelated",
}).result,
).rejects.toThrow("parent exec did not use the exact delivered-completion command");
});
@ -733,42 +786,45 @@ describe("live subagent scenario timeouts", () => {
await expect(runCompletionPolicyFlow(fixture).result).rejects.toThrow(fixture.failure);
});
it("accepts a current requester-owned delivered child with complete transcript proof", async () => {
const {
gatewayCalls,
result,
state,
runRequesterSessionKeys,
transcriptReadOptions,
transcriptSessionKeys,
} = runCompletionPolicyFlow();
it.each(["direct", "code-mode"] as const)(
"accepts a current requester-owned delivered child with complete %s transcript proof",
async (parentExecMode) => {
const {
gatewayCalls,
result,
state,
runRequesterSessionKeys,
transcriptReadOptions,
transcriptSessionKeys,
} = runCompletionPolicyFlow({ parentExecMode });
await expect(result).resolves.toMatchObject({ status: "pass" });
expect(gatewayCalls).toEqual([
{
method: "chat.history",
request: { sessionKey: completionParentSessionKey, limit: 100, maxChars: 131_072 },
},
]);
expect(runRequesterSessionKeys).toEqual([completionParentSessionKey]);
expect(transcriptSessionKeys).toEqual([
completionParentSessionKey,
completionParentSessionKey,
completionChildSessionKey,
]);
expect(transcriptReadOptions).toEqual([
{ allowEmpty: true },
{ afterEventCursor: completionParentTranscriptCursor },
undefined,
]);
expect(state.getSnapshot().messages[0]?.text).toContain("CHILD_DONE");
expect(state.getSnapshot().messages[1]).toMatchObject({
accountId: "qa-channel",
conversation: { id: "issue-109025-completion", kind: "direct" },
direction: "outbound",
text: completionProofMarker,
});
});
await expect(result).resolves.toMatchObject({ status: "pass" });
expect(gatewayCalls).toEqual([
{
method: "chat.history",
request: { sessionKey: completionParentSessionKey, limit: 100, maxChars: 131_072 },
},
]);
expect(runRequesterSessionKeys).toEqual([completionParentSessionKey]);
expect(transcriptSessionKeys).toEqual([
completionParentSessionKey,
completionParentSessionKey,
completionChildSessionKey,
]);
expect(transcriptReadOptions).toEqual([
{ allowEmpty: true },
{ afterEventCursor: completionParentTranscriptCursor },
undefined,
]);
expect(state.getSnapshot().messages[0]?.text).toContain("CHILD_DONE");
expect(state.getSnapshot().messages[1]).toMatchObject({
accountId: "qa-channel",
conversation: { id: "issue-109025-completion", kind: "direct" },
direction: "outbound",
text: completionProofMarker,
});
},
);
it("waits until the exact final requester reply is visible after the exec result", async () => {
const { gatewayCalls, result } = runCompletionPolicyFlow({ delayedParentReplyHistoryReads: 2 });

View file

@ -234,7 +234,7 @@ flow:
expr: |-
(async () => {
const history = await env.gateway.call('chat.history', { sessionKey: parentSessionKey, limit: 100, maxChars: 131072 }, { timeoutMs: liveTurnTimeoutMs(env, 15000) });
const messages = history.messages ?? [];
const messages = projectQaToolMessages(history.messages ?? []);
const execIndex = messages.findLastIndex((message) => message?.role === 'assistant' && Array.isArray(message.content) && message.content.some((item) => (item?.type === 'toolCall' || item?.type === 'toolUse') && (item?.name ?? item?.toolName) === 'exec'));
if (execIndex < 0) return undefined;
const finalReply = messages.slice(execIndex + 1).findLast((message) => message?.role === 'assistant' && message.phase !== 'commentary' && !message.openclawMessageToolMirror && !message.openclawDeliveryMirror && !(message.provider === 'openclaw' && ['delivery-mirror', 'gateway-injected'].includes(message.model)) && (!Array.isArray(message.content) || !message.content.some((item) => ['toolCall', 'toolUse', 'tool_use', 'toolResult', 'tool_result'].includes(item?.type))));
@ -245,7 +245,7 @@ flow:
- expr: liveTurnTimeoutMs(env, 60000)
- 250
- assert:
expr: "(parentHistory.messages ?? []).some((message) => message?.role === 'assistant' && Array.isArray(message.content) && message.content.some((item) => (item?.type === 'toolCall' || item?.type === 'toolUse') && (item?.name ?? item?.toolName) === 'exec' && item?.id === parentSuccessfulToolEvents[2].toolCallId && (item?.arguments ?? item?.input)?.command === execCommand.replace('__CHILD_COMPLETION_TOKEN__', childTerminalMarker)))"
expr: "projectQaToolMessages(parentHistory.messages ?? []).some((message) => message?.role === 'assistant' && Array.isArray(message.content) && message.content.some((item) => (item?.type === 'toolCall' || item?.type === 'toolUse') && (item?.name ?? item?.toolName) === 'exec' && item?.id === parentSuccessfulToolEvents[2].toolCallId && (item?.arguments ?? item?.input)?.command === execCommand.replace('__CHILD_COMPLETION_TOKEN__', childTerminalMarker)))"
message: parent exec did not use the exact delivered-completion command
- call: waitForOutboundMessage
saveAs: parentOutbound