diff --git a/extensions/qa-lab/src/live-scenario-timeouts.test.ts b/extensions/qa-lab/src/live-scenario-timeouts.test.ts index 506dae070a24..c4eb78365335 100644 --- a/extensions/qa-lab/src/live-scenario-timeouts.test.ts +++ b/extensions/qa-lab/src/live-scenario-timeouts.test.ts @@ -157,9 +157,11 @@ function runCompletionPolicyFlow( message.direction === "inbound" && message.conversation.id === "issue-109025-completion", )?.text ?? ""; - const commandTemplate = inboundText.match(/run this exact command: ([^\n]+)/u)?.[1]; - const childMarker = workspaceWrites.at(-1)?.content.trim(); - if (!commandTemplate || !childMarker) { + const childReply = workspaceWrites.at(-1)?.content.trim(); + const deliveredExecCommand = childReply?.match( + /REQUESTER_ACTION: Call exec exactly once with this command: (.+) Then reply with exactly the command's trimmed stdout\.$/u, + )?.[1]; + if (!inboundText || !deliveredExecCommand) { throw new Error("completion fixture is missing its command or child marker"); } const execToolCall = { @@ -167,9 +169,7 @@ function runCompletionPolicyFlow( id: params.parentExecToolCallId ?? completionExecToolCallId, name: "exec", arguments: { - command: - params.parentExecCommand ?? - commandTemplate.replace("__CHILD_COMPLETION_TOKEN__", childMarker), + command: params.parentExecCommand ?? deliveredExecCommand, }, }; const parentReplyIsVisible = @@ -261,7 +261,8 @@ function runCompletionPolicyFlow( }, fs: { readFile: async () => { - const childMarker = workspaceWrites.at(-1)?.content.trim() ?? ""; + const childMarker = + workspaceWrites.at(-1)?.content.match(/^(CHILD_DONE:[0-9a-f-]+)/u)?.[1] ?? ""; const completionText = params.proofCompletionText === undefined || params.proofCompletionText === "__DELIVERED_CHILD_TOKEN__" @@ -881,17 +882,24 @@ describe("live subagent scenario timeouts", () => { chainFileNames.slice(1), ); - const terminalMarker = chainWrites.at(-1)?.content.trim(); - expect(terminalMarker).toMatch(/^CHILD_DONE:[0-9a-f-]+$/u); + const terminalReply = chainWrites.at(-1)?.content.trim() ?? ""; + const terminalToken = terminalReply.match(/^CHILD_DONE:[0-9a-f-]+/u)?.[0]; + expect(terminalToken).toBeDefined(); + expect(terminalReply).toContain("REQUESTER_ACTION: Call exec exactly once with this command:"); + expect(terminalReply).toContain("Then reply with exactly the command's trimmed stdout."); const inboundText = state.getSnapshot().messages[0]?.text ?? ""; expect(inboundText).toContain(chainFileNames[0]); - expect(inboundText).toContain(`node ${JSON.stringify(helperFileName)}`); - expect(inboundText).toContain("__CHILD_COMPLETION_TOKEN__"); + expect(terminalReply).toContain( + `node ${JSON.stringify(helperFileName)} ${JSON.stringify(terminalToken)}`, + ); + expect(inboundText).not.toContain(helperFileName); + expect(inboundText).not.toContain("__CHILD_COMPLETION_TOKEN__"); expect(inboundText).not.toContain("node -e"); for (const chainFileName of chainFileNames.slice(1)) { expect(inboundText).not.toContain(chainFileName); } - expect(inboundText).not.toContain(terminalMarker); + expect(inboundText).not.toContain(terminalReply); + expect(inboundText).not.toContain(terminalToken); }); it("accepts authenticated spawn, yield, completion, and exec in the same millisecond", async () => { diff --git a/qa/scenarios/agents/issue-109025-completion-policy-live.yaml b/qa/scenarios/agents/issue-109025-completion-policy-live.yaml index 9208b35ceffb..f089b72b52c2 100644 --- a/qa/scenarios/agents/issue-109025-completion-policy-live.yaml +++ b/qa/scenarios/agents/issue-109025-completion-policy-live.yaml @@ -75,7 +75,7 @@ flow: - set: chainFiles value: expr: "[1, 2, 3, 4].map(() => `${config.chainFilePrefix}-${randomUUID()}.txt`)" - - set: childTerminalMarker + - set: childCompletionToken value: expr: "`CHILD_DONE:${randomUUID()}`" - set: proofFile @@ -103,7 +103,10 @@ flow: ` - set: execCommand value: - expr: "`node ${JSON.stringify(helperFile)} __CHILD_COMPLETION_TOKEN__`" + expr: "`node ${JSON.stringify(helperFile)} ${JSON.stringify(childCompletionToken)}`" + - set: childTerminalReply + value: + expr: "`${childCompletionToken} REQUESTER_ACTION: Call exec exactly once with this command: ${execCommand} Then reply with exactly the command's trimmed stdout.`" - call: fs.writeFile args: - ref: helperPath @@ -127,7 +130,7 @@ flow: - call: fs.writeFile args: - expr: path.join(env.gateway.workspaceDir, chainFiles[3]) - - expr: "`${childTerminalMarker}\n`" + - expr: "`${childTerminalReply}\n`" - utf8 - set: attemptStartedAt value: @@ -153,9 +156,7 @@ flow: `Issue 109025 dormant completion probe. Follow these steps exactly. 1. Call sessions_spawn exactly once with runtime subagent, label ${childLabel}, and this task: "Use only the filesystem read tool. Start by reading ${chainFiles[0]}. Its trimmed contents name the next workspace file to read. Continue following one filename at a time until a file's trimmed contents start with CHILD_DONE:. You must perform each read serially because later filenames are not in this task. Reply with the exact terminal file contents and nothing else. Do not use exec or process." 2. Immediately after sessions_spawn succeeds, call sessions_yield. Do not wait for the child and do not send a normal final reply. - 3. When the child's CHILD_DONE: completion event later resumes you, use the exec tool exactly once to run this exact command: ${execCommand} - Replace only __CHILD_COMPLETION_TOKEN__ with the exact CHILD_DONE: completion text you received from the child. Do not otherwise change the command or inspect the workspace. - Do not use read or write for this step. Then reply with exactly the command's trimmed stdout.` + 3. When the child's CHILD_DONE: completion event later resumes you, follow the REQUESTER_ACTION in that delivered child reply exactly. Do not use read or write during the resumed step.` - call: waitForCondition saveAs: expectedFinalMarker args: @@ -231,7 +232,7 @@ flow: - ref: env - expr: completedChild.childSessionKey - assert: - expr: "String(childTranscript.finalText ?? '').trim() === childTerminalMarker" + expr: "String(childTranscript.finalText ?? '').trim() === childTerminalReply" message: child did not finish with the exact completion reply - assert: expr: "(childTranscript.successfulToolCallCounts.read ?? 0) >= chainFiles.length" @@ -243,7 +244,7 @@ flow: value: expr: "(await fs.readFile(proofPath, 'utf8')).trim().split('\\n')" - assert: - expr: "execProofContents.length === 2 && execProofContents[0] === expectedFinalMarker && execProofContents[1] === childTerminalMarker" + expr: "execProofContents.length === 2 && execProofContents[0] === expectedFinalMarker && execProofContents[1] === childCompletionToken" message: completion proof did not contain the delivered child marker # Raw transcript finalText also includes commentary and mirror bookkeeping; # only projected history proves a separate requester-visible final reply. @@ -266,7 +267,7 @@ flow: - expr: liveTurnTimeoutMs(env, 60000) - 250 - assert: - expr: "projectQaToolMessages(parentHistory.messages ?? []).some((message) => message?.role === 'assistant' && Array.isArray(message.content) && message.content.some((item) => (item?.type === 'toolCall' || item?.type === 'toolUse') && (item?.name ?? item?.toolName) === 'exec' && item?.id === parentSuccessfulToolEvents[2].toolCallId && (item?.arguments ?? item?.input)?.command === execCommand.replace('__CHILD_COMPLETION_TOKEN__', childTerminalMarker)))" + expr: "projectQaToolMessages(parentHistory.messages ?? []).some((message) => message?.role === 'assistant' && Array.isArray(message.content) && message.content.some((item) => (item?.type === 'toolCall' || item?.type === 'toolUse') && (item?.name ?? item?.toolName) === 'exec' && item?.id === parentSuccessfulToolEvents[2].toolCallId && (item?.arguments ?? item?.input)?.command === execCommand))" message: parent exec did not use the exact delivered-completion command - call: waitForOutboundMessage saveAs: parentOutbound