diff --git a/apps/android/app/src/test/java/ai/openclaw/app/NodeForegroundServiceTest.kt b/apps/android/app/src/test/java/ai/openclaw/app/NodeForegroundServiceTest.kt index 5a3beb88f8ae..9ae7509d85d8 100644 --- a/apps/android/app/src/test/java/ai/openclaw/app/NodeForegroundServiceTest.kt +++ b/apps/android/app/src/test/java/ai/openclaw/app/NodeForegroundServiceTest.kt @@ -153,7 +153,7 @@ class NodeForegroundServiceTest { } @Test - @Config(shadows = [ServiceRuntimePrefsShadow::class]) + @Config(shadows = [ServiceRuntimePrefsShadow::class, SessionDisconnectShadow::class]) fun stopDuringRuntimeConstructionRetiresBackgroundStartup() { val app = RuntimeEnvironment.getApplication() as NodeApp val fixture = Shadow.extract(app) @@ -187,11 +187,11 @@ class NodeForegroundServiceTest { assertFalse(runtime.nodeConnected.value) assertFalse(runtime.isForeground.value) assertEquals("Offline", runtime.gatewayConnectionDisplay.value.statusText) - while (gateway.takeRequest(0, TimeUnit.MILLISECONDS) != null) { - // Construction may have started a socket before Stop retired it. - } + // A startup socket's HTTP upgrade can reach the server after Stop retires it. + // Observe new session admissions instead of the asynchronous server request queue. + fixture.sessionConnections.clear() runtime.setForeground(true) - assertNull("Foreground re-entry must not reconnect a stopped runtime", gateway.takeRequest(10, TimeUnit.SECONDS)) + assertNull("Foreground re-entry must not reconnect a stopped runtime", fixture.sessionConnections.poll(10, TimeUnit.SECONDS)) } finally { gate.release.countDown() fixture.prefsReadGate = null diff --git a/docs/tools/code-mode.md b/docs/tools/code-mode.md index dd9445ef9e03..e18cafc549d3 100644 --- a/docs/tools/code-mode.md +++ b/docs/tools/code-mode.md @@ -265,43 +265,21 @@ one from an unawaited call or timer callback, fails the cell instead of silently reporting success. Handlers attached after a suspension still handle their original promises. -JavaScript syntax errors, TypeScript transform errors, and tool failures proven -to occur before execution become failed `exec` results that the model can read -and correct across successive turns. A failed `exec` or `wait` does not automatically -end the agent run when OpenClaw's host execution record proves that no potentially -mutating nested action started, including before a suspended run resumed. +JavaScript syntax errors, TypeScript transform errors, and uncaught nested tool +failures become failed `exec` or `wait` results. The model can read the error, +correct its code, inspect the current state, and continue with the normal tool +surface. A failed cell does not impose a separate recovery mode or mutation budget. -An exec host-policy rejection can carry this proof even after hooks, approval -resolution, and tool implementation entry: the host owns the narrower fact that -no command process or remote dispatch started. A corrected call runs the ordinary -hooks and approvals again. Consumed voice confirmations stay consumed; recovery -does not restore a grant or authorize replay. +OpenClaw does not automatically replay a failed program. Earlier calls may have +changed state, and a failed call may have partially applied. Inspect authoritative +state before deciding what remains, and do not repeat completed actions. This +also applies when `wait` resumes a suspended cell: its earlier calls belong to the +same program. -Catalog search, handle `describe()`, `skills.list()`, and `skills.read()` are -read-only discovery. A guest error after only these operations still allows -ordinary recovery from a failed `exec`; discovery does not count as a mutation. - -OpenClaw does not automatically replay a failed program. If earlier calls -may have changed state or a failed call may have partially applied, OpenClaw -deliberately permits one temporary read-only recovery attempt to inspect the -current state. The internal instruction identifies OpenClaw as its source. It -does not expose writes, sends, shell commands, or other mutations during that -inspection. - -If inspection finds unfinished work, the model can request one bounded recovery. -Code Mode stays disabled, and OpenClaw restores the normal direct-tool or Tool -Search surface with its real tool names and argument schemas. Host-recorded -nested-call facts block an exact repeat whose earlier effect was committed or -uncertain. The recovery permits one mutation attempt; reads and schema discovery -remain available afterward, but a later mutation does not run blindly when the -first attempt fails. If no work remains, the inspection report ends the run -without another model turn. Cancellation, explicitly terminal tool outcomes, -sandbox restrictions, approval requirements, and tool-policy denials retain -their existing behavior. - -Computer observations, including window and cursor queries, cropped screenshots, -browser state, and dialog inspection, do not spend that mutation attempt. Browser -preparation, input, and dialog acceptance or dismissal still count as mutations. +Every subsequent call runs the ordinary hooks and approvals again. Consumed voice +confirmations stay consumed; continuing after an error does not restore a grant. +Cancellation, explicitly terminal tool outcomes, sandbox restrictions, approval +requirements, and tool-policy denials retain their existing behavior. ### Verify the active surface @@ -1164,15 +1142,11 @@ reconstructed from source code or outer results. Nested tool failures cross into the guest as catchable JavaScript errors. If guest code does not catch an error, `exec` or `wait` returns a failed tool -result. Proven no-start failures and errors after only audited read-only work -allow ordinary model recovery, including when `wait` resumes a suspended cell. -This proof covers the cell's entire execution, not just the latest resume. -Failed waits without that host proof remain terminal; serialized result fields -cannot grant recovery. Possible nested side effects in a failed `exec` require -the [read-only inspection and bounded recovery](#recover-from-tool-errors) flow -before any further action. Network-controlled tool output and errors retain -their existing untrusted-content wrapping and sanitization; recovering from a -failure does not grant new permissions or replay completed side effects. +result and the agent can continue normally. Follow the +[tool-error guidance](#recover-from-tool-errors) to inspect possible partial +effects before choosing another action. Network-controlled tool output and errors +retain their existing untrusted-content wrapping and sanitization; continuing +after a failure does not grant new permissions or replay completed side effects. Parallel nested calls are allowed up to `maxPendingToolCalls`. An oversized raw tool batch fails before any call in that batch is dispatched. [Swarm](/tools/swarm) diff --git a/qa/scenarios/channels/qa-channel-failed-tool-terminal-finalization.yaml b/qa/scenarios/channels/qa-channel-failed-tool-terminal-finalization.yaml index 0f8c1fa0364e..366b67d40fde 100644 --- a/qa/scenarios/channels/qa-channel-failed-tool-terminal-finalization.yaml +++ b/qa/scenarios/channels/qa-channel-failed-tool-terminal-finalization.yaml @@ -18,7 +18,8 @@ scenario: - The failed tool call is never replayed or described as successful. - The model receives one ordinary continuation with its failed result and qa-channel delivers exactly one visible reply. codeRefs: - - src/agents/embedded-agent-runner/run/code-mode-outcome.ts + - packages/agent-core/src/agent-loop.ts + - src/agents/embedded-agent-runner/extensions.ts - extensions/qa-lab/src/providers/mock-openai/server.ts execution: kind: flow diff --git a/qa/scenarios/plugins/clawhub-release-policy-contracts.yaml b/qa/scenarios/plugins/clawhub-release-policy-contracts.yaml index a8e07a598f5f..e4accd1df463 100644 --- a/qa/scenarios/plugins/clawhub-release-policy-contracts.yaml +++ b/qa/scenarios/plugins/clawhub-release-policy-contracts.yaml @@ -18,7 +18,7 @@ scenario: - A committed source change without a version bump fails the ClawHub release check. - The same source change passes after the package version and compatibility metadata are bumped and committed. docsRefs: - - docs/clawhub/publishing.md + - docs/plugins/building-plugins.md - docs/plugins/community.md - docs/help/testing-updates-plugins.md codeRefs: diff --git a/qa/scenarios/scheduling/cron-failed-tool-terminal-finalization.yaml b/qa/scenarios/scheduling/cron-failed-tool-terminal-finalization.yaml index 3d3307eac77e..7055c0424573 100644 --- a/qa/scenarios/scheduling/cron-failed-tool-terminal-finalization.yaml +++ b/qa/scenarios/scheduling/cron-failed-tool-terminal-finalization.yaml @@ -21,7 +21,8 @@ scenario: - The failed tool never replays and qa-channel receives exactly one final reply. codeRefs: - src/cron/isolated-agent/run-executor.ts - - src/agents/embedded-agent-runner/run/code-mode-outcome.ts + - packages/agent-core/src/agent-loop.ts + - src/agents/embedded-agent-runner/extensions.ts - extensions/qa-lab/src/providers/mock-openai/server.ts execution: kind: flow diff --git a/src/agents/code-mode-bridge.ts b/src/agents/code-mode-bridge.ts index d716a3cd33fc..d777c1573607 100644 --- a/src/agents/code-mode-bridge.ts +++ b/src/agents/code-mode-bridge.ts @@ -17,16 +17,7 @@ import { consumeMcpCodeModeGuestResult } from "./mcp-content.js"; import type { AgentToolUpdateCallback } from "./runtime/index.js"; import { isCollectorSpawnTool } from "./subagents/swarm/swarm-collector-capability.js"; import { resolveSwarmConfig } from "./subagents/swarm/swarm-config.js"; -import { - consumeToolEffectReceipt, - registerToolEffectReceipt, - type ToolEffectReceipt, -} from "./tool-effect-receipt.js"; import { isToolExecutionAllowed, TOOL_EXECUTION_GATED_MESSAGE } from "./tool-policy-shared.js"; -import { - consumeTrustedToolNoStartError, - registerTrustedToolNoStartError, -} from "./tool-result-error.js"; import type { ToolSearchRuntime } from "./tool-search-runtime.js"; import type { ToolSearchCatalogEntry, ToolSearchToolContext } from "./tool-search-types.js"; import { ToolInputError } from "./tools/common.js"; @@ -224,7 +215,6 @@ export async function runBridgeRequest(params: { onUpdate?: AgentToolUpdateCallback; }): Promise { const catalogProjection = params.catalogProjection; - let effectReceipt: ToolEffectReceipt | undefined; try { const values = Array.isArray(params.request.args) ? params.request.args : []; let value: unknown; @@ -300,7 +290,6 @@ export async function runBridgeRequest(params: { signal: params.signal, onUpdate: params.onUpdate, }); - effectReceipt = consumeToolEffectReceipt(called.result); value = isRecord(called.result) && "details" in called.result ? called.result.details @@ -351,7 +340,6 @@ export async function runBridgeRequest(params: { signal: params.signal, onUpdate: params.onUpdate, }); - effectReceipt = consumeToolEffectReceipt(called.result); if (request.catalogId) { const guestResult = consumeMcpCodeModeGuestResult(called.result); if (guestResult === undefined) { @@ -366,7 +354,6 @@ export async function runBridgeRequest(params: { : called.result; }, ); - effectReceipt ??= consumeToolEffectReceipt(value); break; } case "agentSpawn": @@ -423,24 +410,16 @@ export async function runBridgeRequest(params: { "Search results exceed the output budget. Narrow the query or lower the limit.", ); } - const settled: SettledBridgeRequest = { id: params.request.id, ok: true, value }; - return effectReceipt ? registerToolEffectReceipt(settled, effectReceipt) : settled; + return { id: params.request.id, ok: true, value }; } catch (error) { const boundedError = boundCodeModeError( redactCodeModeCatalogIds(formatErrorMessage(error), catalogProjection.bindings), params.maxOutputBytes, ); - const settled: SettledBridgeRequest = { + return { id: params.request.id, ok: false, error: boundedError, }; - const trustedNoStart = consumeTrustedToolNoStartError(error); - if (trustedNoStart) { - registerTrustedToolNoStartError(settled); - } - effectReceipt = - consumeToolEffectReceipt(error) ?? (trustedNoStart ? { state: "not_started" } : undefined); - return effectReceipt ? registerToolEffectReceipt(settled, effectReceipt) : settled; } } diff --git a/src/agents/code-mode-execution.ts b/src/agents/code-mode-execution.ts index 20dc19e58808..65f6abde422d 100644 --- a/src/agents/code-mode-execution.ts +++ b/src/agents/code-mode-execution.ts @@ -12,7 +12,6 @@ import { createCodeModeNamespaceRuntime, type CodeModeNamespaceRuntime, } from "./code-mode-namespaces.js"; -import { registerRepairableCodeModeFailure } from "./code-mode-repair-provenance.js"; import { CODE_MODE_WORKER_WATCHDOG_GRACE_MS, codeModeFailureCode, @@ -35,7 +34,6 @@ import { createCodeModeRunOwner, createPendingBridgeStates, disposeCodeModeRun, - isCodeModeBridgeRepairEligible, pendingBridgeRequestsReplaySafe, pendingBridgeStatesForSettlement, pendingToolCalls, @@ -420,7 +418,7 @@ async function settleCodeModeResult(params: { { error: result.pendingRequests.every((request) => request.method === "namespace") ? "restart-safe code mode cannot call namespace tools." - : "restart-safe code mode cannot call tool surfaces that are not proven replay-safe; recovery runs must use audited read, grep, or find tools.", + : "restart-safe code mode cannot call tool surfaces that are not proven replay-safe; use audited read, grep, or find tools.", }, params.runtime.hasNetworkContent(), ); @@ -494,11 +492,7 @@ async function settleCodeModeResult(params: { replaySafe: params.replaySafe, telemetry: telemetry(params.runtime), }; - const finalized = output.takeResult(metadata, channels, params.runtime.hasNetworkContent()); - if (finalized.status === "failed" && isCodeModeBridgeRepairEligible(params.bridgeDispatch)) { - registerRepairableCodeModeFailure(finalized); - } - return finalized; + return output.takeResult(metadata, channels, params.runtime.hasNetworkContent()); } export async function runWait(params: { diff --git a/src/agents/code-mode-namespaces.ts b/src/agents/code-mode-namespaces.ts index e8dfbe49b861..04cd93d3f241 100644 --- a/src/agents/code-mode-namespaces.ts +++ b/src/agents/code-mode-namespaces.ts @@ -18,7 +18,6 @@ import { type CodeModeApiVirtualFile, type McpApiServerDoc, } from "./code-mode-mcp-api.js"; -import { registerToolEffectReceipt } from "./tool-effect-receipt.js"; export type { CodeModeApiVirtualFile } from "./code-mode-mcp-api.js"; @@ -627,17 +626,9 @@ export function createCodeModeNamespaceRuntime( if (!isCodeModeNamespaceToolCall(target)) { throw new Error(`Code mode namespace path is not callable: ${path.join(".")}`); } - let input: unknown; - try { - input = target.input ? await target.input(args) : (args[0] ?? {}); - } catch (error) { - if (target.local) { - throw registerToolEffectReceipt(error, { state: "failed_no_effect" }); - } - throw error; - } + const input = target.input ? await target.input(args) : (args[0] ?? {}); if (target.local) { - return registerToolEffectReceipt(toCodeModeJsonSafe(input), { state: "read_completed" }); + return toCodeModeJsonSafe(input); } if (!target.catalogId) { throw new Error(`Code mode namespace path has no catalog tool: ${path.join(".")}`); diff --git a/src/agents/code-mode-nodes.test.ts b/src/agents/code-mode-nodes.test.ts index 40bb89e5642d..d27a2b8b4bf9 100644 --- a/src/agents/code-mode-nodes.test.ts +++ b/src/agents/code-mode-nodes.test.ts @@ -43,7 +43,6 @@ const nodeSnapshot = [ let applyCodeModeCatalog: typeof import("./code-mode.js").applyCodeModeCatalog; let createCodeModeTools: typeof import("./code-mode.js").createCodeModeTools; -let consumeRepairableCodeModeFailure: typeof import("./code-mode-repair-provenance.js").consumeRepairableCodeModeFailure; let createToolSearchCatalogRef: typeof import("./tool-search.js").createToolSearchCatalogRef; let createNodesTool: typeof import("./tools/nodes-tool.js").createNodesTool; let testing: typeof import("./code-mode.test-support.js").testing; @@ -115,7 +114,6 @@ describe("Code Mode nodes", () => { beforeAll(async () => { vi.resetModules(); ({ applyCodeModeCatalog, createCodeModeTools } = await import("./code-mode.js")); - ({ consumeRepairableCodeModeFailure } = await import("./code-mode-repair-provenance.js")); ({ createToolSearchCatalogRef } = await import("./tool-search.js")); ({ createNodesTool } = await import("./tools/nodes-tool.js")); ({ testing } = await import("./code-mode.test-support.js")); @@ -293,7 +291,7 @@ describe("Code Mode nodes", () => { label: "get", code: `return (await nodes.get("Desk")).describe();`, }, - ])("keeps a guest error after nodes.$label eligible for ordinary recovery", async ({ code }) => { + ])("reports a guest error after nodes.$label", async ({ code }) => { const details = await runUntilCompleted({ ...createHarness(), code }); expect(details).toMatchObject({ @@ -301,10 +299,9 @@ describe("Code Mode nodes", () => { failurePhase: "bridge", bridgeDispatchStarted: true, }); - expect(consumeRepairableCodeModeFailure(details)).toBe(true); }); - it("keeps a guest error after nodes.invoke restricted", async () => { + it("reports a guest error after nodes.invoke without replaying the invocation", async () => { const details = await runUntilCompleted({ ...createHarness(), code: ` @@ -319,7 +316,6 @@ describe("Code Mode nodes", () => { failurePhase: "bridge", bridgeDispatchStarted: true, }); - expect(consumeRepairableCodeModeFailure(details)).toBe(false); expect(gatewayMocks.callGatewayTool).toHaveBeenCalledWith( "node.invoke", expect.anything(), diff --git a/src/agents/code-mode-repair-provenance.ts b/src/agents/code-mode-permission-change.ts similarity index 53% rename from src/agents/code-mode-repair-provenance.ts rename to src/agents/code-mode-permission-change.ts index cd3e7ca13abf..598438b4f315 100644 --- a/src/agents/code-mode-repair-provenance.ts +++ b/src/agents/code-mode-permission-change.ts @@ -1,6 +1,4 @@ -const repairableFailureDetails = new WeakSet(); const permissionChangeReasons = new WeakSet(); -const permissionChangedFailureDetails = new WeakSet(); /** Mint the exact host-owned reason for an operator's permission transition. */ export function createCodeModePermissionChangeReason(): Error { @@ -24,27 +22,5 @@ export function markCodeModePermissionChangeResult( ) { details.error = "Permission change interrupted this Code Mode program. Continue the current task using the updated permissions. Do not replay this program or repeat completed actions. Any in-flight action may have partially applied; inspect authoritative state before deciding what work remains."; - permissionChangedFailureDetails.add(details); } } - -/** Only the exact host-finalized cancellation result may continue under the new policy. */ -export function consumeCodeModePermissionChangeResult(details: unknown): boolean { - return ( - typeof details === "object" && - details !== null && - permissionChangedFailureDetails.delete(details) - ); -} - -/** Attach host-only repair authority to one finalized Code Mode failure payload. */ -export function registerRepairableCodeModeFailure(details: object): void { - repairableFailureDetails.add(details); -} - -/** Consume repair authority from the exact host-created failure payload. */ -export function consumeRepairableCodeModeFailure(details: unknown): boolean { - return ( - typeof details === "object" && details !== null && repairableFailureDetails.delete(details) - ); -} diff --git a/src/agents/code-mode-state.ts b/src/agents/code-mode-state.ts index 85e3a21f01d6..d1d4cb4cb3b3 100644 --- a/src/agents/code-mode-state.ts +++ b/src/agents/code-mode-state.ts @@ -17,19 +17,12 @@ import type { SettledBridgeRequest, } from "./code-mode-runtime.js"; import type { AgentToolUpdateCallback } from "./runtime/index.js"; -import { - consumeToolEffectReceipt, - toolEffectStateProvesNoEffect, - type ToolEffectReceipt, -} from "./tool-effect-receipt.js"; -import { consumeTrustedToolNoStartError } from "./tool-result-error.js"; import type { ToolSearchRuntime } from "./tool-search-runtime.js"; import type { ToolSearchToolContext } from "./tool-search-types.js"; import { ToolInputError } from "./tools/common.js"; export type CodeModeBridgeDispatchState = { started: boolean; - operations: Map; }; export type PendingBridgeState = PendingBridgeRequest & { @@ -137,18 +130,7 @@ export function createCodeModeRunOwner(ctx: ToolSearchToolContext) { } export function createCodeModeBridgeDispatchState(): CodeModeBridgeDispatchState { - return { started: false, operations: new Map() }; -} - -/** Read the host-only side-effect classification for one Code Mode run. */ -export function isCodeModeBridgeRepairEligible(state: CodeModeBridgeDispatchState): boolean { - return ( - state.started && - state.operations.size > 0 && - [...state.operations.values()].every( - (effect) => effect !== "pending" && toolEffectStateProvesNoEffect(effect), - ) - ); + return { started: false }; } // One unreferenced timer owns parked snapshots even when no later exec or wait @@ -385,15 +367,8 @@ export function createPendingBridgeStates( if (params.signal.aborted) { onAbort(); } - const tracksDispatch = request.method !== "sleep"; - // Discovery is read-only; replay-safe actions such as agentSpawn may still mutate. - const recoverySafe = - ["search", "describe", "skillsList", "skillsRead"].includes(request.method) || - (["nodes", "callValue"].includes(request.method) && - isPendingBridgeRequestReplaySafe(request, params.runtime, params.catalogProjection)); - if (tracksDispatch) { + if (request.method !== "sleep") { params.bridgeDispatch.started = true; - params.bridgeDispatch.operations.set(request.id, "pending"); } const bridgeCall = runBridgeRequest({ runtime: params.runtime, @@ -417,17 +392,6 @@ export function createPendingBridgeStates( ...request, promise: completion.then((settled) => { params.signal.removeEventListener("abort", onAbort); - // The effect receipt owns classification; consume the predecessor marker - // so reusing this settled object cannot preserve stale no-start authority. - consumeTrustedToolNoStartError(settled); - const effectReceipt = consumeToolEffectReceipt(settled); - if (tracksDispatch) { - params.bridgeDispatch.operations.set( - request.id, - effectReceipt?.state ?? - (recoverySafe ? (settled.ok ? "read_completed" : "failed_no_effect") : "uncertain"), - ); - } state.settledSequence = ++nextPendingBridgeSettlementSequence; state.settled = settled; // Only the response is needed until guest replay; live calls keep their own request. diff --git a/src/agents/code-mode-swarm.test.ts b/src/agents/code-mode-swarm.test.ts index 4be93f68171c..af73eba627fc 100644 --- a/src/agents/code-mode-swarm.test.ts +++ b/src/agents/code-mode-swarm.test.ts @@ -4,7 +4,6 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; import { createDeferred } from "../../test/helpers/promise.js"; import { CodeModeOutputState } from "./code-mode-json.js"; import { createCodeModeNamespaceRuntime } from "./code-mode-namespaces.js"; -import { consumeRepairableCodeModeFailure } from "./code-mode-repair-provenance.js"; import type { CodeModeWorkerResult } from "./code-mode-runtime.js"; import { applyCodeModeCatalog, resolveCodeModeConfig } from "./code-mode.js"; import { @@ -335,7 +334,7 @@ describe("Code Mode swarm guest", () => { } }); - it("keeps a guest error after a parked agents.run restricted without replaying the collector", async () => { + it("reports a guest error after a parked agents.run without replaying the collector", async () => { const collectorStarted = createDeferred(); const collectorRelease = createDeferred(); swarmMocks.waitForCollectorCompletion.mockImplementation(async () => { @@ -371,7 +370,6 @@ describe("Code Mode swarm guest", () => { bridgeDispatchStarted: true, error: expect.stringContaining("ReferenceError: missingAfterCollector is not defined"), }); - expect(consumeRepairableCodeModeFailure(details)).toBe(false); expect(harness.spawnTool.execute).toHaveBeenCalledOnce(); expect(swarmMocks.waitForCollectorCompletion).toHaveBeenCalledOnce(); expect(testing.activeRuns.size).toBe(0); diff --git a/src/agents/code-mode.agent-loop.test.ts b/src/agents/code-mode.agent-loop.test.ts index 5f4d2001a261..76fd67810b90 100644 --- a/src/agents/code-mode.agent-loop.test.ts +++ b/src/agents/code-mode.agent-loop.test.ts @@ -11,10 +11,7 @@ import { afterEach, describe, expect, it, vi } from "vitest"; import { createDeferred } from "../../test/helpers/promise.js"; import { setPluginToolMeta } from "../plugins/tool-metadata.js"; import { wrapToolWithAbortSignal } from "./agent-tools.abort.js"; -import { - consumeRepairableCodeModeFailure, - createCodeModePermissionChangeReason, -} from "./code-mode-repair-provenance.js"; +import { createCodeModePermissionChangeReason } from "./code-mode-permission-change.js"; import type { CodeModeSkill } from "./code-mode-skills.js"; import { createSubscribedCodeModeHarness } from "./code-mode.bridge.lifecycle.test-support.js"; import { applyCodeModeCatalog, createCodeModeTools } from "./code-mode.js"; @@ -27,7 +24,6 @@ import { resultDetails, testing, } from "./code-mode.test-support.js"; -import { installCodeModeOutcomeHook } from "./embedded-agent-runner/run/code-mode-outcome.js"; import { Agent } from "./runtime/index.js"; import { createReadTool } from "./sessions/tools/read.js"; import { isToolResultError, readToolResultDetails } from "./tool-result-error.js"; @@ -101,7 +97,6 @@ async function runCodeModeAgent(params: { codeModeSkills: params.codeModeSkills, }); const providerContexts: Context[] = []; - let reconciliationCandidates = 0; const agent = new Agent({ initialState: { model, tools }, afterToolCall: async ({ result, isError }) => ({ @@ -139,12 +134,6 @@ async function runCodeModeAgent(params: { }, }); params.configureAgent?.(agent, { tools, ctx }); - installCodeModeOutcomeHook({ - agent, - onReconciliationCandidate: () => { - reconciliationCandidates += 1; - }, - }); try { await agent.prompt("finish the task despite tool errors"); } finally { @@ -153,7 +142,7 @@ async function runCodeModeAgent(params: { } } - return { agent, providerContexts, reconciliationCandidates }; + return { agent, providerContexts }; } type ParkedFailure = @@ -262,7 +251,6 @@ async function runParkedReadFailure(scenario: ParkedFailure) { expect(read.execute).toHaveBeenCalledOnce(); expect(testing.activeRuns.size).toBe(0); expect(testing.resumingRunIds.size).toBe(0); - expect(result.reconciliationCandidates).toBe(0); expect(waitDetails).toMatchObject( scenario === "cancel wait" ? { status: "failed", code: "aborted" } @@ -292,6 +280,60 @@ describe("Code Mode agent-loop error recovery", () => { vi.useRealTimers(); }); + it("inspects source and completes multiple edits after a partially applied program fails", async () => { + const workspace = await realpath(await mkdtemp(join(tmpdir(), "code-mode-continue-"))); + const file = join(workspace, "source.txt"); + await writeFile(file, "original\n"); + const patch = pluginToolWithExecute( + "apply_patch", + "Append a source change", + async (_id, input) => { + await writeFile( + file, + `${await readFile(file, "utf8")}${(input as { value: string }).value}\n`, + ); + return jsonResult({ applied: true }); + }, + ); + const shell = pluginToolWithExecute("shell_command", "Inspect source", async () => + jsonResult({ source: await readFile(file, "utf8") }), + ); + + try { + const { agent, providerContexts } = await runCodeModeAgent({ + hiddenTools: [patch, shell], + programs: [ + 'await apply_patch({ value: "first" }); throw new Error("interrupted after first change");', + "return await shell_command({});", + 'return await apply_patch({ value: "second" });', + 'return await apply_patch({ value: "third" });', + "return await shell_command({});", + ], + }); + + expect(providerContexts).toHaveLength(6); + expect(providerContexts[1]?.messages).toContainEqual( + expect.objectContaining({ + role: "toolResult", + isError: true, + details: expect.objectContaining({ + status: "failed", + error: expect.stringContaining("interrupted after first change"), + }), + }), + ); + expect(patch.execute).toHaveBeenCalledTimes(3); + expect(shell.execute).toHaveBeenCalledTimes(2); + expect(await readFile(file, "utf8")).toBe("original\nfirst\nsecond\nthird\n"); + expect(agent.state.messages.at(-1)).toMatchObject({ + role: "assistant", + content: [{ type: "text", text: "recovered" }], + }); + } finally { + await rm(workspace, { recursive: true, force: true }); + } + }); + it("recovers from a real parked read failure and completes the requested mutation once", async () => { const { agent, providerContexts, complete, effects } = await runParkedReadFailure("read-only"); expect(providerContexts).toHaveLength(4); @@ -303,21 +345,27 @@ describe("Code Mode agent-loop error recovery", () => { }); }); - it.each([ - "earlier mutation", - "terminal read", - "copied details", - "cancel wait", - "abort outcome", - ] as const)("prevents another provider turn after a parked failure with %s", async (scenario) => { - const { providerContexts, complete, effects, waitDetails } = - await runParkedReadFailure(scenario); - expect(providerContexts).toHaveLength(2); - expect(complete.execute).toHaveBeenCalledTimes(scenario === "earlier mutation" ? 1 : 0); - expect(effects).toEqual(scenario === "earlier mutation" ? ["completed"] : []); - // Replacing details before the outcome boundary must leave the original proof unused. - expect(consumeRepairableCodeModeFailure(waitDetails)).toBe(scenario === "copied details"); - }); + it.each(["earlier mutation", "copied details"] as const)( + "continues after a parked failure with %s without replaying prior work", + async (scenario) => { + const { providerContexts, complete, effects } = await runParkedReadFailure(scenario); + expect(providerContexts).toHaveLength(4); + expect(complete.execute).toHaveBeenCalledTimes(scenario === "earlier mutation" ? 2 : 1); + expect(effects).toEqual( + scenario === "earlier mutation" ? ["completed", "completed"] : ["completed"], + ); + }, + ); + + it.each(["terminal read", "cancel wait", "abort outcome"] as const)( + "prevents another provider turn after a parked failure with %s", + async (scenario) => { + const { providerContexts, complete, effects } = await runParkedReadFailure(scenario); + expect(providerContexts).toHaveLength(2); + expect(complete.execute).not.toHaveBeenCalled(); + expect(effects).toEqual([]); + }, + ); it("continues after an operator permission change without replaying earlier mutations", async () => { const generation = new AbortController(); @@ -384,10 +432,9 @@ describe("Code Mode agent-loop error recovery", () => { }); await parked.promise; changePermissions(); - const { agent, providerContexts, reconciliationCandidates } = await running; + const { agent, providerContexts } = await running; expect(providerContexts).toHaveLength(4); - expect(reconciliationCandidates).toBe(0); expect(applied).toEqual(["prior mutation", "remaining mutation"]); expect(recordEffect.execute).toHaveBeenCalledOnce(); expect(pending.execute).toHaveBeenCalledOnce(); @@ -433,7 +480,7 @@ describe("Code Mode agent-loop error recovery", () => { const complete = pluginToolWithExecute("complete_task", "Complete the task", async () => jsonResult({ completed: true }), ); - const { agent, providerContexts, reconciliationCandidates } = await runCodeModeAgent({ + const { agent, providerContexts } = await runCodeModeAgent({ hiddenTools: [complete], codeModeSkills: [ { @@ -460,7 +507,6 @@ describe("Code Mode agent-loop error recovery", () => { expect(agent.state.messages).toContainEqual(failure); expect(providerContexts).toHaveLength(3); expect(complete.execute).toHaveBeenCalledOnce(); - expect(reconciliationCandidates).toBe(0); expect(agent.state.messages.at(-1)).toMatchObject({ role: "assistant", content: [{ type: "text", text: "recovered" }], @@ -479,7 +525,7 @@ describe("Code Mode agent-loop error recovery", () => { jsonResult({ recovered: true }), ); - const { agent, providerContexts, reconciliationCandidates } = await runCodeModeAgent({ + const { agent, providerContexts } = await runCodeModeAgent({ hiddenTools: [terminal, recover], programs: ["return await terminal({});", "return await recover_task({});"], }); @@ -500,7 +546,6 @@ describe("Code Mode agent-loop error recovery", () => { ); expect(terminal.execute).not.toHaveBeenCalled(); expect(recover.execute).toHaveBeenCalledOnce(); - expect(reconciliationCandidates).toBe(0); expect(agent.state.messages.at(-1)).toMatchObject({ role: "assistant", content: [{ type: "text", text: "recovered" }], @@ -515,7 +560,7 @@ describe("Code Mode agent-loop error recovery", () => { jsonResult({ recovered: true }), ); - const { agent, providerContexts, reconciliationCandidates } = await runCodeModeAgent({ + const { agent, providerContexts } = await runCodeModeAgent({ hiddenTools: [terminal, recover], programs: [ "return await terminal({ value: 42 });", @@ -539,7 +584,6 @@ describe("Code Mode agent-loop error recovery", () => { ); expect(terminal.execute).not.toHaveBeenCalled(); expect(recover.execute).toHaveBeenCalledOnce(); - expect(reconciliationCandidates).toBe(0); expect(agent.state.messages.at(-1)).toMatchObject({ role: "assistant", content: [{ type: "text", text: "recovered" }], @@ -555,7 +599,7 @@ describe("Code Mode agent-loop error recovery", () => { jsonResult({ recovered: true }), ); - const { providerContexts, reconciliationCandidates } = await runCodeModeAgent({ + const { providerContexts } = await runCodeModeAgent({ hiddenTools: [readOnly, recover], programs: ["return await sessions_history({});", "return await recover_task({});"], }); @@ -563,10 +607,9 @@ describe("Code Mode agent-loop error recovery", () => { expect(providerContexts).toHaveLength(3); expect(readOnly.execute).toHaveBeenCalledOnce(); expect(recover.execute).toHaveBeenCalledOnce(); - expect(reconciliationCandidates).toBe(0); }); - it("uses the terminal owner's input-aware read receipt for ordinary recovery", async () => { + it("continues after a failed read through a mixed-action tool", async () => { const mixedAction = fakeTool("message", "Read or mutate messages"); mixedAction.parameters = { type: "object", @@ -580,7 +623,7 @@ describe("Code Mode agent-loop error recovery", () => { jsonResult({ recovered: true }), ); - const { providerContexts, reconciliationCandidates } = await runCodeModeAgent({ + const { providerContexts } = await runCodeModeAgent({ hiddenTools: [mixedAction, recover], harness: createSubscribedCodeModeHarness({ name: "input-aware-read-receipt" }), programs: ['return await message({ action: "read" });', "return await recover_task({});"], @@ -589,10 +632,9 @@ describe("Code Mode agent-loop error recovery", () => { expect(providerContexts).toHaveLength(3); expect(mixedAction.execute).toHaveBeenCalledOnce(); expect(recover.execute).toHaveBeenCalledOnce(); - expect(reconciliationCandidates).toBe(0); }); - it("uses the namespace owner's local read receipt for ordinary recovery", async () => { + it("continues after a namespace metadata call and a guest error", async () => { const listResources = mcpTool({ name: "mcp_files_resources_list", serverName: "files", @@ -603,7 +645,7 @@ describe("Code Mode agent-loop error recovery", () => { jsonResult({ recovered: true }), ); - const { providerContexts, reconciliationCandidates } = await runCodeModeAgent({ + const { providerContexts } = await runCodeModeAgent({ hiddenTools: [listResources, recover], programs: ["json(await MCP.$api()); return missingFn();", "return await recover_task({});"], }); @@ -611,10 +653,9 @@ describe("Code Mode agent-loop error recovery", () => { expect(providerContexts).toHaveLength(3); expect(listResources.execute).not.toHaveBeenCalled(); expect(recover.execute).toHaveBeenCalledOnce(); - expect(reconciliationCandidates).toBe(0); }); - it("uses an exact plugin instance's replay-safe receipt for ordinary recovery", async () => { + it("continues after a replay-safe plugin failure", async () => { const readOnly = pluginToolWithExecute("plugin_read", "Read plugin state", async () => { throw new Error("plugin read failed after dispatch"); }); @@ -627,7 +668,7 @@ describe("Code Mode agent-loop error recovery", () => { jsonResult({ recovered: true }), ); - const { providerContexts, reconciliationCandidates } = await runCodeModeAgent({ + const { providerContexts } = await runCodeModeAgent({ hiddenTools: [readOnly, recover], harness: createSubscribedCodeModeHarness({ name: "plugin-read-receipt" }), programs: ["return await plugin_read({});", "return await recover_task({});"], @@ -636,10 +677,9 @@ describe("Code Mode agent-loop error recovery", () => { expect(providerContexts).toHaveLength(3); expect(readOnly.execute).toHaveBeenCalledOnce(); expect(recover.execute).toHaveBeenCalledOnce(); - expect(reconciliationCandidates).toBe(0); }); - it("keeps replay-safe side-effecting plugin failures in restricted reconciliation", async () => { + it("continues after a side-effecting plugin failure without replaying it", async () => { const appliedChanges: string[] = []; const mutation = pluginToolWithExecute("plugin_mutation", "Mutate plugin state", async () => { appliedChanges.push("plugin state changed"); @@ -655,15 +695,14 @@ describe("Code Mode agent-loop error recovery", () => { jsonResult({ recovered: true }), ); - const { providerContexts, reconciliationCandidates } = await runCodeModeAgent({ + const { providerContexts } = await runCodeModeAgent({ hiddenTools: [mutation, recover], programs: ["return await plugin_mutation({});", "return await recover_task({});"], }); - expect(providerContexts).toHaveLength(1); + expect(providerContexts).toHaveLength(3); expect(mutation.execute).toHaveBeenCalledOnce(); - expect(recover.execute).not.toHaveBeenCalled(); - expect(reconciliationCandidates).toBe(1); + expect(recover.execute).toHaveBeenCalledOnce(); expect(appliedChanges).toEqual(["plugin state changed"]); }); @@ -698,7 +737,7 @@ describe("Code Mode agent-loop error recovery", () => { }); }); - it("blocks another action after an earlier side effect and a later tool failure", async () => { + it("continues after an earlier side effect and a later tool failure", async () => { const recordEffect = pluginToolWithExecute("record_effect", "Record an effect", async () => jsonResult({ recorded: true }), ); @@ -709,19 +748,18 @@ describe("Code Mode agent-loop error recovery", () => { jsonResult({ repeated: true }), ); - const { providerContexts, reconciliationCandidates } = await runCodeModeAgent({ + const { providerContexts } = await runCodeModeAgent({ hiddenTools: [recordEffect, terminal, write], programs: ["await record_effect({}); return await terminal({});", "return await write({});"], }); - expect(providerContexts).toHaveLength(1); + expect(providerContexts).toHaveLength(3); expect(recordEffect.execute).toHaveBeenCalledOnce(); expect(terminal.execute).toHaveBeenCalledOnce(); - expect(write.execute).not.toHaveBeenCalled(); - expect(reconciliationCandidates).toBe(1); + expect(write.execute).toHaveBeenCalledOnce(); }); - it("routes a partially applied mutation with an input error to restricted reconciliation", async () => { + it("continues after a partially applied mutation reports an input error", async () => { const appliedChanges: string[] = []; const applyPatch = pluginToolWithExecute("apply_patch", "Apply a patch", async () => { appliedChanges.push("first hunk applied"); @@ -737,17 +775,16 @@ describe("Code Mode agent-loop error recovery", () => { jsonResult({ executed: true }), ); - const { providerContexts, reconciliationCandidates } = await runCodeModeAgent({ + const { providerContexts } = await runCodeModeAgent({ hiddenTools: [applyPatch, write, send, shell], programs: ["return await apply_patch({});", "return await write({});"], }); - expect(providerContexts).toHaveLength(1); + expect(providerContexts).toHaveLength(3); expect(applyPatch.execute).toHaveBeenCalledOnce(); - expect(write.execute).not.toHaveBeenCalled(); + expect(write.execute).toHaveBeenCalledOnce(); expect(send.execute).not.toHaveBeenCalled(); expect(shell.execute).not.toHaveBeenCalled(); - expect(reconciliationCandidates).toBe(1); expect(appliedChanges).toEqual(["first hunk applied"]); }); diff --git a/src/agents/code-mode.bridge.host-denial.test.ts b/src/agents/code-mode.bridge.host-denial.test.ts index 2f0a52672252..27908e3301a1 100644 --- a/src/agents/code-mode.bridge.host-denial.test.ts +++ b/src/agents/code-mode.bridge.host-denial.test.ts @@ -26,8 +26,6 @@ import * as nodeHost from "./bash-tools.exec-host-node.js"; import { createExecTool } from "./bash-tools.exec-run.js"; import type { ExecToolDefaults } from "./bash-tools.exec-types.js"; import * as codeModeBridge from "./code-mode-bridge.js"; -import { consumeRepairableCodeModeFailure } from "./code-mode-repair-provenance.js"; -import { isCodeModeBridgeRepairEligible } from "./code-mode-state.js"; import { createSubscribedCodeModeHarness } from "./code-mode.bridge.lifecycle.test-support.js"; import { applyCodeModeCatalog } from "./code-mode.js"; import { @@ -37,7 +35,6 @@ import { testing, waitUntilCompleted, } from "./code-mode.test-support.js"; -import { consumeToolEffectReceipt } from "./tool-effect-receipt.js"; import { consumeTrustedToolNoStartError } from "./tool-result-error.js"; import { createToolTerminalObserver } from "./tool-terminal-outcome.js"; import { jsonResult, ToolInputError } from "./tools/common.js"; @@ -171,12 +168,6 @@ describe("Code Mode subscribed host denial", () => { completedCount: 2, activeCount: 0, }); - const serialized = JSON.stringify(details); - for (const copy of [{ ...details }, structuredClone(details), JSON.parse(serialized)]) { - expect(consumeRepairableCodeModeFailure(copy)).toBe(false); - } - expect(consumeRepairableCodeModeFailure(details)).toBe(true); - expect(consumeRepairableCodeModeFailure(details)).toBe(false); } finally { harness.dispose(); } @@ -236,9 +227,6 @@ describe("Code Mode subscribed host denial", () => { expect(wait).toHaveBeenCalledOnce(); expect(details.value).toMatchObject({ exitCode: 0, aggregated: "host-corrected" }); } - expect(consumeRepairableCodeModeFailure(details)).toBe( - change === "deny" || change === "invalid", - ); } finally { harness.dispose(); vi.useRealTimers(); @@ -285,7 +273,6 @@ describe("Code Mode subscribed host denial", () => { const allowed = decision === "allow-once" || decision === "allow-always"; const details = await harness.runToCompletion(); expect(details.status).toBe("failed"); - expect(consumeRepairableCodeModeFailure(details)).toBe(allowed); expect(before).toHaveBeenCalledOnce(); expect(resolutions).toEqual([ decision === "cancel" || decision === "unavailable" || decision === "report" @@ -339,7 +326,6 @@ describe("Code Mode subscribed host denial", () => { try { const details = await harness.runToCompletion(); expect(details.error).toContain("requested node"); - expect(consumeRepairableCodeModeFailure(details)).toBe(true); expect(approve).toHaveBeenCalledOnce(); expect(rewrite).toHaveBeenCalledOnce(); expect(harness.spawn).not.toHaveBeenCalled(); @@ -402,7 +388,6 @@ describe("Code Mode subscribed host denial", () => { const details = await running; expect(details).toMatchObject({ status: "failed" }); expect(details.error).toMatch(/abort|cancel/i); - expect(consumeRepairableCodeModeFailure(details)).toBe(false); release.resolve(); await finished.promise; await vi.waitFor(() => expect(harness.subscription.getItemLifecycle().activeCount).toBe(0)); @@ -449,7 +434,6 @@ describe("Code Mode subscribed host denial", () => { expect(producerError).toBeInstanceOf(Error); expect(consumeTrustedToolNoStartError(producerError)).toBe(false); expect(consumeTrustedToolNoStartError(replacement)).toBe(false); - expect(consumeRepairableCodeModeFailure(details)).toBe(false); expect(harness.spawn).not.toHaveBeenCalled(); } finally { harness.dispose(); @@ -466,7 +450,6 @@ describe("Code Mode subscribed host denial", () => { const details = await harness.runToCompletion(); await vi.waitFor(() => expect(after).toHaveBeenCalledOnce()); expect(details.error).toContain("exec host not allowed"); - expect(consumeRepairableCodeModeFailure(details)).toBe(true); } finally { harness.dispose(); } @@ -480,7 +463,7 @@ describe("Code Mode subscribed host denial", () => { "denial-first/settles-first", "denial-first/settles-last", "late-settlement", - ] as const)("retains real whole-cell mutation history: %s", async (order) => { + ] as const)("does not replay completed work around a host denial: %s", async (order) => { const dir = await fs.mkdtemp(path.join(os.tmpdir(), "code-mode-host-marker-")); const marker = path.join(dir, "mutation.txt"); const mutated = createDeferred(); @@ -543,26 +526,14 @@ describe("Code Mode subscribed host denial", () => { let details = await harness.run(code); if (order === "parked") { expect(details.status).toBe("waiting"); - const parked = expectDefined( - testing.activeRuns.get(String(details.runId)), - "real parked history", - ); - expect(isCodeModeBridgeRepairEligible(parked.bridgeDispatch)).toBe(false); details = resultDetails( await expectDefined(harness.tools[1], "wait").execute("resume-denial", { runId: details.runId, }), ); - expect(isCodeModeBridgeRepairEligible(parked.bridgeDispatch)).toBe(false); } if (order === "late-settlement") { expect(details.status).toBe("waiting"); - const pending = expectDefined( - testing.activeRuns.get(String(details.runId)), - "pending mutation", - ); - expect(isCodeModeBridgeRepairEligible(pending.bridgeDispatch)).toBe(false); - expect(consumeRepairableCodeModeFailure(details)).toBe(false); await expect(fs.readFile(marker, "utf8")).rejects.toMatchObject({ code: "ENOENT" }); releaseLateMutation.resolve(); await lateMutationFinished.promise; @@ -572,7 +543,6 @@ describe("Code Mode subscribed host denial", () => { }), ); await vi.waitFor(() => expect(harness.subscription.getItemLifecycle().activeCount).toBe(0)); - expect(isCodeModeBridgeRepairEligible(pending.bridgeDispatch)).toBe(false); } details = await harness.complete(details); expect(details).toMatchObject({ status: "failed", bridgeDispatchStarted: true }); @@ -589,7 +559,6 @@ describe("Code Mode subscribed host denial", () => { mutationFirstSettlement ? ["record_mutation", "exec"] : ["exec", "record_mutation"], ); } - expect(consumeRepairableCodeModeFailure(details)).toBe(false); expect(harness.spawn).not.toHaveBeenCalled(); expect(harness.remote).not.toHaveBeenCalled(); } finally { @@ -602,7 +571,7 @@ describe("Code Mode subscribed host denial", () => { }); it.each(["unbranded", "input-error", "security-deny"] as const)( - "does not grant operation proof to %s failures", + "reports %s failures without starting a shell process", async (kind) => { const harness = createHostHarness({ name: `untrusted-${kind}`, @@ -625,7 +594,6 @@ describe("Code Mode subscribed host denial", () => { : `return await exec({ command: "printf no", host: "gateway" });`, ); expect(details.status).toBe("failed"); - expect(consumeRepairableCodeModeFailure(details)).toBe(false); expect(harness.spawn).not.toHaveBeenCalled(); expect(harness.remote).not.toHaveBeenCalled(); } finally { @@ -678,7 +646,6 @@ describe("Code Mode subscribed host denial", () => { expect(checkClientVoiceToolConfirmationPolicy(policy).allowed).toBe(true); const denied = await harness.runToCompletion(`return await exec(${JSON.stringify(args)});`); expect(denied.error).toContain("exec host not allowed"); - expect(consumeRepairableCodeModeFailure(denied)).toBe(true); expect(checkClientVoiceToolConfirmationPolicy(policy).allowed).toBe(false); for (const input of [args, { ...args, host: "gateway" }]) { const blocked = await harness.runToCompletion( @@ -689,7 +656,6 @@ describe("Code Mode subscribed host denial", () => { value: { status: "blocked", deniedReason: "client-voice-confirmation" }, }); expect(JSON.stringify(blocked.value)).toContain("VOICE_CONFIRMATION_REQUIRED"); - expect(consumeRepairableCodeModeFailure(blocked)).toBe(false); } expect(harness.spawn).not.toHaveBeenCalled(); expect(harness.remote).not.toHaveBeenCalled(); @@ -698,40 +664,4 @@ describe("Code Mode subscribed host denial", () => { resetClientVoiceConfirmationStateForTest(); } }); - it.each(["copy", "json", "reuse"] as const)( - "does not transfer a settled object's proof through %s", - async (mode) => { - const harness = createHostHarness({ name: `settled-${mode}` }); - const runBridge = codeModeBridge.runBridgeRequest; - let first: Awaited> | undefined; - vi.spyOn(codeModeBridge, "runBridgeRequest").mockImplementation(async (params) => { - const settled = await runBridge(params); - if (mode === "reuse") { - if (first) { - return first; - } - first = settled; - return settled; - } - first = settled; - const serialized = JSON.stringify(settled); - return mode === "json" ? JSON.parse(serialized) : { ...settled }; - }); - try { - const details = await harness.runToCompletion(); - expect(details.error).toContain("exec host not allowed"); - expect(consumeRepairableCodeModeFailure(details)).toBe(mode === "reuse"); - if (mode === "reuse") { - const reused = await harness.runToCompletion(); - expect(consumeRepairableCodeModeFailure(reused)).toBe(false); - } else { - expect(consumeToolEffectReceipt(first)).toEqual({ state: "not_started" }); - } - expect(consumeToolEffectReceipt(first)).toBeUndefined(); - expect(harness.spawn).not.toHaveBeenCalled(); - } finally { - harness.dispose(); - } - }, - ); }); diff --git a/src/agents/code-mode.search.test.ts b/src/agents/code-mode.search.test.ts index a754ccf008da..1ab26a1ba275 100644 --- a/src/agents/code-mode.search.test.ts +++ b/src/agents/code-mode.search.test.ts @@ -1,6 +1,5 @@ import { expectDefined } from "@openclaw/normalization-core"; import { afterEach, describe, expect, it } from "vitest"; -import { consumeRepairableCodeModeFailure } from "./code-mode-repair-provenance.js"; import { applyCodeModeCatalog, createCodeModeTools, @@ -52,9 +51,6 @@ describe.each(["interactive", "headless"] as const)("Code Mode %s search", (mode for (const target of targets) { expect(target.execute).not.toHaveBeenCalled(); } - if (mode === "interactive") { - expect(consumeRepairableCodeModeFailure(overflow)).toBe(true); - } const narrowed = await run(` const matches = await catalog.search("shipment", { limit: 1 }); diff --git a/src/agents/code-mode.ts b/src/agents/code-mode.ts index 84e9c8d3b3a8..b72fd8d6a121 100644 --- a/src/agents/code-mode.ts +++ b/src/agents/code-mode.ts @@ -22,7 +22,7 @@ import { import { runCodeModeExec, runWait } from "./code-mode-execution.js"; import { runCodeModeScriptHeadless } from "./code-mode-headless.js"; import { describeCodeModeNamespacesForPrompt } from "./code-mode-namespaces.js"; -import { markCodeModePermissionChangeResult } from "./code-mode-repair-provenance.js"; +import { markCodeModePermissionChangeResult } from "./code-mode-permission-change.js"; import { isCodeModeEngagedForModel, readCode, @@ -171,7 +171,7 @@ function createCodeModeExecDescription( : undefined; const catalogIndex = projection ? formatCodeModeCatalogIndex(projection.bindings) : ""; return ( - `Run JavaScript or TypeScript in OpenClaw code mode. Enabled tools are async global functions listed in the quick index. Await dependent calls in order; independent calls may run with Promise.all. Declared output fields may feed later calls in the same program; do not spend another \`exec\` merely inspecting them. Return the final value; otherwise the result is \`null\`. \`-> ?\` means unknown output: do not feed it into guessed field-dependent logic in the same program. Return the raw value first, observe it, then use a later \`exec\` for dependent composition. If a tool is omitted from the bounded index, use \`catalog.search(query)\`; results are callable: \`const [tool] = await catalog.search("..."); return await tool({...});\`. Handles expose \`describe()\` when a schema is needed. \`setTimeout\` and \`clearTimeout\` work. Nested calls enforce normal tool policy and approvals. Tool failures are catchable JavaScript errors; otherwise, use a safe failed result to correct your code or choose another tool. If an action may have started, inspect its outcome without repeating mutations. Never replay actions that already ran. Each nested result is bounded separately to ${maxOutputBytes} bytes. Cumulative output and the final value or error share ${maxOutputBytes} bytes across waits. Model-facing results may use a smaller allowance to preserve complete status and continuation within model context limits. Output/value truncation reports a prefix and omitted bytes of the original normalized JSON; rerun with narrower args. Ordinary output is incremental; unchanged summaries are suppressed, changed cumulative summaries replace earlier ones. Node.js modules and \`require\`/\`import\` are NOT available; use enabled globals for shell, file, network, or external actions.` + + `Run JavaScript or TypeScript in OpenClaw code mode. Enabled tools are async global functions listed in the quick index. Await dependent calls in order; independent calls may run with Promise.all. Declared output fields may feed later calls in the same program; do not spend another \`exec\` merely inspecting them. Return the final value; otherwise the result is \`null\`. \`-> ?\` means unknown output: do not feed it into guessed field-dependent logic in the same program. Return the raw value first, observe it, then use a later \`exec\` for dependent composition. If a tool is omitted from the bounded index, use \`catalog.search(query)\`; results are callable: \`const [tool] = await catalog.search("..."); return await tool({...});\`. Handles expose \`describe()\` when a schema is needed. \`setTimeout\` and \`clearTimeout\` work. Nested calls enforce normal tool policy and approvals. Tool failures are catchable JavaScript errors; otherwise, use the failed result to correct your code or choose another tool. If an action may have started, inspect its outcome without repeating mutations. Never replay actions that already ran. Each nested result is bounded separately to ${maxOutputBytes} bytes. Cumulative output and the final value or error share ${maxOutputBytes} bytes across waits. Model-facing results may use a smaller allowance to preserve complete status and continuation within model context limits. Output/value truncation reports a prefix and omitted bytes of the original normalized JSON; rerun with narrower args. Ordinary output is incremental; unchanged summaries are suppressed, changed cumulative summaries replace earlier ones. Node.js modules and \`require\`/\`import\` are NOT available; use enabled globals for shell, file, network, or external actions.` + apiGuidance + mcpGuidance + swarmGuidance + diff --git a/src/agents/embedded-agent-runner/run-loop.ts b/src/agents/embedded-agent-runner/run-loop.ts index 30ef7ecd23f5..f4ed563f03ce 100644 --- a/src/agents/embedded-agent-runner/run-loop.ts +++ b/src/agents/embedded-agent-runner/run-loop.ts @@ -29,7 +29,6 @@ import { prepareAndDispatchEmbeddedRunAttempt } from "./run/attempt-dispatch-pre import { normalizeEmbeddedRunAttempt } from "./run/attempt-normalization.js"; import { recoverEmbeddedRunAttempt } from "./run/attempt-recovery.js"; import { createAttemptCarryover } from "./run/attempt-result.js"; -import { advanceCodeModeRecovery } from "./run/code-mode-reconciliation.js"; import { hasCodexAppServerRecoveryRetryBudget } from "./run/codex-app-server-recovery.js"; import { createEmbeddedRunCompactionRuntime } from "./run/compaction-runtime.js"; import { createEmbeddedRunContextRecoveryState } from "./run/context-recovery-state.js"; @@ -552,16 +551,6 @@ export async function runPreparedEmbeddedLoop( if (assistantFailureOutcome.action === "retry") { continue; } - if ( - advanceCodeModeRecovery({ - attempt, - hostOwnsToolSurface: !pluginHarnessOwnsTransport, - retryState: terminalRetryState, - activateInternalPrompt: sessionPromptState.activateInternalPrompt, - }) - ) { - continue; - } let assistantProfileFailureReason = assistantFailureOutcome.assistantProfileFailureReason; const terminalToolPresentationText = terminalToolPresentation.read(); const finalizedTerminal = await prepareTerminalWithSettledTurnFinalization({ diff --git a/src/agents/embedded-agent-runner/run.code-mode-reconciliation.test-support.ts b/src/agents/embedded-agent-runner/run.code-mode-reconciliation.test-support.ts deleted file mode 100644 index 78d4642164b9..000000000000 --- a/src/agents/embedded-agent-runner/run.code-mode-reconciliation.test-support.ts +++ /dev/null @@ -1,161 +0,0 @@ -import { afterEach, beforeAll, beforeEach, describe, expect, it } from "vitest"; -import type { OpenClawTestState } from "../../test-utils/openclaw-test-state.js"; -import { buildEmbeddedRunnerAssistant } from "../test-helpers/embedded-agent-runner-e2e-fixtures.js"; -import { makeAttemptResult } from "./run.overflow-compaction.fixture.js"; -import { - mockedClassifyFailoverReason, - mockedRunEmbeddedAttempt, - createOverflowRunParams, - resetSharedRunIntegrationHarnessMocks, - useOpenAIPlatformAuthFixture, -} from "./run.overflow-compaction.harness.js"; -import { loadSharedRunIntegrationHarness } from "./run.shared-integration-harness.test-support.js"; - -let state: OpenClawTestState; -let runEmbeddedAgent: Awaited>; - -describe("runEmbeddedAgent Code Mode reconciliation", () => { - beforeAll(async () => { - runEmbeddedAgent = await loadSharedRunIntegrationHarness(); - }); - - beforeEach(async () => { - resetSharedRunIntegrationHarnessMocks(); - const { createOpenClawTestState } = await import("../../test-utils/openclaw-test-state.js"); - state = await createOpenClawTestState({ label: "run.code-mode-reconciliation" }); - mockedClassifyFailoverReason.mockReturnValue(null); - useOpenAIPlatformAuthFixture(); - }); - - afterEach(async () => { - await state?.cleanup(); - }); - - it("continues a settled partial mutation through inspection and bounded recovery", async () => { - const mutationAssistant = buildEmbeddedRunnerAssistant({ - stopReason: "toolUse", - content: [ - { - type: "toolCall", - id: "code-mode-mutation", - name: "code_mode", - arguments: { action: "exec" }, - }, - ], - }); - const retryAssistant = buildEmbeddedRunnerAssistant({ - stopReason: "error", - content: [], - }); - mockedRunEmbeddedAttempt - .mockResolvedValueOnce( - makeAttemptResult({ - assistantTexts: [], - lastAssistant: mutationAssistant, - currentAttemptAssistant: mutationAssistant, - currentAttemptCompletedAssistant: mutationAssistant, - codeModeRecoveryCandidate: { blockedActionKeys: ["apply_patch:prior"] }, - itemLifecycle: { startedCount: 1, completedCount: 1, activeCount: 0 }, - }), - ) - .mockResolvedValueOnce( - makeAttemptResult({ - assistantTexts: [], - itemLifecycle: { startedCount: 2, completedCount: 2, activeCount: 0 }, - toolMetas: [ - { toolName: "read", isError: false }, - { toolName: "recovery_resume", isError: false, terminate: true }, - ], - }), - ) - .mockResolvedValueOnce( - makeAttemptResult({ - assistantTexts: [], - lastAssistant: retryAssistant, - currentAttemptAssistant: retryAssistant, - currentAttemptCompletedAssistant: retryAssistant, - }), - ) - .mockResolvedValueOnce(makeAttemptResult({ assistantTexts: ["Recovery completed."] })); - - await runEmbeddedAgent({ - ...createOverflowRunParams(state), - config: { - agents: { - defaults: { - models: { "openai/gpt-5.5": { agentRuntime: { id: "openclaw" } } }, - }, - }, - }, - provider: "openai", - model: "gpt-5.5", - runId: "run-code-mode-reconciliation", - }); - - expect(mockedRunEmbeddedAttempt).toHaveBeenCalledTimes(4); - expect(mockedRunEmbeddedAttempt.mock.calls[1]?.[0]).toMatchObject({ - codeModeRecovery: { kind: "inspect" }, - prompt: expect.stringContaining("may have partially applied"), - }); - expect(mockedRunEmbeddedAttempt.mock.calls[2]?.[0]).toMatchObject({ - codeModeOverride: false, - codeModeRecovery: { kind: "resume" }, - }); - expect(mockedRunEmbeddedAttempt.mock.calls[3]?.[0]).toMatchObject({ - codeModeOverride: false, - codeModeRecovery: { kind: "resume" }, - prompt: expect.stringContaining("at most one mutation attempt"), - }); - }); - - it("ends after inspection when no recovery is requested", async () => { - const mutationAssistant = buildEmbeddedRunnerAssistant({ - stopReason: "toolUse", - content: [ - { - type: "toolCall", - id: "code-mode-mutation", - name: "code_mode", - arguments: { action: "exec" }, - }, - ], - }); - mockedRunEmbeddedAttempt - .mockResolvedValueOnce( - makeAttemptResult({ - assistantTexts: [], - lastAssistant: mutationAssistant, - currentAttemptAssistant: mutationAssistant, - currentAttemptCompletedAssistant: mutationAssistant, - codeModeRecoveryCandidate: { blockedActionKeys: ["apply_patch:prior"] }, - itemLifecycle: { startedCount: 1, completedCount: 1, activeCount: 0 }, - }), - ) - .mockResolvedValueOnce( - makeAttemptResult({ - assistantTexts: ["The requested change already applied."], - itemLifecycle: { startedCount: 1, completedCount: 1, activeCount: 0 }, - toolMetas: [{ toolName: "read", isError: false }], - }), - ); - - await runEmbeddedAgent({ - ...createOverflowRunParams(state), - config: { - agents: { - defaults: { - models: { "openai/gpt-5.5": { agentRuntime: { id: "openclaw" } } }, - }, - }, - }, - provider: "openai", - model: "gpt-5.5", - runId: "run-code-mode-reconciliation-complete", - }); - - expect(mockedRunEmbeddedAttempt).toHaveBeenCalledTimes(2); - expect(mockedRunEmbeddedAttempt.mock.calls[1]?.[0]).toMatchObject({ - codeModeRecovery: { kind: "inspect" }, - }); - }); -}); diff --git a/src/agents/embedded-agent-runner/run.shared-integration.test.ts b/src/agents/embedded-agent-runner/run.shared-integration.test.ts index 3d890ada9f8e..ec8f609b8b36 100644 --- a/src/agents/embedded-agent-runner/run.shared-integration.test.ts +++ b/src/agents/embedded-agent-runner/run.shared-integration.test.ts @@ -1,7 +1,6 @@ // The imported scenario modules share one mocked runEmbeddedAgent module graph. import "./run.before-agent-finalize.test-support.js"; import "./run.before-agent-reply-cron.test-support.js"; -import "./run.code-mode-reconciliation.test-support.js"; import "./run.codex-app-server-recovery.test-support.js"; import "./run.codex-server-error-fallback.test-support.js"; import "./run.compaction-loop-guard.test-support.js"; diff --git a/src/agents/embedded-agent-runner/run/attempt-bundle-tools.test.ts b/src/agents/embedded-agent-runner/run/attempt-bundle-tools.test.ts index 9f185c394ecf..340d6108fe0b 100644 --- a/src/agents/embedded-agent-runner/run/attempt-bundle-tools.test.ts +++ b/src/agents/embedded-agent-runner/run/attempt-bundle-tools.test.ts @@ -155,7 +155,7 @@ describe("prepareEmbeddedAttemptBundleTools", () => { expect(mocks.getOrCreateSessionMcpRuntime).toHaveBeenCalledOnce(); }); - it.each(["disableTools", "raw", "restart", "reconciliation", "model"])( + it.each(["disableTools", "raw", "restart", "model"])( "does not discover matching MCP when tools are disabled by %s", async (mode) => { const input = createInput([], []); @@ -164,9 +164,6 @@ describe("prepareEmbeddedAttemptBundleTools", () => { input.attempt.disableTools = mode === "disableTools"; input.isRawModelRun = mode === "raw"; input.attempt.forceRestartSafeTools = mode === "restart"; - if (mode === "reconciliation") { - input.attempt.codeModeRecovery = { kind: "inspect", phase: "read-required" }; - } input.preparedToolBase.toolsEnabled = mode !== "model"; await prepareEmbeddedAttemptBundleTools(input); diff --git a/src/agents/embedded-agent-runner/run/attempt-bundle-tools.ts b/src/agents/embedded-agent-runner/run/attempt-bundle-tools.ts index c385eaa2f5c9..2b829b422a3c 100644 --- a/src/agents/embedded-agent-runner/run/attempt-bundle-tools.ts +++ b/src/agents/embedded-agent-runner/run/attempt-bundle-tools.ts @@ -75,8 +75,7 @@ export async function prepareEmbeddedAttemptBundleTools(params: { toolsEnabled && !params.attempt.disableTools && !params.isRawModelRun && - !params.attempt.forceRestartSafeTools && - params.attempt.codeModeRecovery?.kind !== "inspect" + !params.attempt.forceRestartSafeTools ? params.attempt.clientTools : undefined; // Client functions share the attempt's authority; filter before their names @@ -105,7 +104,6 @@ export async function prepareEmbeddedAttemptBundleTools(params: { }; const bundleMcpEnabled = !params.attempt.forceRestartSafeTools && - params.attempt.codeModeRecovery?.kind !== "inspect" && shouldCreateBundleMcpRuntimeForAttempt({ toolsEnabled, disableTools: params.attempt.disableTools || params.isRawModelRun, @@ -153,7 +151,6 @@ export async function prepareEmbeddedAttemptBundleTools(params: { try { const bundleLspEnabled = !params.attempt.forceRestartSafeTools && - params.attempt.codeModeRecovery?.kind !== "inspect" && shouldCreateBundleLspRuntimeForAttempt({ toolsEnabled, disableTools: params.attempt.disableTools || params.isRawModelRun, diff --git a/src/agents/embedded-agent-runner/run/attempt-client-tools.test.ts b/src/agents/embedded-agent-runner/run/attempt-client-tools.test.ts index 1d53f741ec3f..efd8d23f7cea 100644 --- a/src/agents/embedded-agent-runner/run/attempt-client-tools.test.ts +++ b/src/agents/embedded-agent-runner/run/attempt-client-tools.test.ts @@ -16,7 +16,6 @@ import { createMockPluginRegistry } from "../../../plugins/hooks.test-fixtures.j import { setPluginToolMeta } from "../../../plugins/tool-metadata.js"; import { createDeferredCore } from "../../../shared/deferred.js"; import { wrapToolWithAbortSignal } from "../../agent-tools.abort.js"; -import { setChannelAgentToolMeta } from "../../channel-tool-metadata.js"; import { createCodeModeCatalogProjection } from "../../code-mode-catalog.js"; import { markCodeModeControlTool } from "../../code-mode-control-tools.js"; import { applyCodeModeCatalog, createCodeModeTools } from "../../code-mode.js"; @@ -188,28 +187,6 @@ describe("prepareEmbeddedAttemptClientTools", () => { }, ); - it("records core read entitlement without plugin or channel shadows", () => { - const coreRead = createStubTool("read"); - const pluginRead = createStubTool("read"); - const channelRead = createStubTool("read"); - const catalogRef = createToolSearchCatalogRef(); - setPluginToolMeta(pluginRead, { pluginId: "example-plugin", optional: false }); - setChannelAgentToolMeta(channelRead as never, { channelId: "example-channel" }); - - expect( - [coreRead, pluginRead, channelRead].map( - (tool) => - prepare({ - codeModeControlsEnabledForRun: false, - attemptConfig: CATALOGS_DISABLED_CONFIG, - toolSearchRuntimeConfig: CATALOGS_DISABLED_CONFIG, - catalogRef, - uncompactedEffectiveTools: [tool], - }).coreReadAuthorized, - ), - ).toEqual([true, false, false]); - }); - it("collects only the marked Code Mode exec as a code-mode exec tool name", () => { const catalogRef = createToolSearchCatalogRef(); const markedExec = markCodeModeControlTool(createStubTool("exec")); diff --git a/src/agents/embedded-agent-runner/run/attempt-client-tools.ts b/src/agents/embedded-agent-runner/run/attempt-client-tools.ts index 41f5496e29bc..53f721ce25aa 100644 --- a/src/agents/embedded-agent-runner/run/attempt-client-tools.ts +++ b/src/agents/embedded-agent-runner/run/attempt-client-tools.ts @@ -9,7 +9,6 @@ import { } from "../../agent-tool-definition-adapter.js"; import { wrapToolWithAbortSignal } from "../../agent-tools.abort.js"; import { resolveToolLoopDetectionConfig } from "../../agent-tools.js"; -import { getChannelAgentToolMeta } from "../../channel-tools.js"; import { isCodeModeExecTool } from "../../code-mode-control-tools.js"; import { addClientToolsToCodeModeCatalog } from "../../code-mode.js"; import type { AgentTool } from "../../runtime/index.js"; @@ -17,7 +16,6 @@ import { createToolDefinitionFromAgentTool, wrapToolDefinition, } from "../../sessions/tools/tool-definition-wrapper.js"; -import { normalizeToolPolicyName } from "../../tool-policy.js"; import { collectReplaySafeToolNames, collectSideEffectToolOwners, @@ -33,7 +31,6 @@ import { } from "../tool-name-allowlist.js"; import { splitSdkTools } from "../tool-split.js"; import type { EmbeddedAttemptClientToolCallSlot } from "./attempt-result.js"; -import { applyCodeModeRecoveryPreparedToolSurface } from "./code-mode-reconciliation.js"; import type { EmbeddedRunAttemptParams } from "./types.js"; export function prepareEmbeddedAttemptClientTools(params: { @@ -125,12 +122,6 @@ export function prepareEmbeddedAttemptClientTools(params: { isPluginTool: (tool) => Boolean(getPluginToolMeta(tool as Parameters[0])), }); - const coreReadAuthorized = params.uncompactedEffectiveTools.some( - (tool) => - normalizeToolPolicyName(tool.name ?? "") === "read" && - !getPluginToolMeta(tool) && - !getChannelAgentToolMeta(tool), - ); const isReplaySafeTool = (tool: { name?: string }) => isAgentToolReplaySafe(tool, params.replaySafetyOptions); const replaySafeTools = new Set(params.uncompactedEffectiveTools.filter(isReplaySafeTool)); @@ -160,12 +151,6 @@ export function prepareEmbeddedAttemptClientTools(params: { wrapToolWithAbortSignal(wrapToolDefinition(definition), params.getToolAbortSignal?.()), ), ); - if (params.attempt.codeModeRecovery?.kind === "resume") { - clientToolDefs = applyCodeModeRecoveryPreparedToolSurface({ - tools: clientToolDefs, - state: params.attempt.codeModeRecovery, - }); - } // Terminal observations are name-only, so ownership is valid only when one // concrete OpenClaw or client tool owns the normalized name. const sideEffectToolOwners = collectSideEffectToolOwners( @@ -214,7 +199,6 @@ export function prepareEmbeddedAttemptClientTools(params: { allCustomTools, builtinToolNames, coreBuiltinToolNames, - coreReadAuthorized, clientToolCallSlots, clientToolDefs, replaySafeToolNames, diff --git a/src/agents/embedded-agent-runner/run/attempt-dispatch-preparation.ts b/src/agents/embedded-agent-runner/run/attempt-dispatch-preparation.ts index fbd454014dd9..55bf0a07b1fb 100644 --- a/src/agents/embedded-agent-runner/run/attempt-dispatch-preparation.ts +++ b/src/agents/embedded-agent-runner/run/attempt-dispatch-preparation.ts @@ -56,15 +56,7 @@ export async function prepareAndDispatchEmbeddedRunAttempt(input: { provider, modelId, } = input; - const codeModeRecovery = terminalRetryState.codeModeRecovery; - const params = - codeModeRecovery.kind === "resume" - ? { - ...runInput.runParams, - codeModeOverride: false, - forceCodeModeTools: false, - } - : runInput.runParams; + const params = runInput.runParams; const { workspaceResolution, workspaceDir, @@ -234,7 +226,6 @@ export async function prepareAndDispatchEmbeddedRunAttempt(input: { }); const dispatchedAttempt = await dispatchEmbeddedRunAttempt({ params, - codeModeRecovery: codeModeRecovery.kind === "idle" ? undefined : codeModeRecovery, permissionChange: input.permissionChange, runStartedAtMs: runInput.startedAtMs, transcriptOwnership: params.sessionManager diff --git a/src/agents/embedded-agent-runner/run/attempt-execution-settle.test.ts b/src/agents/embedded-agent-runner/run/attempt-execution-settle.test.ts index 40a3ec6bd0bd..06f891d0c453 100644 --- a/src/agents/embedded-agent-runner/run/attempt-execution-settle.test.ts +++ b/src/agents/embedded-agent-runner/run/attempt-execution-settle.test.ts @@ -178,11 +178,8 @@ function createFixture() { agentSession: { activeSession, clientToolCallSlots: [], - coreReadAuthorized: true, - getCodeModeRecoveryCandidate: vi.fn(() => undefined), hasDeliveredSourceReply: vi.fn(() => true), hookRunner, - setCodeModeReconciliationReadAuthorized: vi.fn(), setActiveSessionSystemPrompt: vi.fn(), settingsManager: { getCompactionReserveTokens: vi.fn(() => 1_000) }, }, diff --git a/src/agents/embedded-agent-runner/run/attempt-prompt-phase.test.ts b/src/agents/embedded-agent-runner/run/attempt-prompt-phase.test.ts index 97f636339cb0..7289860bbe31 100644 --- a/src/agents/embedded-agent-runner/run/attempt-prompt-phase.test.ts +++ b/src/agents/embedded-agent-runner/run/attempt-prompt-phase.test.ts @@ -121,7 +121,6 @@ function createFixture() { prePromptMessageCount = count; }); const setPromptCacheChangesForTurn = vi.fn(); - const setCodeModeReconciliationReadAuthorized = vi.fn(); const setFinalPromptText = vi.fn(); const markBeforeAgentRunBlocked = vi.fn(); const markYieldAborted = vi.fn(() => { @@ -259,7 +258,6 @@ function createFixture() { effectiveTools: [{ name: "read" }], uncompactedEffectiveTools: [{ name: "read" }], tools: [{ name: "read" }], - coreReadAuthorized: true, }, apply(toolsAllow: string[] | undefined) { Object.assign(this.current, mocks.applyPromptToolsAllow({ toolsAllow })); @@ -289,7 +287,6 @@ function createFixture() { setPrePromptMessageCount, setCurrentUserTimestampOverride: vi.fn(), setPromptCacheChangesForTurn, - setCodeModeReconciliationReadAuthorized, setFinalPromptText, markBeforeAgentRunBlocked, markYieldAborted, @@ -307,7 +304,6 @@ function createFixture() { setFinalPromptText, setPrePromptMessageCount, setPromptCacheChangesForTurn, - setCodeModeReconciliationReadAuthorized, state, yieldState, }; @@ -317,7 +313,6 @@ beforeEach(() => { vi.clearAllMocks(); mocks.applyPromptToolsAllow.mockReturnValue({ activeToolNames: ["read"], - coreReadAuthorized: true, effectiveTools: [{ name: "read" }], uncompactedEffectiveTools: [{ name: "read" }], tools: [{ name: "read" }], @@ -391,7 +386,6 @@ describe("runEmbeddedAttemptPromptPhase", () => { ]); expect(fixture.setPrePromptMessageCount).toHaveBeenCalledWith(2); expect(fixture.setPromptCacheChangesForTurn).toHaveBeenCalledWith([]); - expect(fixture.setCodeModeReconciliationReadAuthorized).toHaveBeenCalledWith(true); expect(fixture.setFinalPromptText).toHaveBeenCalledWith("hello"); expect(mocks.preparePromptExecution).toHaveBeenCalledWith( expect.objectContaining({ @@ -419,21 +413,6 @@ describe("runEmbeddedAttemptPromptPhase", () => { expect(mocks.releasePendingSteering).not.toHaveBeenCalled(); }); - it("records a final prompt policy that removes core read", async () => { - const fixture = createFixture(); - mocks.applyPromptToolsAllow.mockReturnValueOnce({ - activeToolNames: [], - coreReadAuthorized: false, - effectiveTools: [], - uncompactedEffectiveTools: [], - tools: [], - }); - - await runEmbeddedAttemptPromptPhase(fixture.input); - - expect(fixture.setCodeModeReconciliationReadAuthorized).toHaveBeenCalledWith(false); - }); - it("skips before_agent_run for settled-turn finalization", async () => { const fixture = createFixture(); fixture.input.attempt.operation = "settled-tool-finalization"; diff --git a/src/agents/embedded-agent-runner/run/attempt-prompt-phase.ts b/src/agents/embedded-agent-runner/run/attempt-prompt-phase.ts index 27bf6552e240..078ea2c8d833 100644 --- a/src/agents/embedded-agent-runner/run/attempt-prompt-phase.ts +++ b/src/agents/embedded-agent-runner/run/attempt-prompt-phase.ts @@ -129,7 +129,6 @@ export async function runEmbeddedAttemptPromptPhase(input: { setPromptCacheChangesForTurn: ( changes: PromptAssemblyResult["promptCacheChangesForTurn"], ) => void; - setCodeModeReconciliationReadAuthorized: (value: boolean) => void; setFinalPromptText: (prompt: string) => void; markBeforeAgentRunBlocked: (outcome: BeforeAgentRunOutcome) => void; markYieldAborted: () => void; @@ -193,9 +192,7 @@ export async function runEmbeddedAttemptPromptPhase(input: { sessionManager, ...input.assembly, applyPromptBuildToolsAllow: (toolsAllow) => { - const promptToolSurface = input.toolPolicy.apply(toolsAllow); - input.lifecycle.setCodeModeReconciliationReadAuthorized(promptToolSurface.coreReadAuthorized); - return promptToolSurface.activeToolNames; + return input.toolPolicy.apply(toolsAllow).activeToolNames; }, setLeasedSteering: (lease) => { leasedSteering = lease; diff --git a/src/agents/embedded-agent-runner/run/attempt-prompt-support.ts b/src/agents/embedded-agent-runner/run/attempt-prompt-support.ts index df89fadd7047..50706129fa8f 100644 --- a/src/agents/embedded-agent-runner/run/attempt-prompt-support.ts +++ b/src/agents/embedded-agent-runner/run/attempt-prompt-support.ts @@ -63,7 +63,6 @@ export function createPromptBuildToolPolicy< let toolsAllow: string[] | undefined; const current = { activeToolNames: [...baseline.activeToolNames], - coreReadAuthorized: params.coreReadAuthorized, effectiveTools: params.effectiveTools, uncompactedEffectiveTools: params.uncompactedEffectiveTools, tools: params.tools, @@ -124,11 +123,9 @@ export function applyPromptBuildToolsAllow< tools: TTool[]; catalogRef?: ToolSearchCatalogRef; codeModeControlsEnabled: boolean; - coreReadAuthorized: boolean; forceToolNames?: readonly string[]; }): { activeToolNames: string[]; - coreReadAuthorized: boolean; effectiveTools: TEffectiveTool[]; uncompactedEffectiveTools: TUncompactedTool[]; tools: TTool[]; @@ -172,9 +169,6 @@ export function applyPromptBuildToolsAllow< return { activeToolNames, - coreReadAuthorized: - params.coreReadAuthorized && - allowedUncompactedTools.some((tool) => normalizeToolPolicyName(tool.name) === "read"), effectiveTools: promptPolicy.tools, uncompactedEffectiveTools: allowedUncompactedTools, tools: allowedTools, diff --git a/src/agents/embedded-agent-runner/run/attempt-prompt-tool-policy.test.ts b/src/agents/embedded-agent-runner/run/attempt-prompt-tool-policy.test.ts index 0c1072a36018..5f253b0e7361 100644 --- a/src/agents/embedded-agent-runner/run/attempt-prompt-tool-policy.test.ts +++ b/src/agents/embedded-agent-runner/run/attempt-prompt-tool-policy.test.ts @@ -64,7 +64,6 @@ describe("applyPromptBuildToolsAllow", () => { tools, catalogRef, codeModeControlsEnabled: false, - coreReadAuthorized: true, }); if (hookTiming === "before") { policy.apply(["read"]); @@ -131,11 +130,9 @@ describe("applyPromptBuildToolsAllow", () => { tools: [{ name: "read" }, { name: "write" }, { name: "message" }], catalogRef, codeModeControlsEnabled: false, - coreReadAuthorized: true, }); expect(result.activeToolNames).toEqual([]); - expect(result.coreReadAuthorized).toBe(false); expect(result.effectiveTools).toEqual([]); expect(result.uncompactedEffectiveTools).toEqual([]); expect(result.tools).toEqual([]); @@ -162,7 +159,6 @@ describe("applyPromptBuildToolsAllow", () => { uncompactedEffectiveTools: [{ name: "message" }, { name: "read" }], tools: [{ name: "message" }, { name: "read" }], codeModeControlsEnabled: false, - coreReadAuthorized: true, }); expect(result.activeToolNames).toEqual(["message"]); @@ -199,11 +195,9 @@ describe("applyPromptBuildToolsAllow", () => { tools: [{ name: "read" }, { name: "write" }, { name: "message" }], catalogRef, codeModeControlsEnabled: false, - coreReadAuthorized: true, }); expect(result.activeToolNames).toEqual(["tool_search"]); - expect(result.coreReadAuthorized).toBe(true); expect(result.effectiveTools).toEqual([{ name: "tool_search" }]); expect(result.uncompactedEffectiveTools).toEqual([{ name: "read" }]); expect(result.tools).toEqual([{ name: "read" }]); @@ -222,11 +216,9 @@ describe("applyPromptBuildToolsAllow", () => { uncompactedEffectiveTools: [{ name: "read" }], tools: [{ name: "read" }], codeModeControlsEnabled: false, - coreReadAuthorized: true, }); expect(result.activeToolNames).toEqual([]); - expect(result.coreReadAuthorized).toBe(false); expect(result.effectiveTools).toEqual([]); expect(result.uncompactedEffectiveTools).toEqual([]); expect(result.tools).toEqual([]); @@ -258,7 +250,6 @@ describe("applyPromptBuildToolsAllow", () => { tools: [pluginTool], catalogRef, codeModeControlsEnabled: false, - coreReadAuthorized: false, }); expect(result.activeToolNames).toEqual(["tool_search"]); @@ -286,7 +277,6 @@ describe("applyPromptBuildToolsAllow", () => { tools: [{ name: "read" }, { name: "write" }], catalogRef, codeModeControlsEnabled: false, - coreReadAuthorized: true, }; applyPromptBuildToolsAllow({ ...params, toolsAllow: ["read"] }); diff --git a/src/agents/embedded-agent-runner/run/attempt-result.ts b/src/agents/embedded-agent-runner/run/attempt-result.ts index cf049d03a001..3e5dff2fed7c 100644 --- a/src/agents/embedded-agent-runner/run/attempt-result.ts +++ b/src/agents/embedded-agent-runner/run/attempt-result.ts @@ -85,7 +85,6 @@ type EmbeddedAttemptResultState = Pick< | "lastAssistant" | "currentAttemptAssistant" | "currentAttemptCompletedAssistant" - | "codeModeRecoveryCandidate" | "successfulNestedToolNames" | "attemptUsage" | "promptCache" @@ -459,7 +458,6 @@ export function completeEmbeddedAttemptResult( ...(settledTurnFinalizationContext ? { settledTurnFinalizationContext } : {}), replayMetadata, currentAttemptReplayMetadata, - codeModeRecoveryCandidate: state.codeModeRecoveryCandidate, itemLifecycle: getItemLifecycle(), assistantTurns: getAssistantTurnCount(), setTerminalLifecycleMeta, diff --git a/src/agents/embedded-agent-runner/run/attempt-session-prepare.ts b/src/agents/embedded-agent-runner/run/attempt-session-prepare.ts index a10bcbe177b3..07f790d0385c 100644 --- a/src/agents/embedded-agent-runner/run/attempt-session-prepare.ts +++ b/src/agents/embedded-agent-runner/run/attempt-session-prepare.ts @@ -11,7 +11,6 @@ import { import { getGlobalHookRunner } from "../../../plugins/hook-runner-global.js"; import type { PluginMetadataSnapshot } from "../../../plugins/plugin-metadata-snapshot.types.js"; import { isMainSessionRestartRecoveryInputProvenance } from "../../../sessions/input-provenance.js"; -import type { NestedToolActivity } from "../../../sessions/nested-tool-activity.js"; import { createPreparedEmbeddedAgentSettingsManager } from "../../agent-project-settings.js"; import { applyAgentAutoCompactionGuard, @@ -53,8 +52,6 @@ import { buildAfterTurnRuntimeContext } from "./attempt-prompt-helpers.js"; import { resolveExistingAttemptTranscriptState } from "./attempt-transcript-helpers.js"; import type { EmbeddedAttemptTranscriptLifecycle } from "./attempt-transcript-lifecycle.js"; import { createUserTranscriptContextRegistry } from "./attempt-user-transcript-context-registry.js"; -import { installCodeModeOutcomeHook } from "./code-mode-outcome.js"; -import { buildCodeModeRecoveryCandidate } from "./code-mode-reconciliation.js"; import { installMessageToolOnlyTerminalHook } from "./message-tool-terminal.js"; import { preparePersistedCurrentUserTurn, @@ -97,7 +94,6 @@ export async function prepareEmbeddedAttemptAgentSession(input: { transcriptLifecycle: EmbeddedAttemptTranscriptLifecycle; sessionManager: AttemptSessionManager; assertInitialUserTurnReplay?: () => void; - nestedToolActivities: readonly NestedToolActivity[]; }) { const { attempt } = input; const settingsManager = createPreparedEmbeddedAgentSettingsManager({ @@ -221,10 +217,7 @@ export async function prepareEmbeddedAttemptAgentSession(input: { // Without a resolved model budget, the outer loop cannot own bounded recovery. contextOverflowRecoveryOwner: attempt.contextTokenBudget === undefined ? "session" : "caller", beforeToolBatch: input.clientToolPreparation.catalogToolHookContext - ? createToolLoopBatchAdmission( - input.clientToolPreparation.catalogToolHookContext, - attempt.codeModeRecovery, - ) + ? createToolLoopBatchAdmission(input.clientToolPreparation.catalogToolHookContext) : undefined, }); const activeSession = createdSession.session; @@ -302,8 +295,6 @@ export async function prepareEmbeddedAttemptAgentSession(input: { }; setActiveSessionSystemPrompt(input.initialSystemPrompt); let didDeliverSourceReplyViaMessageTool = false; - let codeModeRecoveryCandidate: ReturnType | undefined; - let codeModeReconciliationReadAuthorized = false; const markSourceReplyDelivered = () => { didDeliverSourceReplyViaMessageTool = true; }; @@ -322,32 +313,15 @@ export async function prepareEmbeddedAttemptAgentSession(input: { hasRepliedRef: attempt.hasRepliedRef, sessionKey: attempt.sessionKey, }); - if (input.clientToolPreparation.codeModeControlsEnabledForRun) { - installCodeModeOutcomeHook({ - agent: activeSession.agent, - onReconciliationCandidate: (parentToolCallId) => { - if (codeModeReconciliationReadAuthorized) { - codeModeRecoveryCandidate = buildCodeModeRecoveryCandidate({ - parentToolCallId, - nestedToolActivities: input.nestedToolActivities, - }); - } - }, - }); - } input.markStage("agent-session"); return { activeSession, allCustomTools, ...clientToolRuntime, - getCodeModeRecoveryCandidate: () => codeModeRecoveryCandidate, hasDeliveredSourceReply: () => didDeliverSourceReplyViaMessageTool, hookRunner, markSourceReplyDelivered, - setCodeModeReconciliationReadAuthorized: (value: boolean) => { - codeModeReconciliationReadAuthorized = clientToolRuntime.coreReadAuthorized && value; - }, setActiveSessionSystemPrompt, settingsManager, refreshTools: () => { diff --git a/src/agents/embedded-agent-runner/run/attempt-session-runtime-prepare.test.ts b/src/agents/embedded-agent-runner/run/attempt-session-runtime-prepare.test.ts index 74eadc2a56d8..dc656bfc3a8b 100644 --- a/src/agents/embedded-agent-runner/run/attempt-session-runtime-prepare.test.ts +++ b/src/agents/embedded-agent-runner/run/attempt-session-runtime-prepare.test.ts @@ -168,7 +168,6 @@ function createFixture() { effectiveWorkspace: "/workspace", initialSystemPrompt: "initial prompt", isRawModelRun: false, - nestedToolActivities: [], sessionManager: { replayAllowedToolNames: new Set(["read"]), resolveActiveContextEnginePluginId: vi.fn(), diff --git a/src/agents/embedded-agent-runner/run/attempt-session-runtime-prepare.ts b/src/agents/embedded-agent-runner/run/attempt-session-runtime-prepare.ts index 65d65b1ac500..1425b7e9ad94 100644 --- a/src/agents/embedded-agent-runner/run/attempt-session-runtime-prepare.ts +++ b/src/agents/embedded-agent-runner/run/attempt-session-runtime-prepare.ts @@ -50,7 +50,6 @@ export async function prepareEmbeddedAttemptSessionRuntime(input: { effectiveWorkspace: string; initialSystemPrompt: string; isRawModelRun: boolean; - nestedToolActivities: AgentSessionInput["nestedToolActivities"]; sessionManager: Pick< SessionManagerInput, | "replayAllowedToolNames" @@ -141,7 +140,6 @@ export async function prepareEmbeddedAttemptSessionRuntime(input: { transcriptLifecycle: input.sessionManager.transcriptLifecycle, sessionManager, assertInitialUserTurnReplay: preparedSessionManager.assertInitialUserTurnReplay, - nestedToolActivities: input.nestedToolActivities, }); const { activeSession, setActiveSessionSystemPrompt, settingsManager } = preparedAgentSession; const recordCurrentTurnImageFailure = (count: number) => { diff --git a/src/agents/embedded-agent-runner/run/attempt-session.test.ts b/src/agents/embedded-agent-runner/run/attempt-session.test.ts index 65d934d4cf43..7d76a9a0a358 100644 --- a/src/agents/embedded-agent-runner/run/attempt-session.test.ts +++ b/src/agents/embedded-agent-runner/run/attempt-session.test.ts @@ -22,7 +22,6 @@ const hoisted = vi.hoisted(() => ({ createEmbeddedAgentResourceLoader: vi.fn(), createPreparedEmbeddedAgentSettingsManager: vi.fn(), getGlobalHookRunner: vi.fn(), - installCodeModeOutcomeHook: vi.fn(), installMessageToolOnlyTerminalHook: vi.fn(), prepareEmbeddedAttemptClientTools: vi.fn(), resolveEffectiveCompactionMode: vi.fn(), @@ -70,9 +69,6 @@ vi.mock("../system-prompt.js", () => ({ vi.mock("./attempt-client-tools.js", () => ({ prepareEmbeddedAttemptClientTools: hoisted.prepareEmbeddedAttemptClientTools, })); -vi.mock("./code-mode-outcome.js", () => ({ - installCodeModeOutcomeHook: hoisted.installCodeModeOutcomeHook, -})); vi.mock("./message-tool-terminal.js", () => ({ installMessageToolOnlyTerminalHook: hoisted.installMessageToolOnlyTerminalHook, })); @@ -98,11 +94,7 @@ const attempt = { workspaceDir: "/workspace", } as unknown as EmbeddedRunAttemptParams; -function createInput(options?: { - activationError?: Error; - codeModeControlsEnabledForRun?: boolean; - coreReadAllowed?: boolean; -}) { +function createInput(options?: { activationError?: Error }) { const events: string[] = []; const settingsManager = { id: "settings" }; const resourceLoader = { @@ -132,15 +124,13 @@ function createInput(options?: { const allCustomTools = [{ name: "custom" }]; const clientToolRuntime = { builtinToolNames: new Set(["read"]), - coreBuiltinToolNames: new Set(options?.coreReadAllowed === false ? [] : ["read"]), - coreReadAuthorized: options?.coreReadAllowed !== false, + coreBuiltinToolNames: new Set(["read"]), clientToolCallSlots: [], clientToolDefs: [], replaySafeToolNames: new Set(["read"]), replaySafeTools: new Set(allCustomTools), }; let onDeliveredSourceReply: (() => void) | undefined; - let onReconciliationCandidate: ((parentToolCallId: string) => void) | undefined; hoisted.createPreparedEmbeddedAgentSettingsManager.mockReturnValue(settingsManager); hoisted.resolveEffectiveCompactionMode.mockReturnValue("safeguard"); @@ -168,12 +158,6 @@ function createInput(options?: { onDeliveredSourceReply = input.onDeliveredSourceReply; }, ); - hoisted.installCodeModeOutcomeHook.mockImplementation( - (input: { onReconciliationCandidate?: (parentToolCallId: string) => void }) => { - onReconciliationCandidate = input.onReconciliationCandidate; - events.push("install-code-mode-outcome"); - }, - ); return { activeSession, @@ -187,7 +171,7 @@ function createInput(options?: { agentCoreThinkingLevel: "high" as const, agentDir: "/agent", clientToolPreparation: { - codeModeControlsEnabledForRun: options?.codeModeControlsEnabledForRun ?? true, + codeModeControlsEnabledForRun: true, deferredDirectoryToolsCallable: false, } as never, effectiveCwd: "/workspace", @@ -206,9 +190,7 @@ function createInput(options?: { sessionAgentId: "agent-1", transcriptLifecycle: transcriptLifecycle as never, sessionManager: sessionManager as never, - nestedToolActivities: [], }, - markCodeModeReconciliationCandidate: () => onReconciliationCandidate?.("code-mode-call"), onDeliveredSourceReply: () => onDeliveredSourceReply?.(), resourceLoader, setActiveToolsByName, @@ -358,7 +340,6 @@ describe("prepareEmbeddedAttemptAgentSession", () => { "publish-system-prompt", "apply-system-prompt", "install-terminal-hook", - "install-code-mode-outcome", "stage:agent-session", ]); expect(hoisted.applyAgentAutoCompactionGuard).toHaveBeenCalledTimes(2); @@ -386,10 +367,6 @@ describe("prepareEmbeddedAttemptAgentSession", () => { expect(result.hasDeliveredSourceReply()).toBe(false); fixture.onDeliveredSourceReply(); expect(result.hasDeliveredSourceReply()).toBe(true); - expect(result.getCodeModeRecoveryCandidate()).toBeUndefined(); - result.setCodeModeReconciliationReadAuthorized(true); - fixture.markCodeModeReconciliationCandidate(); - expect(result.getCodeModeRecoveryCandidate()).toEqual({}); }); it.each(["replace", "replace-reject", "replace-pending", "abort", "current-error"] as const)( @@ -470,32 +447,6 @@ describe("prepareEmbeddedAttemptAgentSession", () => { }, ); - it("does not install Code Mode outcome handling when the run kept direct tools", async () => { - const fixture = createInput({ codeModeControlsEnabledForRun: false }); - - await prepareEmbeddedAttemptAgentSession(fixture.input); - - expect(hoisted.installCodeModeOutcomeHook).not.toHaveBeenCalled(); - expect(fixture.events).not.toContain("install-code-mode-outcome"); - }); - - it.each([ - ["the effective core tools exclude read", false, true], - ["the final prompt policy removes read", true, false], - ])("withholds reconciliation when %s", async (_label, coreReadAllowed, finalReadAllowed) => { - const fixture = createInput({ coreReadAllowed }); - - const result = await prepareEmbeddedAttemptAgentSession(fixture.input); - - expect(hoisted.installCodeModeOutcomeHook).toHaveBeenCalledWith({ - agent: fixture.activeSession.agent, - onReconciliationCandidate: expect.any(Function), - }); - result.setCodeModeReconciliationReadAuthorized(finalReadAllowed); - fixture.markCodeModeReconciliationCandidate(); - expect(result.getCodeModeRecoveryCandidate()).toBeUndefined(); - }); - it("leaves overflow recovery with the session when no model budget was resolved", async () => { const fixture = createInput(); fixture.input.attempt = { diff --git a/src/agents/embedded-agent-runner/run/attempt-settle.ts b/src/agents/embedded-agent-runner/run/attempt-settle.ts index c72a1587696e..10b8fd3fb451 100644 --- a/src/agents/embedded-agent-runner/run/attempt-settle.ts +++ b/src/agents/embedded-agent-runner/run/attempt-settle.ts @@ -138,10 +138,8 @@ export async function runEmbeddedAttemptSettledPhase( agentSession: { activeSession, clientToolCallSlots, - getCodeModeRecoveryCandidate, hasDeliveredSourceReply, hookRunner, - setCodeModeReconciliationReadAuthorized, setActiveSessionSystemPrompt, settingsManager, }, @@ -328,7 +326,6 @@ export async function runEmbeddedAttemptSettledPhase( setPromptCacheChangesForTurn: (changes) => { promptCacheChangesForTurn = changes; }, - setCodeModeReconciliationReadAuthorized, setFinalPromptText: (prompt) => { finalPromptText = prompt; }, @@ -626,7 +623,6 @@ export async function runEmbeddedAttemptSettledPhase( lastAssistant, currentAttemptAssistant, currentAttemptCompletedAssistant, - codeModeRecoveryCandidate: getCodeModeRecoveryCandidate(), successfulNestedToolNames, attemptUsage, promptCache: sessionRuntimeState.promptCache, diff --git a/src/agents/embedded-agent-runner/run/attempt-stream-finalize.test.ts b/src/agents/embedded-agent-runner/run/attempt-stream-finalize.test.ts index ee87cf8bc088..684b8e2b379a 100644 --- a/src/agents/embedded-agent-runner/run/attempt-stream-finalize.test.ts +++ b/src/agents/embedded-agent-runner/run/attempt-stream-finalize.test.ts @@ -122,11 +122,8 @@ function createFixture(overrides: FixtureOverrides = {}) { agentSession: { activeSession, clientToolCallSlots: [], - coreReadAuthorized: true, - getCodeModeRecoveryCandidate: vi.fn(() => undefined), hasDeliveredSourceReply: vi.fn(() => false), hookRunner: {}, - setCodeModeReconciliationReadAuthorized: vi.fn(), setActiveSessionSystemPrompt: vi.fn(), settingsManager: { getCompactionReserveTokens: vi.fn(() => 1_000) }, }, diff --git a/src/agents/embedded-agent-runner/run/attempt-stream-prepare.ts b/src/agents/embedded-agent-runner/run/attempt-stream-prepare.ts index 2ac313814683..d2ffb69a5c98 100644 --- a/src/agents/embedded-agent-runner/run/attempt-stream-prepare.ts +++ b/src/agents/embedded-agent-runner/run/attempt-stream-prepare.ts @@ -46,8 +46,6 @@ import { getInternalToolExecutionPreparer, } from "../../runtime/internal-hooks.js"; import type { AgentSession } from "../../sessions/index.js"; -import { hashToolCall } from "../../tool-loop-detection.js"; -import { normalizeToolPolicyName } from "../../tool-policy.js"; import type { ToolSearchCatalogToolExecutor } from "../../tool-search.js"; import { redactTranscriptMessage } from "../../transcript-redact.js"; import { log } from "../logger.js"; @@ -73,7 +71,6 @@ import { steerActiveSessionWithOptionalDeliveryWait, } from "./attempt-queue-message.js"; import type { EmbeddedAttemptClientToolCallSlot } from "./attempt-result.js"; -import { registerCodeModeRecoveryJournalEntry } from "./code-mode-recovery-journal.js"; import { createEmbeddedAttemptDeferredLifecycleOwner, type EmbeddedAttemptDeferredLifecycleOwner, @@ -441,13 +438,6 @@ export function prepareEmbeddedAttemptStream(input: { if (!recorded) { throw new Error("Nested activity became invalid during transcript redaction"); } - registerCodeModeRecoveryJournalEntry(recorded, { - actionKey: hashToolCall( - normalizeToolPolicyName(toolParams.toolName), - terminal.executedArguments, - ), - effectState: terminal.effectReceipt.state, - }); input.nestedToolActivities.push(recorded); }, ); diff --git a/src/agents/embedded-agent-runner/run/attempt-tool-catalog.ts b/src/agents/embedded-agent-runner/run/attempt-tool-catalog.ts index 81ea6835955f..9663409e12ec 100644 --- a/src/agents/embedded-agent-runner/run/attempt-tool-catalog.ts +++ b/src/agents/embedded-agent-runner/run/attempt-tool-catalog.ts @@ -12,11 +12,6 @@ import { markAgentToolExecutionUnavailable, } from "../../agent-tool-availability.js"; import { wrapToolWithAbortSignal } from "../../agent-tools.abort.js"; -import { - isToolWrappedWithBeforeToolCallHook, - rewrapToolWithBeforeToolCallHook, - wrapToolWithBeforeToolCallHook, -} from "../../agent-tools.before-tool-call.js"; import { resolveToolLoopDetectionConfig } from "../../agent-tools.js"; import { CODE_MODE_EXEC_TOOL_NAME, @@ -43,7 +38,6 @@ import type { prepareEmbeddedAttemptBundleTools } from "./attempt-bundle-tools.j import { collectAttemptExplicitToolAllowlistSources } from "./attempt-tool-allowlist.js"; import type { prepareEmbeddedAttemptToolBase } from "./attempt-tool-prepare.js"; import { buildToolSearchRunPlan } from "./attempt-tool-search-run-plan.js"; -import { applyCodeModeRecoveryToolSurface } from "./code-mode-reconciliation.js"; import { wrapEmbeddedAttemptToolWithActivity } from "./tool-activity-heartbeat.js"; import type { EmbeddedRunAttemptParams } from "./types.js"; @@ -103,29 +97,6 @@ export function prepareEmbeddedAttemptToolCatalog(input: { onToolOutcome: attempt.onToolOutcome, allocateToolOutcomeOrdinal: attempt.allocateToolOutcomeOrdinal, }; - if (attempt.codeModeRecovery?.kind === "resume") { - if (toolSearchControlsEnabledForRun) { - effectiveTools = effectiveTools.map((tool) => { - const prepareInput = typeof tool.prepareBeforeToolCallParams === "function"; - if (!isToolWrappedWithBeforeToolCallHook(tool)) { - return wrapToolWithBeforeToolCallHook( - tool, - catalogToolHookContext, - prepareInput ? { protectNetworkErrors: false } : undefined, - ); - } - return prepareInput - ? rewrapToolWithBeforeToolCallHook(tool, catalogToolHookContext, { - protectNetworkErrors: false, - }) - : tool; - }); - } - effectiveTools = applyCodeModeRecoveryToolSurface({ - tools: effectiveTools, - state: attempt.codeModeRecovery, - }); - } const codeModeTools = codeModeControlsEnabledForRun ? createCodeModeTools({ config: attempt.config, @@ -187,12 +158,6 @@ export function prepareEmbeddedAttemptToolCatalog(input: { attempt.runId, ), ); - if (attempt.codeModeRecovery?.kind === "inspect") { - effectiveTools = applyCodeModeRecoveryToolSurface({ - tools: effectiveTools, - state: attempt.codeModeRecovery, - }); - } if (codeModeControlsEnabledForRun && isCodeModeDiagnosticEnabled()) { logCodeModeDiagnostic(log, "final-surface", { runId: attempt.runId, diff --git a/src/agents/embedded-agent-runner/run/attempt-tool-prepare.ts b/src/agents/embedded-agent-runner/run/attempt-tool-prepare.ts index 8242a7a2c5c3..f3ec50db77fa 100644 --- a/src/agents/embedded-agent-runner/run/attempt-tool-prepare.ts +++ b/src/agents/embedded-agent-runner/run/attempt-tool-prepare.ts @@ -17,7 +17,7 @@ import type { NestedToolActivity } from "../../../sessions/nested-tool-activity. import { createOpenClawCodingTools } from "../../agent-tools.js"; import { createSkillInstructionDeliveryCache } from "../../agent-tools.read.js"; import { getChannelAgentToolMeta } from "../../channel-tools.js"; -import { createCodeModePermissionChangeReason } from "../../code-mode-repair-provenance.js"; +import { createCodeModePermissionChangeReason } from "../../code-mode-permission-change.js"; import type { CodeModeSkill } from "../../code-mode-skills.js"; import { resolveConversationCapabilityProfile } from "../../conversation-capability-profile.js"; import { @@ -31,7 +31,7 @@ import { resolveSessionPermissionExecMode, type PreparedSessionPermissionPolicy, } from "../../tool-fs-policy.js"; -import { normalizeToolPolicyName, toolPolicyRestrictsTools } from "../../tool-policy.js"; +import { toolPolicyRestrictsTools } from "../../tool-policy.js"; import { isAgentToolRestartSafe } from "../../tool-replay-safety.js"; import { createToolSearchCatalogRef, @@ -78,18 +78,13 @@ export function prepareEmbeddedAttemptToolBase(params: { toolSearchCatalogExecutor: ToolSearchCatalogToolExecutor; }) { const { attempt } = params; - const inspectingCodeModeRecovery = attempt.codeModeRecovery?.kind === "inspect"; - const forceDirectMessageTool = inspectingCodeModeRecovery - ? false - : messageToolOwnsVisibleReply(attempt); + const forceDirectMessageTool = messageToolOwnsVisibleReply(attempt); const toolRunContext = buildEmbeddedAttemptToolRunContext({ ...attempt, forceMessageTool: forceDirectMessageTool, trace: params.runTrace, }); - const toolsAllowWithForcedRuntimeTools = inspectingCodeModeRecovery - ? ["read"] - : toolRunContext.runtimeToolAllowlist; + const toolsAllowWithForcedRuntimeTools = toolRunContext.runtimeToolAllowlist; const toolsEnabled = supportsModelTools(attempt.model); const isRawModelRun = attempt.modelRun === true || attempt.promptMode === "none"; const toolConstructionPlan = resolveEmbeddedAttemptToolConstructionPlan({ @@ -117,7 +112,6 @@ export function prepareEmbeddedAttemptToolBase(params: { isRawModelRun, toolsAllow: attempt.toolsAllow, forceCodeModeControls: attempt.forceCodeModeTools, - forceDirectTools: inspectingCodeModeRecovery, }); if (isCodeModeDiagnosticEnabled()) { logCodeModeDiagnostic(log, "activation", { @@ -392,11 +386,9 @@ export function prepareEmbeddedAttemptToolBase(params: { params.markCoreToolStage("attempt:tools-allow"); return filteredTools; })(); - const toolsRaw = inspectingCodeModeRecovery - ? constructedToolsRaw.filter((tool) => normalizeToolPolicyName(tool.name) === "read") - : attempt.forceRestartSafeTools - ? constructedToolsRaw.filter((tool) => isAgentToolRestartSafe(tool, restartSafetyOptions)) - : constructedToolsRaw; + const toolsRaw = attempt.forceRestartSafeTools + ? constructedToolsRaw.filter((tool) => isAgentToolRestartSafe(tool, restartSafetyOptions)) + : constructedToolsRaw; if (attempt.forceRestartSafeTools) { log.info( `restart-safe recovery tool policy retained ${toolsRaw.length}/${constructedToolsRaw.length} concrete tools`, diff --git a/src/agents/embedded-agent-runner/run/attempt.code-mode-continuation.test.ts b/src/agents/embedded-agent-runner/run/attempt.code-mode-continuation.test.ts new file mode 100644 index 000000000000..8b9e078347e1 --- /dev/null +++ b/src/agents/embedded-agent-runner/run/attempt.code-mode-continuation.test.ts @@ -0,0 +1,209 @@ +import { + createAssistantMessageEventStream, + type AssistantMessage, + type Context, + type Model, +} from "openclaw/plugin-sdk/llm"; +import { afterEach, beforeAll, beforeEach, describe, expect, it } from "vitest"; +import { readNestedToolActivity } from "../../../sessions/nested-tool-activity.js"; +import { + fakeTool, + pluginToolWithExecute, + resetCodeModeTestState, +} from "../../code-mode.test-support.js"; +import { Agent, type AgentTool } from "../../runtime/index.js"; +import { SessionManager } from "../../sessions/session-manager.js"; +import { isToolResultError } from "../../tool-result-error.js"; +import { jsonResult } from "../../tools/common.js"; +import { + cleanupTempPaths, + createContextEngineAttemptRunner, + createContextEngineBootstrapAndAssemble, + createDefaultEmbeddedSession, + getHoisted, + preloadRunEmbeddedAttemptForTests, + resetEmbeddedAttemptHarness, +} from "./attempt-spawn-workspace.test-support.js"; + +const hoisted = getHoisted(); +const tempPaths: string[] = []; +const model: Model = { + id: "test-model", + name: "Test Model", + api: "openai-responses", + provider: "openai", + baseUrl: "https://example.test", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 8_192, + maxTokens: 8_192, +}; + +function streamAssistant(content: AssistantMessage["content"]) { + const message: AssistantMessage = { + role: "assistant", + content, + api: model.api, + provider: model.provider, + model: model.id, + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: content.some((entry) => entry.type === "toolCall") ? "toolUse" : "stop", + timestamp: Date.now(), + }; + const stream = createAssistantMessageEventStream(); + queueMicrotask(() => { + stream.push({ + type: "done", + reason: message.stopReason === "toolUse" ? "toolUse" : "stop", + message, + }); + stream.end(); + }); + return stream; +} + +describe("runEmbeddedAttempt Code Mode recovery boundary", () => { + beforeAll(async () => { + await preloadRunEmbeddedAttemptForTests(); + }); + + beforeEach(() => { + resetEmbeddedAttemptHarness(); + }); + + afterEach(async () => { + resetCodeModeTestState(); + await cleanupTempPaths(tempPaths); + }); + + it("keeps the normal Code Mode surface through failure, inspection, and multiple edits", async () => { + const sessionManager = SessionManager.inMemory(); + const appliedChanges: string[] = []; + const read = fakeTool("read", "Inspect current file contents"); + const shell = fakeTool("shell_command", "Inspect source through a shell"); + const applyPatch = pluginToolWithExecute("apply_patch", "Apply a patch", async () => { + appliedChanges.push("first hunk applied"); + throw new Error("second hunk is ambiguous"); + }); + const write = pluginToolWithExecute("write", "Write a file", async (_id, input) => { + appliedChanges.push((input as { value: string }).value); + return jsonResult({ written: true }); + }); + hoisted.createOpenClawCodingToolsMock.mockReturnValue([read, shell, applyPatch, write]); + + const programs = [ + "return await apply_patch({});", + "return await shell_command({});", + 'return await write({ value: "second hunk applied" });', + 'return await write({ value: "third hunk applied" });', + "return await read({});", + ]; + const providerContexts: Context[] = []; + const createSession = () => { + const session = createDefaultEmbeddedSession(); + const options = hoisted.createAgentSessionMock.mock.calls.at(-1)?.[0] as { + customTools: AgentTool[]; + }; + const allTools = options.customTools; + const agent = new Agent({ + initialState: { model, tools: allTools }, + afterToolCall: async ({ result, isError }) => ({ + isError: isError || isToolResultError(result), + }), + streamFn: (_activeModel, context) => { + const turn = providerContexts.length; + providerContexts.push(context); + const code = programs[turn]; + return streamAssistant( + code === undefined + ? [{ type: "text", text: "all changes verified" }] + : [{ type: "toolCall", id: `program-${turn}`, name: "exec", arguments: { code } }], + ); + }, + }); + session.agent = agent as typeof session.agent; + Object.defineProperty(session, "messages", { + get: () => agent.state.messages, + set: (messages) => { + agent.state.messages = messages; + }, + }); + session.setActiveToolsByName = (toolNames) => { + agent.state.tools = allTools.filter((tool) => toolNames.includes(tool.name)); + }; + session.getActiveToolNames = () => agent.state.tools.map((tool) => tool.name); + session.prompt = async (prompt, promptOptions) => { + promptOptions?.preflightResult?.(true); + await agent.prompt(prompt); + }; + return session; + }; + + const result = await createContextEngineAttemptRunner({ + contextEngine: createContextEngineBootstrapAndAssemble(), + createSession, + sessionKey: "agent:main:main", + tempPaths, + attemptOverrides: { + config: { tools: { codeMode: true } }, + sessionManager, + disableMessageTool: false, + disableTools: false, + model, + }, + }); + + expect(providerContexts).toHaveLength(6); + for (const context of providerContexts) { + expect(context.tools?.map((tool) => tool.name)).toContain("exec"); + } + expect(providerContexts[1]?.messages).toContainEqual( + expect.objectContaining({ + role: "toolResult", + isError: true, + content: [ + expect.objectContaining({ text: expect.stringContaining("second hunk is ambiguous") }), + ], + }), + ); + expect(appliedChanges).toEqual([ + "first hunk applied", + "second hunk applied", + "third hunk applied", + ]); + expect(applyPatch.execute).toHaveBeenCalledOnce(); + expect(shell.execute).toHaveBeenCalledOnce(); + expect(write.execute).toHaveBeenCalledTimes(2); + expect(read.execute).toHaveBeenCalledOnce(); + expect(result.messagesSnapshot.at(-1)).toMatchObject({ + role: "assistant", + content: [{ type: "text", text: "all changes verified" }], + }); + const activities = sessionManager.getEntries().flatMap((entry) => { + const activity = entry.type === "message" && readNestedToolActivity(entry.message); + return activity ? [activity.details] : []; + }); + expect(activities).toMatchObject([ + { toolName: "apply_patch", isError: true }, + { toolName: "shell_command", isError: false }, + { toolName: "write", isError: false }, + { toolName: "write", isError: false }, + { toolName: "read", isError: false }, + ]); + expect(activities.map((activity) => activity.parentToolCallId)).toEqual( + result.messagesSnapshot.flatMap((message) => + message.role === "assistant" + ? message.content.flatMap((entry) => (entry.type === "toolCall" ? [entry.id] : [])) + : [], + ), + ); + }); +}); diff --git a/src/agents/embedded-agent-runner/run/attempt.code-mode-reconciliation.test.ts b/src/agents/embedded-agent-runner/run/attempt.code-mode-reconciliation.test.ts deleted file mode 100644 index 7d672d4795a4..000000000000 --- a/src/agents/embedded-agent-runner/run/attempt.code-mode-reconciliation.test.ts +++ /dev/null @@ -1,392 +0,0 @@ -import { - createAssistantMessageEventStream, - type AssistantMessage, - type Context, - type Model, -} from "openclaw/plugin-sdk/llm"; -import { afterEach, beforeAll, beforeEach, describe, expect, it } from "vitest"; -import { readNestedToolActivity } from "../../../sessions/nested-tool-activity.js"; -import { - fakeTool, - mcpTool, - pluginToolWithExecute, - resetCodeModeTestState, -} from "../../code-mode.test-support.js"; -import { Agent, type AgentTool } from "../../runtime/index.js"; -import { SessionManager } from "../../sessions/session-manager.js"; -import { createToolSearchTools } from "../../tool-search.js"; -import { jsonResult } from "../../tools/common.js"; -import { - cleanupTempPaths, - createContextEngineAttemptRunner, - createContextEngineBootstrapAndAssemble, - createDefaultEmbeddedSession, - getHoisted, - preloadRunEmbeddedAttemptForTests, - resetEmbeddedAttemptHarness, -} from "./attempt-spawn-workspace.test-support.js"; -import { advanceCodeModeRecovery } from "./code-mode-reconciliation.js"; -import { createEmbeddedRunTerminalRetryState } from "./terminal-retry-state.js"; - -const hoisted = getHoisted(); -const tempPaths: string[] = []; -const model: Model = { - id: "test-model", - name: "Test Model", - api: "openai-responses", - provider: "openai", - baseUrl: "https://example.test", - reasoning: false, - input: ["text"], - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, - contextWindow: 8_192, - maxTokens: 8_192, -}; - -function streamAssistant(content: AssistantMessage["content"]) { - const message: AssistantMessage = { - role: "assistant", - content, - api: model.api, - provider: model.provider, - model: model.id, - usage: { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - totalTokens: 0, - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - }, - stopReason: content.some((entry) => entry.type === "toolCall") ? "toolUse" : "stop", - timestamp: Date.now(), - }; - const stream = createAssistantMessageEventStream(); - queueMicrotask(() => { - stream.push({ - type: "done", - reason: message.stopReason === "toolUse" ? "toolUse" : "stop", - message, - }); - stream.end(); - }); - return stream; -} - -describe("runEmbeddedAttempt Code Mode recovery boundary", () => { - beforeAll(async () => { - await preloadRunEmbeddedAttemptForTests(); - }); - - beforeEach(() => { - resetEmbeddedAttemptHarness(); - }); - - afterEach(async () => { - resetCodeModeTestState(); - await cleanupTempPaths(tempPaths); - }); - - it("inspects a partial mutation, then resumes through Tool Search behind a replay fence", async () => { - const sessionManager = SessionManager.inMemory(); - const appliedChanges: string[] = []; - const read = fakeTool("read", "Inspect current file contents"); - const computer = fakeTool("computer", "Observe the computer"); - computer.catalogMode = "direct-only"; - computer.parameters = { type: "object", properties: { action: { type: "string" } } }; - const applyPatch = pluginToolWithExecute("apply_patch", "Apply a patch", async () => { - appliedChanges.push("first hunk applied"); - throw new Error("second hunk is ambiguous"); - }); - const write = pluginToolWithExecute("write", "Write a file", async () => { - throw new Error("recovery write failed"); - }); - const message = pluginToolWithExecute("message", "Send a message", async () => jsonResult({})); - const shell = pluginToolWithExecute("shell_command", "Run a shell", async () => jsonResult({})); - const remoteMutation = mcpTool({ - name: "remote_mutation", - serverName: "remote", - toolName: "mutate", - }); - const coreTools = [read, computer, applyPatch, write, message, shell, remoteMutation]; - hoisted.createOpenClawCodingToolsMock.mockImplementation((rawOptions) => { - const options = rawOptions as { - includeToolSearchControls?: boolean; - config?: Parameters[0]["config"]; - toolSearchCatalogRef?: Parameters[0]["catalogRef"]; - toolSearchCatalogExecutor?: Parameters[0]["executeTool"]; - }; - return [ - ...coreTools, - ...(options.includeToolSearchControls - ? createToolSearchTools({ - config: options.config, - runtimeConfig: options.config, - agentId: "main", - sessionKey: "agent:main:main", - sessionId: "session-code-mode-recovery", - runId: "run-code-mode-recovery", - catalogRef: options.toolSearchCatalogRef, - executeTool: options.toolSearchCatalogExecutor, - }) - : []), - ]; - }); - - const providerContexts: Context[] = []; - const retryState = createEmbeddedRunTerminalRetryState(); - let phase: "mutation" | "inspection" | "resume" = "mutation"; - const baseSubscribe = hoisted.subscribeEmbeddedAgentSessionMock.getMockImplementation(); - if (!baseSubscribe) { - throw new Error("Missing embedded subscription test implementation"); - } - hoisted.subscribeEmbeddedAgentSessionMock.mockImplementation((params) => { - const subscription = baseSubscribe(params); - if (phase === "inspection") { - subscription.toolMetas.push( - { toolName: "read", isError: false }, - { toolName: "recovery_resume", isError: false, terminate: true }, - ); - } - return subscription; - }); - const createSession = () => { - const session = createDefaultEmbeddedSession(); - const options = hoisted.createAgentSessionMock.mock.calls.at(-1)?.[0] as { - customTools: AgentTool[]; - }; - const allTools = options.customTools; - let assistantTurn = 0; - const agent = new Agent({ - initialState: { model, tools: allTools }, - streamFn: (_activeModel, context) => { - providerContexts.push(context); - const turn = assistantTurn++; - if (phase === "inspection") { - if (turn === 0) { - return streamAssistant([ - { type: "toolCall", id: "observe", name: "read", arguments: { value: "file" } }, - ]); - } - if (turn === 1) { - return streamAssistant([ - { type: "toolCall", id: "resume", name: "recovery_resume", arguments: {} }, - ]); - } - } - if (phase === "resume") { - if (turn === 0 || turn === 7) { - return streamAssistant([ - { - type: "toolCall", - id: `computer-observe-${turn}`, - name: "computer", - arguments: { action: "list_windows" }, - }, - ]); - } - if (turn === 1) { - return streamAssistant([ - { - type: "toolCall", - id: "search-patch", - name: "tool_search", - arguments: { query: "apply_patch", limit: 1 }, - }, - ]); - } - if (turn === 2) { - return streamAssistant([ - { - type: "toolCall", - id: "describe-patch", - name: "tool_describe", - arguments: { id: "apply_patch" }, - }, - ]); - } - if (turn === 3) { - return streamAssistant([ - { - type: "toolCall", - id: "replay", - name: "tool_call", - arguments: { id: "apply_patch", args: {} }, - }, - ]); - } - if (turn === 4) { - return streamAssistant([ - { - type: "toolCall", - id: "continue", - name: "tool_call", - arguments: { id: "write", args: { value: "remaining work" } }, - }, - ]); - } - if (turn === 5) { - return streamAssistant([ - { - type: "toolCall", - id: "blind-later-work", - name: "tool_call", - arguments: { id: "remote_mutation", args: { value: "later work" } }, - }, - ]); - } - if (turn === 6) { - return streamAssistant([ - { - type: "toolCall", - id: "verify", - name: "tool_call", - arguments: { id: "read", args: { value: "file" } }, - }, - ]); - } - return streamAssistant([{ type: "text", text: "recovery completed" }]); - } - if (turn === 0) { - return streamAssistant([ - { - type: "toolCall", - id: "mutate", - name: "exec", - arguments: { code: "return await apply_patch({});" }, - }, - ]); - } - return streamAssistant([{ type: "text", text: "first hunk applied" }]); - }, - }); - session.agent = agent as typeof session.agent; - Object.defineProperty(session, "messages", { - get: () => agent.state.messages, - set: (messages) => { - agent.state.messages = messages; - }, - }); - session.setActiveToolsByName = (toolNames) => { - agent.state.tools = allTools.filter((tool) => toolNames.includes(tool.name)); - }; - session.getActiveToolNames = () => agent.state.tools.map((tool) => tool.name); - session.prompt = async (prompt, promptOptions) => { - promptOptions?.preflightResult?.(true); - await agent.prompt(prompt); - }; - return session; - }; - - const runAttempt = (overrides = {}) => - createContextEngineAttemptRunner({ - contextEngine: createContextEngineBootstrapAndAssemble(), - createSession, - sessionKey: "agent:main:main", - tempPaths, - attemptOverrides: { - config: { tools: { codeMode: true } }, - sessionManager, - disableMessageTool: false, - disableTools: false, - model, - ...overrides, - }, - }); - - const firstAttempt = await runAttempt(); - expect(firstAttempt.codeModeRecoveryCandidate?.blockedActionKeys).toHaveLength(1); - expect(appliedChanges).toEqual(["first hunk applied"]); - expect(applyPatch.execute).toHaveBeenCalledOnce(); - expect(remoteMutation.execute).not.toHaveBeenCalled(); - const activities = sessionManager.getEntries().flatMap((entry) => { - const activity = entry.type === "message" && readNestedToolActivity(entry.message); - return activity ? [activity.details] : []; - }); - expect(activities).toMatchObject([ - { - parentToolCallId: "mutate", - toolName: "apply_patch", - isError: true, - }, - ]); - - let inspectionPrompt = ""; - expect( - advanceCodeModeRecovery({ - attempt: firstAttempt, - hostOwnsToolSurface: true, - retryState, - activateInternalPrompt: (prompt) => { - inspectionPrompt = prompt; - }, - }), - ).toBe(true); - - phase = "inspection"; - const inspectionAttempt = await runAttempt({ - codeModeRecovery: retryState.codeModeRecovery, - prompt: inspectionPrompt, - }); - expect(providerContexts[2]?.tools?.map((tool) => tool.name)).toEqual([ - "read", - "recovery_resume", - ]); - expect(read.execute).toHaveBeenCalledOnce(); - expect(write.execute).not.toHaveBeenCalled(); - expect(retryState.codeModeRecovery).toMatchObject({ - kind: "inspect", - phase: "ready", - }); - expect(inspectionAttempt.toolMetas).toEqual( - expect.arrayContaining([ - expect.objectContaining({ toolName: "read", isError: false }), - expect.objectContaining({ - toolName: "recovery_resume", - isError: false, - terminate: true, - }), - ]), - ); - - let resumePrompt = ""; - expect( - advanceCodeModeRecovery({ - attempt: inspectionAttempt, - hostOwnsToolSurface: true, - retryState, - activateInternalPrompt: (prompt) => { - resumePrompt = prompt; - }, - }), - ).toBe(true); - expect(retryState.codeModeRecovery.kind).toBe("resume"); - - phase = "resume"; - await runAttempt({ - codeModeOverride: false, - codeModeRecovery: retryState.codeModeRecovery, - disableMessageTool: true, - config: { - tools: { - codeMode: false, - toolSearch: { enabled: true, mode: "tools" }, - }, - }, - prompt: resumePrompt, - }); - const resumeTools = providerContexts.at(-1)?.tools?.map((tool) => tool.name) ?? []; - expect(resumeTools).toEqual( - expect.arrayContaining(["computer", "tool_search", "tool_describe", "tool_call"]), - ); - expect(resumeTools).not.toContain("write"); - expect(resumeTools).not.toContain("apply_patch"); - expect(resumeTools).not.toContain("exec"); - expect(write.execute).toHaveBeenCalledOnce(); - expect(computer.execute).toHaveBeenCalledTimes(2); - expect(applyPatch.execute).toHaveBeenCalledOnce(); - expect(read.execute).toHaveBeenCalledTimes(2); - expect(message.execute).not.toHaveBeenCalled(); - expect(shell.execute).not.toHaveBeenCalled(); - }); -}); diff --git a/src/agents/embedded-agent-runner/run/attempt.ts b/src/agents/embedded-agent-runner/run/attempt.ts index 68d642ce878e..d669b9bb6348 100644 --- a/src/agents/embedded-agent-runner/run/attempt.ts +++ b/src/agents/embedded-agent-runner/run/attempt.ts @@ -373,7 +373,6 @@ export async function runEmbeddedAttempt( effectiveWorkspace, initialSystemPrompt: preparedSystemPrompt.systemPromptText, isRawModelRun, - nestedToolActivities: preparedToolBase.nestedToolActivities, sessionManager: { replayAllowedToolNames: toolSearchRunPlan.replayAllowedToolNames, resolveActiveContextEnginePluginId, @@ -453,16 +452,12 @@ export async function runEmbeddedAttempt( tools: preparedBundleTools.tools, catalogRef: preparedToolBase.toolSearchCatalogRef, codeModeControlsEnabled: preparedToolBase.codeModeControlsEnabledForRun, - coreReadAuthorized: preparedSessionRuntime.agentSession.coreReadAuthorized, onApplied: (surface) => { const allowedNames = new Set([ ...surface.activeToolNames, ...surface.uncompactedEffectiveTools.map((tool) => tool.name), ]); preparedToolCatalog.applyPromptToolPolicy(allowedNames); - preparedSessionRuntime.agentSession.setCodeModeReconciliationReadAuthorized( - surface.coreReadAuthorized, - ); }, forceToolNames: [ ...(preparedToolBase.forceDirectMessageTool ? ["message"] : []), diff --git a/src/agents/embedded-agent-runner/run/code-mode-outcome.test.ts b/src/agents/embedded-agent-runner/run/code-mode-outcome.test.ts deleted file mode 100644 index 4341280c360a..000000000000 --- a/src/agents/embedded-agent-runner/run/code-mode-outcome.test.ts +++ /dev/null @@ -1,186 +0,0 @@ -import { describe, expect, it, vi } from "vitest"; -import { registerRepairableCodeModeFailure } from "../../code-mode-repair-provenance.js"; -import type { Agent } from "../../runtime/index.js"; -import { readToolResultDetails } from "../../tool-result-error.js"; -import { installCodeModeOutcomeHook } from "./code-mode-outcome.js"; - -type AfterToolOutcomeContext = Parameters>[0]; - -function createOutcome( - options: { - bridgeStarted?: boolean; - repairableFailure?: boolean; - toolName?: "exec" | "wait"; - terminal?: boolean; - } = {}, -): AfterToolOutcomeContext { - const details = { - status: "failed", - error: "execution failed", - bridgeDispatchStarted: options.bridgeStarted ?? false, - }; - if (options.repairableFailure) { - registerRepairableCodeModeFailure(details); - } - const toolCall = { - type: "toolCall" as const, - id: "call-1", - name: options.toolName ?? "exec", - arguments: {}, - }; - return { - assistantMessage: { role: "assistant", content: [toolCall], timestamp: 1 }, - toolCall, - args: {}, - result: { - content: [{ type: "text", text: JSON.stringify(details) }], - details, - ...(options.terminal ? { terminate: true } : {}), - }, - isError: true, - executionStarted: true, - context: { systemPrompt: "", messages: [], tools: [] }, - } as unknown as AfterToolOutcomeContext; -} - -function createAgent(previous?: Agent["afterToolOutcome"]) { - const agent = { afterToolOutcome: previous } as Agent; - const onReconciliationCandidate = vi.fn(); - installCodeModeOutcomeHook({ agent, onReconciliationCandidate }); - return { agent, onReconciliationCandidate }; -} - -describe("Code Mode outcome safety", () => { - it("allows successive guest failures without a repair-attempt counter", async () => { - const { agent, onReconciliationCandidate } = createAgent(); - - for (let attempt = 0; attempt < 3; attempt += 1) { - const result = await agent.afterToolOutcome?.(createOutcome()); - expect(result).toMatchObject({ isError: true }); - expect(result).not.toHaveProperty("terminate"); - } - expect(onReconciliationCandidate).not.toHaveBeenCalled(); - }); - - it.each(["exec", "wait"] as const)( - "accepts exact host proof once for %s, never copied or serialized proof", - async (toolName) => { - const outcome = createOutcome({ toolName, bridgeStarted: true, repairableFailure: true }); - const details = readToolResultDetails(outcome.result); - for (const copied of [ - { ...details }, - structuredClone(details), - // oxlint-disable-next-line unicorn/prefer-structured-clone -- Exercise serialized tool results separately from in-memory clones. - JSON.parse(JSON.stringify(details)), - ]) { - const { agent } = createAgent(); - await expect( - agent.afterToolOutcome?.({ - ...outcome, - result: { ...outcome.result, details: copied }, - }), - ).resolves.toMatchObject({ isError: true, terminate: true }); - } - - const { agent, onReconciliationCandidate } = createAgent(); - const result = await agent.afterToolOutcome?.(outcome); - expect(result).toMatchObject({ isError: true }); - expect.soft(result).not.toHaveProperty("terminate"); - expect(onReconciliationCandidate).not.toHaveBeenCalled(); - // The real consumer used the proof above; never mint it again for the replay. - await expect(createAgent().agent.afterToolOutcome?.(outcome)).resolves.toMatchObject({ - isError: true, - terminate: true, - }); - }, - ); - - it("sends uncertain bridge side effects to read-only reconciliation", async () => { - const { agent, onReconciliationCandidate } = createAgent(); - - await expect( - agent.afterToolOutcome?.(createOutcome({ bridgeStarted: true })), - ).resolves.toMatchObject({ isError: true, terminate: true }); - expect(onReconciliationCandidate).toHaveBeenCalledOnce(); - }); - - it.each([ - { bridgeStarted: true, executionStarted: true }, - { bridgeStarted: false, executionStarted: false }, - { bridgeStarted: undefined, executionStarted: false }, - ])("keeps unproven wait failures closed: %j", async ({ bridgeStarted, executionStarted }) => { - const { agent, onReconciliationCandidate } = createAgent(); - const outcome = createOutcome({ toolName: "wait" }); - outcome.result.details = { - status: "failed", - ...(bridgeStarted === undefined ? {} : { bridgeDispatchStarted: bridgeStarted }), - }; - outcome.executionStarted = executionStarted; - - await expect(agent.afterToolOutcome?.(outcome)).resolves.toMatchObject({ - isError: true, - terminate: true, - }); - expect(onReconciliationCandidate).not.toHaveBeenCalled(); - }); - - it.each(["exec", "wait"] as const)( - "does not trust permission-change strings from a failed %s", - async (toolName) => { - const { agent } = createAgent(); - const outcome = createOutcome({ toolName, bridgeStarted: true }); - outcome.result.details = { - status: "failed", - code: "aborted", - error: "Permission change", - permissionChanged: true, - bridgeDispatchStarted: true, - }; - await expect(agent.afterToolOutcome?.(outcome)).resolves.toMatchObject({ - isError: true, - terminate: true, - }); - }, - ); - - it("preserves original dispatch evidence when another hook rewrites the result", async () => { - const { agent, onReconciliationCandidate } = createAgent(async () => ({ - content: [{ type: "text", text: "looks successful" }], - details: { status: "completed" }, - isError: false, - terminate: false, - })); - - await expect( - agent.afterToolOutcome?.(createOutcome({ bridgeStarted: true })), - ).resolves.toMatchObject({ - details: { status: "failed", bridgeDispatchStarted: true }, - isError: true, - terminate: true, - }); - expect(onReconciliationCandidate).toHaveBeenCalledOnce(); - }); - - it("keeps explicit terminal outcomes and hook failures closed", async () => { - const terminalAgent = createAgent(async () => ({ terminate: false })); - await expect( - terminalAgent.agent.afterToolOutcome?.( - createOutcome({ - toolName: "wait", - terminal: true, - bridgeStarted: true, - repairableFailure: true, - }), - ), - ).resolves.toMatchObject({ terminate: true }); - - const brokenHook = createAgent(async () => { - throw new Error("hook failed"); - }); - await expect(brokenHook.agent.afterToolOutcome?.(createOutcome())).resolves.toMatchObject({ - isError: true, - terminate: true, - details: { status: "failed", error: "hook failed" }, - }); - }); -}); diff --git a/src/agents/embedded-agent-runner/run/code-mode-outcome.ts b/src/agents/embedded-agent-runner/run/code-mode-outcome.ts deleted file mode 100644 index c56723462c47..000000000000 --- a/src/agents/embedded-agent-runner/run/code-mode-outcome.ts +++ /dev/null @@ -1,85 +0,0 @@ -import { - CODE_MODE_EXEC_TOOL_NAME, - CODE_MODE_WAIT_TOOL_NAME, -} from "../../code-mode-control-tools.js"; -import { - consumeCodeModePermissionChangeResult, - consumeRepairableCodeModeFailure, -} from "../../code-mode-repair-provenance.js"; -import type { AfterToolCallResult, Agent } from "../../runtime/index.js"; -import { readToolResultDetails } from "../../tool-result-error.js"; - -/** Preserve the model's ordinary error recovery without replaying uncertain mutations. */ -export function installCodeModeOutcomeHook(params: { - agent: Agent; - onReconciliationCandidate?: (parentToolCallId: string) => void; -}): void { - const previousAfterToolOutcome = params.agent.afterToolOutcome?.bind(params.agent); - - params.agent.afterToolOutcome = async (context, signal) => { - const isCodeModeExec = context.toolCall.name === CODE_MODE_EXEC_TOOL_NAME; - const isCodeModeWait = context.toolCall.name === CODE_MODE_WAIT_TOOL_NAME; - if (!isCodeModeExec && !isCodeModeWait) { - return await previousAfterToolOutcome?.(context, signal); - } - - const details = readToolResultDetails(context.result); - // Exact host proof covers the full cell history, including work before wait; copies cannot grant it. - const repairableFailure = consumeRepairableCodeModeFailure(details); - const permissionChanged = consumeCodeModePermissionChangeResult(details); - let prior: AfterToolCallResult | undefined; - try { - prior = await previousAfterToolOutcome?.(context, signal); - } catch (error) { - const message = error instanceof Error ? error.message : String(error); - return { - content: [{ type: "text", text: `Code Mode outcome hook failed: ${message}` }], - details: { ...details, status: "failed", error: message }, - isError: true, - terminate: true, - }; - } - - if (context.result.terminate === true || prior?.terminate === true) { - return { ...prior, terminate: true }; - } - if (signal?.aborted && !context.executionStarted) { - return prior; - } - if ( - (details?.status === "blocked" && details.deniedReason === "tool-loop") || - (details?.status === "skipped" && details.deniedReason === "steering") - ) { - return prior; - } - - const failed = context.isError || details?.status === "failed" || prior?.isError === true; - if (!failed) { - return prior; - } - - const bridgeStarted = details?.bridgeDispatchStarted === true; - const dispatchUnknown = - context.executionStarted && typeof details?.bridgeDispatchStarted !== "boolean"; - const unsafeToContinue = - (!permissionChanged || signal?.aborted === true) && - (isCodeModeWait || bridgeStarted || dispatchUnknown) && - !repairableFailure; - if ( - unsafeToContinue && - isCodeModeExec && - context.assistantMessage.content.filter((entry) => entry.type === "toolCall").length === 1 - ) { - params.onReconciliationCandidate?.(context.toolCall.id); - } - - // Agent core owns ordinary continuation; only uncertain side effects need a restricted retry. - return { - ...prior, - content: context.result.content, - details: context.result.details, - isError: true, - ...(unsafeToContinue ? { terminate: true } : {}), - }; - }; -} diff --git a/src/agents/embedded-agent-runner/run/code-mode-reconciliation.test.ts b/src/agents/embedded-agent-runner/run/code-mode-reconciliation.test.ts deleted file mode 100644 index eb08310fe7b4..000000000000 --- a/src/agents/embedded-agent-runner/run/code-mode-reconciliation.test.ts +++ /dev/null @@ -1,310 +0,0 @@ -import { describe, expect, it, vi } from "vitest"; -import { createNestedToolActivity } from "../../../sessions/nested-tool-activity.js"; -import { fakeTool, pluginToolWithExecute } from "../../code-mode.test-support.js"; -import { - getInternalToolExecutionPreparer, - type InternalToolExecutionPreparer, -} from "../../runtime/internal-hooks.js"; -import { makeEmbeddedRunnerAttempt } from "../../test-helpers/embedded-agent-runner-e2e-fixtures.js"; -import { hashToolCall } from "../../tool-loop-detection.js"; -import { jsonResult } from "../../tools/common.js"; -import { - advanceCodeModeRecovery, - applyCodeModeRecoveryToolSurface, - buildCodeModeRecoveryCandidate, -} from "./code-mode-reconciliation.js"; -import { registerCodeModeRecoveryJournalEntry } from "./code-mode-recovery-journal.js"; -import { - createEmbeddedRunTerminalRetryState, - type CodeModeRecoveryState, -} from "./terminal-retry-state.js"; - -function eligibleAttempt() { - return makeEmbeddedRunnerAttempt({ - codeModeRecoveryCandidate: { blockedActionKeys: ["write:prior"] }, - itemLifecycle: { startedCount: 2, completedCount: 2, activeCount: 0 }, - }); -} - -function prepare( - tool: ReturnType, - args: Record, -): Promise>> { - const preparer = getInternalToolExecutionPreparer(tool); - if (!preparer) { - throw new Error("Expected recovery preparer"); - } - return preparer({ toolCallId: "call", args }); -} - -describe("Code Mode recovery", () => { - it("allows recovery when every recorded nested call proves no effect", () => { - const activity = createNestedToolActivity({ - runId: "run", - scopeId: "scope", - afterEntryId: null, - startOrder: 0, - parentToolCallId: "exec", - toolCallId: "nested", - toolName: "write", - input: { value: "test" }, - result: jsonResult({}), - isError: true, - startedAt: 1, - timestamp: 2, - }); - registerCodeModeRecoveryJournalEntry(activity, { - actionKey: hashToolCall("write", { value: "test" }), - effectState: "failed_no_effect", - }); - - expect( - buildCodeModeRecoveryCandidate({ - parentToolCallId: "exec", - nestedToolActivities: [activity], - }), - ).toEqual({ blockedActionKeys: [] }); - }); - - it("moves one quiescent candidate into read-only inspection", () => { - const retryState = createEmbeddedRunTerminalRetryState(); - let prompt = ""; - expect( - advanceCodeModeRecovery({ - attempt: eligibleAttempt(), - hostOwnsToolSurface: true, - retryState, - activateInternalPrompt: (value) => { - prompt = value; - }, - }), - ).toBe(true); - expect(retryState.codeModeRecovery).toEqual({ - kind: "inspect", - phase: "read-required", - blockedActionKeys: ["write:prior"], - }); - expect(prompt).toContain("recovery_resume"); - }); - - it.each([ - ["active tool", { itemLifecycle: { startedCount: 2, completedCount: 1, activeCount: 1 } }], - ["async work", { toolMetas: [{ toolName: "exec", asyncStarted: true }] }], - ["message delivery", { didSendViaMessagingTool: true }], - ["child session", { acceptedSessionSpawns: [{ runId: "child" }] }], - ["approval", { didSendDeterministicApprovalPrompt: true }], - ["yield", { yieldDetected: true }], - ["plugin-owned transport", {}, false], - ])("rejects a candidate with %s", (_label, overrides, hostOwnsToolSurface = true) => { - expect( - advanceCodeModeRecovery({ - attempt: { ...eligibleAttempt(), ...overrides } as ReturnType, - hostOwnsToolSurface, - retryState: createEmbeddedRunTerminalRetryState(), - activateInternalPrompt: () => undefined, - }), - ).toBe(false); - }); - - it("ends on the inspection report when no recovery is requested", () => { - const retryState = createEmbeddedRunTerminalRetryState(); - retryState.codeModeRecovery = { - kind: "inspect", - phase: "ready", - blockedActionKeys: ["write:prior"], - }; - const activateInternalPrompt = vi.fn(); - expect( - advanceCodeModeRecovery({ - attempt: makeEmbeddedRunnerAttempt({ - itemLifecycle: { startedCount: 1, completedCount: 1, activeCount: 0 }, - toolMetas: [{ toolName: "read", isError: false }], - }), - hostOwnsToolSurface: true, - retryState, - activateInternalPrompt, - }), - ).toBe(false); - expect(retryState.codeModeRecovery).toEqual({ kind: "idle" }); - expect(activateInternalPrompt).not.toHaveBeenCalled(); - }); - - it("requires a completed read before requesting bounded recovery", async () => { - const state: Extract = { - kind: "inspect", - phase: "read-required", - blockedActionKeys: ["write:prior"], - }; - const read = pluginToolWithExecute("read", "Read", async () => jsonResult({ value: "ok" })); - const tools = applyCodeModeRecoveryToolSurface({ - tools: [read, fakeTool("write", "Write")], - state, - }); - expect(tools.map((tool) => tool.name)).toEqual(["read", "recovery_resume"]); - await expect(tools[1]?.execute("resume", {})).rejects.toThrow("Use read by itself"); - const prepared = await prepare(read, { path: "proof.txt" }); - expect(prepared.kind).toBe("ready"); - if (prepared.kind === "ready") { - await prepared.execute(() => undefined); - } - expect(state.phase).toBe("ready"); - await expect(tools[1]?.execute("resume", {})).resolves.toMatchObject({ terminate: true }); - }); - - it("enters one normal-tool recovery after read and resume request", () => { - const retryState = createEmbeddedRunTerminalRetryState(); - retryState.codeModeRecovery = { - kind: "inspect", - phase: "ready", - blockedActionKeys: ["write:prior"], - }; - expect( - advanceCodeModeRecovery({ - attempt: makeEmbeddedRunnerAttempt({ - itemLifecycle: { startedCount: 2, completedCount: 2, activeCount: 0 }, - toolMetas: [ - { toolName: "read", isError: false }, - { toolName: "recovery_resume", isError: false, terminate: true }, - ], - }), - hostOwnsToolSurface: true, - retryState, - activateInternalPrompt: () => undefined, - }), - ).toBe(true); - expect(retryState.codeModeRecovery).toMatchObject({ - kind: "resume", - mutationAttempt: "available", - }); - }); - - it("blocks exact unsafe replays without consuming the mutation budget", async () => { - const args = { path: "proof.txt", content: "alpha=applied" }; - const state: Extract = { - kind: "resume", - blockedActionKeys: new Set([hashToolCall("write", args)]), - mutationAttempt: "available", - }; - const write = applyCodeModeRecoveryToolSurface({ - tools: [fakeTool("write", "Write")], - state, - })[0]!; - const blocked = await prepare(write, args); - expect(blocked).toMatchObject({ - kind: "immediate", - outcome: { kind: "result", isError: true }, - }); - expect(state.mutationAttempt).toBe("available"); - }); - - it("allows reads after one mutation and blocks later mutation work", async () => { - const firstExecute = vi.fn(async () => { - throw new Error("first recovery mutation failed"); - }); - const state: Extract = { - kind: "resume", - blockedActionKeys: new Set(), - mutationAttempt: "available", - }; - const [write, read, applyPatch] = applyCodeModeRecoveryToolSurface({ - tools: [ - pluginToolWithExecute("write", "Write", firstExecute), - fakeTool("read", "Read"), - fakeTool("apply_patch", "Patch"), - ], - state, - }); - const first = await prepare(write!, { path: "proof.txt" }); - expect(first.kind).toBe("ready"); - expect(state.mutationAttempt).toBe("reserved"); - if (first.kind === "ready") { - await expect(first.execute(() => undefined)).rejects.toThrow( - "first recovery mutation failed", - ); - } - expect(state.mutationAttempt).toBe("consumed"); - expect((await prepare(read!, { path: "proof.txt" })).kind).toBe("ready"); - expect(await prepare(applyPatch!, { input: "later work" })).toMatchObject({ - kind: "immediate", - outcome: { kind: "result", isError: true }, - }); - expect(firstExecute).toHaveBeenCalledOnce(); - }); - - it("releases a reserved mutation when the prepared call does not start", async () => { - const state: Extract = { - kind: "resume", - blockedActionKeys: new Set(), - mutationAttempt: "available", - }; - const write = applyCodeModeRecoveryToolSurface({ - tools: [fakeTool("write", "Write")], - state, - })[0]!; - const prepared = await prepare(write, { path: "proof.txt" }); - expect(prepared.kind).toBe("ready"); - expect(state.mutationAttempt).toBe("reserved"); - prepared.dispose(); - expect(state.mutationAttempt).toBe("available"); - }); - - it.each([ - ["get_cursor_position", false], - ["list_windows", true], - ])( - "keeps computer %s available around one input (observation error: %s)", - async (action, fails) => { - const unsafe = { action: "key", text: "ENTER" }; - const state: Extract = { - kind: "resume", - blockedActionKeys: new Set([hashToolCall("computer", unsafe)]), - mutationAttempt: "available", - }; - const execute = vi.fn[2]>(async () => - jsonResult({}), - ); - const computer = applyCodeModeRecoveryToolSurface({ - tools: [pluginToolWithExecute("computer", "Computer", execute)], - state, - })[0]!; - const observe = async () => { - const prepared = await prepare(computer, { action }); - expect(prepared.kind).toBe("ready"); - if (prepared.kind !== "ready") { - throw new Error("Observation was blocked"); - } - try { - if (fails) { - execute.mockRejectedValueOnce(new Error("observation unavailable")); - } - const result = prepared.execute(() => undefined); - if (fails) { - await expect(result).rejects.toThrow("observation unavailable"); - } else { - await result; - } - } finally { - prepared.dispose(); - } - }; - await observe(); - expect(state.mutationAttempt).toBe("available"); - expect((await prepare(computer, unsafe)).kind).toBe("immediate"); - const input = await prepare(computer, { action: "key", text: "ESC" }); - expect(input.kind).toBe("ready"); - if (input.kind === "ready") { - await input.execute(() => undefined); - input.dispose(); - } - expect(state.mutationAttempt).toBe("consumed"); - await observe(); - expect(state.mutationAttempt).toBe("consumed"); - expect((await prepare(computer, { action: "type", text: "later" })).kind).toBe("immediate"); - expect(execute.mock.calls.map(([, args]) => args)).toEqual([ - { action }, - { action: "key", text: "ESC" }, - { action }, - ]); - }, - ); -}); diff --git a/src/agents/embedded-agent-runner/run/code-mode-reconciliation.ts b/src/agents/embedded-agent-runner/run/code-mode-reconciliation.ts deleted file mode 100644 index bcfb29e56fb3..000000000000 --- a/src/agents/embedded-agent-runner/run/code-mode-reconciliation.ts +++ /dev/null @@ -1,335 +0,0 @@ -import { Type } from "typebox"; -import { getPluginToolSideEffectOwnerKey } from "../../../plugins/tool-metadata.js"; -import type { NestedToolActivity } from "../../../sessions/nested-tool-activity.js"; -import { - attachInternalToolExecutionPreparer, - getInternalToolExecutionPreparer, - type InternalToolExecutionPreparer, -} from "../../runtime/internal-hooks.js"; -import { toolEffectStateProvesNoEffect } from "../../tool-effect-receipt.js"; -import { hashToolCall } from "../../tool-loop-detection.js"; -import { buildToolMutationState } from "../../tool-mutation.js"; -import { normalizeToolPolicyName } from "../../tool-policy.js"; -import { isToolResultError } from "../../tool-result-error.js"; -import { TOOL_SEARCH_CONTROL_TOOL_NAMES } from "../../tool-search-types.js"; -import type { AnyAgentTool } from "../../tools/common.js"; -import { textResult, ToolInputError } from "../../tools/common.js"; -import { readCodeModeRecoveryJournalEntry } from "./code-mode-recovery-journal.js"; -import type { - CodeModeRecoveryCandidate, - CodeModeRecoveryState, - EmbeddedRunTerminalRetryState, -} from "./terminal-retry-state.js"; -import type { EmbeddedRunAttemptResult } from "./types.js"; - -const CODE_MODE_RECOVERY_RESUME_TOOL_NAME = "recovery_resume"; -type ToolExecutionPreparation = Awaited>; - -export function isCodeModeRecoveryResumeTool(tool: { name?: string }): boolean { - return normalizeToolPolicyName(tool.name ?? "") === CODE_MODE_RECOVERY_RESUME_TOOL_NAME; -} - -const CODE_MODE_POST_RECONCILIATION_INSTRUCTION = - "The previous uncertain Code Mode mutation was inspected. Code Mode is disabled for this bounded recovery. Use the available normal tools and their real schemas. OpenClaw permits at most one mutation attempt, blocks exact repeats whose earlier effect was committed or uncertain, and keeps reads and schema discovery available so you can verify and report the result."; - -function reconciliationPrompt(canResume: boolean): string { - const resume = - " If work remains, call recovery_resume by itself after the read result. It performs no mutation and starts one bounded recovery with the normal tool surface."; - return ( - "OpenClaw activated this temporary read-only recovery because the previous Code Mode mutation may have partially applied. First use read by itself to determine the authoritative current state." + - (canResume ? resume : "") + - " If no work remains, report the authoritative state. Do not repeat or finish a mutation during inspection." - ); -} - -function recoveryBlocked(message: string): ToolExecutionPreparation { - return { - kind: "immediate", - outcome: { - kind: "result", - result: textResult(message, { - status: "blocked", - deniedReason: "code-mode-recovery", - }), - isError: true, - }, - dispose() {}, - }; -} - -function createReadyToolExecution( - tool: AnyAgentTool, - params: Parameters>>[0], -): ToolExecutionPreparation { - return { - kind: "ready", - args: params.args, - execute: async (onImplementationStart) => { - onImplementationStart?.(); - return await tool.execute( - params.toolCallId, - params.args as never, // SAFETY: AnyAgentTool erases concrete input after schema validation. - params.signal, - params.onUpdate, - ); - }, - dispose() {}, - }; -} - -function gatePreparedRecoveryTool( - tool: T, - state: Extract, - originalPreparer: InternalToolExecutionPreparer, - ownerKey?: string, -): T { - attachInternalToolExecutionPreparer(tool, async (params) => { - const prepared = await originalPreparer(params); - if (prepared.kind === "immediate") { - return prepared; - } - const mutation = buildToolMutationState( - tool.name, - prepared.args, - ownerKey ? { ownerKey } : undefined, - ); - if (mutation.replaySafe) { - return prepared; - } - const actionKey = hashToolCall(normalizeToolPolicyName(tool.name), prepared.args); - if (state.blockedActionKeys.has(actionKey)) { - prepared.dispose(); - return recoveryBlocked( - "Blocked an exact repeat of a Code Mode call whose earlier effect was committed or uncertain. Inspect the current state and choose a different operation.", - ); - } - if (state.mutationAttempt !== "available") { - prepared.dispose(); - return recoveryBlocked( - "This recovery already attempted one mutation. Use read-only tools to inspect the result and report any remaining work.", - ); - } - state.mutationAttempt = "reserved"; - let started = false; - return { - ...prepared, - execute: async (onImplementationStart) => { - return await prepared.execute(() => { - state.mutationAttempt = "consumed"; - started = true; - onImplementationStart?.(); - }); - }, - dispose: () => { - prepared.dispose(); - if (!started && state.mutationAttempt === "reserved") { - state.mutationAttempt = "available"; - } - }, - }; - }); - return tool; -} - -function gateRecoveryTool( - tool: T, - state: Extract, -): T { - if (TOOL_SEARCH_CONTROL_TOOL_NAMES.has(normalizeToolPolicyName(tool.name))) { - return tool; - } - const originalPreparer = getInternalToolExecutionPreparer(tool); - return gatePreparedRecoveryTool( - tool, - state, - originalPreparer ?? (async (params) => createReadyToolExecution(tool, params)), - getPluginToolSideEffectOwnerKey(tool), - ); -} - -export function applyCodeModeRecoveryPreparedToolSurface(params: { - tools: T[]; - state: Extract; -}): T[] { - return params.tools.map((tool) => { - const preparer = getInternalToolExecutionPreparer(tool); - if (!preparer) { - throw new Error(`Code Mode recovery tool ${tool.name} has no execution preparer`); - } - return gatePreparedRecoveryTool(tool, params.state, preparer); - }); -} - -function createRecoveryResumeTool( - state: Extract, -): AnyAgentTool { - return { - name: CODE_MODE_RECOVERY_RESUME_TOOL_NAME, - label: "Resume recovery", - description: - "After a completed read, end inspection and start one bounded recovery with the normal tool surface.", - parameters: Type.Object({}, { additionalProperties: false }), - executionMode: "sequential", - execute: async () => { - if (state.phase !== "ready") { - throw new ToolInputError("Use read by itself and wait for its result before resuming."); - } - return { - ...textResult("Read-only inspection completed; bounded recovery requested.", { - status: "ok", - }), - terminate: true, - }; - }, - }; -} - -function gateInspectionRead( - tool: T, - state: Extract, -): T { - const originalPreparer = getInternalToolExecutionPreparer(tool); - attachInternalToolExecutionPreparer(tool, async (params) => { - const prepared = originalPreparer - ? await originalPreparer(params) - : createReadyToolExecution(tool, params); - if (prepared.kind === "immediate") { - return prepared; - } - return { - ...prepared, - execute: async (onImplementationStart) => { - const result = await prepared.execute(onImplementationStart); - if (!isToolResultError(result)) { - state.phase = "ready"; - } - return result; - }, - }; - }); - tool.executionMode = "sequential"; - return tool; -} - -export function applyCodeModeRecoveryToolSurface(params: { - tools: T[]; - state: Exclude; -}): T[] { - const state = params.state; - if (state.kind === "inspect") { - const read = params.tools.find((tool) => normalizeToolPolicyName(tool.name) === "read"); - return [ - ...(read ? [gateInspectionRead(read, state)] : []), - ...(state.blockedActionKeys - ? [ - createRecoveryResumeTool(state) as T, // SAFETY: T is the erased AgentTool surface. - ] - : []), - ]; - } - return params.tools.map((tool) => gateRecoveryTool(tool, state)); -} - -export function buildCodeModeRecoveryCandidate(params: { - parentToolCallId: string; - nestedToolActivities: readonly NestedToolActivity[]; -}): CodeModeRecoveryCandidate { - const calls = params.nestedToolActivities.filter( - (activity) => activity.details.parentToolCallId === params.parentToolCallId, - ); - const journal = calls.map(readCodeModeRecoveryJournalEntry); - if (calls.length === 0 || journal.some((entry) => entry === undefined)) { - return {}; - } - const blockedActionKeys = [ - ...new Set( - journal.flatMap((entry) => - entry && !toolEffectStateProvesNoEffect(entry.effectState) ? [entry.actionKey] : [], - ), - ), - ]; - return { blockedActionKeys }; -} - -function isQuiescentRecoveryAttempt(params: { - attempt: EmbeddedRunAttemptResult; - hostOwnsToolSurface: boolean; -}): boolean { - const { attempt } = params; - return ( - attempt.terminal.kind === "ok" && - params.hostOwnsToolSurface && - attempt.itemLifecycle.activeCount === 0 && - attempt.itemLifecycle.startedCount === attempt.itemLifecycle.completedCount && - !attempt.clientToolCalls && - !attempt.yieldDetected && - !attempt.didSendDeterministicApprovalPrompt && - !attempt.runtimeContinuationStarted && - !attempt.toolMetas.some((entry) => entry.asyncStarted === true) && - (attempt.acceptedSessionSpawns?.length ?? 0) === 0 && - !attempt.didSendViaMessagingTool && - (attempt.successfulCronAdds ?? 0) === 0 - ); -} - -function hasSuccessfulInspectionRead(attempt: EmbeddedRunAttemptResult): boolean { - return attempt.toolMetas.some( - (entry) => - normalizeToolPolicyName(entry.toolName) === "read" && - entry.isError !== true && - entry.terminate !== true && - entry.asyncStarted !== true, - ); -} - -function hasSuccessfulResumeRequest(attempt: EmbeddedRunAttemptResult): boolean { - return attempt.toolMetas.some( - (entry) => - normalizeToolPolicyName(entry.toolName) === CODE_MODE_RECOVERY_RESUME_TOOL_NAME && - entry.isError !== true && - entry.terminate === true, - ); -} - -export function advanceCodeModeRecovery(params: { - attempt: EmbeddedRunAttemptResult; - hostOwnsToolSurface: boolean; - retryState: EmbeddedRunTerminalRetryState; - activateInternalPrompt: (prompt: string) => void; -}): boolean { - const state = params.retryState.codeModeRecovery; - if (state.kind === "idle") { - const candidate = params.attempt.codeModeRecoveryCandidate; - if (!candidate || !isQuiescentRecoveryAttempt(params)) { - return false; - } - params.retryState.codeModeRecovery = { - kind: "inspect", - phase: "read-required", - ...(candidate.blockedActionKeys ? { blockedActionKeys: candidate.blockedActionKeys } : {}), - }; - params.activateInternalPrompt(reconciliationPrompt(Boolean(candidate.blockedActionKeys))); - return true; - } - if (state.kind === "inspect") { - const resume = - state.blockedActionKeys && - isQuiescentRecoveryAttempt(params) && - hasSuccessfulInspectionRead(params.attempt) && - hasSuccessfulResumeRequest(params.attempt); - params.retryState.codeModeRecovery = resume - ? { - kind: "resume", - blockedActionKeys: new Set(state.blockedActionKeys), - mutationAttempt: "available", - } - : { kind: "idle" }; - if (!resume) { - return false; - } - params.activateInternalPrompt(CODE_MODE_POST_RECONCILIATION_INSTRUCTION); - return true; - } - params.retryState.codeModeRecovery = { kind: "idle" }; - return false; -} diff --git a/src/agents/embedded-agent-runner/run/code-mode-recovery-journal.ts b/src/agents/embedded-agent-runner/run/code-mode-recovery-journal.ts deleted file mode 100644 index 2b20a7c2ebf2..000000000000 --- a/src/agents/embedded-agent-runner/run/code-mode-recovery-journal.ts +++ /dev/null @@ -1,23 +0,0 @@ -import type { NestedToolActivity } from "../../../sessions/nested-tool-activity.js"; -import type { ToolEffectReceipt } from "../../tool-effect-receipt.js"; - -export type CodeModeRecoveryJournalEntry = { - actionKey: string; - effectState: ToolEffectReceipt["state"]; -}; - -const recoveryFacts = new WeakMap(); - -/** Bind host-only execution facts to the nested activity recorded for that exact call. */ -export function registerCodeModeRecoveryJournalEntry( - activity: NestedToolActivity, - entry: CodeModeRecoveryJournalEntry, -): void { - recoveryFacts.set(activity, entry); -} - -export function readCodeModeRecoveryJournalEntry( - activity: NestedToolActivity, -): CodeModeRecoveryJournalEntry | undefined { - return recoveryFacts.get(activity); -} diff --git a/src/agents/embedded-agent-runner/run/run-attempt-dispatch.ts b/src/agents/embedded-agent-runner/run/run-attempt-dispatch.ts index 6955b39b169c..20173827d5d5 100644 --- a/src/agents/embedded-agent-runner/run/run-attempt-dispatch.ts +++ b/src/agents/embedded-agent-runner/run/run-attempt-dispatch.ts @@ -37,7 +37,6 @@ import type { PreparedNativeSessionRuntime } from "./model-setup.js"; import type { RunEmbeddedAgentParams } from "./params.js"; import { prepareEmbeddedAttemptPromptExecution } from "./prompt-image-preparation.js"; import { resolveSkillWorkshopAttemptParams } from "./skill-workshop-attempt-params.js"; -import type { CodeModeRecoveryState } from "./terminal-retry-state.js"; import type { EmbeddedRunAttemptParams, EmbeddedRunAttemptTrajectoryRecorder } from "./types.js"; type InternalRunParams = RunEmbeddedAgentInternalParams & { @@ -126,7 +125,6 @@ type AttemptControl = { export async function dispatchEmbeddedRunAttempt(input: { params: InternalRunParams; - codeModeRecovery?: Exclude; permissionChange?: EmbeddedRunAttemptParams["permissionChange"]; /** Run-owned start timestamp captured before admission; projected on recovery. */ runStartedAtMs: number; @@ -493,7 +491,6 @@ export async function dispatchEmbeddedRunAttempt(input: { disableMessageTool: params.disableMessageTool, swarmCollector: params.swarmCollector, swarmOutputSchema: params.swarmOutputSchema, - codeModeRecovery: input.codeModeRecovery, forceRestartSafeTools: params.forceRestartSafeTools, forceCodeModeTools: params.forceCodeModeTools, codeModeOverride: params.codeModeOverride, diff --git a/src/agents/embedded-agent-runner/run/terminal-retry-state.ts b/src/agents/embedded-agent-runner/run/terminal-retry-state.ts index 0981042076db..d14193ccf0b9 100644 --- a/src/agents/embedded-agent-runner/run/terminal-retry-state.ts +++ b/src/agents/embedded-agent-runner/run/terminal-retry-state.ts @@ -1,22 +1,5 @@ export const MAX_BEFORE_AGENT_FINALIZE_REVISIONS = 3; -export type CodeModeRecoveryCandidate = { - blockedActionKeys?: readonly string[]; -}; - -export type CodeModeRecoveryState = - | { kind: "idle" } - | { - kind: "inspect"; - phase: "read-required" | "ready"; - blockedActionKeys?: readonly string[]; - } - | { - kind: "resume"; - blockedActionKeys: ReadonlySet; - mutationAttempt: "available" | "reserved" | "consumed"; - }; - export type EmbeddedRunTerminalRetryState = { reasoningOnlyAttempts: number; emptyResponseAttempts: number; @@ -24,7 +7,6 @@ export type EmbeddedRunTerminalRetryState = { compactionContinuationAttempts: number; compactionContinuationInstruction: string | null; beforeFinalizeRevisionAttempts: number; - codeModeRecovery: CodeModeRecoveryState; }; export function createEmbeddedRunTerminalRetryState(): EmbeddedRunTerminalRetryState { @@ -35,6 +17,5 @@ export function createEmbeddedRunTerminalRetryState(): EmbeddedRunTerminalRetryS compactionContinuationAttempts: 0, compactionContinuationInstruction: null, beforeFinalizeRevisionAttempts: 0, - codeModeRecovery: { kind: "idle" }, }; } diff --git a/src/agents/embedded-agent-runner/run/tool-loop-recovery.test.ts b/src/agents/embedded-agent-runner/run/tool-loop-recovery.test.ts index 0e6e2aab3c07..288d4a7ce644 100644 --- a/src/agents/embedded-agent-runner/run/tool-loop-recovery.test.ts +++ b/src/agents/embedded-agent-runner/run/tool-loop-recovery.test.ts @@ -55,56 +55,6 @@ function batchCall(id: string, args: Record): InternalToolBatch } describe("tool-loop recovery batch admission", () => { - it("returns an ordinary intervention when resume is requested before a read result", async () => { - const admission = createToolLoopBatchAdmission( - { runId: "recovery", loopDetection: { enabled: false } }, - { kind: "inspect", phase: "read-required", blockedActionKeys: ["write:prior"] }, - ); - if (!admission) { - throw new Error("Expected recovery batch admission"); - } - const tool = { ...codeModeExecTool(), name: "recovery_resume" }; - await expect( - admission({ - assistantMessage: { - role: "assistant", - content: [], - api: "openai-responses", - provider: "test", - model: "test", - usage: { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - totalTokens: 0, - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - }, - stopReason: "toolUse", - timestamp: 1, - }, - calls: [ - { - toolCall: { - type: "toolCall", - id: "resume", - name: "recovery_resume", - arguments: {}, - }, - args: {}, - tool, - }, - ], - context: { systemPrompt: "", messages: [] }, - }), - ).resolves.toMatchObject({ - intervention: { - toolCallId: "resume", - reason: expect.stringContaining("Use read by itself"), - }, - }); - }); - it("canonicalizes equivalent Code Mode exec aliases before loop detection", async () => { mocks.committedArgs.length = 0; mocks.releasedIds.length = 0; diff --git a/src/agents/embedded-agent-runner/run/tool-loop-recovery.ts b/src/agents/embedded-agent-runner/run/tool-loop-recovery.ts index 2aaa7067e8f9..5ac886b51170 100644 --- a/src/agents/embedded-agent-runner/run/tool-loop-recovery.ts +++ b/src/agents/embedded-agent-runner/run/tool-loop-recovery.ts @@ -9,16 +9,12 @@ import { import { admitToolCallBatch } from "../../tool-loop-admission.js"; import { hashToolCall } from "../../tool-loop-detection.js"; import { log } from "../logger.js"; -import { isCodeModeRecoveryResumeTool } from "./code-mode-reconciliation.js"; -import type { CodeModeRecoveryState } from "./terminal-retry-state.js"; /** Build the embedded-runner's private bridge into agent-core loop recovery. */ export function createToolLoopBatchAdmission( ctx: HookContext, - codeModeRecovery?: Exclude, ): InternalBeforeToolBatchHook | undefined { - const loopDetectionEnabled = ctx.loopDetection?.enabled === true; - if (!loopDetectionEnabled && codeModeRecovery?.kind !== "inspect") { + if (ctx.loopDetection?.enabled !== true) { return undefined; } return async ({ calls }) => { @@ -29,25 +25,7 @@ export function createToolLoopBatchAdmission( : call.args, })); try { - if (codeModeRecovery?.kind === "inspect" && codeModeRecovery.phase === "read-required") { - const resumeCall = canonicalCalls.find((call) => - isCodeModeRecoveryResumeTool(call.toolCall), - ); - if (resumeCall) { - return { - intervention: { - kind: "critical-tool-loop", - toolCallId: resumeCall.toolCall.id, - toolName: resumeCall.toolCall.name, - actionKey: hashToolCall(resumeCall.toolCall.name, resumeCall.args), - detector: "loop_admission_failure", - count: 1, - reason: "Use read by itself and wait for its result before resuming.", - }, - }; - } - } - const admission = loopDetectionEnabled ? await admitToolCallBatch(canonicalCalls, ctx) : {}; + const admission = await admitToolCallBatch(canonicalCalls, ctx); const { commitReadyCalls, releaseSkippedCalls, ...result } = admission; return commitReadyCalls && releaseSkippedCalls ? attachInternalToolBatchLifecycle(result, { diff --git a/src/agents/embedded-agent-runner/run/types.ts b/src/agents/embedded-agent-runner/run/types.ts index 12b5a2718810..dd559087493b 100644 --- a/src/agents/embedded-agent-runner/run/types.ts +++ b/src/agents/embedded-agent-runner/run/types.ts @@ -106,11 +106,6 @@ export type EmbeddedRunAttemptTrajectoryRecorder = { export type EmbeddedRunAttemptParams = EmbeddedRunAttemptBase & { admittedRunContext: NonNullable; - /** Host-private bounded recovery state for this exact attempt. */ - codeModeRecovery?: Exclude< - import("./terminal-retry-state.js").CodeModeRecoveryState, - { kind: "idle" } - >; /** * Run-owned start timestamp captured by the embedded-run orchestrator before * admission. Flows onto the queue handle so recovery can project the active @@ -397,8 +392,6 @@ export type EmbeddedRunAttemptResult = { * how config-enabled code mode stays visible as a no-op on harness routes. */ codeModeEngaged?: boolean; - /** Host-authenticated facts for bounded post-mutation inspection and recovery. */ - codeModeRecoveryCandidate?: import("./terminal-retry-state.js").CodeModeRecoveryCandidate; /** Completed assistant round trips observed during this attempt. */ assistantTurns?: number; /** Inner bridge call counts from this attempt's tool-search/code-mode catalog. */ diff --git a/src/agents/embedded-agent-subscribe.tool-lifecycle.ts b/src/agents/embedded-agent-subscribe.tool-lifecycle.ts index 77a5d2dcb054..54c6c4ee01a2 100644 --- a/src/agents/embedded-agent-subscribe.tool-lifecycle.ts +++ b/src/agents/embedded-agent-subscribe.tool-lifecycle.ts @@ -4,7 +4,7 @@ import { } from "./embedded-agent-subscribe.handlers.tools.js"; import type { EmbeddedAgentSubscribeContext } from "./embedded-agent-subscribe.handlers.types.js"; import { buildToolLifecycleErrorResult } from "./embedded-agent-tool-results.js"; -import { registerToolEffectReceipt, type ToolEffectReceipt } from "./tool-effect-receipt.js"; +import type { ToolEffectReceipt } from "./tool-effect-receipt.js"; import { consumeTrustedToolNoStartError } from "./tool-result-error.js"; type ToolTerminal = { @@ -58,16 +58,16 @@ export function createEmbeddedToolLifecycleRunner( const effectReceipt = trustedNoStart ? ({ state: "not_started" } as const) : terminal.effectReceipt; - await notifyTerminal(toolParams.onTerminal, { ...terminal, effectReceipt }); - throw registerToolEffectReceipt(error, effectReceipt); + await toolParams.onTerminal?.({ ...terminal, effectReceipt }); + throw error; } const terminal = await finishToolLifecycle(ctx, toolParams, { executionStarted, isError: false, result: completedResult, }); - await notifyTerminal(toolParams.onTerminal, terminal); - return registerToolEffectReceipt(completedResult, terminal.effectReceipt); + await toolParams.onTerminal?.(terminal); + return completedResult; }; } @@ -93,14 +93,3 @@ async function finishToolLifecycle( effectReceipt: terminal.effectReceipt, }; } - -async function notifyTerminal( - callback: ((terminal: ToolTerminal) => void | Promise) | undefined, - terminal: ToolTerminal, -): Promise { - try { - await callback?.(terminal); - } catch (error) { - throw registerToolEffectReceipt(error, terminal.effectReceipt); - } -} diff --git a/src/agents/harness/selection.test.ts b/src/agents/harness/selection.test.ts index de6cbab16616..50af6c00baa2 100644 --- a/src/agents/harness/selection.test.ts +++ b/src/agents/harness/selection.test.ts @@ -114,10 +114,6 @@ const contextEngineTurnAttemptMocks = vi.hoisted(() => ({ const builtInHarnesses = vi.hoisted(() => new WeakSet()); const privateHarnessParamCases = [ { field: "__openclawSourceReplyDeliveryRuntime", value: { currentMode: "automatic" } }, - { - field: "codeModeRecovery", - value: { kind: "resume", blockedActionKeys: new Set(), mutationAttempt: "available" }, - }, { field: "compactionCountOwner", value: "caller" }, { field: "onContextAccountingEvent", value: () => undefined }, ] as const; @@ -1547,15 +1543,9 @@ describe("runAgentHarnessAttempt", () => { ); const params = createAttemptParams(); - params.codeModeRecovery = { - kind: "resume", - blockedActionKeys: new Set(), - mutationAttempt: "available", - }; const result = await runAgentHarnessAttempt(params); const classifyCall = classify.mock.calls.at(0); - expect(runAttempt.mock.calls[0]?.[0]).not.toHaveProperty("codeModeRecovery"); expect(classifyCall?.[0].sessionIdUsed).toBe("codex"); expect(classifyCall?.[1]).toEqual( expect.objectContaining({ @@ -1566,7 +1556,6 @@ describe("runAgentHarnessAttempt", () => { }), ); expect(classifyCall?.[1]).not.toHaveProperty("admittedRunContext"); - expect(classifyCall?.[1]).not.toHaveProperty("codeModeRecovery"); expect(classifyCall?.[1]).not.toHaveProperty("operationalRunInstance"); expect(result.agentHarnessId).toBe("codex"); expect(result.agentHarnessResultClassification).toBe("empty"); diff --git a/src/agents/harness/selection.ts b/src/agents/harness/selection.ts index 29ac28b2f739..52f27a075426 100644 --- a/src/agents/harness/selection.ts +++ b/src/agents/harness/selection.ts @@ -704,7 +704,6 @@ function withoutPluginHarnessPrivateState( // separate projections can drift and expose authority on less common operations. const { admittedRunContext: _admittedRunContext, - codeModeRecovery: _codeModeRecovery, compactionCountOwner: _compactionCountOwner, onContextAccountingEvent: _onContextAccountingEvent, contextEngineLogicalTurnLease: _contextEngineLogicalTurnLease, diff --git a/src/agents/harness/types.ts b/src/agents/harness/types.ts index 852f15be6f56..2eb2af240043 100644 --- a/src/agents/harness/types.ts +++ b/src/agents/harness/types.ts @@ -106,7 +106,6 @@ type AgentHarnessLegacyAttemptResult = Omit< type AgentHarnessAttemptParamsBase = Omit< InternalEmbeddedRunAttemptParams, | "admittedRunContext" - | "codeModeRecovery" | "contextEngineLogicalTurnLease" | "onContextEngineTurnCandidate" | "trajectoryRecorder" diff --git a/src/agents/tool-effect-receipt.ts b/src/agents/tool-effect-receipt.ts index 72d3dcc8e647..f86cdf226b83 100644 --- a/src/agents/tool-effect-receipt.ts +++ b/src/agents/tool-effect-receipt.ts @@ -3,8 +3,6 @@ export type ToolEffectReceipt = Readonly<{ state: "not_started" | "read_completed" | "failed_no_effect" | "mutation_committed" | "uncertain"; }>; -const toolEffectReceipts = new WeakMap(); - /** Resolve the strongest effect fact available at the terminal lifecycle owner. */ export function buildToolEffectReceipt(params: { executionStarted: boolean; @@ -27,34 +25,3 @@ export function buildToolEffectReceipt(params: { params.mutatingAction && params.outcome === "success" ? "mutation_committed" : "uncertain", }; } - -/** Bind provenance to the exact host-owned value crossing the next boundary. */ -export function registerToolEffectReceipt(target: T, receipt: ToolEffectReceipt): T { - if ((typeof target === "object" && target !== null) || typeof target === "function") { - toolEffectReceipts.set(target, receipt); - } - return target; -} - -/** Move one receipt across a host-owned projection without making it model-visible. */ -export function transferToolEffectReceipt(source: unknown, target: unknown): void { - const receipt = consumeToolEffectReceipt(source); - if (receipt) { - registerToolEffectReceipt(target, receipt); - } -} - -/** Consume provenance once so copied or replayed values cannot inherit authority. */ -export function consumeToolEffectReceipt(target: unknown): ToolEffectReceipt | undefined { - if ((typeof target !== "object" || target === null) && typeof target !== "function") { - return undefined; - } - const receipt = toolEffectReceipts.get(target); - toolEffectReceipts.delete(target); - return receipt; -} - -/** Return whether one recorded operation state proves that no mutation could have occurred. */ -export function toolEffectStateProvesNoEffect(state: ToolEffectReceipt["state"]): boolean { - return state === "not_started" || state === "read_completed" || state === "failed_no_effect"; -} diff --git a/src/agents/tool-search-transcript.ts b/src/agents/tool-search-transcript.ts index 9be945ef195a..065c71715030 100644 --- a/src/agents/tool-search-transcript.ts +++ b/src/agents/tool-search-transcript.ts @@ -2,7 +2,6 @@ import { isRecord } from "@openclaw/normalization-core/record-coerce"; import { transferMcpCodeModeGuestResult } from "./mcp-content.js"; import type { AgentToolResult } from "./runtime/index.js"; import { copyInternalToolResultState } from "./runtime/internal-hooks.js"; -import { transferToolEffectReceipt } from "./tool-effect-receipt.js"; import { toToolSearchJsonSafe } from "./tool-search-json.js"; function freezeJsonSnapshot(value: unknown): unknown { @@ -31,6 +30,5 @@ export function snapshotToolSearchTargetTranscriptResult( result.details === undefined ? undefined : toToolSearchJsonSafe(result.details); } const target = freezeJsonSnapshot(snapshot) as AgentToolResult; - transferToolEffectReceipt(result, target); return transferMcpCodeModeGuestResult(result, copyInternalToolResultState(result, target)); } diff --git a/src/agents/tool-surface-plan.ts b/src/agents/tool-surface-plan.ts index b7f711e4f0df..0e30160b4023 100644 --- a/src/agents/tool-surface-plan.ts +++ b/src/agents/tool-surface-plan.ts @@ -28,7 +28,6 @@ type AgentToolSurfacePlanParams = { isRawModelRun: boolean; toolsAllow?: readonly string[]; forceCodeModeControls?: boolean; - forceDirectTools?: boolean; }; export function resolveAgentToolSurfacePlan(params: AgentToolSurfacePlanParams) { @@ -62,16 +61,12 @@ export function resolveAgentToolSurfacePlan(params: AgentToolSurfacePlanParams) ); const codeModeControlsEnabled = toolsAvailable && - params.forceDirectTools !== true && // Restart recovery continues one provider turn. Keep its original control // schema even when the reloaded config disables Code Mode for new turns. (params.forceCodeModeControls === true || isCodeModeEngagedForModel(codeModeConfig, params.model)); const toolSearchControlsEnabled = - toolsAvailable && - params.forceDirectTools !== true && - !codeModeControlsEnabled && - toolSearchConfig.enabled; + toolsAvailable && !codeModeControlsEnabled && toolSearchConfig.enabled; return { codeModeControlsEnabled, toolSearchControlsEnabled, diff --git a/src/agents/tool-terminal-outcome.test.ts b/src/agents/tool-terminal-outcome.test.ts index 482c76db477b..bd9c6a9cab42 100644 --- a/src/agents/tool-terminal-outcome.test.ts +++ b/src/agents/tool-terminal-outcome.test.ts @@ -9,7 +9,6 @@ import { } from "./agent-tools.before-tool-call.state.js"; import { buildPayloads } from "./embedded-agent-runner/run/payloads.test-helpers.js"; import { inferToolMetaFromArgsCore } from "./tool-display.js"; -import { consumeToolEffectReceipt, registerToolEffectReceipt } from "./tool-effect-receipt.js"; import { createToolTerminalObserver } from "./tool-terminal-outcome.js"; describe("tool terminal outcome observer", () => { @@ -195,14 +194,6 @@ describe("tool terminal outcome observer", () => { }); }); - it("binds effect receipts to one exact host-owned result", () => { - const result = registerToolEffectReceipt({ status: "failed" }, { state: "failed_no_effect" }); - - expect(consumeToolEffectReceipt({ ...result })).toBeUndefined(); - expect(consumeToolEffectReceipt(result)).toEqual({ state: "failed_no_effect" }); - expect(consumeToolEffectReceipt(result)).toBeUndefined(); - }); - it("clears a failed sessions_spawn once a retry with adjusted arguments succeeds", () => { const observe = createToolTerminalObserver("run-spawn-retry"); const failedArgs = { diff --git a/src/agents/tools/terminal-tool.test.ts b/src/agents/tools/terminal-tool.test.ts index e08ef4c62445..0c906b76092c 100644 --- a/src/agents/tools/terminal-tool.test.ts +++ b/src/agents/tools/terminal-tool.test.ts @@ -14,7 +14,6 @@ import { } from "../../infra/agent-run-registry.js"; import { GATEWAY_OWNER_ONLY_CORE_TOOLS } from "../../security/dangerous-tools.js"; import { wrapToolWithBeforeToolCallHook } from "../agent-tools.before-tool-call.js"; -import { consumeRepairableCodeModeFailure } from "../code-mode-repair-provenance.js"; import { createSubscribedCodeModeHarness } from "../code-mode.bridge.lifecycle.test-support.js"; import { applyCodeModeCatalog } from "../code-mode.js"; import { runUntilCompleted } from "../code-mode.test-support.js"; @@ -334,7 +333,6 @@ describe("terminal tool", () => { expect(backend.writes).toEqual([]); expect(approvalMocks.register).not.toHaveBeenCalled(); expect(approvalMocks.decide).not.toHaveBeenCalled(); - expect(consumeRepairableCodeModeFailure(details)).toBe(true); } finally { harness.dispose(); manager.closeAgent(agentOwner, sessionId); diff --git a/src/plugin-sdk/agent-harness-runtime.test.ts b/src/plugin-sdk/agent-harness-runtime.test.ts index 570ca532cbb4..01e5a1357afc 100644 --- a/src/plugin-sdk/agent-harness-runtime.test.ts +++ b/src/plugin-sdk/agent-harness-runtime.test.ts @@ -198,12 +198,6 @@ describe("agent harness runtime SDK facade", () => { ? true : false >().toEqualTypeOf(); - expectTypeOf< - "codeModeRecovery" extends keyof AgentHarnessAttemptParamsV2 ? true : false - >().toEqualTypeOf(); - expectTypeOf< - "codeModeRecovery" extends keyof EmbeddedRunAttemptParamsV2 ? true : false - >().toEqualTypeOf(); expectTypeOf< Omit< AgentHarnessSideQuestionParamsV2, diff --git a/src/plugin-sdk/agent-harness-runtime.ts b/src/plugin-sdk/agent-harness-runtime.ts index d0b6fe7f0c02..21a3a70dd7a0 100644 --- a/src/plugin-sdk/agent-harness-runtime.ts +++ b/src/plugin-sdk/agent-harness-runtime.ts @@ -133,7 +133,6 @@ type EmbeddedRunAttemptParamsBase = Omit< CoreEmbeddedRunAttemptParams, | "admittedRunContext" | "authoredContextTokenCap" - | "codeModeRecovery" | "contextEngineLogicalTurnLease" | "onContextEngineTurnCandidate" | "pluginHarnessToolPolicySafeDeniedTools"