From 26f87a37009bb452afae8523cbaf4063fbca247e Mon Sep 17 00:00:00 2001 From: Peter Steinberger Date: Fri, 4 Sep 2026 11:37:51 -0700 Subject: [PATCH] fix: allow environment-only agent runs without a native harness (#138449) * fix: allow environment-only agent runs without a native harness * chore(ui): reconcile measured main startup baseline Unmodified main b6a4d0dad430 fails the startup gzip check at 350377 B in Linux CI; a pristine source build measures 350363 B. Record the measured main baseline while retaining the fixed cap and existing growth and variance allowances. --- .../control-ui-startup-budget-baseline.json | 6 +- docs/ci.md | 2 + docs/cli/agent.md | 2 + scripts/lib/vitest-build-prerequisites.mts | 6 + src/agents/harness/runtime-plugin.test.ts | 17 ++- src/agents/harness/runtime-plugin.ts | 9 +- src/agents/harness/selection.test.ts | 17 ++- test/agent-exec-code-mode.live.test.ts | 128 ++++++++++++++++++ 8 files changed, 177 insertions(+), 10 deletions(-) create mode 100644 test/agent-exec-code-mode.live.test.ts diff --git a/config/control-ui-startup-budget-baseline.json b/config/control-ui-startup-budget-baseline.json index 43345a2e4743..a758c3080a38 100644 --- a/config/control-ui-startup-budget-baseline.json +++ b/config/control-ui-startup-budget-baseline.json @@ -1,5 +1,5 @@ { - "startupJsGzipBytes": 349565, - "reason": "Integrated Control UI and dependency changes since the 348668 B baseline, including chat face/visibility, locale, first-render, Meetings, and Appearance work, now measure 349565 B (+897 B, 0.26%); an independent exact-head build measured 349533 B (32 B lower, within the 64 B build-variance allowance); fixed 358400 B cap and existing allowances retained", - "updatedAt": "2026-09-03" + "startupJsGzipBytes": 350377, + "reason": "Current main b6a4d0dad430 measures 350377 B startup gzip in Linux CI after integrated UI changes (+812 B, 0.23%); pristine-main and candidate builds measured 350363 B and 350380 B, within the existing 64 B variance allowance. Fixed 358400 B cap and existing growth allowance retained.", + "updatedAt": "2026-09-04" } diff --git a/docs/ci.md b/docs/ci.md index 40dd78608c36..272d835ac0bb 100644 --- a/docs/ci.md +++ b/docs/ci.md @@ -1034,6 +1034,8 @@ The release live/E2E child keeps broad native `pnpm test:live` coverage, but it That keeps the same file coverage while making slow live provider failures easier to rerun and diagnose. The aggregate `native-live-src-gateway`, `native-live-extensions-o-z`, `native-live-extensions-media`, and `native-live-extensions-media-music` shard names remain valid for manual one-shot reruns. +Stable/full release validation includes the configless `agent exec --auth-env-only` Code Mode smoke in `native-live-test`. The test runner builds the runtime before starting workers. The smoke copies that built distribution outside the source checkout, applies the package's plugin exclusions, and reuses installed dependencies. It supplies only `OPENAI_API_KEY` to a fresh CLI environment, runs `openai/gpt-5.6-sol` without a runtime override, and verifies Code Mode engagement, nested tool calls, and an exact read-to-write artifact. This proves built-distribution behavior; Package Acceptance owns tarball installation proof. The shard requires passing evidence from this test; a missing key or skipped test cannot satisfy the release gate. + Gateway-profile shards and shards containing the image-tool provider or OpenAI plugin live tests prepare the `sourcePerformance` build profile before starting Vitest. This supplies executable provider and agent runtime artifacts without building declarations or the Control UI. Provider requests, assertions, and test deadlines remain unchanged; gateway diagnostic environment settings apply only to gateway-profile shards. Cold source-plugin Jiti import cost remains a separate performance follow-up, not live provider latency. Stable/full release runs explicitly enable OpenAI AgentSession repeated compaction in `native-live-src-agents` with `OPENCLAW_LIVE_OPENAI_COMPACTION=1` and `OPENCLAW_LIVE_OPENAI_COMPACTION_FULL=0`. This uses the bounded 48k context profile and requires multiple compactions plus durable-marker recall. Manual shard runs retain the explicit opt-in; once enabled, a skipped compaction test fails the shard's pass-evidence gate. The separate 922k full-context stress profile remains a manual opt-in. diff --git a/docs/cli/agent.md b/docs/cli/agent.md index 8d13fdb66611..f98737a3f25f 100644 --- a/docs/cli/agent.md +++ b/docs/cli/agent.md @@ -40,6 +40,8 @@ For reproducible runs, pin the config instead of inheriting it. `--config Stored credentials are used by default, so a folder-scoped run reaches the same logins as the rest of the CLI. Pass `--auth-env-only` to restrict the run to provider keys already present in the process environment. That mode loads no config at all, and pairing it with `--config` is rejected rather than silently ignored, because a config supplies provider credentials through several surfaces at once: [inline keys and secret headers](/reference/secretref-credential-surface), an `env` block, and login-shell import. It also skips OpenClaw auth profiles and external Codex, Claude, or other CLI credential stores. Provider auth variables remain available to model authentication but are omitted from agent-launched host commands. +On a clean installation without the Codex plugin, OpenAI API-key runs use the built-in OpenClaw runtime. The implicit Codex preference does not require installing a native harness before `--auth-env-only` can run. An explicitly configured Codex runtime or an existing Codex session pin still requires that harness. + Select a primary and ordered fallback chain with repeatable flags: ```bash diff --git a/scripts/lib/vitest-build-prerequisites.mts b/scripts/lib/vitest-build-prerequisites.mts index 77ca7eebdbfc..da95a4405ebe 100644 --- a/scripts/lib/vitest-build-prerequisites.mts +++ b/scripts/lib/vitest-build-prerequisites.mts @@ -22,6 +22,12 @@ type TestSelection = { // prerequisite before admitting any workers: a child build invalidates dist // while unrelated workers may still be importing its public plugin facades. const runtimeConsumers = [ + { + file: "test/agent-exec-code-mode.live.test.ts", + configs: ["test/vitest/vitest.live.config.ts"], + mode: "runtime", + dir: "", + }, { file: "extensions/qa-lab/src/suite-process-lifecycle.test.ts", configs: ["test/vitest/vitest.extension-qa.config.ts"], diff --git a/src/agents/harness/runtime-plugin.test.ts b/src/agents/harness/runtime-plugin.test.ts index 37dcc341017e..764e903dbd1f 100644 --- a/src/agents/harness/runtime-plugin.test.ts +++ b/src/agents/harness/runtime-plugin.test.ts @@ -118,14 +118,27 @@ describe("harness runtime plugins", () => { expect(pluginRegistry.agentHarnesses).toHaveLength(1); }); - it("carries the missing Codex owner reason and remediation into the turn guard", async () => { + it.each([ + { name: "runtime override", selection: { agentHarnessRuntimeOverride: "codex" } }, + { name: "session pin", selection: { agentHarnessId: "codex" } }, + { + name: "model policy", + selection: { + config: { + agents: { + defaults: { models: { "openai/gpt-5.5": { agentRuntime: { id: "codex" } } } }, + }, + }, + }, + }, + ])("keeps a missing explicit Codex $name fatal with remediation", async ({ selection }) => { const pluginRegistry = createEmptyPluginRegistry(); attachPreparedPluginFacts(pluginRegistry, {}, makeRegistry([])); const error = await ensureSelectedAgentHarnessPlugin({ provider: "openai", modelId: "gpt-5.5", - agentHarnessRuntimeOverride: "codex", + ...selection, workspaceDir: "/tmp/workspace", pluginRegistry, }).catch((cause: unknown) => cause); diff --git a/src/agents/harness/runtime-plugin.ts b/src/agents/harness/runtime-plugin.ts index d3c57c441adb..e13ca9bfb4ed 100644 --- a/src/agents/harness/runtime-plugin.ts +++ b/src/agents/harness/runtime-plugin.ts @@ -203,11 +203,12 @@ export async function ensureSelectedAgentHarnessPlugin(params: { requestTransportOverrides: params.requestTransportOverrides, }); const requestedRuntime = pinnedHarnessId ?? runtimeOverride; - const runtime = - requestedRuntime && !isDefaultAgentRuntimeId(requestedRuntime) - ? requestedRuntime - : policy.runtime; + const explicitRuntime = isDefaultAgentRuntimeId(requestedRuntime) ? undefined : requestedRuntime; + const runtime = explicitRuntime ?? policy.runtime; + // Harness selection owns implicit preferences and their unavailable-runtime fallback. + // Authored policies and session pins still require their selected registration. if ( + (!explicitRuntime && policy.runtimeSource === "implicit") || isDefaultAgentRuntimeId(runtime) || runtime === OPENCLAW_AGENT_RUNTIME_ID || isCliRuntimeAliasForProvider({ diff --git a/src/agents/harness/selection.test.ts b/src/agents/harness/selection.test.ts index 50af6c00baa2..40b83af9b90e 100644 --- a/src/agents/harness/selection.test.ts +++ b/src/agents/harness/selection.test.ts @@ -70,6 +70,7 @@ import { maybeCompactAgentHarnessSession as maybeCompactAgentHarnessSessionImpl import type { ContextEngineLogicalTurnLease } from "./context-engine-logical-turn.js"; import { resolveAgentHarnessPolicy } from "./policy.js"; import { clearAgentHarnesses, registerAgentHarness } from "./registry.js"; +import { ensureSelectedAgentHarnessPlugin } from "./runtime-plugin.js"; import { agentHarnessBuildsOpenClawTools, agentHarnessExposesOpenClawTools, @@ -1495,6 +1496,13 @@ describe("runAgentHarnessAttempt", () => { runtimeSource: "implicit", }); + await ensureSelectedAgentHarnessPlugin({ + provider: "openai", + modelId: "gpt-5.4", + workspaceDir: "/tmp/workspace", + pluginRegistry: getActivePluginRegistry() ?? undefined, + }); + const result = await runAgentHarnessAttempt({ ...createAttemptParams(), provider: "openai", @@ -3118,11 +3126,18 @@ describe("selectAgentHarness", () => { it.each(["default", "auto"] as const)( "falls back from configured %s to OpenClaw when implicit Codex is unavailable or unsupported", - (runtime) => { + async (runtime) => { const config = providerRuntimeConfig("openai", runtime); expect(resolveAgentHarnessPolicy({ provider: "openai", modelId: "gpt-5.4", config })).toEqual( { runtime: "codex", runtimeSource: "implicit" }, ); + await ensureSelectedAgentHarnessPlugin({ + provider: "openai", + modelId: "gpt-5.4", + config, + workspaceDir: "/tmp/workspace", + pluginRegistry: getActivePluginRegistry() ?? undefined, + }); expect(selectAgentHarness({ provider: "openai", modelId: "gpt-5.4", config }).id).toBe( "openclaw", ); diff --git a/test/agent-exec-code-mode.live.test.ts b/test/agent-exec-code-mode.live.test.ts new file mode 100644 index 000000000000..24375c799509 --- /dev/null +++ b/test/agent-exec-code-mode.live.test.ts @@ -0,0 +1,128 @@ +import { execFile } from "node:child_process"; +import { randomUUID } from "node:crypto"; +import { constants } from "node:fs"; +import fs from "node:fs/promises"; +import path from "node:path"; +import { promisify } from "node:util"; +import { afterEach, describe, expect, it } from "vitest"; +import { collectRootPackageExcludedExtensionDirs } from "../scripts/lib/bundled-plugin-build-entries.mjs"; +import { isLiveTestEnabled } from "../src/agents/live-test-helpers.js"; +import type { AgentExecEnvelope } from "../src/commands/agent-exec-result.js"; +import { useAutoCleanupTempDirTracker } from "./helpers/temp-dir.js"; + +const execFileAsync = promisify(execFile); +const tempDirs = useAutoCleanupTempDirTracker(afterEach); +const openAiApiKey = process.env.OPENAI_API_KEY?.trim() ?? ""; +const describeLive = isLiveTestEnabled() && openAiApiKey.length > 0 ? describe : describe.skip; + +describeLive("agent exec Code Mode with environment authentication", () => { + it("starts without configuration and completes dependent filesystem calls", async () => { + const root = tempDirs.make("openclaw-agent-exec-code-mode-live-"); + const home = path.join(root, "home"); + const stateDir = path.join(root, "state"); + const workspace = path.join(root, "workspace"); + const tmpDir = path.join(root, "tmp"); + await Promise.all([home, stateDir, workspace, tmpDir].map((dir) => fs.mkdir(dir))); + const repoRoot = path.resolve(import.meta.dirname, ".."); + const installedRoot = path.join(root, "node_modules", "openclaw"); + const sourceDist = path.join(repoRoot, "dist"); + const excludedPlugins = collectRootPackageExcludedExtensionDirs({ cwd: repoRoot }); + await fs.mkdir(installedRoot, { recursive: true }); + await fs.copyFile( + path.join(repoRoot, "package.json"), + path.join(installedRoot, "package.json"), + ); + // A source checkout discovers external plugins beside dist. Copy the built + // distribution with its package exclusions so discovery sees a clean install. + await fs.cp(sourceDist, path.join(installedRoot, "dist"), { + recursive: true, + mode: constants.COPYFILE_FICLONE, + filter: (source) => { + const [directory, pluginId] = path.relative(sourceDist, source).split(path.sep); + return ( + directory !== "extensions" || pluginId === undefined || !excludedPlugins.has(pluginId) + ); + }, + }); + await fs.symlink( + await fs.realpath(path.join(repoRoot, "node_modules")), + path.join(installedRoot, "node_modules"), + process.platform === "win32" ? "junction" : "dir", + ); + const input = `code-mode-${randomUUID()}\n`; + await fs.writeFile(path.join(workspace, "input.txt"), input); + + // Exercise production discovery: no inherited test shortcuts, authored + // config, installed harness, external CLI credentials, or runtime override. + const env: NodeJS.ProcessEnv = { + PATH: process.env.PATH, + SystemRoot: process.env.SystemRoot, + HOME: home, + USERPROFILE: home, + TMPDIR: tmpDir, + TMP: tmpDir, + TEMP: tmpDir, + OPENCLAW_STATE_DIR: stateDir, + OPENAI_API_KEY: openAiApiKey, + }; + let stdout: string; + let stderr: string; + try { + ({ stdout, stderr } = await execFileAsync( + process.execPath, + [ + path.join(installedRoot, "dist", "entry.js"), + "agent", + "exec", + "Use Code Mode and the filesystem read and write tools to read input.txt, then write " + + "output.txt containing exactly the input bytes followed by processed and a newline. " + + "Do not use a shell command. Reply with exactly DONE after the write completes.", + "--auth-env-only", + "--model", + "openai/gpt-5.6-sol", + "--code-mode", + "code", + "--local-model-lean", + "--thinking", + "low", + "--cwd", + workspace, + "--state-dir", + stateDir, + "--timeout", + "180", + "--json", + ], + { + cwd: workspace, + env, + encoding: "utf8", + maxBuffer: 2 * 1024 * 1024, + timeout: 300_000, + }, + )); + } catch (error) { + // Child-process failures retain stdout/stderr; never attach that raw + // error as a cause because live credentials could appear in diagnostics. + const message = error instanceof Error ? error.message : String(error); + // eslint-disable-next-line preserve-caught-error -- The raw cause can expose live credentials. + throw new Error(message.replaceAll(openAiApiKey, "[REDACTED]")); + } + + expect(stdout.includes(openAiApiKey) || stderr.includes(openAiApiKey)).toBe(false); + const result = JSON.parse(stdout) as AgentExecEnvelope; + expect(result).toMatchObject({ + ok: true, + status: "ok", + provider: "openai", + model: "gpt-5.6-sol", + codeModeEngaged: true, + final: "DONE", + }); + expect(result.toolSummary?.tools).toContain("exec"); + expect(result.bridgeCalls?.call).toBeGreaterThanOrEqual(2); + await expect(fs.readFile(path.join(workspace, "output.txt"), "utf8")).resolves.toBe( + `${input}processed\n`, + ); + }, 330_000); +});