From 21cf514b0f6c809ef1089bef5e18192be898bcfd Mon Sep 17 00:00:00 2001 From: Peter Steinberger Date: Mon, 28 Sep 2026 17:48:54 -0700 Subject: [PATCH] fix(onboarding): fail fast on unavailable model endpoints (#160813) Preserve transport error codes through embedded recovery and let one-shot setup verification reject definite connection failures without spending the normal session retry budget. Keep ordinary agent retries unchanged and report actionable server, URL, and DNS guidance through the shared UI/CLI owner. Remote proof: 167 focused tests; both regressions fail on baseline. Real Gateway/Chromium medians (five fresh runs): closed port 79.766s to 0.720s, DNS NXDOMAIN 69.449s to 0.849s, HTTP 401 control 1.003s to 0.860s. CLI closed-port verification 66.655s to 1.313s. Successful activation, chat, and reload pass. Changed checks pass after removing the obsolete status-mapper export; the exact dead-export gate and setup tests were rerun after that final cleanup. Co-authored-by: Peter Steinberger --- docs/start/onboarding-overview.md | 5 +++ .../run/assistant-failure.ts | 1 + .../run/attempt-recovery.test-support.ts | 9 ++-- .../run/attempt-recovery.test.ts | 24 ++++++++++ .../run/attempt-recovery.ts | 2 + .../run/failover-retry-controller.ts | 10 +++++ .../embedded-agent-runner/run/params.ts | 2 + src/system-agent/setup-inference-core.ts | 25 ++++++++++- src/system-agent/setup-inference-turn.test.ts | 44 +++++++++++++++++++ src/system-agent/setup-inference-turn.ts | 8 ++-- 10 files changed, 122 insertions(+), 8 deletions(-) diff --git a/docs/start/onboarding-overview.md b/docs/start/onboarding-overview.md index f398ae4738bf..19d8336efd9a 100644 --- a/docs/start/onboarding-overview.md +++ b/docs/start/onboarding-overview.md @@ -139,6 +139,11 @@ setup verifies a real model reply before saving the provider and activating its model. A failed or cancelled check preserves the previous configuration. The classic wizard also retains its custom-provider setup. +If the endpoint refuses the connection or its hostname cannot be found, setup +reports the failed connection immediately instead of waiting through normal +chat retries. Start the server or correct the URL and network settings on the +Gateway host, then retry. Ordinary agent sessions keep their connection retries. + ## Related - [Getting started](/start/getting-started) diff --git a/src/agents/embedded-agent-runner/run/assistant-failure.ts b/src/agents/embedded-agent-runner/run/assistant-failure.ts index fafa6858a3fc..bd756aff9e78 100644 --- a/src/agents/embedded-agent-runner/run/assistant-failure.ts +++ b/src/agents/embedded-agent-runner/run/assistant-failure.ts @@ -489,6 +489,7 @@ export async function handleEmbeddedAssistantFailure(input: { profileId: input.authProfileId, authMode, status, + code: failedAssistant?.errorCode, rawError: failedAssistant?.errorMessage?.trim(), // Retry reason "timeout" also includes 5xx; only the terminal owner records a deadline. timeout: diff --git a/src/agents/embedded-agent-runner/run/attempt-recovery.test-support.ts b/src/agents/embedded-agent-runner/run/attempt-recovery.test-support.ts index 7aef8ca51571..95bc5f192122 100644 --- a/src/agents/embedded-agent-runner/run/attempt-recovery.test-support.ts +++ b/src/agents/embedded-agent-runner/run/attempt-recovery.test-support.ts @@ -36,6 +36,7 @@ export type TransportDropScenario = { lastToolError?: Parameters[0]["lastToolError"]; pluginHarnessOwnsTransport?: boolean; retryAvailable?: boolean; + retryConnectionErrors?: boolean; replaySafe?: boolean; fallbackConfigured?: boolean; providerRetryMaxDelayMs?: number; @@ -136,9 +137,11 @@ export async function recoverAfterTransportDrop(scenario: TransportDropScenario const continueFromCurrentTranscript = vi.fn(); const contextRecoveryState = createEmbeddedRunContextRecoveryState(); const failoverRetryController = createEmbeddedRunFailoverRetryController({ - runParams: { runId: "run:transport-drop", config: scenario.config } as Parameters< - typeof createEmbeddedRunFailoverRetryController - >[0]["runParams"], + runParams: { + runId: "run:transport-drop", + config: scenario.config, + retryConnectionErrors: scenario.retryConnectionErrors, + } as Parameters[0]["runParams"], provider, modelId, globalLane: "test", diff --git a/src/agents/embedded-agent-runner/run/attempt-recovery.test.ts b/src/agents/embedded-agent-runner/run/attempt-recovery.test.ts index efe4f9df5d70..5aab5579184f 100644 --- a/src/agents/embedded-agent-runner/run/attempt-recovery.test.ts +++ b/src/agents/embedded-agent-runner/run/attempt-recovery.test.ts @@ -35,6 +35,30 @@ const tempDirs = createTempDirTracker(); const requireRecord = createRequireRecord("record", "expected-label-object-capitalized"); afterEach(() => tempDirs.cleanup()); +it.each(["ECONNREFUSED", "ENOTFOUND", "EHOSTUNREACH", "ENETUNREACH"])( + "fails fast on %s for setup while normal runs retain connection retries", + async (errorCode) => { + const scenario = { + errorMessage: "Connection error.", + errorCode, + noTools: true, + replaySafe: true, + content: [], + } satisfies TransportDropScenario; + vi.mocked(sleepWithAbort).mockClear(); + const setup = await recoverAfterTransportDrop({ ...scenario, retryConnectionErrors: false }); + expect(setup.recovery.action).toBe("proceed"); + expect(sleepWithAbort).not.toHaveBeenCalled(); + await expect(handleAssistantFailureAfterRecovery(setup)).rejects.toMatchObject({ + reason: "timeout", + code: errorCode, + }); + const ordinary = await recoverAfterTransportDrop(scenario); + expect(ordinary.recovery.action).toBe("retry"); + expect(sleepWithAbort).toHaveBeenCalledOnce(); + }, +); + function handleAssistantFailureAfterRecovery( fixture: Awaited>, previousRetryFailoverReason: Parameters< diff --git a/src/agents/embedded-agent-runner/run/attempt-recovery.ts b/src/agents/embedded-agent-runner/run/attempt-recovery.ts index 6315fd372386..1374d24da810 100644 --- a/src/agents/embedded-agent-runner/run/attempt-recovery.ts +++ b/src/agents/embedded-agent-runner/run/attempt-recovery.ts @@ -13,6 +13,7 @@ import { } from "../../failover-error.js"; import { failoverReasonFromClassification } from "../../failover/classification-rules.js"; import { classifyFailoverSignal } from "../../failover/classify.js"; +import { getFailoverErrorCode } from "../../failover/error.js"; import { resolveRetryAfterMs } from "../../failover/retry-evidence.js"; import { LiveSessionModelSwitchError } from "../../live-model-switch-error.js"; import { shouldSwitchToLiveModel, clearLiveModelSwitchPending } from "../../live-model-switch.js"; @@ -394,6 +395,7 @@ export async function recoverEmbeddedRunAttempt(input: { recoveryReason && (await failoverRetryController.maybeRetryTransient({ reason: recoveryReason, + code: promptError ? getFailoverErrorCode(promptError) : assistantSignal?.code, message: promptError ? formatErrorMessage(promptError) : assistantSignal?.message, retryAfterMs: promptError ? resolveRetryAfterMs(formatErrorMessage(promptError), Date.now(), promptError) diff --git a/src/agents/embedded-agent-runner/run/failover-retry-controller.ts b/src/agents/embedded-agent-runner/run/failover-retry-controller.ts index f841272d23de..e51f2192e5db 100644 --- a/src/agents/embedded-agent-runner/run/failover-retry-controller.ts +++ b/src/agents/embedded-agent-runner/run/failover-retry-controller.ts @@ -252,6 +252,7 @@ export function createEmbeddedRunFailoverRetryController(input: { maybeRetryTransient: async (retry: { reason: TransientRetryReason; message?: string; + code?: string; retryAfterMs?: number; /** Saved retry.provider.maxRetryDelayMs; undefined or 0 disables the cap. */ maxRetryDelayMs?: number; @@ -268,6 +269,7 @@ export function createEmbeddedRunFailoverRetryController(input: { decision: "accepted" | "rejected", reason: | "non_transient" + | "connection_retry_disabled" | "long_window_rate_limit" | "retry_budget_exhausted" | "retry_delay_unavailable" @@ -284,6 +286,14 @@ export function createEmbeddedRunFailoverRetryController(input: { }, { config: params.config }, ); + if ( + params.retryConnectionErrors === false && + retry.code !== undefined && + ["ECONNREFUSED", "ENOTFOUND", "EHOSTUNREACH", "ENETUNREACH"].includes(retry.code) + ) { + recordDecision("rejected", "connection_retry_disabled"); + return false; + } if ( retry.reason !== "rate_limit" && retry.reason !== "overloaded" && diff --git a/src/agents/embedded-agent-runner/run/params.ts b/src/agents/embedded-agent-runner/run/params.ts index a85f31b0423a..3fdec09e6950 100644 --- a/src/agents/embedded-agent-runner/run/params.ts +++ b/src/agents/embedded-agent-runner/run/params.ts @@ -116,6 +116,8 @@ export type RunEmbeddedAgentParams = { codeModeOverride?: boolean | "auto"; /** Internal one-shot model probe mode: no tools, no workspace/chat prompt policy. */ modelRun?: boolean; + /** Setup can reject unavailable endpoints without spending the session retry budget. */ + retryConnectionErrors?: boolean; /** Disable trajectory persistence for auxiliary runs with no durable session owner. */ disableTrajectory?: boolean; /** Restrict Skill Workshop to a bounded pending-proposal budget for an internal review run. */ diff --git a/src/system-agent/setup-inference-core.ts b/src/system-agent/setup-inference-core.ts index ef0a6c115783..3863b4cf8368 100644 --- a/src/system-agent/setup-inference-core.ts +++ b/src/system-agent/setup-inference-core.ts @@ -8,6 +8,7 @@ import type { AgentRunResultView } from "../agents/agent-run-result.js"; import type { loadAuthProfileStoreForRuntime } from "../agents/auth-profiles/store-runtime.js"; import type { readCodexCliActiveApiKey } from "../agents/cli-credentials.js"; import type { AgentExecutionAuthBinding } from "../agents/execution-auth-binding.js"; +import { describeFailoverError } from "../agents/failover-error.js"; import type { FailoverReason } from "../agents/failover/signal.js"; import { DEFAULT_AGENT_WORKSPACE_DIR } from "../agents/workspace-default.js"; import type { @@ -402,12 +403,34 @@ const SETUP_STATUS_BY_FAILOVER_REASON = { unknown: "unknown", } satisfies Record; -export function mapFailoverReasonToSetupStatus( +function mapFailoverReasonToSetupStatus( reason?: FailoverReason | null, ): SetupInferenceFailureStatus { return reason ? SETUP_STATUS_BY_FAILOVER_REASON[reason] : "unknown"; } +export function describeSetupInferenceError( + error: unknown, + route: SystemAgentConfiguredRoute, +): { status: SetupInferenceFailureStatus; error: string } { + const described = describeFailoverError(error); + const origin = URL.parse( + route.runConfig.models?.providers?.[route.provider]?.baseUrl ?? "", + )?.origin; + const connectionError = !origin + ? undefined + : described.code === "ECONNREFUSED" + ? `Nothing is listening at ${origin}. Start the server or check the URL, then retry setup.` + : described.code === "ENOTFOUND" + ? `The server name in ${origin} could not be found. Check the URL and DNS settings, then retry setup.` + : described.code === "EHOSTUNREACH" || described.code === "ENETUNREACH" + ? `Cannot reach ${origin}. Check the URL and network connection from the Gateway host, then retry setup.` + : undefined; + return connectionError + ? { status: "unavailable", error: `${connectionError} No default model was changed.` } + : { status: mapFailoverReasonToSetupStatus(described.reason), error: described.message }; +} + export function validateSetupInferenceOwnerEvidence(params: { runner: "cli" | "embedded"; configuredHarnessId?: string; diff --git a/src/system-agent/setup-inference-turn.test.ts b/src/system-agent/setup-inference-turn.test.ts index ba77518df8b6..d2a3da205072 100644 --- a/src/system-agent/setup-inference-turn.test.ts +++ b/src/system-agent/setup-inference-turn.test.ts @@ -1,5 +1,6 @@ import { describe, expect, it, vi } from "vitest"; import { createDeferred } from "../../test/helpers/promise.js"; +import { FailoverError } from "../agents/failover-error.js"; import { createPluginMetadataSnapshot, makeRegistry, @@ -67,6 +68,49 @@ function embeddedRoute(runtime: "codex" | "openclaw" = "codex"): SystemAgentConf } describe("setup inference plugin ownership", () => { + it.each([ + ["ECONNREFUSED", "Nothing is listening at http://127.0.0.1:43210", "Start the server"], + ["ENOTFOUND", "The server name in http://127.0.0.1:43210 could not be found", "DNS"], + ["EHOSTUNREACH", "Cannot reach http://127.0.0.1:43210", "network"], + ])( + "explains a failed connection without calling it a timeout (%s)", + async (code, message, nextStep) => { + const route = embeddedRoute("openclaw"); + route.runConfig.models = { + providers: { + openai: { + baseUrl: "http://127.0.0.1:43210/v1?key=synthetic-private-value", + models: [], + }, + }, + }; + let retryConnectionErrors: boolean | undefined; + const result = await runSetupInferenceTurn({ + route, + requireExecutionOwner: false, + deps: { + createTempDir: async () => "/tmp/openclaw-setup-inference-test", + removeTempDir: async () => {}, + runEmbeddedAgent: async (params) => { + retryConnectionErrors = params.retryConnectionErrors; + throw new FailoverError("Connection error.", { reason: "timeout", code }); + }, + }, + }); + expect(result).toMatchObject({ + ok: false, + status: "unavailable", + error: expect.stringContaining(message), + }); + expect(retryConnectionErrors).toBe(false); + if (!result.ok) { + expect(result.error).toContain(nextStep); + expect(result.error).toContain("No default model was changed."); + expect(result.error).not.toContain("synthetic-private-value"); + } + }, + ); + it("waits for the isolated probe runtime to release its plugin work", async () => { const route = embeddedRoute(); const cleanupStarted = createDeferred(); diff --git a/src/system-agent/setup-inference-turn.ts b/src/system-agent/setup-inference-turn.ts index 08c21aa0e280..3aac5bd177de 100644 --- a/src/system-agent/setup-inference-turn.ts +++ b/src/system-agent/setup-inference-turn.ts @@ -11,7 +11,6 @@ import { } from "../agents/agent-run-result.js"; import { resolveAgentWorkspaceDir } from "../agents/agent-scope.js"; import type { AgentExecutionAuthBinding } from "../agents/execution-auth-binding.js"; -import { describeFailoverError } from "../agents/failover-error.js"; import type { AgentHarnessPluginSelection } from "../agents/harness/runtime-plugin-load-plan.js"; import { loadAgentRuntimePluginRegistryHandle } from "../agents/runtime-plugins.js"; import { SessionManager } from "../agents/sessions/index.js"; @@ -38,8 +37,8 @@ import { type ActivateSetupInferenceDeps, type BoundVerifySetupInferenceResult, type CompleteSetupInferenceResult, + describeSetupInferenceError, invalidSetupConfigError, - mapFailoverReasonToSetupStatus, parseInferenceRef, redactSetupInferenceError, resolveSetupInferenceWinnerError, @@ -178,6 +177,7 @@ export async function runSetupInferenceTurn(params: { ...(route.authProfileId ? { authProfileIdSource: "user" as const } : {}), authProfileStateMode: "read-only", allowAuthProfileFallback: false, + retryConnectionErrors: false, preparedModelRuntimeMode: "isolated-read-only", ...(harness === "codex" ? { cleanupBundleMcpOnRunEnd: true } : {}), ...(harness ? { agentHarnessRuntimeOverride: harness } : {}), @@ -230,8 +230,8 @@ export async function runSetupInferenceTurn(params: { auth: successfulAuth ?? (route.authProfileId ? { authProfileId: route.authProfileId } : {}), }; } catch (error) { - const described = describeFailoverError(error); - return failed(mapFailoverReasonToSetupStatus(described.reason), described.message); + const described = describeSetupInferenceError(error, route); + return failed(described.status, described.error); } finally { preparedRunAdmission.close(); clearAgentRunContext(runId);