diff --git a/docs/start/onboarding-overview.md b/docs/start/onboarding-overview.md index f398ae4738bf..19d8336efd9a 100644 --- a/docs/start/onboarding-overview.md +++ b/docs/start/onboarding-overview.md @@ -139,6 +139,11 @@ setup verifies a real model reply before saving the provider and activating its model. A failed or cancelled check preserves the previous configuration. The classic wizard also retains its custom-provider setup. +If the endpoint refuses the connection or its hostname cannot be found, setup +reports the failed connection immediately instead of waiting through normal +chat retries. Start the server or correct the URL and network settings on the +Gateway host, then retry. Ordinary agent sessions keep their connection retries. + ## Related - [Getting started](/start/getting-started) diff --git a/src/agents/embedded-agent-runner/run/assistant-failure.ts b/src/agents/embedded-agent-runner/run/assistant-failure.ts index fafa6858a3fc..bd756aff9e78 100644 --- a/src/agents/embedded-agent-runner/run/assistant-failure.ts +++ b/src/agents/embedded-agent-runner/run/assistant-failure.ts @@ -489,6 +489,7 @@ export async function handleEmbeddedAssistantFailure(input: { profileId: input.authProfileId, authMode, status, + code: failedAssistant?.errorCode, rawError: failedAssistant?.errorMessage?.trim(), // Retry reason "timeout" also includes 5xx; only the terminal owner records a deadline. timeout: diff --git a/src/agents/embedded-agent-runner/run/attempt-recovery.test-support.ts b/src/agents/embedded-agent-runner/run/attempt-recovery.test-support.ts index 7aef8ca51571..95bc5f192122 100644 --- a/src/agents/embedded-agent-runner/run/attempt-recovery.test-support.ts +++ b/src/agents/embedded-agent-runner/run/attempt-recovery.test-support.ts @@ -36,6 +36,7 @@ export type TransportDropScenario = { lastToolError?: Parameters[0]["lastToolError"]; pluginHarnessOwnsTransport?: boolean; retryAvailable?: boolean; + retryConnectionErrors?: boolean; replaySafe?: boolean; fallbackConfigured?: boolean; providerRetryMaxDelayMs?: number; @@ -136,9 +137,11 @@ export async function recoverAfterTransportDrop(scenario: TransportDropScenario const continueFromCurrentTranscript = vi.fn(); const contextRecoveryState = createEmbeddedRunContextRecoveryState(); const failoverRetryController = createEmbeddedRunFailoverRetryController({ - runParams: { runId: "run:transport-drop", config: scenario.config } as Parameters< - typeof createEmbeddedRunFailoverRetryController - >[0]["runParams"], + runParams: { + runId: "run:transport-drop", + config: scenario.config, + retryConnectionErrors: scenario.retryConnectionErrors, + } as Parameters[0]["runParams"], provider, modelId, globalLane: "test", diff --git a/src/agents/embedded-agent-runner/run/attempt-recovery.test.ts b/src/agents/embedded-agent-runner/run/attempt-recovery.test.ts index efe4f9df5d70..5aab5579184f 100644 --- a/src/agents/embedded-agent-runner/run/attempt-recovery.test.ts +++ b/src/agents/embedded-agent-runner/run/attempt-recovery.test.ts @@ -35,6 +35,30 @@ const tempDirs = createTempDirTracker(); const requireRecord = createRequireRecord("record", "expected-label-object-capitalized"); afterEach(() => tempDirs.cleanup()); +it.each(["ECONNREFUSED", "ENOTFOUND", "EHOSTUNREACH", "ENETUNREACH"])( + "fails fast on %s for setup while normal runs retain connection retries", + async (errorCode) => { + const scenario = { + errorMessage: "Connection error.", + errorCode, + noTools: true, + replaySafe: true, + content: [], + } satisfies TransportDropScenario; + vi.mocked(sleepWithAbort).mockClear(); + const setup = await recoverAfterTransportDrop({ ...scenario, retryConnectionErrors: false }); + expect(setup.recovery.action).toBe("proceed"); + expect(sleepWithAbort).not.toHaveBeenCalled(); + await expect(handleAssistantFailureAfterRecovery(setup)).rejects.toMatchObject({ + reason: "timeout", + code: errorCode, + }); + const ordinary = await recoverAfterTransportDrop(scenario); + expect(ordinary.recovery.action).toBe("retry"); + expect(sleepWithAbort).toHaveBeenCalledOnce(); + }, +); + function handleAssistantFailureAfterRecovery( fixture: Awaited>, previousRetryFailoverReason: Parameters< diff --git a/src/agents/embedded-agent-runner/run/attempt-recovery.ts b/src/agents/embedded-agent-runner/run/attempt-recovery.ts index 6315fd372386..1374d24da810 100644 --- a/src/agents/embedded-agent-runner/run/attempt-recovery.ts +++ b/src/agents/embedded-agent-runner/run/attempt-recovery.ts @@ -13,6 +13,7 @@ import { } from "../../failover-error.js"; import { failoverReasonFromClassification } from "../../failover/classification-rules.js"; import { classifyFailoverSignal } from "../../failover/classify.js"; +import { getFailoverErrorCode } from "../../failover/error.js"; import { resolveRetryAfterMs } from "../../failover/retry-evidence.js"; import { LiveSessionModelSwitchError } from "../../live-model-switch-error.js"; import { shouldSwitchToLiveModel, clearLiveModelSwitchPending } from "../../live-model-switch.js"; @@ -394,6 +395,7 @@ export async function recoverEmbeddedRunAttempt(input: { recoveryReason && (await failoverRetryController.maybeRetryTransient({ reason: recoveryReason, + code: promptError ? getFailoverErrorCode(promptError) : assistantSignal?.code, message: promptError ? formatErrorMessage(promptError) : assistantSignal?.message, retryAfterMs: promptError ? resolveRetryAfterMs(formatErrorMessage(promptError), Date.now(), promptError) diff --git a/src/agents/embedded-agent-runner/run/failover-retry-controller.ts b/src/agents/embedded-agent-runner/run/failover-retry-controller.ts index f841272d23de..e51f2192e5db 100644 --- a/src/agents/embedded-agent-runner/run/failover-retry-controller.ts +++ b/src/agents/embedded-agent-runner/run/failover-retry-controller.ts @@ -252,6 +252,7 @@ export function createEmbeddedRunFailoverRetryController(input: { maybeRetryTransient: async (retry: { reason: TransientRetryReason; message?: string; + code?: string; retryAfterMs?: number; /** Saved retry.provider.maxRetryDelayMs; undefined or 0 disables the cap. */ maxRetryDelayMs?: number; @@ -268,6 +269,7 @@ export function createEmbeddedRunFailoverRetryController(input: { decision: "accepted" | "rejected", reason: | "non_transient" + | "connection_retry_disabled" | "long_window_rate_limit" | "retry_budget_exhausted" | "retry_delay_unavailable" @@ -284,6 +286,14 @@ export function createEmbeddedRunFailoverRetryController(input: { }, { config: params.config }, ); + if ( + params.retryConnectionErrors === false && + retry.code !== undefined && + ["ECONNREFUSED", "ENOTFOUND", "EHOSTUNREACH", "ENETUNREACH"].includes(retry.code) + ) { + recordDecision("rejected", "connection_retry_disabled"); + return false; + } if ( retry.reason !== "rate_limit" && retry.reason !== "overloaded" && diff --git a/src/agents/embedded-agent-runner/run/params.ts b/src/agents/embedded-agent-runner/run/params.ts index a85f31b0423a..3fdec09e6950 100644 --- a/src/agents/embedded-agent-runner/run/params.ts +++ b/src/agents/embedded-agent-runner/run/params.ts @@ -116,6 +116,8 @@ export type RunEmbeddedAgentParams = { codeModeOverride?: boolean | "auto"; /** Internal one-shot model probe mode: no tools, no workspace/chat prompt policy. */ modelRun?: boolean; + /** Setup can reject unavailable endpoints without spending the session retry budget. */ + retryConnectionErrors?: boolean; /** Disable trajectory persistence for auxiliary runs with no durable session owner. */ disableTrajectory?: boolean; /** Restrict Skill Workshop to a bounded pending-proposal budget for an internal review run. */ diff --git a/src/system-agent/setup-inference-core.ts b/src/system-agent/setup-inference-core.ts index ef0a6c115783..3863b4cf8368 100644 --- a/src/system-agent/setup-inference-core.ts +++ b/src/system-agent/setup-inference-core.ts @@ -8,6 +8,7 @@ import type { AgentRunResultView } from "../agents/agent-run-result.js"; import type { loadAuthProfileStoreForRuntime } from "../agents/auth-profiles/store-runtime.js"; import type { readCodexCliActiveApiKey } from "../agents/cli-credentials.js"; import type { AgentExecutionAuthBinding } from "../agents/execution-auth-binding.js"; +import { describeFailoverError } from "../agents/failover-error.js"; import type { FailoverReason } from "../agents/failover/signal.js"; import { DEFAULT_AGENT_WORKSPACE_DIR } from "../agents/workspace-default.js"; import type { @@ -402,12 +403,34 @@ const SETUP_STATUS_BY_FAILOVER_REASON = { unknown: "unknown", } satisfies Record; -export function mapFailoverReasonToSetupStatus( +function mapFailoverReasonToSetupStatus( reason?: FailoverReason | null, ): SetupInferenceFailureStatus { return reason ? SETUP_STATUS_BY_FAILOVER_REASON[reason] : "unknown"; } +export function describeSetupInferenceError( + error: unknown, + route: SystemAgentConfiguredRoute, +): { status: SetupInferenceFailureStatus; error: string } { + const described = describeFailoverError(error); + const origin = URL.parse( + route.runConfig.models?.providers?.[route.provider]?.baseUrl ?? "", + )?.origin; + const connectionError = !origin + ? undefined + : described.code === "ECONNREFUSED" + ? `Nothing is listening at ${origin}. Start the server or check the URL, then retry setup.` + : described.code === "ENOTFOUND" + ? `The server name in ${origin} could not be found. Check the URL and DNS settings, then retry setup.` + : described.code === "EHOSTUNREACH" || described.code === "ENETUNREACH" + ? `Cannot reach ${origin}. Check the URL and network connection from the Gateway host, then retry setup.` + : undefined; + return connectionError + ? { status: "unavailable", error: `${connectionError} No default model was changed.` } + : { status: mapFailoverReasonToSetupStatus(described.reason), error: described.message }; +} + export function validateSetupInferenceOwnerEvidence(params: { runner: "cli" | "embedded"; configuredHarnessId?: string; diff --git a/src/system-agent/setup-inference-turn.test.ts b/src/system-agent/setup-inference-turn.test.ts index ba77518df8b6..d2a3da205072 100644 --- a/src/system-agent/setup-inference-turn.test.ts +++ b/src/system-agent/setup-inference-turn.test.ts @@ -1,5 +1,6 @@ import { describe, expect, it, vi } from "vitest"; import { createDeferred } from "../../test/helpers/promise.js"; +import { FailoverError } from "../agents/failover-error.js"; import { createPluginMetadataSnapshot, makeRegistry, @@ -67,6 +68,49 @@ function embeddedRoute(runtime: "codex" | "openclaw" = "codex"): SystemAgentConf } describe("setup inference plugin ownership", () => { + it.each([ + ["ECONNREFUSED", "Nothing is listening at http://127.0.0.1:43210", "Start the server"], + ["ENOTFOUND", "The server name in http://127.0.0.1:43210 could not be found", "DNS"], + ["EHOSTUNREACH", "Cannot reach http://127.0.0.1:43210", "network"], + ])( + "explains a failed connection without calling it a timeout (%s)", + async (code, message, nextStep) => { + const route = embeddedRoute("openclaw"); + route.runConfig.models = { + providers: { + openai: { + baseUrl: "http://127.0.0.1:43210/v1?key=synthetic-private-value", + models: [], + }, + }, + }; + let retryConnectionErrors: boolean | undefined; + const result = await runSetupInferenceTurn({ + route, + requireExecutionOwner: false, + deps: { + createTempDir: async () => "/tmp/openclaw-setup-inference-test", + removeTempDir: async () => {}, + runEmbeddedAgent: async (params) => { + retryConnectionErrors = params.retryConnectionErrors; + throw new FailoverError("Connection error.", { reason: "timeout", code }); + }, + }, + }); + expect(result).toMatchObject({ + ok: false, + status: "unavailable", + error: expect.stringContaining(message), + }); + expect(retryConnectionErrors).toBe(false); + if (!result.ok) { + expect(result.error).toContain(nextStep); + expect(result.error).toContain("No default model was changed."); + expect(result.error).not.toContain("synthetic-private-value"); + } + }, + ); + it("waits for the isolated probe runtime to release its plugin work", async () => { const route = embeddedRoute(); const cleanupStarted = createDeferred(); diff --git a/src/system-agent/setup-inference-turn.ts b/src/system-agent/setup-inference-turn.ts index 08c21aa0e280..3aac5bd177de 100644 --- a/src/system-agent/setup-inference-turn.ts +++ b/src/system-agent/setup-inference-turn.ts @@ -11,7 +11,6 @@ import { } from "../agents/agent-run-result.js"; import { resolveAgentWorkspaceDir } from "../agents/agent-scope.js"; import type { AgentExecutionAuthBinding } from "../agents/execution-auth-binding.js"; -import { describeFailoverError } from "../agents/failover-error.js"; import type { AgentHarnessPluginSelection } from "../agents/harness/runtime-plugin-load-plan.js"; import { loadAgentRuntimePluginRegistryHandle } from "../agents/runtime-plugins.js"; import { SessionManager } from "../agents/sessions/index.js"; @@ -38,8 +37,8 @@ import { type ActivateSetupInferenceDeps, type BoundVerifySetupInferenceResult, type CompleteSetupInferenceResult, + describeSetupInferenceError, invalidSetupConfigError, - mapFailoverReasonToSetupStatus, parseInferenceRef, redactSetupInferenceError, resolveSetupInferenceWinnerError, @@ -178,6 +177,7 @@ export async function runSetupInferenceTurn(params: { ...(route.authProfileId ? { authProfileIdSource: "user" as const } : {}), authProfileStateMode: "read-only", allowAuthProfileFallback: false, + retryConnectionErrors: false, preparedModelRuntimeMode: "isolated-read-only", ...(harness === "codex" ? { cleanupBundleMcpOnRunEnd: true } : {}), ...(harness ? { agentHarnessRuntimeOverride: harness } : {}), @@ -230,8 +230,8 @@ export async function runSetupInferenceTurn(params: { auth: successfulAuth ?? (route.authProfileId ? { authProfileId: route.authProfileId } : {}), }; } catch (error) { - const described = describeFailoverError(error); - return failed(mapFailoverReasonToSetupStatus(described.reason), described.message); + const described = describeSetupInferenceError(error, route); + return failed(described.status, described.error); } finally { preparedRunAdmission.close(); clearAgentRunContext(runId);