diff --git a/docs/automation/cron-jobs/payloads.md b/docs/automation/cron-jobs/payloads.md index 472ac34ed5fe..4fd77b3e3f3c 100644 --- a/docs/automation/cron-jobs/payloads.md +++ b/docs/automation/cron-jobs/payloads.md @@ -164,6 +164,8 @@ If a run hits a live model-switch handoff, the scheduler retries with the switch Before an isolated run starts, OpenClaw checks reachable local endpoints for configured `api: "ollama"` and `api: "openai-completions"` providers whose `baseUrl` is loopback, private-network, or `.local`. This preflight walks the job's configured fallback chain and only marks the run `skipped` once every candidate is unreachable; `--fallbacks ""` keeps that walk strict to just the primary model. A down endpoint records the run as `skipped` with a clear error instead of starting a model call. The result is cached for 5 minutes per endpoint (not per job or model), so many due jobs sharing a dead local Ollama/vLLM/SGLang/LM Studio server cost one probe instead of a request storm. Skipped preflight runs do not increment execution-error backoff; set `failureAlert.includeSkipped` to opt into repeated skip alerts. +Client-side preflight timeouts are not cached. The next scheduled run probes the endpoint again instead of inheriting a timeout from another run. + ### Command payloads Command payloads run deterministic scripts inside the Gateway scheduler without starting a model-backed turn. They execute on the Gateway host, capture stdout/stderr, record the run in the job's run history, and reuse the same `announce`, `webhook`, and `none` delivery modes as agent-turn jobs. diff --git a/src/cron/isolated-agent/model-preflight.runtime.test.ts b/src/cron/isolated-agent/model-preflight.runtime.test.ts index fb34278fbb42..28821213a5fa 100644 --- a/src/cron/isolated-agent/model-preflight.runtime.test.ts +++ b/src/cron/isolated-agent/model-preflight.runtime.test.ts @@ -283,6 +283,41 @@ describe("preflightCronModelProvider", () => { expect(request.auditContext).toBe("cron-model-provider-preflight"); }); + it.each([false, true])("reprobes after a client timeout (nested: %s)", async (nested) => { + const timeout = new DOMException("request timed out", "TimeoutError"); + fetchWithSsrFGuardMock.mockRejectedValueOnce( + nested ? new TypeError("fetch failed", { cause: timeout }) : timeout, + ); + mockReachableResponse(); + const cfg = { + models: { + providers: { + vllm: { + api: "openai-completions" as const, + baseUrl: "http://127.0.0.1:8000/v1", + models: [], + }, + }, + }, + }; + + const first = await preflightCronModelProvider({ + cfg, + provider: "vllm", + model: "first", + nowMs: 1000, + }); + const next = await preflightCronModelProvider({ + cfg, + provider: "vllm", + model: "next", + nowMs: 2000, + }); + + expect(first.status).toBe("unavailable"); + expect(next).toEqual({ status: "available" }); + }); + it("reports a nested guarded-fetch deadline separately from endpoint failures", async () => { const timeoutError = new Error("request timed out"); timeoutError.name = "TimeoutError"; diff --git a/src/cron/isolated-agent/model-preflight.runtime.ts b/src/cron/isolated-agent/model-preflight.runtime.ts index 358f25c924c7..c44c7461313f 100644 --- a/src/cron/isolated-agent/model-preflight.runtime.ts +++ b/src/cron/isolated-agent/model-preflight.runtime.ts @@ -123,13 +123,16 @@ function collectPreflightErrorCauseChain(error: unknown): unknown[] { return chain; } -function formatPreflightError(error: unknown): string { - const causeChain = collectPreflightErrorCauseChain(error); - const causeDetails = formatErrorMessageWithCode(error); +function isPreflightTimeout(error: unknown): boolean { // fetchWithSsrFGuard propagates only its owned deadline as TimeoutError. - const classified = causeChain.some( + return collectPreflightErrorCauseChain(error).some( (candidate) => readErrorProperty(candidate, "name") === "TimeoutError", - ) + ); +} + +function formatPreflightError(error: unknown): string { + const causeDetails = formatErrorMessageWithCode(error); + const classified = isPreflightTimeout(error) ? `Local provider preflight exceeded its configured ${PREFLIGHT_TIMEOUT_MS}ms deadline | ${causeDetails}` : causeDetails; return classified.length <= MAX_PREFLIGHT_ERROR_CHARS @@ -240,7 +243,9 @@ export async function preflightCronModelProvider(params: { } catch (error) { result = { status: "unavailable", error }; } - preflightCache.set(cacheKey, { checkedAtMs: nowMs, result }); + if (result.status === "available" || !isPreflightTimeout(result.error)) { + preflightCache.set(cacheKey, { checkedAtMs: nowMs, result }); + } if (result.status === "available") { return { status: "available" }; }