mirror of
https://github.com/openclaw/openclaw.git
synced 2026-10-03 09:39:25 +00:00
fix(cron): transient provider timeouts skip later scheduled runs (#156237)
Closes #156163 ## What Problem This Solves Fixes later scheduled runs being skipped for five minutes after one local-provider preflight hits its client-side deadline, even when the endpoint has recovered. ## User Impact The next run can reach a recovered endpoint without waiting for the negative cache to expire. The timed-out run still skips; hard endpoint failures remain cached. No configuration or migration is needed. ## Why This Change Was Made Reuse the existing guarded-fetch timeout classification for the cache-write decision. Successful probes and non-timeout failures retain their existing five-minute cache behavior. This does not add an immediate retry or change the timeout. ## Evidence - A real local HTTP server withheld the first `/v1/models` response beyond 2500ms and answered subsequent requests. On the unchanged baseline, both consecutive preflights returned unavailable with only one HTTP request; with this change, the next preflight returned available after a second request. - Through an isolated Gateway and `openclaw automations`, the first run skipped on the real HTTP timeout; the next run in the same Gateway, 11.949 seconds later, completed successfully with `PROOF_OK`. The provider recorded two model probes and one completion request. The baseline comparison above exercises the preflight owner rather than a full baseline Gateway run. - Direct and nested timeout-recovery regression cases fail on the baseline and pass with the fix. Both cases take approximately 1ms each. The four focused preflight, transport, fallback-policy, and isolated-agent suites passed: 47 tests, 41.25s wall including cold worker preparation. Existing hard-failure caching and cache-expiry cases pass. - `pnpm test src/cron/isolated-agent/model-preflight.runtime.test.ts --maxWorkers=1`: 23 tests passed, 1.92s wall. - `git diff --check` passed. Co-authored-by: Ayaan Zaidi <hi@obviy.us>
This commit is contained in:
parent
f421207ecf
commit
0864fb6f86
3 changed files with 48 additions and 6 deletions
|
|
@ -164,6 +164,8 @@ If a run hits a live model-switch handoff, the scheduler retries with the switch
|
|||
|
||||
Before an isolated run starts, OpenClaw checks reachable local endpoints for configured `api: "ollama"` and `api: "openai-completions"` providers whose `baseUrl` is loopback, private-network, or `.local`. This preflight walks the job's configured fallback chain and only marks the run `skipped` once every candidate is unreachable; `--fallbacks ""` keeps that walk strict to just the primary model. A down endpoint records the run as `skipped` with a clear error instead of starting a model call. The result is cached for 5 minutes per endpoint (not per job or model), so many due jobs sharing a dead local Ollama/vLLM/SGLang/LM Studio server cost one probe instead of a request storm. Skipped preflight runs do not increment execution-error backoff; set `failureAlert.includeSkipped` to opt into repeated skip alerts.
|
||||
|
||||
Client-side preflight timeouts are not cached. The next scheduled run probes the endpoint again instead of inheriting a timeout from another run.
|
||||
|
||||
### Command payloads
|
||||
|
||||
Command payloads run deterministic scripts inside the Gateway scheduler without starting a model-backed turn. They execute on the Gateway host, capture stdout/stderr, record the run in the job's run history, and reuse the same `announce`, `webhook`, and `none` delivery modes as agent-turn jobs.
|
||||
|
|
|
|||
|
|
@ -283,6 +283,41 @@ describe("preflightCronModelProvider", () => {
|
|||
expect(request.auditContext).toBe("cron-model-provider-preflight");
|
||||
});
|
||||
|
||||
it.each([false, true])("reprobes after a client timeout (nested: %s)", async (nested) => {
|
||||
const timeout = new DOMException("request timed out", "TimeoutError");
|
||||
fetchWithSsrFGuardMock.mockRejectedValueOnce(
|
||||
nested ? new TypeError("fetch failed", { cause: timeout }) : timeout,
|
||||
);
|
||||
mockReachableResponse();
|
||||
const cfg = {
|
||||
models: {
|
||||
providers: {
|
||||
vllm: {
|
||||
api: "openai-completions" as const,
|
||||
baseUrl: "http://127.0.0.1:8000/v1",
|
||||
models: [],
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
const first = await preflightCronModelProvider({
|
||||
cfg,
|
||||
provider: "vllm",
|
||||
model: "first",
|
||||
nowMs: 1000,
|
||||
});
|
||||
const next = await preflightCronModelProvider({
|
||||
cfg,
|
||||
provider: "vllm",
|
||||
model: "next",
|
||||
nowMs: 2000,
|
||||
});
|
||||
|
||||
expect(first.status).toBe("unavailable");
|
||||
expect(next).toEqual({ status: "available" });
|
||||
});
|
||||
|
||||
it("reports a nested guarded-fetch deadline separately from endpoint failures", async () => {
|
||||
const timeoutError = new Error("request timed out");
|
||||
timeoutError.name = "TimeoutError";
|
||||
|
|
|
|||
|
|
@ -123,13 +123,16 @@ function collectPreflightErrorCauseChain(error: unknown): unknown[] {
|
|||
return chain;
|
||||
}
|
||||
|
||||
function formatPreflightError(error: unknown): string {
|
||||
const causeChain = collectPreflightErrorCauseChain(error);
|
||||
const causeDetails = formatErrorMessageWithCode(error);
|
||||
function isPreflightTimeout(error: unknown): boolean {
|
||||
// fetchWithSsrFGuard propagates only its owned deadline as TimeoutError.
|
||||
const classified = causeChain.some(
|
||||
return collectPreflightErrorCauseChain(error).some(
|
||||
(candidate) => readErrorProperty(candidate, "name") === "TimeoutError",
|
||||
)
|
||||
);
|
||||
}
|
||||
|
||||
function formatPreflightError(error: unknown): string {
|
||||
const causeDetails = formatErrorMessageWithCode(error);
|
||||
const classified = isPreflightTimeout(error)
|
||||
? `Local provider preflight exceeded its configured ${PREFLIGHT_TIMEOUT_MS}ms deadline | ${causeDetails}`
|
||||
: causeDetails;
|
||||
return classified.length <= MAX_PREFLIGHT_ERROR_CHARS
|
||||
|
|
@ -240,7 +243,9 @@ export async function preflightCronModelProvider(params: {
|
|||
} catch (error) {
|
||||
result = { status: "unavailable", error };
|
||||
}
|
||||
preflightCache.set(cacheKey, { checkedAtMs: nowMs, result });
|
||||
if (result.status === "available" || !isPreflightTimeout(result.error)) {
|
||||
preflightCache.set(cacheKey, { checkedAtMs: nowMs, result });
|
||||
}
|
||||
if (result.status === "available") {
|
||||
return { status: "available" };
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue