mirror of
https://github.com/razzant/ouroboros.git
synced 2026-10-03 04:07:04 +00:00
Module side (23 D02 owners): 5 byte-identical across tip/reference/base (fallback_cooldown, local_model, local_model_autostart, model_concurrency, pricing); 2 pure upstream drift - upstream bytes stand (llm_observability, vision_routing); 12 NEW post-cutoff upstream modules with no ledger rows - upstream bytes stand untouched (anthropic_native_custody, net_transport, openai_chat_custom/dispatch, openrouter_attribution, request_wire_* x6, route_spec, transport_custody; transport/timeout contracts not touched). The split: llm.py 4414 -> 727 composes LLMClient from ten mixins per ledger rows 1666-1793/4001-4003 (131 rows). 100 spans byte-identical to the oracle leaves, 28 byte-falsified by pure upstream drift (request-wire custody041e6e39, attribution9a20df6a, custody hardening 802f1056/f702439f) and re-emitted from tip bytes. Proof green: transplant-tool verify per leaf (ast=tokens=bytes on every module-level span, undeclared_top_level=[], leaf_invariants=[]) + member-level byte proof (117 mixin members == tip LLMClient members) + facade audit (14 kept members and 4 kept top-level defs byte-identical to tip; every tip top-level name and member accounted). Two non-tip-byte spans, both ledger-sanctioned: row 1674 (the documented one-identifier requalification) and row 1784 - the LIVE D09 delta: the local lane's reintroduced `for attempt in range(3)` retry loop is deleted (one physical attempt per call; transient failures surface to call_llm_with_retry), RE-DERIVED on tip bytes to preserve upstream 802f1056's exception-owned capture custody clause that a verbatim oracle replay would have reverted. The D09 typed-policy-refusal subfamily (rows 1706/1749/1751/1759/1760 + ProviderPolicyRefusal machinery) is HOT-DEFERRED with evidence: zero raisers/classifiers at tip, all consuming rungs upstream-reworked. provider_models: ledger note-contract completed (lazy config imports -> top-level model_slots/settings_defaults leaf imports, cycle-free); the reference pin restored under its ledger name. llm_probe: reference delta adopted (executor import named at its owner leaf, tip==base). Unrowed tip symbols _RESPONSE_METADATA_LABEL_MAX_CHARS/_bounded_response_metadata_label ride with their only reader into llm_openai_compatible; facade re-exports. Facade keeps the full tip import surface (noqa discipline). Test side: identity suite tests/test_llm_extraction.py (7) and provider-route goldens tests/test_llm_provider_golden.py + 9 fixtures carried; goldens re-baselined from tip behaviour via the suite's own --write (drift classes: attribution headers, request_wire disclosure, response metadata labels, effort-ladder; the volatile request_wire.attempt_id projected to a presence flag; the 2 typed-refusal cases removed - they pin the deferred organ, as does test_llm_typed_policy_refusal.py, not carried). D09 pin test_local_transport_makes_exactly_one_physical_attempt added; the three local-lane retry tests re-pinned 3 -> 1 attempts. Dead-patch class closed: execute_physical_attempt patches -> llm_attempt (6 files), chat-path _execute_candidate/last_physical_attempt_capture -> llm_fallback (2 post-cutoff files, disclosed in-file), local executor -> llm_local; TTL-consumer path pin -> llm_attempt. D01-owned oracle adaptations (loop_messages/loop_round_limits) reverse-mapped to tip spellings, not carried. Test-name accounting: 11 touched files lossless, +1 pin test, 1 pin renamed to its ledger name, 2 new files. size-ratchet manifest regenerated with the official tool (llm.py leaves GIANT_PATHS; no new band entries); ratchet 5 passed; ruff F clean tree-wide; receipts: 324 (targeted battery) + 1788/-n16-loadscope + 1 serial across all 65 llm-importing test files, all rc=0; HEAD held through every pytest run. Ledger corrections: docs/v7next/LEDGER_CORRECTIONS.md D02 section, entries 1-10. (cherry picked from commit 1c25ee661df2745c6d5d1346d996d036cb031f40)
309 lines
11 KiB
Python
309 lines
11 KiB
Python
"""Regression tests: a retried LLM call must not be served from a response cache.
|
|
|
|
`provider_incomplete_response` is classified as transient (`_TRANSIENT_RETRY_KINDS`),
|
|
i.e. the retry loop assumes a repeat MAY produce a different result. That assumption
|
|
only holds when nothing between the client and the model caches responses. A gateway
|
|
response cache (LiteLLM `cache: true`) replays the identical failed body for every
|
|
attempt, so the whole transient-retry budget is spent without ever reaching the model
|
|
and the task ends as `infra_failed` / `provider_unavailable`.
|
|
|
|
Observed in the field: six attempts, one shared `response_id`, each returned in ~0.0s;
|
|
the same request replayed later succeeded.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import pytest
|
|
|
|
|
|
class TestBuildRemoteKwargsCacheOptOut:
|
|
def test_no_cache_field_absent_by_default(self):
|
|
from ouroboros.llm import LLMClient
|
|
|
|
client = LLMClient(api_key="test")
|
|
target = {
|
|
"provider": "openai-compatible",
|
|
"resolved_model": "local-reason",
|
|
"usage_model": "local-reason",
|
|
"api_key": "test",
|
|
"base_url": "http://127.0.0.1:4000/v1",
|
|
"default_headers": {},
|
|
}
|
|
kwargs = client._build_remote_kwargs(
|
|
target=target,
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
reasoning_effort="medium",
|
|
max_tokens=256,
|
|
tool_choice="auto",
|
|
temperature=None,
|
|
tools=None,
|
|
)
|
|
assert "cache" not in (kwargs.get("extra_body") or {}), (
|
|
"the first attempt must stay cacheable; only retries opt out"
|
|
)
|
|
assert "cache" not in kwargs, "cache must never be a top-level kwarg"
|
|
|
|
def test_cache_field_present_when_bypassing(self):
|
|
from ouroboros.llm import LLMClient
|
|
|
|
client = LLMClient(api_key="test")
|
|
target = {
|
|
"provider": "openai-compatible",
|
|
"resolved_model": "local-reason",
|
|
"usage_model": "local-reason",
|
|
"api_key": "test",
|
|
"base_url": "http://127.0.0.1:4000/v1",
|
|
"default_headers": {},
|
|
}
|
|
kwargs = client._build_remote_kwargs(
|
|
target=target,
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
reasoning_effort="medium",
|
|
max_tokens=256,
|
|
tool_choice="auto",
|
|
temperature=None,
|
|
tools=None,
|
|
bypass_response_cache=True,
|
|
)
|
|
assert (kwargs.get("extra_body") or {}).get("cache") == {"no-cache": True}, (
|
|
"a retry must carry LiteLLM's documented per-request cache opt-out, "
|
|
"otherwise the gateway replays the cached failed response"
|
|
)
|
|
assert "cache" not in kwargs, (
|
|
"the OpenAI SDK raises TypeError on unknown top-level kwargs, so the "
|
|
"opt-out must ride in extra_body"
|
|
)
|
|
|
|
def test_litellm_control_is_not_sent_to_direct_providers_or_openrouter(self):
|
|
from ouroboros.llm import LLMClient
|
|
|
|
client = LLMClient(api_key="test")
|
|
targets = [
|
|
{
|
|
"provider": "openai",
|
|
"resolved_model": "gpt-5.5",
|
|
"usage_model": "openai/gpt-5.5",
|
|
"supports_openrouter_extensions": False,
|
|
},
|
|
{
|
|
"provider": "minimax",
|
|
"resolved_model": "MiniMax-M2.5",
|
|
"usage_model": "minimax/MiniMax-M2.5",
|
|
"supports_openrouter_extensions": False,
|
|
},
|
|
{
|
|
"provider": "cloudru",
|
|
"resolved_model": "foundation-model",
|
|
"usage_model": "cloudru/foundation-model",
|
|
"supports_openrouter_extensions": False,
|
|
},
|
|
{
|
|
"provider": "openrouter",
|
|
"resolved_model": "openai/gpt-5.5",
|
|
"usage_model": "openai/gpt-5.5",
|
|
"supports_openrouter_extensions": True,
|
|
},
|
|
]
|
|
|
|
for target in targets:
|
|
kwargs = client._build_remote_kwargs(
|
|
target=target,
|
|
messages=[{"role": "user", "content": "hi"}],
|
|
reasoning_effort="medium",
|
|
max_tokens=256,
|
|
tool_choice="auto",
|
|
temperature=None,
|
|
tools=None,
|
|
skip_capability_fetch=True,
|
|
bypass_response_cache=True,
|
|
)
|
|
assert "cache" not in (kwargs.get("extra_body") or {}), target["provider"]
|
|
|
|
|
|
class TestRetryLoopRequestsCacheOptOut:
|
|
def test_incomplete_response_arms_only_the_following_attempt(self, tmp_path, monkeypatch):
|
|
import time
|
|
|
|
from ouroboros.loop_llm_call import call_llm_with_retry
|
|
|
|
monkeypatch.setattr(time, "sleep", lambda _seconds: None)
|
|
calls = []
|
|
|
|
class IncompleteThenSuccessfulLLM:
|
|
def chat(self, **kwargs):
|
|
calls.append(kwargs)
|
|
if len(calls) == 1:
|
|
return {"content": "", "tool_calls": [], "finish_reason": None}, {}
|
|
return (
|
|
{"content": "recovered", "tool_calls": [], "finish_reason": "stop"},
|
|
{"provider": "openai-compatible", "resolved_model": "local-reason"},
|
|
)
|
|
|
|
msg, _cost = call_llm_with_retry(
|
|
IncompleteThenSuccessfulLLM(),
|
|
[{"role": "user", "content": "hi"}],
|
|
"openai-compatible::local-reason",
|
|
None,
|
|
"medium",
|
|
2,
|
|
tmp_path,
|
|
"task-incomplete-cache",
|
|
1,
|
|
None,
|
|
{},
|
|
)
|
|
|
|
assert msg["content"] == "recovered"
|
|
assert [call["bypass_response_cache"] for call in calls] == [False, True]
|
|
|
|
def test_transport_retry_does_not_request_litellm_cache_bypass(self, tmp_path, monkeypatch):
|
|
import time
|
|
|
|
from ouroboros.loop_llm_call import call_llm_with_retry
|
|
|
|
monkeypatch.setattr(time, "sleep", lambda _seconds: None)
|
|
calls = []
|
|
|
|
class TransientThenSuccessfulLLM:
|
|
def chat(self, **kwargs):
|
|
calls.append(kwargs)
|
|
if len(calls) == 1:
|
|
error = RuntimeError("503 service unavailable")
|
|
error.status_code = 503
|
|
raise error
|
|
return (
|
|
{"content": "recovered", "tool_calls": [], "finish_reason": "stop"},
|
|
{"provider": "openai-compatible", "resolved_model": "local-reason"},
|
|
)
|
|
|
|
msg, _cost = call_llm_with_retry(
|
|
TransientThenSuccessfulLLM(),
|
|
[{"role": "user", "content": "hi"}],
|
|
"openai-compatible::local-reason",
|
|
None,
|
|
"medium",
|
|
2,
|
|
tmp_path,
|
|
"task-transient-cache",
|
|
1,
|
|
None,
|
|
{},
|
|
)
|
|
|
|
assert msg["content"] == "recovered"
|
|
assert [call["bypass_response_cache"] for call in calls] == [False, False]
|
|
|
|
@pytest.mark.parametrize(("kind", "code"), [("rate_limit", 429), ("provider_transient", 503)])
|
|
def test_transient_body_error_does_not_request_litellm_cache_bypass(
|
|
self, kind, code, tmp_path, monkeypatch,
|
|
):
|
|
import time
|
|
|
|
from ouroboros.llm import LLMClient
|
|
from ouroboros.loop_llm_call import call_llm_with_retry
|
|
|
|
monkeypatch.setattr(time, "sleep", lambda _seconds: None)
|
|
target = {
|
|
"provider": "openai-compatible",
|
|
"resolved_model": "local-reason",
|
|
"usage_model": "openai-compatible/local-reason",
|
|
"supports_openrouter_extensions": False,
|
|
}
|
|
body_error_msg, body_error_usage = LLMClient(api_key="test")._normalize_remote_response(
|
|
{
|
|
"id": f"body-error-{code}",
|
|
"choices": [],
|
|
"error": {"code": code, "message": "transient provider error"},
|
|
"usage": {},
|
|
},
|
|
target,
|
|
skip_cost_fetch=True,
|
|
)
|
|
assert body_error_usage["provider_error"]["kind"] == kind
|
|
calls = []
|
|
|
|
class BodyErrorThenSuccessfulLLM:
|
|
def chat(self, **kwargs):
|
|
calls.append(kwargs)
|
|
if len(calls) == 1:
|
|
return body_error_msg, body_error_usage
|
|
return (
|
|
{"content": "recovered", "tool_calls": [], "finish_reason": "stop"},
|
|
{"provider": "openai-compatible", "resolved_model": "local-reason"},
|
|
)
|
|
|
|
msg, _cost = call_llm_with_retry(
|
|
BodyErrorThenSuccessfulLLM(),
|
|
[{"role": "user", "content": "hi"}],
|
|
"openai-compatible::local-reason",
|
|
None,
|
|
"medium",
|
|
2,
|
|
tmp_path,
|
|
f"task-body-error-{code}",
|
|
1,
|
|
None,
|
|
{},
|
|
)
|
|
|
|
assert msg["content"] == "recovered"
|
|
assert [call["bypass_response_cache"] for call in calls] == [False, False]
|
|
|
|
|
|
class TestStrictCompatibleRecovery:
|
|
def test_explicit_cache_rejection_gets_one_exact_retry(self, monkeypatch):
|
|
import ouroboros.llm_attempt as llm_attempt_mod
|
|
from ouroboros.llm import LLMClient
|
|
|
|
monkeypatch.setattr(
|
|
llm_attempt_mod,
|
|
"execute_physical_attempt",
|
|
lambda _request, send: send(),
|
|
)
|
|
client = LLMClient(api_key="unused")
|
|
calls = []
|
|
expected = object()
|
|
|
|
def create(**kwargs):
|
|
calls.append(kwargs)
|
|
if len(calls) == 1:
|
|
raise RuntimeError("400 unknown parameter: cache")
|
|
return expected
|
|
|
|
target = {
|
|
"provider": "openai-compatible",
|
|
"resolved_model": "strict-model",
|
|
"usage_model": "openai-compatible/strict-model",
|
|
"supports_openrouter_extensions": False,
|
|
}
|
|
kwargs = {
|
|
"model": "strict-model",
|
|
"messages": [{"role": "user", "content": "hi"}],
|
|
"extra_body": {
|
|
"cache": {"no-cache": True},
|
|
"future_extension": {"keep": True},
|
|
},
|
|
}
|
|
|
|
result = client._create_chat_completion_with_retries(create, kwargs, target)
|
|
|
|
assert result is expected
|
|
assert len(calls) == 2
|
|
assert calls[0]["extra_body"]["cache"] == {"no-cache": True}
|
|
assert "cache" not in calls[1]["extra_body"]
|
|
assert calls[1]["extra_body"]["future_extension"] == {"keep": True}
|
|
|
|
|
|
class TestChatSignatureThreadsFlag:
|
|
def test_chat_accepts_bypass_response_cache(self):
|
|
import inspect
|
|
|
|
from ouroboros.llm import LLMClient
|
|
|
|
params = inspect.signature(LLMClient.chat).parameters
|
|
assert "bypass_response_cache" in params, (
|
|
"LLM.chat must expose the flag so the retry loop can request a fresh call"
|
|
)
|
|
assert params["bypass_response_cache"].default is False, (
|
|
"bypassing the cache must be opt-in"
|
|
)
|