Share one Codex prompt-cache key per install and model across executions

Every main-loop execution sent the Codex backend its own prompt_cache_key
(the execution id, which the Claudexor adapter also puts in the session_id
header). The backend reuses a cached prefix across conversations only under
the same key and session, so a new root task, subagent child or consciousness
cycle paid the governance prefix cold on its first round although consecutive
executions share it byte-identically (about 286k tokens between two
consciousness cycles, about 215k between two root tasks). In one measured day
those cold first rounds were 63 % of all uncached codex tokens.

`llm_claudexor.cache_key_for_model` now derives one header-safe key per data
root and model, and the three producers of the main-loop affinity use it: the
main call, the prepared and re-prepared call, and the prospective wrap-up
pricing build that must bind the same payload the send produces. Two direct
backend probes (2026-09-17) showed a second conversation under the shared key
receives 99 % of the prefix from cache, a different session_id receives none,
and a conversation's own turn state stays valid after another conversation's
turn under the shared session. API-compatible lanes keep their prefix-derived
session identity; reviewer surfaces keep their own affinities.

Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com>
This commit is contained in:
Anton Razzhigaev 2026-09-17 02:41:03 +03:00
parent fce6b350cb
commit d426bea3cd
8 changed files with 73 additions and 19 deletions

View file

@ -319,6 +319,12 @@ engine, not an Agent Run. `LLMClient` dispatches sync and async calls through
`llm_claudexor.py` before OpenAI-compatible filtering/retries. Ouroboros retains
its SYSTEM, BIBLE, canonical messages, tool selection and execution. The raw-model adapter is Codex; connected Claude/Cursor and other harnesses keep
their existing Agent capabilities. Direct API keys keep their existing routes.
Every main-loop execution of one install sends the same Codex cache key
(`llm_claudexor.cache_key_for_model`: per data root and model, projected to
`prompt_cache_key` and the `session_id` header), so a new task, child or
consciousness cycle is served the governance prefix its predecessors already
cached on its first round; per-execution turn states ride `nativeContinuation`
unchanged under that shared session.
Before creating a model operation, the host discovers the existing operation
catalog's `captureFailureEvidence=true` query and freezes that choice for the

View file

@ -1042,7 +1042,13 @@ by "Provider Independence" above. Call-site imperatives:
direct-Anthropic lane by `_anthropic_blocks_from_content` and on
OpenRouter by `supports_message_cache_control`, and pinned by
`tests/test_review_prompt_caching.py`. The main loop declares an
execution-scoped cache affinity only for subscription transport; API-compatible
install-scoped cache affinity for the subscription transport
(`llm_claudexor.cache_key_for_model`: one Codex `prompt_cache_key`, and so one
`session_id`, per data root and model, shared by every task, child and
consciousness cycle), because the Codex backend reuses a cached prefix across
conversations only under the same session and per-conversation turn states stay
valid under a shared one (measured 2026-09-17); a per-execution key paid the
shared governance prefix cold on every task start. API-compatible
lanes retain their prefix-derived session identity. A consciousness wake-up is one
of those main-loop executions: its schema array and cached system prefix are
byte-identical to an owner turn's, so everything the level or the wake reason

View file

@ -42,7 +42,9 @@ import asyncio
import copy
import contextvars
from dataclasses import replace
import hashlib
import json
import re
import logging
import threading
import time
@ -301,6 +303,27 @@ def _remember_failed_profile(target: dict, parameters: dict, error: ClaudexorMod
_FAILED_PROFILE.set((*key, route["credentialProfileId"]))
def cache_key_for_model(model: str) -> str:
"""The Codex prompt-cache key every main-loop execution of this install shares.
The Codex backend reuses a cached prefix across conversations only when both
``prompt_cache_key`` and the ``session_id`` header match (the adapter sets
both from ``cacheKey``), and per-conversation turn states stay valid under a
shared session (measured 2026-09-17). One key per data root and model
therefore lets a new task, child or consciousness cycle be served the
governance prefix it shares with its predecessors on its very first round,
instead of paying it cold under a per-execution key. Empty for every other
provider: API-compatible lanes keep their prefix-derived session identity.
"""
from ouroboros.provider_models import provider_for_model
if provider_for_model(model) != "claudexor":
return ""
label = re.sub(r"[^A-Za-z0-9._-]+", "-", str(model).rsplit("=", 1)[-1]).strip("-")[:40] or "model"
digest = hashlib.sha256(f"{config.DATA_DIR}\0{model}".encode("utf-8")).hexdigest()[:16]
return f"ouroboros-{label}-{digest}"
def _request(target: dict, messages: list, tools: list | None, parameters: dict) -> dict:
from ouroboros.llm_messages import _MessageShapingMixin

View file

@ -29,7 +29,7 @@ from ouroboros.deadline_utils import (
main_transport_timeout_sec as _main_transport_timeout,
)
from ouroboros.llm import LLMClient, LocalContextTooLargeError, add_usage
from ouroboros.llm_claudexor import propagate_model_error
from ouroboros.llm_claudexor import cache_key_for_model, propagate_model_error
from ouroboros.model_wait import propagate_model_control
from ouroboros.openai_chat_dispatch import CUSTOM_RECEIPTS_USAGE_KEY, pop_custom_validation_receipts
from ouroboros.llm_attempt import PROVIDER_POLICY_REFUSAL, _is_provider_policy_refusal # typed-refusal contract owner
@ -1382,7 +1382,7 @@ def call_llm_with_retry(
"max_tokens": MAIN_LOOP_MAX_TOKENS,
"stream": True, "caller_deadline_ts": (None if deadline_ts is None
else float(deadline_ts) - float(transport_reserve_sec or 0.0)),
"use_local": use_local, "cache_affinity": execution_id if provider_for_model(model) == "claudexor" else "",
"use_local": use_local, "cache_affinity": cache_key_for_model(model),
"allow_server_web_search": bool(allow_server_web_search) and provider_for_model(model) != "claudexor",
"bypass_response_cache": response_cache_bypass_requested and provider_for_model(model) != "claudexor",
"timeout": _main_transport_timeout(model, deadline_ts, reserve_sec=transport_reserve_sec),

View file

@ -555,11 +555,9 @@ def _reprepare_waiting_main(ctx: _RoundModelCallContext, kwargs: dict):
_loop()._run_main_reclaim(ctx, disposition)
disposition = _measure_after_reclaim(ctx)
kwargs["messages"] = ctx.messages
from ouroboros.llm_claudexor import cache_key_for_model
from ouroboros.provider_models import provider_for_model
kwargs["cache_affinity"] = (
str(ctx.accumulated_usage.get("execution_id") or "")
if not use_local and provider_for_model(model) == "claudexor" else ""
)
kwargs["cache_affinity"] = "" if use_local else cache_key_for_model(model)
kwargs["allow_server_web_search"] = (_loop()._server_web_allowed_by_task(ctx.tools._ctx)
and not use_local and provider_for_model(model) != "claudexor")
if provider_for_model(model) == "claudexor":

View file

@ -718,10 +718,10 @@ def prepared_wrapup_candidate(
The forced send this candidate admits continues the loop's active transport
turn, so the candidate is built from that same owner slot."""
from ouroboros.llm_claudexor import cache_key_for_model
from ouroboros.loop_llm_call import _prepare_main_messages
from ouroboros.model_slots import task_model_binding, task_processing_preference
from ouroboros.model_wait import current_model_wait
from ouroboros.observability import new_execution_id
owner_ctx = getattr(getattr(ctx, "tools", None), "_ctx", None)
waiter = current_model_wait()
@ -750,9 +750,9 @@ def prepared_wrapup_candidate(
model_account_override=account,
model_turn_state=getattr(owner_ctx, "model_turn_state", None),
# The admitted candidate must be the payload the send will produce: the
# main loop declares the same execution-scoped cache affinity, so this
# prepared copy binds the execution id exactly as that dispatch does.
cache_affinity=str(ctx.accumulated_usage.setdefault("execution_id", new_execution_id())),
# main loop declares the same install-scoped cache affinity, so this
# prepared copy binds the same key as that dispatch does.
cache_affinity="" if getattr(ctx, "active_use_local", False) else cache_key_for_model(ctx.active_model),
processing_preference=task_processing_preference(
{"task_metadata": getattr(owner_ctx, "task_metadata", {})}, model_role=role),
)

View file

@ -2,7 +2,8 @@
import queue
from ouroboros.llm_claudexor import _request
from ouroboros import config
from ouroboros.llm_claudexor import _request, cache_key_for_model
from ouroboros.loop_llm_call import call_llm_with_retry
@ -16,7 +17,19 @@ def test_claudexor_request_projects_explicit_affinity_to_cache_key():
assert payload["options"]["cacheKey"] == "execution-7"
def test_main_loop_scopes_execution_affinity_to_claudexor(tmp_path):
def test_cache_key_is_install_scoped_stable_and_header_safe(monkeypatch, tmp_path):
key = cache_key_for_model("claudexor::codex=gpt-6-astra")
assert key == cache_key_for_model("claudexor::codex=gpt-6-astra"), "same install + model -> same key"
assert key.startswith("ouroboros-gpt-6-astra-")
assert all(ch.isalnum() or ch in "._-" for ch in key), "must be a valid HTTP header value"
assert key != cache_key_for_model("claudexor::codex=gpt-5.6-sol"), "one shard per model"
assert cache_key_for_model("openrouter::openai/model") == ""
assert cache_key_for_model("openai/gpt-5.6-sol") == ""
monkeypatch.setattr(config, "DATA_DIR", tmp_path / "other-install")
assert cache_key_for_model("claudexor::codex=gpt-6-astra") != key, "another data root is another install"
def test_main_loop_shares_one_install_affinity_across_executions_on_claudexor(tmp_path):
captured = []
class LLM:
@ -29,16 +42,21 @@ def test_main_loop_scopes_execution_affinity_to_claudexor(tmp_path):
logs = tmp_path / "logs"
logs.mkdir()
for model in ("claudexor::codex=model", "openrouter::openai/model"):
usage = {"execution_id": "execution-7"}
for model, execution in (("claudexor::codex=model", "execution-7"), ("claudexor::codex=model", "execution-8"),
("openrouter::openai/model", "execution-9")):
usage = {"execution_id": execution}
message, _cost = call_llm_with_retry(
LLM(), [{"role": "user", "content": "work"}], model, None, "medium", 1,
logs, "task", 1, queue.Queue(), usage,
)
assert message["content"] == "done"
assert captured[0]["cache_affinity"] == "execution-7"
assert captured[1]["cache_affinity"] == ""
# Two executions of one install share the key, so the second one's first
# round is served the governance prefix the first one already paid for.
assert captured[0]["cache_affinity"] == cache_key_for_model("claudexor::codex=model")
assert captured[1]["cache_affinity"] == captured[0]["cache_affinity"]
assert "execution-7" not in captured[0]["cache_affinity"]
assert captured[2]["cache_affinity"] == ""
def test_main_loop_projects_claudexor_options_outside_the_route(tmp_path):

View file

@ -11,6 +11,7 @@ import pytest
from ouroboros import loop, model_wait, usage_accounting as ua
from ouroboros.llm_attempt import _attempt_request, _candidate_before_dispatch
from ouroboros.llm_claudexor import cache_key_for_model
from ouroboros.loop_model_call import _reprepare_waiting_main
from ouroboros.model_slots import MODEL_ACCOUNTS_KEY
from tests.test_context_fit_integration import _plan
@ -539,7 +540,7 @@ def test_live_owner_wait_reprojects_affinity(main_call, monkeypatch, destination
}
assert facts["completed_tool_texts"] == ["verified read A", "completed review B"]
assert ctx.accumulated_usage["execution_id"] == CACHE_REPREPARE_EXECUTION
assert prepared[-1]["cache_affinity"] == ("" if use_local or destination != MODEL else CACHE_REPREPARE_EXECUTION), (
assert prepared[-1]["cache_affinity"] == ("" if use_local or destination != MODEL else cache_key_for_model(MODEL)), (
facts
)
if not use_local:
@ -581,4 +582,6 @@ def test_recorded_wait_override_reprojects_affinity_before_send(main_call, initi
}
assert facts["completed_tool_texts"] == ["verified read A", "completed review B"]
assert ctx.accumulated_usage["execution_id"] == CACHE_REPREPARE_EXECUTION
assert payload["options"].get("cacheKey") == CACHE_REPREPARE_EXECUTION, facts
# The override re-prepared the send for the subscription route, so the wire
# carries the install-scoped Codex key that route shares across executions.
assert payload["options"].get("cacheKey") == cache_key_for_model(MODEL), facts