mirror of
https://github.com/razzant/ouroboros.git
synced 2026-10-03 04:07:04 +00:00
Share one Codex prompt-cache key per install and model across executions
Every main-loop execution sent the Codex backend its own prompt_cache_key (the execution id, which the Claudexor adapter also puts in the session_id header). The backend reuses a cached prefix across conversations only under the same key and session, so a new root task, subagent child or consciousness cycle paid the governance prefix cold on its first round although consecutive executions share it byte-identically (about 286k tokens between two consciousness cycles, about 215k between two root tasks). In one measured day those cold first rounds were 63 % of all uncached codex tokens. `llm_claudexor.cache_key_for_model` now derives one header-safe key per data root and model, and the three producers of the main-loop affinity use it: the main call, the prepared and re-prepared call, and the prospective wrap-up pricing build that must bind the same payload the send produces. Two direct backend probes (2026-09-17) showed a second conversation under the shared key receives 99 % of the prefix from cache, a different session_id receives none, and a conversation's own turn state stays valid after another conversation's turn under the shared session. API-compatible lanes keep their prefix-derived session identity; reviewer surfaces keep their own affinities. Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com>
This commit is contained in:
parent
fce6b350cb
commit
d426bea3cd
8 changed files with 73 additions and 19 deletions
|
|
@ -319,6 +319,12 @@ engine, not an Agent Run. `LLMClient` dispatches sync and async calls through
|
|||
`llm_claudexor.py` before OpenAI-compatible filtering/retries. Ouroboros retains
|
||||
its SYSTEM, BIBLE, canonical messages, tool selection and execution. The raw-model adapter is Codex; connected Claude/Cursor and other harnesses keep
|
||||
their existing Agent capabilities. Direct API keys keep their existing routes.
|
||||
Every main-loop execution of one install sends the same Codex cache key
|
||||
(`llm_claudexor.cache_key_for_model`: per data root and model, projected to
|
||||
`prompt_cache_key` and the `session_id` header), so a new task, child or
|
||||
consciousness cycle is served the governance prefix its predecessors already
|
||||
cached on its first round; per-execution turn states ride `nativeContinuation`
|
||||
unchanged under that shared session.
|
||||
|
||||
Before creating a model operation, the host discovers the existing operation
|
||||
catalog's `captureFailureEvidence=true` query and freezes that choice for the
|
||||
|
|
|
|||
|
|
@ -1042,7 +1042,13 @@ by "Provider Independence" above. Call-site imperatives:
|
|||
direct-Anthropic lane by `_anthropic_blocks_from_content` and on
|
||||
OpenRouter by `supports_message_cache_control`, and pinned by
|
||||
`tests/test_review_prompt_caching.py`. The main loop declares an
|
||||
execution-scoped cache affinity only for subscription transport; API-compatible
|
||||
install-scoped cache affinity for the subscription transport
|
||||
(`llm_claudexor.cache_key_for_model`: one Codex `prompt_cache_key`, and so one
|
||||
`session_id`, per data root and model, shared by every task, child and
|
||||
consciousness cycle), because the Codex backend reuses a cached prefix across
|
||||
conversations only under the same session and per-conversation turn states stay
|
||||
valid under a shared one (measured 2026-09-17); a per-execution key paid the
|
||||
shared governance prefix cold on every task start. API-compatible
|
||||
lanes retain their prefix-derived session identity. A consciousness wake-up is one
|
||||
of those main-loop executions: its schema array and cached system prefix are
|
||||
byte-identical to an owner turn's, so everything the level or the wake reason
|
||||
|
|
|
|||
|
|
@ -42,7 +42,9 @@ import asyncio
|
|||
import copy
|
||||
import contextvars
|
||||
from dataclasses import replace
|
||||
import hashlib
|
||||
import json
|
||||
import re
|
||||
import logging
|
||||
import threading
|
||||
import time
|
||||
|
|
@ -301,6 +303,27 @@ def _remember_failed_profile(target: dict, parameters: dict, error: ClaudexorMod
|
|||
_FAILED_PROFILE.set((*key, route["credentialProfileId"]))
|
||||
|
||||
|
||||
def cache_key_for_model(model: str) -> str:
|
||||
"""The Codex prompt-cache key every main-loop execution of this install shares.
|
||||
|
||||
The Codex backend reuses a cached prefix across conversations only when both
|
||||
``prompt_cache_key`` and the ``session_id`` header match (the adapter sets
|
||||
both from ``cacheKey``), and per-conversation turn states stay valid under a
|
||||
shared session (measured 2026-09-17). One key per data root and model
|
||||
therefore lets a new task, child or consciousness cycle be served the
|
||||
governance prefix it shares with its predecessors on its very first round,
|
||||
instead of paying it cold under a per-execution key. Empty for every other
|
||||
provider: API-compatible lanes keep their prefix-derived session identity.
|
||||
"""
|
||||
from ouroboros.provider_models import provider_for_model
|
||||
|
||||
if provider_for_model(model) != "claudexor":
|
||||
return ""
|
||||
label = re.sub(r"[^A-Za-z0-9._-]+", "-", str(model).rsplit("=", 1)[-1]).strip("-")[:40] or "model"
|
||||
digest = hashlib.sha256(f"{config.DATA_DIR}\0{model}".encode("utf-8")).hexdigest()[:16]
|
||||
return f"ouroboros-{label}-{digest}"
|
||||
|
||||
|
||||
def _request(target: dict, messages: list, tools: list | None, parameters: dict) -> dict:
|
||||
from ouroboros.llm_messages import _MessageShapingMixin
|
||||
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@ from ouroboros.deadline_utils import (
|
|||
main_transport_timeout_sec as _main_transport_timeout,
|
||||
)
|
||||
from ouroboros.llm import LLMClient, LocalContextTooLargeError, add_usage
|
||||
from ouroboros.llm_claudexor import propagate_model_error
|
||||
from ouroboros.llm_claudexor import cache_key_for_model, propagate_model_error
|
||||
from ouroboros.model_wait import propagate_model_control
|
||||
from ouroboros.openai_chat_dispatch import CUSTOM_RECEIPTS_USAGE_KEY, pop_custom_validation_receipts
|
||||
from ouroboros.llm_attempt import PROVIDER_POLICY_REFUSAL, _is_provider_policy_refusal # typed-refusal contract owner
|
||||
|
|
@ -1382,7 +1382,7 @@ def call_llm_with_retry(
|
|||
"max_tokens": MAIN_LOOP_MAX_TOKENS,
|
||||
"stream": True, "caller_deadline_ts": (None if deadline_ts is None
|
||||
else float(deadline_ts) - float(transport_reserve_sec or 0.0)),
|
||||
"use_local": use_local, "cache_affinity": execution_id if provider_for_model(model) == "claudexor" else "",
|
||||
"use_local": use_local, "cache_affinity": cache_key_for_model(model),
|
||||
"allow_server_web_search": bool(allow_server_web_search) and provider_for_model(model) != "claudexor",
|
||||
"bypass_response_cache": response_cache_bypass_requested and provider_for_model(model) != "claudexor",
|
||||
"timeout": _main_transport_timeout(model, deadline_ts, reserve_sec=transport_reserve_sec),
|
||||
|
|
|
|||
|
|
@ -555,11 +555,9 @@ def _reprepare_waiting_main(ctx: _RoundModelCallContext, kwargs: dict):
|
|||
_loop()._run_main_reclaim(ctx, disposition)
|
||||
disposition = _measure_after_reclaim(ctx)
|
||||
kwargs["messages"] = ctx.messages
|
||||
from ouroboros.llm_claudexor import cache_key_for_model
|
||||
from ouroboros.provider_models import provider_for_model
|
||||
kwargs["cache_affinity"] = (
|
||||
str(ctx.accumulated_usage.get("execution_id") or "")
|
||||
if not use_local and provider_for_model(model) == "claudexor" else ""
|
||||
)
|
||||
kwargs["cache_affinity"] = "" if use_local else cache_key_for_model(model)
|
||||
kwargs["allow_server_web_search"] = (_loop()._server_web_allowed_by_task(ctx.tools._ctx)
|
||||
and not use_local and provider_for_model(model) != "claudexor")
|
||||
if provider_for_model(model) == "claudexor":
|
||||
|
|
|
|||
|
|
@ -718,10 +718,10 @@ def prepared_wrapup_candidate(
|
|||
|
||||
The forced send this candidate admits continues the loop's active transport
|
||||
turn, so the candidate is built from that same owner slot."""
|
||||
from ouroboros.llm_claudexor import cache_key_for_model
|
||||
from ouroboros.loop_llm_call import _prepare_main_messages
|
||||
from ouroboros.model_slots import task_model_binding, task_processing_preference
|
||||
from ouroboros.model_wait import current_model_wait
|
||||
from ouroboros.observability import new_execution_id
|
||||
|
||||
owner_ctx = getattr(getattr(ctx, "tools", None), "_ctx", None)
|
||||
waiter = current_model_wait()
|
||||
|
|
@ -750,9 +750,9 @@ def prepared_wrapup_candidate(
|
|||
model_account_override=account,
|
||||
model_turn_state=getattr(owner_ctx, "model_turn_state", None),
|
||||
# The admitted candidate must be the payload the send will produce: the
|
||||
# main loop declares the same execution-scoped cache affinity, so this
|
||||
# prepared copy binds the execution id exactly as that dispatch does.
|
||||
cache_affinity=str(ctx.accumulated_usage.setdefault("execution_id", new_execution_id())),
|
||||
# main loop declares the same install-scoped cache affinity, so this
|
||||
# prepared copy binds the same key as that dispatch does.
|
||||
cache_affinity="" if getattr(ctx, "active_use_local", False) else cache_key_for_model(ctx.active_model),
|
||||
processing_preference=task_processing_preference(
|
||||
{"task_metadata": getattr(owner_ctx, "task_metadata", {})}, model_role=role),
|
||||
)
|
||||
|
|
|
|||
|
|
@ -2,7 +2,8 @@
|
|||
|
||||
import queue
|
||||
|
||||
from ouroboros.llm_claudexor import _request
|
||||
from ouroboros import config
|
||||
from ouroboros.llm_claudexor import _request, cache_key_for_model
|
||||
from ouroboros.loop_llm_call import call_llm_with_retry
|
||||
|
||||
|
||||
|
|
@ -16,7 +17,19 @@ def test_claudexor_request_projects_explicit_affinity_to_cache_key():
|
|||
assert payload["options"]["cacheKey"] == "execution-7"
|
||||
|
||||
|
||||
def test_main_loop_scopes_execution_affinity_to_claudexor(tmp_path):
|
||||
def test_cache_key_is_install_scoped_stable_and_header_safe(monkeypatch, tmp_path):
|
||||
key = cache_key_for_model("claudexor::codex=gpt-6-astra")
|
||||
assert key == cache_key_for_model("claudexor::codex=gpt-6-astra"), "same install + model -> same key"
|
||||
assert key.startswith("ouroboros-gpt-6-astra-")
|
||||
assert all(ch.isalnum() or ch in "._-" for ch in key), "must be a valid HTTP header value"
|
||||
assert key != cache_key_for_model("claudexor::codex=gpt-5.6-sol"), "one shard per model"
|
||||
assert cache_key_for_model("openrouter::openai/model") == ""
|
||||
assert cache_key_for_model("openai/gpt-5.6-sol") == ""
|
||||
monkeypatch.setattr(config, "DATA_DIR", tmp_path / "other-install")
|
||||
assert cache_key_for_model("claudexor::codex=gpt-6-astra") != key, "another data root is another install"
|
||||
|
||||
|
||||
def test_main_loop_shares_one_install_affinity_across_executions_on_claudexor(tmp_path):
|
||||
captured = []
|
||||
|
||||
class LLM:
|
||||
|
|
@ -29,16 +42,21 @@ def test_main_loop_scopes_execution_affinity_to_claudexor(tmp_path):
|
|||
|
||||
logs = tmp_path / "logs"
|
||||
logs.mkdir()
|
||||
for model in ("claudexor::codex=model", "openrouter::openai/model"):
|
||||
usage = {"execution_id": "execution-7"}
|
||||
for model, execution in (("claudexor::codex=model", "execution-7"), ("claudexor::codex=model", "execution-8"),
|
||||
("openrouter::openai/model", "execution-9")):
|
||||
usage = {"execution_id": execution}
|
||||
message, _cost = call_llm_with_retry(
|
||||
LLM(), [{"role": "user", "content": "work"}], model, None, "medium", 1,
|
||||
logs, "task", 1, queue.Queue(), usage,
|
||||
)
|
||||
assert message["content"] == "done"
|
||||
|
||||
assert captured[0]["cache_affinity"] == "execution-7"
|
||||
assert captured[1]["cache_affinity"] == ""
|
||||
# Two executions of one install share the key, so the second one's first
|
||||
# round is served the governance prefix the first one already paid for.
|
||||
assert captured[0]["cache_affinity"] == cache_key_for_model("claudexor::codex=model")
|
||||
assert captured[1]["cache_affinity"] == captured[0]["cache_affinity"]
|
||||
assert "execution-7" not in captured[0]["cache_affinity"]
|
||||
assert captured[2]["cache_affinity"] == ""
|
||||
|
||||
|
||||
def test_main_loop_projects_claudexor_options_outside_the_route(tmp_path):
|
||||
|
|
|
|||
|
|
@ -11,6 +11,7 @@ import pytest
|
|||
|
||||
from ouroboros import loop, model_wait, usage_accounting as ua
|
||||
from ouroboros.llm_attempt import _attempt_request, _candidate_before_dispatch
|
||||
from ouroboros.llm_claudexor import cache_key_for_model
|
||||
from ouroboros.loop_model_call import _reprepare_waiting_main
|
||||
from ouroboros.model_slots import MODEL_ACCOUNTS_KEY
|
||||
from tests.test_context_fit_integration import _plan
|
||||
|
|
@ -539,7 +540,7 @@ def test_live_owner_wait_reprojects_affinity(main_call, monkeypatch, destination
|
|||
}
|
||||
assert facts["completed_tool_texts"] == ["verified read A", "completed review B"]
|
||||
assert ctx.accumulated_usage["execution_id"] == CACHE_REPREPARE_EXECUTION
|
||||
assert prepared[-1]["cache_affinity"] == ("" if use_local or destination != MODEL else CACHE_REPREPARE_EXECUTION), (
|
||||
assert prepared[-1]["cache_affinity"] == ("" if use_local or destination != MODEL else cache_key_for_model(MODEL)), (
|
||||
facts
|
||||
)
|
||||
if not use_local:
|
||||
|
|
@ -581,4 +582,6 @@ def test_recorded_wait_override_reprojects_affinity_before_send(main_call, initi
|
|||
}
|
||||
assert facts["completed_tool_texts"] == ["verified read A", "completed review B"]
|
||||
assert ctx.accumulated_usage["execution_id"] == CACHE_REPREPARE_EXECUTION
|
||||
assert payload["options"].get("cacheKey") == CACHE_REPREPARE_EXECUTION, facts
|
||||
# The override re-prepared the send for the subscription route, so the wire
|
||||
# carries the install-scoped Codex key that route shares across executions.
|
||||
assert payload["options"].get("cacheKey") == cache_key_for_model(MODEL), facts
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue