diff --git a/docs/architecture/06-agent-core.md b/docs/architecture/06-agent-core.md index ba4c2eeb8..94a4987b9 100644 --- a/docs/architecture/06-agent-core.md +++ b/docs/architecture/06-agent-core.md @@ -319,6 +319,12 @@ engine, not an Agent Run. `LLMClient` dispatches sync and async calls through `llm_claudexor.py` before OpenAI-compatible filtering/retries. Ouroboros retains its SYSTEM, BIBLE, canonical messages, tool selection and execution. The raw-model adapter is Codex; connected Claude/Cursor and other harnesses keep their existing Agent capabilities. Direct API keys keep their existing routes. +Every main-loop execution of one install sends the same Codex cache key +(`llm_claudexor.cache_key_for_model`: per data root and model, projected to +`prompt_cache_key` and the `session_id` header), so a new task, child or +consciousness cycle is served the governance prefix its predecessors already +cached on its first round; per-execution turn states ride `nativeContinuation` +unchanged under that shared session. Before creating a model operation, the host discovers the existing operation catalog's `captureFailureEvidence=true` query and freezes that choice for the diff --git a/docs/development/06-rules-by-change-class.md b/docs/development/06-rules-by-change-class.md index 5f7139877..e8dc31dfe 100644 --- a/docs/development/06-rules-by-change-class.md +++ b/docs/development/06-rules-by-change-class.md @@ -1042,7 +1042,13 @@ by "Provider Independence" above. Call-site imperatives: direct-Anthropic lane by `_anthropic_blocks_from_content` and on OpenRouter by `supports_message_cache_control`, and pinned by `tests/test_review_prompt_caching.py`. The main loop declares an - execution-scoped cache affinity only for subscription transport; API-compatible + install-scoped cache affinity for the subscription transport + (`llm_claudexor.cache_key_for_model`: one Codex `prompt_cache_key`, and so one + `session_id`, per data root and model, shared by every task, child and + consciousness cycle), because the Codex backend reuses a cached prefix across + conversations only under the same session and per-conversation turn states stay + valid under a shared one (measured 2026-09-17); a per-execution key paid the + shared governance prefix cold on every task start. API-compatible lanes retain their prefix-derived session identity. A consciousness wake-up is one of those main-loop executions: its schema array and cached system prefix are byte-identical to an owner turn's, so everything the level or the wake reason diff --git a/ouroboros/llm_claudexor.py b/ouroboros/llm_claudexor.py index 21e8cb0d2..3d9891407 100644 --- a/ouroboros/llm_claudexor.py +++ b/ouroboros/llm_claudexor.py @@ -42,7 +42,9 @@ import asyncio import copy import contextvars from dataclasses import replace +import hashlib import json +import re import logging import threading import time @@ -301,6 +303,27 @@ def _remember_failed_profile(target: dict, parameters: dict, error: ClaudexorMod _FAILED_PROFILE.set((*key, route["credentialProfileId"])) +def cache_key_for_model(model: str) -> str: + """The Codex prompt-cache key every main-loop execution of this install shares. + + The Codex backend reuses a cached prefix across conversations only when both + ``prompt_cache_key`` and the ``session_id`` header match (the adapter sets + both from ``cacheKey``), and per-conversation turn states stay valid under a + shared session (measured 2026-09-17). One key per data root and model + therefore lets a new task, child or consciousness cycle be served the + governance prefix it shares with its predecessors on its very first round, + instead of paying it cold under a per-execution key. Empty for every other + provider: API-compatible lanes keep their prefix-derived session identity. + """ + from ouroboros.provider_models import provider_for_model + + if provider_for_model(model) != "claudexor": + return "" + label = re.sub(r"[^A-Za-z0-9._-]+", "-", str(model).rsplit("=", 1)[-1]).strip("-")[:40] or "model" + digest = hashlib.sha256(f"{config.DATA_DIR}\0{model}".encode("utf-8")).hexdigest()[:16] + return f"ouroboros-{label}-{digest}" + + def _request(target: dict, messages: list, tools: list | None, parameters: dict) -> dict: from ouroboros.llm_messages import _MessageShapingMixin diff --git a/ouroboros/loop_llm_call.py b/ouroboros/loop_llm_call.py index 682cab52d..79513e5e6 100644 --- a/ouroboros/loop_llm_call.py +++ b/ouroboros/loop_llm_call.py @@ -29,7 +29,7 @@ from ouroboros.deadline_utils import ( main_transport_timeout_sec as _main_transport_timeout, ) from ouroboros.llm import LLMClient, LocalContextTooLargeError, add_usage -from ouroboros.llm_claudexor import propagate_model_error +from ouroboros.llm_claudexor import cache_key_for_model, propagate_model_error from ouroboros.model_wait import propagate_model_control from ouroboros.openai_chat_dispatch import CUSTOM_RECEIPTS_USAGE_KEY, pop_custom_validation_receipts from ouroboros.llm_attempt import PROVIDER_POLICY_REFUSAL, _is_provider_policy_refusal # typed-refusal contract owner @@ -1382,7 +1382,7 @@ def call_llm_with_retry( "max_tokens": MAIN_LOOP_MAX_TOKENS, "stream": True, "caller_deadline_ts": (None if deadline_ts is None else float(deadline_ts) - float(transport_reserve_sec or 0.0)), - "use_local": use_local, "cache_affinity": execution_id if provider_for_model(model) == "claudexor" else "", + "use_local": use_local, "cache_affinity": cache_key_for_model(model), "allow_server_web_search": bool(allow_server_web_search) and provider_for_model(model) != "claudexor", "bypass_response_cache": response_cache_bypass_requested and provider_for_model(model) != "claudexor", "timeout": _main_transport_timeout(model, deadline_ts, reserve_sec=transport_reserve_sec), diff --git a/ouroboros/loop_model_call.py b/ouroboros/loop_model_call.py index f52b6355f..e3663ace9 100644 --- a/ouroboros/loop_model_call.py +++ b/ouroboros/loop_model_call.py @@ -555,11 +555,9 @@ def _reprepare_waiting_main(ctx: _RoundModelCallContext, kwargs: dict): _loop()._run_main_reclaim(ctx, disposition) disposition = _measure_after_reclaim(ctx) kwargs["messages"] = ctx.messages + from ouroboros.llm_claudexor import cache_key_for_model from ouroboros.provider_models import provider_for_model - kwargs["cache_affinity"] = ( - str(ctx.accumulated_usage.get("execution_id") or "") - if not use_local and provider_for_model(model) == "claudexor" else "" - ) + kwargs["cache_affinity"] = "" if use_local else cache_key_for_model(model) kwargs["allow_server_web_search"] = (_loop()._server_web_allowed_by_task(ctx.tools._ctx) and not use_local and provider_for_model(model) != "claudexor") if provider_for_model(model) == "claudexor": diff --git a/ouroboros/task_pacing.py b/ouroboros/task_pacing.py index 7dfeb1ae4..baedc6673 100644 --- a/ouroboros/task_pacing.py +++ b/ouroboros/task_pacing.py @@ -718,10 +718,10 @@ def prepared_wrapup_candidate( The forced send this candidate admits continues the loop's active transport turn, so the candidate is built from that same owner slot.""" + from ouroboros.llm_claudexor import cache_key_for_model from ouroboros.loop_llm_call import _prepare_main_messages from ouroboros.model_slots import task_model_binding, task_processing_preference from ouroboros.model_wait import current_model_wait - from ouroboros.observability import new_execution_id owner_ctx = getattr(getattr(ctx, "tools", None), "_ctx", None) waiter = current_model_wait() @@ -750,9 +750,9 @@ def prepared_wrapup_candidate( model_account_override=account, model_turn_state=getattr(owner_ctx, "model_turn_state", None), # The admitted candidate must be the payload the send will produce: the - # main loop declares the same execution-scoped cache affinity, so this - # prepared copy binds the execution id exactly as that dispatch does. - cache_affinity=str(ctx.accumulated_usage.setdefault("execution_id", new_execution_id())), + # main loop declares the same install-scoped cache affinity, so this + # prepared copy binds the same key as that dispatch does. + cache_affinity="" if getattr(ctx, "active_use_local", False) else cache_key_for_model(ctx.active_model), processing_preference=task_processing_preference( {"task_metadata": getattr(owner_ctx, "task_metadata", {})}, model_role=role), ) diff --git a/tests/test_review_prompt_caching_p3.py b/tests/test_review_prompt_caching_p3.py index 91f23302d..927b7ee5e 100644 --- a/tests/test_review_prompt_caching_p3.py +++ b/tests/test_review_prompt_caching_p3.py @@ -2,7 +2,8 @@ import queue -from ouroboros.llm_claudexor import _request +from ouroboros import config +from ouroboros.llm_claudexor import _request, cache_key_for_model from ouroboros.loop_llm_call import call_llm_with_retry @@ -16,7 +17,19 @@ def test_claudexor_request_projects_explicit_affinity_to_cache_key(): assert payload["options"]["cacheKey"] == "execution-7" -def test_main_loop_scopes_execution_affinity_to_claudexor(tmp_path): +def test_cache_key_is_install_scoped_stable_and_header_safe(monkeypatch, tmp_path): + key = cache_key_for_model("claudexor::codex=gpt-6-astra") + assert key == cache_key_for_model("claudexor::codex=gpt-6-astra"), "same install + model -> same key" + assert key.startswith("ouroboros-gpt-6-astra-") + assert all(ch.isalnum() or ch in "._-" for ch in key), "must be a valid HTTP header value" + assert key != cache_key_for_model("claudexor::codex=gpt-5.6-sol"), "one shard per model" + assert cache_key_for_model("openrouter::openai/model") == "" + assert cache_key_for_model("openai/gpt-5.6-sol") == "" + monkeypatch.setattr(config, "DATA_DIR", tmp_path / "other-install") + assert cache_key_for_model("claudexor::codex=gpt-6-astra") != key, "another data root is another install" + + +def test_main_loop_shares_one_install_affinity_across_executions_on_claudexor(tmp_path): captured = [] class LLM: @@ -29,16 +42,21 @@ def test_main_loop_scopes_execution_affinity_to_claudexor(tmp_path): logs = tmp_path / "logs" logs.mkdir() - for model in ("claudexor::codex=model", "openrouter::openai/model"): - usage = {"execution_id": "execution-7"} + for model, execution in (("claudexor::codex=model", "execution-7"), ("claudexor::codex=model", "execution-8"), + ("openrouter::openai/model", "execution-9")): + usage = {"execution_id": execution} message, _cost = call_llm_with_retry( LLM(), [{"role": "user", "content": "work"}], model, None, "medium", 1, logs, "task", 1, queue.Queue(), usage, ) assert message["content"] == "done" - assert captured[0]["cache_affinity"] == "execution-7" - assert captured[1]["cache_affinity"] == "" + # Two executions of one install share the key, so the second one's first + # round is served the governance prefix the first one already paid for. + assert captured[0]["cache_affinity"] == cache_key_for_model("claudexor::codex=model") + assert captured[1]["cache_affinity"] == captured[0]["cache_affinity"] + assert "execution-7" not in captured[0]["cache_affinity"] + assert captured[2]["cache_affinity"] == "" def test_main_loop_projects_claudexor_options_outside_the_route(tmp_path): diff --git a/tests/test_subscription_main_wait.py b/tests/test_subscription_main_wait.py index 3b1fe75fb..784b76775 100644 --- a/tests/test_subscription_main_wait.py +++ b/tests/test_subscription_main_wait.py @@ -11,6 +11,7 @@ import pytest from ouroboros import loop, model_wait, usage_accounting as ua from ouroboros.llm_attempt import _attempt_request, _candidate_before_dispatch +from ouroboros.llm_claudexor import cache_key_for_model from ouroboros.loop_model_call import _reprepare_waiting_main from ouroboros.model_slots import MODEL_ACCOUNTS_KEY from tests.test_context_fit_integration import _plan @@ -539,7 +540,7 @@ def test_live_owner_wait_reprojects_affinity(main_call, monkeypatch, destination } assert facts["completed_tool_texts"] == ["verified read A", "completed review B"] assert ctx.accumulated_usage["execution_id"] == CACHE_REPREPARE_EXECUTION - assert prepared[-1]["cache_affinity"] == ("" if use_local or destination != MODEL else CACHE_REPREPARE_EXECUTION), ( + assert prepared[-1]["cache_affinity"] == ("" if use_local or destination != MODEL else cache_key_for_model(MODEL)), ( facts ) if not use_local: @@ -581,4 +582,6 @@ def test_recorded_wait_override_reprojects_affinity_before_send(main_call, initi } assert facts["completed_tool_texts"] == ["verified read A", "completed review B"] assert ctx.accumulated_usage["execution_id"] == CACHE_REPREPARE_EXECUTION - assert payload["options"].get("cacheKey") == CACHE_REPREPARE_EXECUTION, facts + # The override re-prepared the send for the subscription route, so the wire + # carries the install-scoped Codex key that route shares across executions. + assert payload["options"].get("cacheKey") == cache_key_for_model(MODEL), facts