From b14ba39775a97e68cbed7ef311eae7046f7c309a Mon Sep 17 00:00:00 2001 From: Ouroboros Date: Thu, 20 Aug 2026 03:24:22 +0300 Subject: [PATCH] feat: expose available subagents in runtime context Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> --- ouroboros/context.py | 46 +++-- ouroboros/subagent_runtime.py | 67 +++++++ tests/test_available_subagents_catalog.py | 214 ++++++++++++++++++++++ tests/test_context.py | 32 ++-- 4 files changed, 328 insertions(+), 31 deletions(-) create mode 100644 tests/test_available_subagents_catalog.py diff --git a/ouroboros/context.py b/ouroboros/context.py index 09a081f91..e1af39fe7 100644 --- a/ouroboros/context.py +++ b/ouroboros/context.py @@ -477,9 +477,7 @@ def _promoted_task_toolset(env: Any) -> Dict[str, Any]: def _delegation_capability_fact() -> Optional[Dict[str, Any]]: - """B4-lite: the CONFIGURED delegation route plus honestly-labeled HISTORICAL - observations (last recorded execution per reviewer slot and the last - delegated run). + """B4-lite: honestly-labeled HISTORICAL delegation observations. Deliberately NOT live health — receipts prove what the last execution did, not what a lane can do now; live lane facts arrive from plan-review wave @@ -491,24 +489,14 @@ def _delegation_capability_fact() -> Optional[Dict[str, Any]]: """ try: from ouroboros.reviewer_slot_config import reviewer_slot_last_executions - from ouroboros.subagents import get_subagent_harness, subagent_last_delegation + from ouroboros.subagents import subagent_last_delegation def _observed_label(ts: Any) -> str: # Timestamp only: the verbatim "historical, not live health" disclaimer # lives ONCE in the note below, never repeated per row. return f"last observed at {str(ts or '').strip() or 'unknown time'}" - route = get_subagent_harness() delegation: Dict[str, Any] = { - "configured_route": ( - { - "harness": route.route_id, - "model": route.model, - "effort": route.effort, - } - if route is not None - else "not configured" - ), "note": ( "Every row here is historical, not live health (the last " "recorded execution per reviewer slot / delegated run): " @@ -528,6 +516,12 @@ def _delegation_capability_fact() -> Optional[Dict[str, Any]]: else "unknown"), "observed": _observed_label(row.get("ts")), } + requested = row.get("requested") if isinstance(row.get("requested"), dict) else {} + effective = row.get("effective") if isinstance(row.get("effective"), dict) else {} + if requested.get("profile_id"): + fact["requested_profile"] = str(requested["profile_id"]) + if effective.get("profile_id"): + fact["applied_profile"] = str(effective["profile_id"]) # B1's typed failure facts, forwarded only when recorded (a dated # window carries reset_at without a code and an undated one the # code without a reset — read both independently). @@ -539,12 +533,19 @@ def _delegation_capability_fact() -> Optional[Dict[str, Any]]: delegation["reviewer_slots_last"] = slot_rows last = subagent_last_delegation() if isinstance(last, dict) and last: - delegation["subagent_last_delegation"] = { + last_fact = { "route": str(last.get("route") or ""), "requested_model": str(last.get("requested_model") or ""), "applied_model": str(last.get("applied_model") or ""), "observed": _observed_label(last.get("ts")), } + if last.get("requested_profile"): + last_fact["requested_profile"] = str(last["requested_profile"]) + if last.get("applied_profile"): + last_fact["applied_profile"] = str(last["applied_profile"]) + delegation["subagent_last_delegation"] = last_fact + if len(delegation) == 1: + return None return delegation except Exception: log.debug("Failed to build delegation capability fact", exc_info=True) @@ -650,8 +651,8 @@ def build_runtime_section(env: Any, task: Dict[str, Any], *, ctx: Any = None) -> "spawn acting subagents." ), } - # B4-lite: configured route + honestly-labeled HISTORY, never live - # health. Own nested fail-soft inside the helper: a failure there must + # B4-lite: honestly-labeled HISTORY, never live health. Own nested + # fail-soft inside the helper: a failure there must # never drop the whole capabilities digest above. _delegation_fact = _delegation_capability_fact() if _delegation_fact is not None: @@ -1356,6 +1357,17 @@ def _capture_context_core( docs_need_development = _task_requires_development_context(task) semi_stable_parts = [] + try: + from ouroboros.subagent_runtime import current_model_visible_subagent_catalog + + subagent_catalog = current_model_visible_subagent_catalog() + if subagent_catalog: + semi_stable_parts.append( + "## Available subagents\n\n" + + json.dumps(subagent_catalog, ensure_ascii=False, indent=1) + ) + except Exception: + log.debug("Failed to build Available subagents catalog", exc_info=True) semi_stable_parts.extend(build_memory_sections(memory, partition="stable")) semi_stable_parts.extend(build_knowledge_sections(env, project_id=resolve_project_id(task))) diff --git a/ouroboros/subagent_runtime.py b/ouroboros/subagent_runtime.py index c9350122b..510744bd6 100644 --- a/ouroboros/subagent_runtime.py +++ b/ouroboros/subagent_runtime.py @@ -72,6 +72,71 @@ def effective_runtime_subagent_settings(settings: Mapping[str, Any]) -> dict[str return effective +def model_visible_subagent_catalog(settings: Mapping[str, Any]) -> dict[str, Any]: + """Project saved, dispatchable rows without probing or ranking them.""" + + resolution = resolve_configured_subagents(settings) + config = resolution.config + if ( + resolution.source in {SOURCE_INVALID, SOURCE_UNDECIDED} + or config is None + or not config.enabled + or not config.items + ): + return {} + + rows: list[dict[str, Any]] = [] + for row in config.items: + session = row.route.is_session + projected: dict[str, Any] = { + "subagent_id": row.subagent_id, + "name": row.name, + "recommended_use": row.recommended_use, + "route_class": "Agent session" if session else "API model", + "requested_effort": row.effort or "(not explicitly set)", + } + if session: + projected["requested_target"] = row.route.target_id + if row.route.credential_profile_id: + projected["account_policy"] = "explicit profile pin" + projected["credential_profile_id"] = row.route.credential_profile_id + else: + projected["account_policy"] = "automatic compatible account selection" + else: + projected["requested_model"] = row.route.target_id + projected["account_policy"] = ( + "API provider credentials (session account selection does not apply)" + ) + rows.append(projected) + + return { + "source": resolution.source, + "config_fingerprint": configured_subagents_fingerprint(config), + "rows": rows, + "selection_guidance": ( + "Choose subagent_id from the owner descriptions and saved route intent. " + "Prefer suitable Agent session choices often when they fit, to reduce " + "incremental API spend; use API model choices when their described strengths " + "fit. The host does not rank or substitute rows." + ), + "dispatch_contract": ( + "schedule_subagent attempts the exact selected row. If live dispatch finds it " + "unavailable, it returns a typed refusal; choose the next action or another " + "subagent_id." + ), + } + + +def current_model_visible_subagent_catalog() -> dict[str, Any]: + """Read the current normalized settings and return the stable catalog.""" + + from ouroboros.config import load_settings + + return model_visible_subagent_catalog( + effective_runtime_subagent_settings(load_settings()) + ) + + def apply_task_start_settings() -> None: """Project the provider-normalized in-memory snapshot for one task start.""" @@ -557,11 +622,13 @@ __all__ = [ "SubagentSelectionError", "apply_task_start_settings", "current_exact_start_selection", + "current_model_visible_subagent_catalog", "current_subagent_alternatives", "delegate_start_entry", "effective_runtime_subagent_settings", "exact_session_binding", "exact_start", + "model_visible_subagent_catalog", "prepare_delegate_start_actor", "resolve_configured_actor_dispatch", "select_subagent_snapshot", diff --git a/tests/test_available_subagents_catalog.py b/tests/test_available_subagents_catalog.py new file mode 100644 index 000000000..ecfc07dc3 --- /dev/null +++ b/tests/test_available_subagents_catalog.py @@ -0,0 +1,214 @@ +"""Focused prompt-catalog projection tests for Available subagents.""" + +from __future__ import annotations + +import json + +import pytest + + +def _row( + row_id: str, + *, + kind: str, + target: str, + recommendation: str, + effort: str = "", + profile: str = "", +) -> dict: + route = {"kind": kind, "target_id": target} + if kind == "agent_session": + route["credential_profile_id"] = profile + return { + "subagent_id": row_id, + "name": row_id.replace("-", " ").title(), + "recommended_use": recommendation, + "route": route, + "effort": effort, + } + + +def _settings(*rows: dict, enabled: bool = True) -> dict: + return { + "OUROBOROS_SUBAGENTS": json.dumps({"enabled": enabled, "items": list(rows)}), + } + + +def test_catalog_projects_every_saved_row_in_owner_order_verbatim(): + from ouroboros.configured_subagents import ( + configured_subagents_fingerprint, + parse_configured_subagents, + ) + from ouroboros.subagent_runtime import model_visible_subagent_catalog + + verbatim = "Use exact owner wording.\nKeep punctuation: a/b, quotes, and cost $0." + settings = _settings( + _row( + "api-scout", + kind="api_model", + target="google/gemini-3.7-flash", + recommendation=verbatim, + effort="low", + ), + _row( + "auto-session", + kind="agent_session", + target="claude=claude-fable-5", + recommendation="Use the automatic account pool.", + effort="high", + ), + _row( + "pinned-session", + kind="agent_session", + target="cursor=cursor-grok-4.6-high", + recommendation="Use this pinned account.", + profile="cursor-owner", + ), + ) + + catalog = model_visible_subagent_catalog(settings) + + config = parse_configured_subagents(settings["OUROBOROS_SUBAGENTS"]) + assert catalog["source"] == "configured" + assert catalog["config_fingerprint"] == configured_subagents_fingerprint(config) + assert [row["subagent_id"] for row in catalog["rows"]] == [ + "api-scout", "auto-session", "pinned-session", + ] + assert catalog["rows"][0]["recommended_use"] == verbatim + assert catalog["rows"][0]["route_class"] == "API model" + assert catalog["rows"][0]["requested_model"] == "google/gemini-3.7-flash" + assert catalog["rows"][0]["requested_effort"] == "low" + assert "session account selection does not apply" in catalog["rows"][0]["account_policy"] + assert catalog["rows"][1]["route_class"] == "Agent session" + assert catalog["rows"][1]["requested_target"] == "claude=claude-fable-5" + assert catalog["rows"][1]["account_policy"] == "automatic compatible account selection" + assert "credential_profile_id" not in catalog["rows"][1] + assert catalog["rows"][2]["requested_effort"] == "(not explicitly set)" + assert catalog["rows"][2]["account_policy"] == "explicit profile pin" + assert catalog["rows"][2]["credential_profile_id"] == "cursor-owner" + assert "does not rank or substitute" in catalog["selection_guidance"] + assert "typed refusal" in catalog["dispatch_contract"] + + +@pytest.mark.parametrize( + "settings", + [ + {"OUROBOROS_SUBAGENTS": "not-json"}, + { + **_settings(enabled=False), + "OUROBOROS_SUBAGENT_HARNESS": "codex=gpt-5.6-sol:high", + }, + _settings(), + ], +) +def test_catalog_omits_unsaved_invalid_disabled_and_empty(settings): + from ouroboros.subagent_runtime import model_visible_subagent_catalog + + assert model_visible_subagent_catalog(settings) == {} + + +def test_catalog_omits_a_real_undecided_candidate_rejected_by_new_id_dispatch(): + from ouroboros.configured_subagents import SOURCE_UNDECIDED, resolve_configured_subagents + from ouroboros.subagent_runtime import ( + SubagentSelectionError, + model_visible_subagent_catalog, + select_subagent_snapshot, + ) + + settings = {"OUROBOROS_MODEL_HEAVY": "owner/unsaved-candidate"} + resolution = resolve_configured_subagents(settings) + assert resolution.source == SOURCE_UNDECIDED + assert resolution.config is not None and resolution.config.items + assert model_visible_subagent_catalog(settings) == {} + with pytest.raises(SubagentSelectionError) as refused: + select_subagent_snapshot(settings, subagent_id="legacy-heavy") + assert refused.value.code == "subagent_configuration_unsaved" + + +def _context_env(tmp_path): + class FakeEnv: + def drive_path(self, path): + return tmp_path / path + + def repo_path(self, path): + return tmp_path / "repo" / path + + @property + def repo_dir(self): + return tmp_path / "repo" + + @property + def drive_root(self): + return tmp_path + + for path in ("state", "logs", "memory", "repo/docs", "repo/prompts", "repo/web"): + (tmp_path / path).mkdir(parents=True, exist_ok=True) + (tmp_path / "repo/prompts/SYSTEM.md").write_text("System", encoding="utf-8") + (tmp_path / "repo/BIBLE.md").write_text("Bible", encoding="utf-8") + (tmp_path / "repo/docs/ARCHITECTURE.md").write_text("Architecture", encoding="utf-8") + (tmp_path / "repo/docs/DEVELOPMENT.md").write_text("Development", encoding="utf-8") + (tmp_path / "state/state.json").write_text("{}", encoding="utf-8") + (tmp_path / "logs/events.jsonl").write_text("", encoding="utf-8") + return FakeEnv() + + +def test_catalog_is_semi_stable_while_dated_history_stays_dynamic(tmp_path, monkeypatch): + from ouroboros.context import _capture_context_core + from ouroboros.context_fit import _render_context_system_content + from ouroboros.memory import Memory + + env = _context_env(tmp_path) + monkeypatch.setattr("ouroboros.config.DATA_DIR", tmp_path) + owner_text = "Use this exact owner description, verbatim.\nSecond line stays intact." + saved = _settings(_row( + "builder", + kind="agent_session", + target="codex=gpt-5.6-sol", + recommendation=owner_text, + effort="high", + )) + monkeypatch.setattr("ouroboros.config.load_settings", lambda: saved) + (tmp_path / "state/reviewer_slot_last_execution.json").write_text(json.dumps({ + "triad": { + "ts": "2026-08-18T01:02:03+00:00", + "status": "ok", + "requested": {"profile_id": "review-requested"}, + "effective": {"profile_id": "review-applied"}, + }, + }), encoding="utf-8") + (tmp_path / "state/subagent_last_delegation.json").write_text(json.dumps({ + "ts": "2026-08-18T02:00:00+00:00", + "route": "codex", + "requested_model": "gpt-5.6-sol", + "applied_model": "gpt-5.6-sol", + "requested_profile": "delegate-requested", + "applied_profile": "delegate-applied", + "run_id": "run-1", + }), encoding="utf-8") + + core = _capture_context_core( + env, Memory(drive_root=tmp_path), + {"id": "task-1", "type": "task", "text": "work"}, + None, None, + ) + blocks = _render_context_system_content(env, core, mode="max") + catalog_text = core.semi_stable_text.split("## Available subagents\n\n", 1)[1] + catalog, _end = json.JSONDecoder().raw_decode(catalog_text) + + assert blocks[1]["text"] == core.semi_stable_text + assert blocks[1]["cache_control"] == {"type": "ephemeral"} + assert blocks[2]["text"] == core.dynamic_text + assert "cache_control" not in blocks[2] + assert catalog["rows"][0]["recommended_use"] == owner_text + assert '"subagent_id": "builder"' in core.semi_stable_text + assert "2026-08-18T01:02:03+00:00" not in core.semi_stable_text + assert "reviewer_slots_last" not in core.semi_stable_text + assert "subagent_last_delegation" not in core.semi_stable_text + assert "2026-08-18T01:02:03+00:00" in core.dynamic_text + assert "reviewer_slots_last" in core.dynamic_text + assert "subagent_last_delegation" in core.dynamic_text + for profile in ( + "review-requested", "review-applied", "delegate-requested", "delegate-applied", + ): + assert profile in core.dynamic_text + assert "configured_route" not in core.dynamic_text diff --git a/tests/test_context.py b/tests/test_context.py index 3b5aa1d0e..1cc2035e6 100644 --- a/tests/test_context.py +++ b/tests/test_context.py @@ -1698,8 +1698,8 @@ def test_settled_continuation_with_open_obligations_survives_age_retirement(tmp_ # --------------------------------------------------------------------------- -# B4-lite: capabilities["delegation"] — configured route + honestly-labeled -# HISTORICAL observations (never live health). +# B4-lite: capabilities["delegation"] — honestly-labeled HISTORICAL +# observations only (never saved intent or live health). # --------------------------------------------------------------------------- @@ -1718,7 +1718,7 @@ def _delegation_fact(tmp_path, monkeypatch): return payload["capabilities"] -def test_delegation_fact_carries_configured_route_and_historical_rows(tmp_path, monkeypatch): +def test_delegation_fact_carries_historical_rows_and_profile_evidence(tmp_path, monkeypatch): root = _delegation_data_root(tmp_path, monkeypatch) monkeypatch.setenv("OUROBOROS_SUBAGENT_HARNESS", "claudexor=opus-5:high") (root / "state" / "reviewer_slot_last_execution.json").write_text(json.dumps({ @@ -1726,7 +1726,12 @@ def test_delegation_fact_carries_configured_route_and_historical_rows(tmp_path, "ts": "2026-08-18T01:02:03+00:00", "surface": "triad", "status": "ok", - "effective": {"route": "agent_session:claudexor", "model": "opus-5"}, + "requested": {"profile_id": "requested-review-profile"}, + "effective": { + "route": "agent_session:claudexor", + "model": "opus-5", + "profile_id": "applied-review-profile", + }, }, "triad_2": { "ts": "2026-08-18T01:02:04+00:00", @@ -1743,17 +1748,19 @@ def test_delegation_fact_carries_configured_route_and_historical_rows(tmp_path, "route": "claudexor", "requested_model": "opus-5", "applied_model": "claude-opus-5", + "requested_profile": "requested-delegate-profile", + "applied_profile": "applied-delegate-profile", "run_id": "run-1", }), encoding="utf-8") capabilities = _delegation_fact(tmp_path, monkeypatch) delegation = capabilities["delegation"] - assert delegation["configured_route"] == { - "harness": "claudexor", "model": "opus-5", "effort": "high", - } + assert "configured_route" not in delegation rows = {row["slot"]: row for row in delegation["reviewer_slots_last"]} assert rows["triad_1"]["outcome"] == "ok" + assert rows["triad_1"]["requested_profile"] == "requested-review-profile" + assert rows["triad_1"]["applied_profile"] == "applied-review-profile" assert "failure_code" not in rows["triad_1"] assert rows["triad_2"]["outcome"] == "failed" assert rows["triad_2"]["failure_code"] == "subscription_window_exhausted" @@ -1764,6 +1771,8 @@ def test_delegation_fact_carries_configured_route_and_historical_rows(tmp_path, last = delegation["subagent_last_delegation"] assert last["route"] == "claudexor" assert last["applied_model"] == "claude-opus-5" + assert last["requested_profile"] == "requested-delegate-profile" + assert last["applied_profile"] == "applied-delegate-profile" assert last["observed"] == "last observed at 2026-08-18T02:00:00+00:00" assert "historical" not in rows["triad_1"]["observed"] # The prompt-visible note teaches the semantics ONCE: rows are history, live @@ -1797,14 +1806,9 @@ def test_delegation_fact_absent_files_mean_absent_observations_not_health(tmp_pa _delegation_data_root(tmp_path, monkeypatch) monkeypatch.delenv("OUROBOROS_SUBAGENT_HARNESS", raising=False) - delegation = _delegation_fact(tmp_path, monkeypatch)["delegation"] + capabilities = _delegation_fact(tmp_path, monkeypatch) - assert delegation["configured_route"] == "not configured" - assert "reviewer_slots_last" not in delegation - assert "subagent_last_delegation" not in delegation - # Nothing in the fact may read as a live-health claim. - assert "healthy" not in json.dumps( - {k: v for k, v in delegation.items() if k != "note"}) + assert "delegation" not in capabilities def test_delegation_fact_failure_never_drops_capability_digest(tmp_path, monkeypatch):