diff --git a/ouroboros/post_task_synthesis.py b/ouroboros/post_task_synthesis.py index 34a8f3f08..ccd9427f2 100644 --- a/ouroboros/post_task_synthesis.py +++ b/ouroboros/post_task_synthesis.py @@ -332,6 +332,39 @@ def _child_failure_classes(rows: Any) -> list: return sorted(classes) +def _child_engine_facts(item: Dict[str, Any]) -> Dict[str, Any]: + """WHO ran a child and for how long, from the child's OWN stored record. + + Facts only: the frozen ``configured_subagent`` snapshot names the engine + (never the live roster, which would relabel the past), under the same handle + the catalog shows. ``used_model`` is reported for an API child alone — a + session child's ``model_execution`` describes its nanny's rounds, not the + leaf. A duration needs both stamps: only the ordinary terminal write stamps + ``ts``, so a row whose ``ts`` does not follow its start yields none. + """ + from ouroboros.deadline_utils import parse_deadline_ts + from ouroboros.subagent_history import execution_identity, snapshot_handle + + facts: Dict[str, Any] = {} + snapshot = item.get("configured_subagent") + if isinstance(snapshot, dict) and isinstance(snapshot.get("route"), dict): + identity = execution_identity(snapshot) + facts["engine"] = { + "subagent_id": snapshot_handle(snapshot), "kind": identity["kind"], + "target": identity["target_id"], + **{key: identity[key] for key in ("effort", "access") if identity.get(key)}, + } + execution = item.get("model_execution") + if identity["kind"] == "api_model" and isinstance(execution, dict) and execution.get("used_model"): + facts["used_model"] = execution["used_model"] + started, finished = parse_deadline_ts(item.get("started_at")), parse_deadline_ts(item.get("ts")) + if started is not None: + facts["started_at"] = item["started_at"] + if finished is not None and finished > started: + facts["duration_sec"] = round((finished - started).total_seconds(), 1) + return facts + + def _child_task_evidence(env: Any, task: Dict[str, Any], limit: int = 6000) -> tuple: """Compact evidence from child/subagent results for parent experience review. @@ -360,6 +393,7 @@ def _child_task_evidence(env: Any, task: Dict[str, Any], limit: int = 6000) -> t "task_id": item.get("task_id") or item.get("id"), "status": item.get("status"), "role": item.get("role"), + **_child_engine_facts(item), "outcome_axes": normalize_outcome_axes(item), "accounted_upper_bound_usd": child_cost, "trace_summary": _truncate_with_notice(item.get("trace_summary", ""), 800), diff --git a/tests/test_process_memory.py b/tests/test_process_memory.py index 01a1cec96..1360e48f5 100644 --- a/tests/test_process_memory.py +++ b/tests/test_process_memory.py @@ -909,3 +909,59 @@ def test_a_failed_child_alone_triggers_the_roots_reflection(tmp_path, monkeypatc assert entry is not None and entry["child_failure_classes"] == [] assert "completed without hard errors" in prompts[0] assert "pattern_register_update" not in calls + + +def test_reflection_evidence_names_who_ran_each_child_as_facts(tmp_path): + """Children do not reflect, so the root's reflection is the only place an + engine's conduct can be learned from - and its evidence rows used to say + nothing about WHO ran a child. Facts only, from each child's OWN stored + record: the frozen snapshot names the engine (never the live roster), the + served model is stated for an API child alone, and a duration needs a + terminal stamp that follows the start.""" + import json + from types import SimpleNamespace + + from ouroboros import post_task_synthesis + from ouroboros.subagent_runtime import select_subagent_snapshot + from ouroboros.task_results import write_task_result + + settings = {"OUROBOROS_SUBAGENTS": json.dumps({"enabled": True, "items": [ + {"subagent_id": "primary-builder", "recommended_use": "Builds.", "effort": "xhigh", + "route": {"kind": "agent_session", "target_id": "codex=gpt-6-astra"}}, + {"subagent_id": "fast-scout", "recommended_use": "Scouts.", "effort": "low", + "route": {"kind": "api_model", "target_id": "x-ai/grok-4.6"}}, + ]})} + session, _ = select_subagent_snapshot(settings, subagent_id="primary-builder") + api, _ = select_subagent_snapshot(settings, subagent_id="fast-scout") + common = {"parent_task_id": "root", "root_task_id": "root", "delegation_role": "subagent"} + write_task_result(tmp_path, "root", "completed", result="done", root_task_id="root", + parent_task_id="", delegation_role="root") + write_task_result(tmp_path, "kid-session", "completed", result="built", configured_subagent=session, + started_at="2026-09-20T10:00:00+00:00", ts="2026-09-20T10:04:30+00:00", + model_execution={"used_model": "nanny/host-model", "source": "usable_solve_response"}, + **common) + write_task_result(tmp_path, "kid-api", "completed", result="scouted", configured_subagent=api, + started_at="2026-09-20T10:00:00+00:00", ts="2026-09-20T09:59:00+00:00", + model_execution={"used_model": "x-ai/grok-4.6", "source": "usable_solve_response"}, + **common) + write_task_result(tmp_path, "kid-plain", "completed", result="no snapshot", **common) + + text, children = post_task_synthesis._child_task_evidence( + SimpleNamespace(drive_root=tmp_path), {"id": "root"}) + by_id = {row["task_id"]: row for row in children} + + assert by_id["kid-session"]["engine"] == { + "subagent_id": "codex=gpt-6-astra/xhigh", "kind": "agent_session", + "target": "codex=gpt-6-astra", "effort": "xhigh", "access": "full", + } + assert by_id["kid-session"]["duration_sec"] == 270.0 + assert "used_model" not in by_id["kid-session"], "a nanny's rounds are not the leaf's served model" + assert by_id["kid-api"]["engine"] == { + "subagent_id": "x-ai/grok-4.6/low", "kind": "api_model", "target": "x-ai/grok-4.6", "effort": "low", + } + assert by_id["kid-api"]["used_model"] == "x-ai/grok-4.6" + assert "duration_sec" not in by_id["kid-api"], "a stamp that does not follow the start yields no duration" + assert by_id["kid-api"]["started_at"] == "2026-09-20T10:00:00+00:00" + for absent in ("engine", "used_model", "started_at", "duration_sec"): + assert absent not in by_id["kid-plain"], "a record without the facts states none" + assert "primary-builder" not in text and "fast-scout" not in text, "stored keys are not evidence"