Tell post-task reflection which engine ran each child

Children do not reflect, so the root's reflection is the only place an
engine's conduct can be learned from - yet its child evidence rows carried
task id, status, role, outcome, cost, trace and result, and nothing about WHO
ran the child. Rows now add facts from the child's own stored record: the
engine (handle, kind, target, effort, access) from the frozen
`configured_subagent` snapshot, never the live roster; `used_model` for an API
child only, because a session child's `model_execution` describes its nanny's
rounds and not the leaf; `started_at`, and `duration_sec` only when the
terminal stamp really follows the start (only the ordinary terminal write
stamps `ts`). Facts only: no sentence inviting reflection was added.

Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com>
This commit is contained in:
Ouroboros 2026-09-21 01:55:58 +03:00
parent 867e6d1474
commit 455eb16d7e
2 changed files with 90 additions and 0 deletions

View file

@ -332,6 +332,39 @@ def _child_failure_classes(rows: Any) -> list:
return sorted(classes)
def _child_engine_facts(item: Dict[str, Any]) -> Dict[str, Any]:
"""WHO ran a child and for how long, from the child's OWN stored record.
Facts only: the frozen ``configured_subagent`` snapshot names the engine
(never the live roster, which would relabel the past), under the same handle
the catalog shows. ``used_model`` is reported for an API child alone — a
session child's ``model_execution`` describes its nanny's rounds, not the
leaf. A duration needs both stamps: only the ordinary terminal write stamps
``ts``, so a row whose ``ts`` does not follow its start yields none.
"""
from ouroboros.deadline_utils import parse_deadline_ts
from ouroboros.subagent_history import execution_identity, snapshot_handle
facts: Dict[str, Any] = {}
snapshot = item.get("configured_subagent")
if isinstance(snapshot, dict) and isinstance(snapshot.get("route"), dict):
identity = execution_identity(snapshot)
facts["engine"] = {
"subagent_id": snapshot_handle(snapshot), "kind": identity["kind"],
"target": identity["target_id"],
**{key: identity[key] for key in ("effort", "access") if identity.get(key)},
}
execution = item.get("model_execution")
if identity["kind"] == "api_model" and isinstance(execution, dict) and execution.get("used_model"):
facts["used_model"] = execution["used_model"]
started, finished = parse_deadline_ts(item.get("started_at")), parse_deadline_ts(item.get("ts"))
if started is not None:
facts["started_at"] = item["started_at"]
if finished is not None and finished > started:
facts["duration_sec"] = round((finished - started).total_seconds(), 1)
return facts
def _child_task_evidence(env: Any, task: Dict[str, Any], limit: int = 6000) -> tuple:
"""Compact evidence from child/subagent results for parent experience review.
@ -360,6 +393,7 @@ def _child_task_evidence(env: Any, task: Dict[str, Any], limit: int = 6000) -> t
"task_id": item.get("task_id") or item.get("id"),
"status": item.get("status"),
"role": item.get("role"),
**_child_engine_facts(item),
"outcome_axes": normalize_outcome_axes(item),
"accounted_upper_bound_usd": child_cost,
"trace_summary": _truncate_with_notice(item.get("trace_summary", ""), 800),

View file

@ -909,3 +909,59 @@ def test_a_failed_child_alone_triggers_the_roots_reflection(tmp_path, monkeypatc
assert entry is not None and entry["child_failure_classes"] == []
assert "completed without hard errors" in prompts[0]
assert "pattern_register_update" not in calls
def test_reflection_evidence_names_who_ran_each_child_as_facts(tmp_path):
"""Children do not reflect, so the root's reflection is the only place an
engine's conduct can be learned from - and its evidence rows used to say
nothing about WHO ran a child. Facts only, from each child's OWN stored
record: the frozen snapshot names the engine (never the live roster), the
served model is stated for an API child alone, and a duration needs a
terminal stamp that follows the start."""
import json
from types import SimpleNamespace
from ouroboros import post_task_synthesis
from ouroboros.subagent_runtime import select_subagent_snapshot
from ouroboros.task_results import write_task_result
settings = {"OUROBOROS_SUBAGENTS": json.dumps({"enabled": True, "items": [
{"subagent_id": "primary-builder", "recommended_use": "Builds.", "effort": "xhigh",
"route": {"kind": "agent_session", "target_id": "codex=gpt-6-astra"}},
{"subagent_id": "fast-scout", "recommended_use": "Scouts.", "effort": "low",
"route": {"kind": "api_model", "target_id": "x-ai/grok-4.6"}},
]})}
session, _ = select_subagent_snapshot(settings, subagent_id="primary-builder")
api, _ = select_subagent_snapshot(settings, subagent_id="fast-scout")
common = {"parent_task_id": "root", "root_task_id": "root", "delegation_role": "subagent"}
write_task_result(tmp_path, "root", "completed", result="done", root_task_id="root",
parent_task_id="", delegation_role="root")
write_task_result(tmp_path, "kid-session", "completed", result="built", configured_subagent=session,
started_at="2026-09-20T10:00:00+00:00", ts="2026-09-20T10:04:30+00:00",
model_execution={"used_model": "nanny/host-model", "source": "usable_solve_response"},
**common)
write_task_result(tmp_path, "kid-api", "completed", result="scouted", configured_subagent=api,
started_at="2026-09-20T10:00:00+00:00", ts="2026-09-20T09:59:00+00:00",
model_execution={"used_model": "x-ai/grok-4.6", "source": "usable_solve_response"},
**common)
write_task_result(tmp_path, "kid-plain", "completed", result="no snapshot", **common)
text, children = post_task_synthesis._child_task_evidence(
SimpleNamespace(drive_root=tmp_path), {"id": "root"})
by_id = {row["task_id"]: row for row in children}
assert by_id["kid-session"]["engine"] == {
"subagent_id": "codex=gpt-6-astra/xhigh", "kind": "agent_session",
"target": "codex=gpt-6-astra", "effort": "xhigh", "access": "full",
}
assert by_id["kid-session"]["duration_sec"] == 270.0
assert "used_model" not in by_id["kid-session"], "a nanny's rounds are not the leaf's served model"
assert by_id["kid-api"]["engine"] == {
"subagent_id": "x-ai/grok-4.6/low", "kind": "api_model", "target": "x-ai/grok-4.6", "effort": "low",
}
assert by_id["kid-api"]["used_model"] == "x-ai/grok-4.6"
assert "duration_sec" not in by_id["kid-api"], "a stamp that does not follow the start yields no duration"
assert by_id["kid-api"]["started_at"] == "2026-09-20T10:00:00+00:00"
for absent in ("engine", "used_model", "started_at", "duration_sec"):
assert absent not in by_id["kid-plain"], "a record without the facts states none"
assert "primary-builder" not in text and "fast-scout" not in text, "stored keys are not evidence"