ouroboros/tests/test_store_task_result.py
Ouroboros 57230ee2ef v7next F1: domain D17 - headless split and three test-giant splits, proof-green
headless.py (1573 at tip) gives up its two ledger-assigned leaves again:
ouroboros/headless_status.py (50, 11 symbols) and
ouroboros/workspace_patch_capture.py (668, 19 symbols). All 30 spans are
byte-identical between the reference leaves and git show HEAD bytes — the
hardened transplant --check (mandatory byte gate, undeclared-top-level check,
def681bd) is green on every symbol of both leaves. The facade (947) replays
the oracle's exact edit script over tip bytes: its only divergence from the
reference facade is genuine upstream residue drift (child_ref promotion
machinery, import changes), verified hunk by hunk.

Test splits per the ledger, upstream bytes as truth:
- test_headless_cli.py 2824 -> 462 + five themed siblings + shared fixtures;
  93 test functions preserved exactly (lossless set equality), one adapted
  span kept (the _PATCH_MAX_UNTRACKED_FILE_BYTES monkeypatch retargeted to
  the new leaf, ledger row 739's own adaptation); nine oracle spans carrying
  OTHER domains' v7 spellings reverse-mapped to upstream signatures keyed to
  git show HEAD (registry._run_shell_safety_check string form, tools.core
  _repo_read, queue.init 3-arg, queue.QUEUE_SNAPSHOT_PATH).
- test_workspace_executor.py 1995 -> 541 + three siblings + shared builder;
  two reference spans byte-falsified by upstream drift (06339bb7 readiness
  truth, a849c9a6 probe uncertainty) re-emitted from tip bytes and recorded
  in docs/v7next/LEDGER_CORRECTIONS.md.
- test_agent_task_pipeline.py 1658 -> 1515: only the five ledger-assigned
  _store_task_result rows carved into tests/test_store_task_result.py; the
  other siblings belong to their own lanes.
Thirteen post-cutoff upstream tests have no ledger rows; placed by the
split's theme rule (task_api x4, task_artifacts x1, docker x6, services x2),
disclosed in LEDGER_CORRECTIONS for F5 row-minting.

Pins and mirrors: test_headless_extraction.py transplanted with the
tool_module_inventory clause reduced under an oracle-SHA note (that leaf
belongs to the tools lane); conftest serial table gains the executor family;
the process-custody Popen allowlist row moves headless.py ->
workspace_patch_capture.py with the oracle's justification. All 14 non-split
D17 runtime modules re-proven zero-v7-delta (task_results.py included:
upstream-hot drift stands, no ledger split assigned).

size-ratchet manifest regenerated with the official tool (three test giants
leave GIANT_PATHS, no new band entries); --check green. ruff F clean.
135/105/244/113 tests green in 4-var isolation; HEAD held after every run.

(cherry picked from commit 8dac8303006085bfdb636bacbfc402d6905fae7c)
2026-08-30 17:27:56 +00:00

155 lines
6.1 KiB
Python

"""``_store_task_result`` persistence semantics.
Split out of ``tests/test_agent_task_pipeline.py`` when that module was divided
by theme; every moved block is verbatim. Covers review-evidence persistence,
the compact review projection (no raw model text), failed-status preservation,
and the unresolved-vs-recovered tool-failure outcome axes.
"""
import json
from types import SimpleNamespace
import ouroboros.agent_task_pipeline as pipeline
def test_store_task_result_persists_review_evidence(tmp_path):
env = SimpleNamespace(drive_root=tmp_path)
pipeline._store_task_result(
env=env,
task={"id": "task-store", "type": "task", "text": "hi"},
text="done",
usage={"rounds": 2, "cost": 0.1},
llm_trace={"tool_calls": [], "reasoning_notes": []},
review_evidence={"has_evidence": True, "open_obligations": [{"item": "tests_affected"}]},
)
payload = json.loads((tmp_path / "task_results" / "task-store.json").read_text(encoding="utf-8"))
assert payload["review_evidence"]["has_evidence"] is True
assert payload["review_evidence"]["open_obligations"][0]["item"] == "tests_affected"
def test_store_task_result_persists_only_compact_review_projection(tmp_path):
env = SimpleNamespace(drive_root=tmp_path)
trace = {
"tool_calls": [],
"review_runs": [{
"request": {"surface": "task_acceptance", "policy": {"min_successful_slots": 1}},
"authority": "host_root",
"aggregate_signal": "DEGRADED",
"actors": [{
"slot_id": "slot_1", "model": "openai/gpt-5.6-sol", "status": "ok",
"parsed": {"verdict": "DEGRADED", "summary": "not enough evidence"},
"signal": "DEGRADED", "raw_text": "PRIVATE RAW MODEL RESPONSE",
}],
}],
}
pipeline._store_task_result(
env=env,
task={"id": "task-review-projection", "type": "task", "text": "hi"},
text="done",
usage={"rounds": 1, "cost": 0.0},
llm_trace=trace,
review_evidence={},
)
payload = json.loads(
(tmp_path / "task_results" / "task-review-projection.json").read_text(encoding="utf-8")
)
actor = payload["review_projection"]["panels"][0]["actors"][0]
assert actor["model"] == "openai/gpt-5.6-sol"
assert actor["parse_status"] == "valid"
assert actor["semantic_verdict"] == "DEGRADED"
assert "raw_text" not in actor
assert "PRIVATE RAW MODEL RESPONSE" not in json.dumps(payload)
def test_store_task_result_preserves_failed_status(tmp_path):
from ouroboros.task_results import STATUS_FAILED, write_task_result
env = SimpleNamespace(drive_root=tmp_path)
write_task_result(tmp_path, "task-failed", STATUS_FAILED, result="initial failure")
pipeline._store_task_result(
env=env,
task={"id": "task-failed", "type": "task", "text": "hi"},
text="final failure reply",
usage={"rounds": 1, "cost": 0.0},
llm_trace={"tool_calls": [], "reasoning_notes": []},
review_evidence={},
)
payload = json.loads((tmp_path / "task_results" / "task-failed.json").read_text(encoding="utf-8"))
assert payload["status"] == STATUS_FAILED
assert payload["result"] == "final failure reply"
def test_store_task_result_marks_unresolved_tool_failure_failed(tmp_path):
from ouroboros.task_results import STATUS_COMPLETED
env = SimpleNamespace(drive_root=tmp_path)
pipeline._store_task_result(
env=env,
task={"id": "task-tool-failed", "type": "task", "text": "make file"},
text="Created the file.",
usage={"rounds": 2, "cost": 0.0},
llm_trace={
"tool_calls": [{
"tool": "run_command",
"args": {"cmd": "python3 -c ..."},
"result": "⚠️ ARTIFACT_OUTPUT_ERROR: command succeeded but declared output registration failed.",
"is_error": True,
"status": "artifact_output_error",
}],
"reasoning_notes": [],
},
review_evidence={},
)
payload = json.loads((tmp_path / "task_results" / "task-tool-failed.json").read_text(encoding="utf-8"))
assert payload["status"] == STATUS_COMPLETED
assert payload["outcome_axes"]["execution"]["status"] == "degraded"
assert payload["outcome_axes"]["objective"]["status"] == "not_evaluated"
assert payload["reason_code"] == "tool_failure"
assert payload["loop_outcome"]["failure"]["tool_errors"][0]["status"] == "artifact_output_error"
def test_store_task_result_allows_recovered_tool_failure_success(tmp_path):
from ouroboros.task_results import STATUS_COMPLETED
env = SimpleNamespace(drive_root=tmp_path)
pipeline._store_task_result(
env=env,
task={"id": "task-tool-recovered", "type": "task", "text": "make file"},
text="Created the file.",
usage={"rounds": 3, "cost": 0.0},
llm_trace={
"tool_calls": [
{
"tool": "edit_text",
"args": {"path": "Desktop/report.html"},
"result": "⚠️ EDIT_TEXT_ERROR: old_str matched 0 times",
"is_error": True,
"status": "edit_text_blocked",
},
{
"tool": "write_file",
"args": {"root": "user_files", "path": "Desktop/report.html"},
"result": "OK: wrote user_files:Desktop/report.html\nARTIFACT_OUTPUTS: registered user file -> artifact_store:report.html",
"is_error": False,
"status": "ok",
"artifact_registered": True,
},
],
"reasoning_notes": [],
},
review_evidence={},
)
payload = json.loads((tmp_path / "task_results" / "task-tool-recovered.json").read_text(encoding="utf-8"))
assert payload["status"] == STATUS_COMPLETED
assert payload["outcome_axes"]["execution"]["status"] == "ok"
assert payload["outcome_axes"]["objective"]["status"] == "not_evaluated"
assert payload["loop_outcome"]["failure"] is None