mirror of
https://github.com/razzant/ouroboros.git
synced 2026-10-03 04:07:04 +00:00
headless.py (1573 at tip) gives up its two ledger-assigned leaves again: ouroboros/headless_status.py (50, 11 symbols) and ouroboros/workspace_patch_capture.py (668, 19 symbols). All 30 spans are byte-identical between the reference leaves and git show HEAD bytes — the hardened transplant --check (mandatory byte gate, undeclared-top-level check,def681bd) is green on every symbol of both leaves. The facade (947) replays the oracle's exact edit script over tip bytes: its only divergence from the reference facade is genuine upstream residue drift (child_ref promotion machinery, import changes), verified hunk by hunk. Test splits per the ledger, upstream bytes as truth: - test_headless_cli.py 2824 -> 462 + five themed siblings + shared fixtures; 93 test functions preserved exactly (lossless set equality), one adapted span kept (the _PATCH_MAX_UNTRACKED_FILE_BYTES monkeypatch retargeted to the new leaf, ledger row 739's own adaptation); nine oracle spans carrying OTHER domains' v7 spellings reverse-mapped to upstream signatures keyed to git show HEAD (registry._run_shell_safety_check string form, tools.core _repo_read, queue.init 3-arg, queue.QUEUE_SNAPSHOT_PATH). - test_workspace_executor.py 1995 -> 541 + three siblings + shared builder; two reference spans byte-falsified by upstream drift (06339bb7readiness truth,a849c9a6probe uncertainty) re-emitted from tip bytes and recorded in docs/v7next/LEDGER_CORRECTIONS.md. - test_agent_task_pipeline.py 1658 -> 1515: only the five ledger-assigned _store_task_result rows carved into tests/test_store_task_result.py; the other siblings belong to their own lanes. Thirteen post-cutoff upstream tests have no ledger rows; placed by the split's theme rule (task_api x4, task_artifacts x1, docker x6, services x2), disclosed in LEDGER_CORRECTIONS for F5 row-minting. Pins and mirrors: test_headless_extraction.py transplanted with the tool_module_inventory clause reduced under an oracle-SHA note (that leaf belongs to the tools lane); conftest serial table gains the executor family; the process-custody Popen allowlist row moves headless.py -> workspace_patch_capture.py with the oracle's justification. All 14 non-split D17 runtime modules re-proven zero-v7-delta (task_results.py included: upstream-hot drift stands, no ledger split assigned). size-ratchet manifest regenerated with the official tool (three test giants leave GIANT_PATHS, no new band entries); --check green. ruff F clean. 135/105/244/113 tests green in 4-var isolation; HEAD held after every run. (cherry picked from commit 8dac8303006085bfdb636bacbfc402d6905fae7c)
155 lines
6.1 KiB
Python
155 lines
6.1 KiB
Python
"""``_store_task_result`` persistence semantics.
|
|
|
|
Split out of ``tests/test_agent_task_pipeline.py`` when that module was divided
|
|
by theme; every moved block is verbatim. Covers review-evidence persistence,
|
|
the compact review projection (no raw model text), failed-status preservation,
|
|
and the unresolved-vs-recovered tool-failure outcome axes.
|
|
"""
|
|
|
|
import json
|
|
from types import SimpleNamespace
|
|
|
|
import ouroboros.agent_task_pipeline as pipeline
|
|
|
|
|
|
def test_store_task_result_persists_review_evidence(tmp_path):
|
|
env = SimpleNamespace(drive_root=tmp_path)
|
|
|
|
pipeline._store_task_result(
|
|
env=env,
|
|
task={"id": "task-store", "type": "task", "text": "hi"},
|
|
text="done",
|
|
usage={"rounds": 2, "cost": 0.1},
|
|
llm_trace={"tool_calls": [], "reasoning_notes": []},
|
|
review_evidence={"has_evidence": True, "open_obligations": [{"item": "tests_affected"}]},
|
|
)
|
|
|
|
payload = json.loads((tmp_path / "task_results" / "task-store.json").read_text(encoding="utf-8"))
|
|
assert payload["review_evidence"]["has_evidence"] is True
|
|
assert payload["review_evidence"]["open_obligations"][0]["item"] == "tests_affected"
|
|
|
|
|
|
def test_store_task_result_persists_only_compact_review_projection(tmp_path):
|
|
env = SimpleNamespace(drive_root=tmp_path)
|
|
trace = {
|
|
"tool_calls": [],
|
|
"review_runs": [{
|
|
"request": {"surface": "task_acceptance", "policy": {"min_successful_slots": 1}},
|
|
"authority": "host_root",
|
|
"aggregate_signal": "DEGRADED",
|
|
"actors": [{
|
|
"slot_id": "slot_1", "model": "openai/gpt-5.6-sol", "status": "ok",
|
|
"parsed": {"verdict": "DEGRADED", "summary": "not enough evidence"},
|
|
"signal": "DEGRADED", "raw_text": "PRIVATE RAW MODEL RESPONSE",
|
|
}],
|
|
}],
|
|
}
|
|
pipeline._store_task_result(
|
|
env=env,
|
|
task={"id": "task-review-projection", "type": "task", "text": "hi"},
|
|
text="done",
|
|
usage={"rounds": 1, "cost": 0.0},
|
|
llm_trace=trace,
|
|
review_evidence={},
|
|
)
|
|
|
|
payload = json.loads(
|
|
(tmp_path / "task_results" / "task-review-projection.json").read_text(encoding="utf-8")
|
|
)
|
|
actor = payload["review_projection"]["panels"][0]["actors"][0]
|
|
assert actor["model"] == "openai/gpt-5.6-sol"
|
|
assert actor["parse_status"] == "valid"
|
|
assert actor["semantic_verdict"] == "DEGRADED"
|
|
assert "raw_text" not in actor
|
|
assert "PRIVATE RAW MODEL RESPONSE" not in json.dumps(payload)
|
|
|
|
|
|
def test_store_task_result_preserves_failed_status(tmp_path):
|
|
from ouroboros.task_results import STATUS_FAILED, write_task_result
|
|
|
|
env = SimpleNamespace(drive_root=tmp_path)
|
|
write_task_result(tmp_path, "task-failed", STATUS_FAILED, result="initial failure")
|
|
|
|
pipeline._store_task_result(
|
|
env=env,
|
|
task={"id": "task-failed", "type": "task", "text": "hi"},
|
|
text="final failure reply",
|
|
usage={"rounds": 1, "cost": 0.0},
|
|
llm_trace={"tool_calls": [], "reasoning_notes": []},
|
|
review_evidence={},
|
|
)
|
|
|
|
payload = json.loads((tmp_path / "task_results" / "task-failed.json").read_text(encoding="utf-8"))
|
|
assert payload["status"] == STATUS_FAILED
|
|
assert payload["result"] == "final failure reply"
|
|
|
|
|
|
def test_store_task_result_marks_unresolved_tool_failure_failed(tmp_path):
|
|
from ouroboros.task_results import STATUS_COMPLETED
|
|
|
|
env = SimpleNamespace(drive_root=tmp_path)
|
|
|
|
pipeline._store_task_result(
|
|
env=env,
|
|
task={"id": "task-tool-failed", "type": "task", "text": "make file"},
|
|
text="Created the file.",
|
|
usage={"rounds": 2, "cost": 0.0},
|
|
llm_trace={
|
|
"tool_calls": [{
|
|
"tool": "run_command",
|
|
"args": {"cmd": "python3 -c ..."},
|
|
"result": "⚠️ ARTIFACT_OUTPUT_ERROR: command succeeded but declared output registration failed.",
|
|
"is_error": True,
|
|
"status": "artifact_output_error",
|
|
}],
|
|
"reasoning_notes": [],
|
|
},
|
|
review_evidence={},
|
|
)
|
|
|
|
payload = json.loads((tmp_path / "task_results" / "task-tool-failed.json").read_text(encoding="utf-8"))
|
|
assert payload["status"] == STATUS_COMPLETED
|
|
assert payload["outcome_axes"]["execution"]["status"] == "degraded"
|
|
assert payload["outcome_axes"]["objective"]["status"] == "not_evaluated"
|
|
assert payload["reason_code"] == "tool_failure"
|
|
assert payload["loop_outcome"]["failure"]["tool_errors"][0]["status"] == "artifact_output_error"
|
|
|
|
|
|
def test_store_task_result_allows_recovered_tool_failure_success(tmp_path):
|
|
from ouroboros.task_results import STATUS_COMPLETED
|
|
|
|
env = SimpleNamespace(drive_root=tmp_path)
|
|
|
|
pipeline._store_task_result(
|
|
env=env,
|
|
task={"id": "task-tool-recovered", "type": "task", "text": "make file"},
|
|
text="Created the file.",
|
|
usage={"rounds": 3, "cost": 0.0},
|
|
llm_trace={
|
|
"tool_calls": [
|
|
{
|
|
"tool": "edit_text",
|
|
"args": {"path": "Desktop/report.html"},
|
|
"result": "⚠️ EDIT_TEXT_ERROR: old_str matched 0 times",
|
|
"is_error": True,
|
|
"status": "edit_text_blocked",
|
|
},
|
|
{
|
|
"tool": "write_file",
|
|
"args": {"root": "user_files", "path": "Desktop/report.html"},
|
|
"result": "OK: wrote user_files:Desktop/report.html\nARTIFACT_OUTPUTS: registered user file -> artifact_store:report.html",
|
|
"is_error": False,
|
|
"status": "ok",
|
|
"artifact_registered": True,
|
|
},
|
|
],
|
|
"reasoning_notes": [],
|
|
},
|
|
review_evidence={},
|
|
)
|
|
|
|
payload = json.loads((tmp_path / "task_results" / "task-tool-recovered.json").read_text(encoding="utf-8"))
|
|
assert payload["status"] == STATUS_COMPLETED
|
|
assert payload["outcome_axes"]["execution"]["status"] == "ok"
|
|
assert payload["outcome_axes"]["objective"]["status"] == "not_evaluated"
|
|
assert payload["loop_outcome"]["failure"] is None
|