Merge PR #1283: retain Cowork evaluator evidence (#1261 A)

fix: retain Cowork evaluator evidence independently of execution (#1261 A)
This commit is contained in:
Ouroboros 2026-09-25 03:51:28 +03:00 • committed by GitHub
commit 829164b908
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
8 changed files with 542 additions and 18 deletions

View file

@ -73,7 +73,9 @@ CALL) — NEVER STAY SILENT.**
every requested instance, including setup failures, timeouts, and empty
patches, even when the official benchmark prediction/submission format only
accepts successful rows. Defaults are adapter-specific (`result_index.jsonl`,
`<predictions>.ledger.jsonl`, or `osworld_preflight.ledger.jsonl`).
`<predictions>.ledger.jsonl`, or `osworld_preflight.ledger.jsonl`). A row's
`official_eval_status` is `unreported` unless its adapter states one;
`not_run` is an explicit claim that the official evaluator never ran.
These sidecars are audit artifacts, not replacement scoring. Official benchmark
harnesses and official result files remain the scoring authority.

View file

@ -128,7 +128,10 @@ def task_result_row(
Pass ``runtime_result=<task result payload>`` (metadata or keyword) wherever the adapter
holds one: ``runtime_outcome`` then discloses why the RUNTIME stopped, independently of
the adapter-stage ``reason_code`` this row's ``status`` describes."""
the adapter-stage ``reason_code`` this row's ``status`` describes.
An omitted or empty ``official_eval_status`` is ``unreported``: the caller made no claim.
``not_run`` asserts the official evaluator never ran, so only an explicit value says it."""
meta = dict(metadata or {})
for key, value in overrides.items():
if value is not None:
@ -142,7 +145,7 @@ def task_result_row(
"reason_code": str(meta.get("reason_code") or ""),
"runtime_outcome": runtime_terminal_disclosure(meta.get("runtime_result")),
"prediction_written": bool(meta.get("prediction_written")),
"official_eval_status": str(meta.get("official_eval_status") or "not_run"),
"official_eval_status": str(meta.get("official_eval_status") or "unreported"),
"output_paths": meta.get("output_paths") or {},
"error": str(meta.get("error") or ""),
"details": meta.get("details") or {},

View file

@ -152,6 +152,41 @@ with remaining work are permitted; settled successes and genuine failures
from every ancestor are skipped, never repeated for best-of selection. Any final
scoring overlay must retain provenance to the original attempts.
Every ledger row, whatever its execution outcome, carries the official
evaluator's receipt: the exact `eval_res.json` path, byte count and SHA-256,
the literal verdict and a bounded cause. The launcher never writes or re-runs
that file, never copies it into the receipt, and never infers a verdict from
runner output. As before, only a scored row also keeps the parsed evaluator
result in its details. `official_eval_status` is `completed` only for a literal JSON boolean.
`declined` means `pass: null` that exactly matches the evaluator's status gate
text, with its linked `traj_log.json` recording a non-success status. Any other
null is `unknown`. A missing or non-boolean `pass` is `invalid`. An I/O,
UTF-8, JSON or non-object failure is `unreadable`. No file is `unreported`, not
proof that the evaluator never ran. Only a successful agent phase with a
literal boolean is scored. `false` with evaluator output remains a genuine
verdict. Any other result there stays an infrastructure row, never
`bool(value)`. A verdict on an agent- or infrastructure-failed row is disclosed
by the ledger and audit but never promoted into the score. The shared ledger
default is `unreported`; `not_run` appears only when an adapter asserts it.
The audit's `official_pass` reports the literal receipt verdict independently
of the unchanged execution/scoring classification. Runtime stop disclosure is
read only from the summary-named exported task result whose embedded identity
matches; missing or mismatched sources remain explicit gaps, never a glob-selected
neighbour's outcome.
Ledgers written before this revision recorded `not_run` for every row without
a completed verdict, including rows where the evaluator ran and declined at its
status gate. The planning report dated 2026-09-24 records a prior SHA-256 match
between reconstructed gate receipts and the `audit_eval_sha256` values in
`combined-index-20260922T102550Z.jsonl` for all 22 affected `agent_failed` rows:
14 `deadline_local` and 8 `wall_clock_timeout`. This implementation did not
re-read those historical receipts or establish their current availability;
the prior hash comparison is not a fresh file-access check. The planning report
also records that the tasks' PostgreSQL state was destroyed, so historical
re-evaluation is unavailable. No historical index or score is rewritten and no
checker is rerun. For new receipts, the gate link is a fixed path plus exact gate
text; the evaluator records no hash of the log it read.
Phase-aware mounts omit the task's evaluator and ground-truth workspace from the
agent's task view. Ouroboros settings and provider credentials remain outside the
shared dump directory; the run-local credential file is mode 0600 and is cleared

View file

@ -179,14 +179,27 @@ def audit_task(task_dump: Path, ledger: dict[str, Any]) -> dict[str, Any]:
# Lack of telemetry is an audit gap, never a reason to rewrite its verdict.
log_incomplete = any(g["source"] == "events.jsonl" for g in gaps)
cost_complete = bool(activity["usage_records"]) and not unknown_cost and not log_incomplete
official_status = str(ledger.get("official_eval_status") or "not_run")
official_status = str(ledger.get("official_eval_status") or "unreported")
# Literal evaluator truth is independent of agent execution and the scoring classification.
# Legacy rows without receipt metadata retain their previously recorded scored verdict.
details = ledger.get("details") if isinstance(ledger.get("details"), dict) else {}
receipt = details.get("official_receipt") if isinstance(details.get("official_receipt"), dict) else {}
receipt_bytes = receipt.get("bytes")
return {
"instance_id": str(ledger.get("instance_id") or task_dump.name.removeprefix("SingleUserTurn-")),
"ledger_status": status,
"classification": classification,
"reason_code": str(ledger.get("reason_code") or summary.get("reason_code") or ""),
"official_eval_status": official_status,
"official_pass": status == "passed" if official_status == "completed" and status in {"passed", "failed"} else None,
"official_pass": (receipt.get("pass") if receipt else status == "passed")
if official_status == "completed" and (isinstance(receipt.get("pass"), bool)
or not receipt and status in {"passed", "failed"}) else None,
"official_receipt": {
"available": bool(receipt),
"pass": receipt.get("pass") if isinstance(receipt.get("pass"), bool) else None,
"bytes": receipt_bytes if isinstance(receipt_bytes, int) and not isinstance(receipt_bytes, bool) else None,
"sha256": str(receipt.get("sha256") or "") or None,
},
"activity": activity,
"model_activity_observed": bool(activity["nonempty_usage_records"]),
"mcp_activity_observed": bool(activity["mcp_calls"]),
@ -225,7 +238,8 @@ def audit_run(run_root: Path | str) -> dict[str, Any]:
"limitations": ["Diagnostic argument references require manual review, not automatic disqualification.",
"No flags do not prove no contamination; copied logs may omit full arguments or responses.",
"llm_usage cost is a compatibility projection, not authoritative provider billing.",
"Billing provider names do not identify OpenRouter upstream endpoints."],
"Billing provider names do not identify OpenRouter upstream endpoints.",
"A receipt verdict on an agent/infrastructure-failed row is disclosure, never a scored pass."],
"task_count": len(rows), "classifications": dict(Counter(row["classification"] for row in rows)),
"manual_review_tasks": [row["instance_id"] for row in rows if row["manual_review"]],
"cost": {"known_usd": known, "total_usd": known if complete else None, "complete": complete},

View file

@ -0,0 +1,133 @@
"""Read-only receipt of the official Cowork evaluator output for one task dump.
Standard library only, so the same reader stays usable inside the benchmark's own
evaluator environment. It reads exactly ``eval_res.json`` and, for a ``pass: null``
decline, exactly the ``traj_log.json`` that the upstream evaluator gated on. It never
globs, never writes either file and never copies the evaluator payload: the receipt
keeps the path, exact byte count and SHA-256, the literal verdict and a bounded cause.
``official_eval_status`` vocabulary produced here:
* ``completed`` — ``pass`` is a literal JSON boolean; ``false`` stays a negative verdict.
* ``declined`` — ``pass: null`` proven to be the upstream status gate: the linked log
is non-success and the file is exactly that gate's two-key payload.
* ``unknown`` — ``pass: null`` without that proof.
* ``invalid`` — a JSON object whose ``pass`` is missing or not boolean/null.
* ``unreadable`` — I/O, UTF-8, JSON or non-object failure.
* ``unreported`` — no file: the evaluator left no receipt, which is not proof it never ran.
"""
from __future__ import annotations
import hashlib
import json
import pathlib
from typing import Any
RECEIPT_SCHEMA = "ouroboros.cowork.official_receipt.v1"
RESULT_NAME = "eval_res.json"
# ``TaskConfig.log_file`` default under the task dump; the evaluator's status gate reads it.
GATE_LOG_NAME = "traj_log.json"
UPSTREAM_SUCCESS = "success"
_CAUSE_LIMIT = 200
def _gate_details(status: str) -> str:
# utils/evaluation/evaluator.py::TaskEvaluator.evaluate_one at the pinned benchmark commit.
return f"Task status: {status}, only SUCCESS counts as pass; pass is null"
def _cause(text: str) -> str:
return text[:_CAUSE_LIMIT]
def _reject_non_json_constant(value: str) -> None:
raise ValueError(f"non_json_constant:{value}")
def _unique_keys(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
result: dict[str, Any] = {}
for key, value in pairs:
if key in result:
raise ValueError("duplicate_json_key")
result[key] = value
return result
def _read_object(path: pathlib.Path) -> tuple[dict[str, Any], Any]:
"""Return ``(facts, value)``; ``facts['state']`` is ``absent``, ``unreadable`` or ``parsed``."""
facts: dict[str, Any] = {"path": str(path), "bytes": None, "sha256": None}
try:
raw = path.read_bytes()
except FileNotFoundError:
return {**facts, "state": "absent"}, None
except OSError as exc:
return {**facts, "state": "unreadable", "cause": _cause(f"os_error:{type(exc).__name__}")}, None
facts.update(bytes=len(raw), sha256=hashlib.sha256(raw).hexdigest())
try:
text = raw.decode("utf-8")
except UnicodeDecodeError as exc:
return {**facts, "state": "unreadable", "cause": _cause(f"utf8_error:byte_{exc.start}")}, None
try:
value = json.loads(text, parse_constant=_reject_non_json_constant,
object_pairs_hook=_unique_keys)
except (ValueError, RecursionError) as exc:
where = f"line_{exc.lineno}_col_{exc.colno}" if isinstance(exc, json.JSONDecodeError) else type(exc).__name__
return {**facts, "state": "unreadable", "cause": _cause(f"json_error:{where}")}, None
if not isinstance(value, dict):
return {**facts, "state": "unreadable", "cause": _cause(f"not_object:{type(value).__name__}")}, None
return {**facts, "state": "parsed"}, value
def _status_gate(task_dump: pathlib.Path, payload: dict[str, Any]) -> dict[str, Any] | None:
"""Evidence that ``pass: null`` came from the evaluator's non-success status gate, else None."""
log, record = _read_object(task_dump / GATE_LOG_NAME)
if log["state"] != "parsed":
return None
status = record.get("status")
if not isinstance(status, str) or not status or status == UPSTREAM_SUCCESS:
return None
if set(payload) != {"pass", "details"} or payload["details"] != _gate_details(status):
return None
return {"log_path": log["path"], "log_bytes": log["bytes"], "log_sha256": log["sha256"],
"log_status": status, "rule": "upstream_status_gate"}
def read_official_receipt(task_dump: pathlib.Path) -> tuple[dict[str, Any], dict[str, Any] | None]:
"""Return ``(receipt, verdict_payload)``; the payload is returned only for a literal verdict.
The caller decides what, if anything, of the payload to retain; the receipt itself is
always safe to persist and publish."""
facts, value = _read_object(task_dump / RESULT_NAME)
receipt = {"schema": RECEIPT_SCHEMA, "path": facts["path"], "bytes": facts["bytes"],
"sha256": facts["sha256"], "pass": None, "cause": facts.get("cause", "")}
if facts["state"] == "absent":
return {**receipt, "official_eval_status": "unreported", "cause": "file_absent"}, None
if facts["state"] == "unreadable":
return {**receipt, "official_eval_status": "unreadable"}, None
if "pass" not in value:
return {**receipt, "official_eval_status": "invalid", "cause": "pass_missing"}, None
verdict = value["pass"]
if verdict is True or verdict is False:
return {**receipt, "official_eval_status": "completed", "pass": verdict}, value
if verdict is not None:
return {**receipt, "official_eval_status": "invalid",
"cause": _cause(f"pass_not_boolean:{type(verdict).__name__}")}, None
gate = _status_gate(task_dump, value)
if gate is None:
return {**receipt, "official_eval_status": "unknown", "cause": "pass_null_unlinked"}, None
return {**receipt, "official_eval_status": "declined", "cause": "status_gate", "gate": gate}, None
def read_linked_runtime_result(task_dump: pathlib.Path, summary: dict[str, Any]) -> tuple[dict[str, Any], dict[str, Any]]:
"""Read only the summary-named exported task, never select an arbitrary JSON file."""
task_id = summary.get("ouroboros_task_id")
if (not isinstance(task_id, str) or not task_id
or any(c not in "abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789_.-" for c in task_id)):
return {}, {"state": "unavailable", "cause": "task_id_unavailable"}
facts, value = _read_object(task_dump / "ouroboros" / f"{task_id}.json")
if facts["state"] != "parsed":
return {}, facts
if value.get("task_id", value.get("id")) != task_id:
return {}, {**facts, "state": "unavailable", "cause": "task_id_mismatch"}
return value, {**facts, "state": "linked", "task_id": task_id}

View file

@ -45,6 +45,7 @@ from devtools.benchmarks.common.result_index import (
from devtools.benchmarks.common.run_roots import assert_outside_repo, repo_root_from_devtools, run_root, timestamp_run_id
from devtools.benchmarks.common.secrets import credential_fingerprint
from devtools.benchmarks.cowork_bench.campaign import CampaignBudget, campaign_lock, key_usage, validate_usage
from devtools.benchmarks.cowork_bench.official_receipt import read_linked_runtime_result, read_official_receipt
from devtools.benchmarks.cowork_bench.resource_limits import LABEL_KEY, prepare_resource_env
from ouroboros.platform_layer import kill_process_group_id, terminate_process_group_id
from ouroboros.process_custody import spawn_supervised
@ -258,35 +259,48 @@ def _load(path: pathlib.Path) -> dict[str, Any]:
def ledger_row(task: str, task_dump: pathlib.Path, runner_row: dict[str, str]) -> dict[str, Any]:
"""One denominator-preserving row. The runner's exit code and CSV are NOT the status: the
adapter summary says how the agent phase ended and ``eval_res.json`` is the verdict."""
adapter summary says how the agent phase ended and ``eval_res.json`` is the verdict.
The official receipt is attached on EVERY branch, independently of that status: a
timed-out agent phase keeps the evaluator's own record instead of a claimed ``not_run``,
and only a literal boolean verdict on a successful agent phase is scored."""
summary = _load(task_dump / "ouroboros_summary.json")
eval_res = _load(task_dump / "eval_res.json")
receipt, eval_res = read_official_receipt(task_dump)
runtime_result, runtime_source = read_linked_runtime_result(task_dump, summary)
official = receipt["official_eval_status"]
paths = {"task_dump": str(task_dump)}
details = {"runner": runner_row, "adapter": summary}
details = {"runner": runner_row, "adapter": summary, "official_receipt": receipt,
"runtime_result_source": runtime_source}
runtime = {"runtime_result": runtime_result}
if runner_row.get("status") == "pg_fail":
return task_result_row(benchmark=BENCHMARK, instance_id=task, status="infra_failed",
reason_code="pg_fail", output_paths=paths, details=details)
reason_code="pg_fail", output_paths=paths, details=details,
official_eval_status=official, **runtime)
if not summary:
status = "infra_failed" if runner_row else "not_attempted"
return task_result_row(benchmark=BENCHMARK, instance_id=task, status=status,
reason_code="missing_adapter_summary" if runner_row else "missing_result",
output_paths=paths, details=details)
output_paths=paths, details=details, official_eval_status=official, **runtime)
reason = str(summary.get("reason_code") or "")
if summary.get("infra_failed"):
return task_result_row(benchmark=BENCHMARK, instance_id=task, status="infra_failed",
reason_code=reason or "infra_failed", output_paths=paths,
error=str(summary.get("error") or ""), details=details)
error=str(summary.get("error") or ""), details=details,
official_eval_status=official, **runtime)
if summary.get("bench_status") != "success":
return task_result_row(benchmark=BENCHMARK, instance_id=task, status="agent_failed",
reason_code=reason or "agent_not_finished", output_paths=paths, details=details)
if "pass" not in eval_res or eval_res.get("pass") is None:
reason_code=reason or "agent_not_finished", output_paths=paths,
details=details, official_eval_status=official, **runtime)
if eval_res is None:
# No literal boolean verdict: never coerce a string/number `pass` into a score.
return task_result_row(benchmark=BENCHMARK, instance_id=task, status="infra_failed",
reason_code="missing_eval_result", output_paths=paths, details=details)
passed = bool(eval_res.get("pass"))
reason_code="missing_eval_result", output_paths=paths, details=details,
official_eval_status=official, **runtime)
passed = receipt["pass"]
return task_result_row(
benchmark=BENCHMARK, instance_id=task, status="passed" if passed else "failed",
reason_code="passed" if passed else "verifier_failed", official_eval_status="completed",
output_paths=paths, details={**details, "eval": eval_res},
output_paths=paths, details={**details, "eval": eval_res}, **runtime,
)

View file

@ -507,7 +507,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de
`devtools/` (including `devtools/benchmarks/cybergym/`) lives outside the runtime and package discovery: no runtime imports, normal review, artifacts in an external output root; a sentinel-marked isolated root suppresses rotation warnings. CyberGym keeps budget/result settlement in `cybergym_adapter`; `cybergym_custody._CustodyMixin` owns gateway admission, normal and cancelled waits, and their shared terminal-custody transfer, while `cybergym_lifecycle._LifecycleMixin` owns startup, official-verifier delivery and cleanup. Both wait paths retain their distinct bounds around the same attempt identity rather than creating another scheduler. `devtools/e2e_live/` is the live E2E stand — K staggered isolated real servers running the owner-shaped scenarios SM1, SW1 and SK1, accepted over durable artifacts and a browser probe, admitted through the same seed gate and manifest seams as the benchmark launchers. Its operator manual is `devtools/e2e_live/README.md`; the one rule binding runtime changes is DEVELOPMENT "Live E2E stand".
`devtools/benchmarks/cowork_bench/` runs a clean seed inside the pinned upstream task containers, preserving MCP process state through a local proxy; its launcher owns campaign spending, resource limits and result ledgers, while its offline audit preserves the official evaluator as scoring authority (see its `METHODOLOGY.md`).
`devtools/benchmarks/cowork_bench/` runs a clean seed inside pinned upstream containers with a persistent MCP proxy. Its launcher owns campaign spending, limits and result ledgers; `official_receipt.py` reads exact evaluator bytes independently of execution, and the offline audit preserves official scoring authority (contracts: its `METHODOLOGY.md`).
### Gateway Boundary v1

View file

@ -0,0 +1,323 @@
"""The official evaluator receipt survives every Cowork outcome without changing its bytes or the score."""
from __future__ import annotations
import hashlib
import json
import pathlib
import pytest
from devtools.benchmarks.common.result_index import task_result_row
from devtools.benchmarks.cowork_bench import run_cowork_bench as launcher
from devtools.benchmarks.cowork_bench.audit_cowork_bench import audit_run
from devtools.benchmarks.cowork_bench.official_receipt import read_official_receipt
GATE = "Task status: {}, only SUCCESS counts as pass; pass is null"
SECRET = "PRIVATE_EVALUATOR_OUTPUT_ЖЁЛТЫЙ"
# Status each execution branch had BEFORE receipts existed; a receipt never changes it.
BRANCHES = {
"pg_fail": ({}, {"status": "pg_fail"}, "infra_failed", "pg_fail"),
"no_summary": ({}, {"status": "failed"}, "infra_failed", "missing_adapter_summary"),
"not_attempted": ({}, {}, "not_attempted", "missing_result"),
"adapter_infra": ({"bench_status": "failed", "infra_failed": True, "reason_code": "llm_api_error"},
{"status": "failed"}, "infra_failed", "llm_api_error"),
"deadline_local": ({"bench_status": "max_turns_reached", "reason_code": "deadline_local"},
{"status": "failed"}, "agent_failed", "deadline_local"),
"wall_clock_timeout": ({"bench_status": "failed", "reason_code": "wall_clock_timeout"},
{"status": "failed"}, "agent_failed", "wall_clock_timeout"),
}
def write_raw(path: pathlib.Path, data: bytes) -> bytes:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(data)
return data
def dump_json(value) -> bytes:
return json.dumps(value, ensure_ascii=False, indent=2).encode("utf-8")
def gate_fixture(dump: pathlib.Path, log_status: str, details_status: str | None = None, **extra) -> bytes:
write_raw(dump / "traj_log.json", dump_json({"config": {}, "status": log_status}))
return write_raw(dump / "eval_res.json",
dump_json({"pass": None, "details": GATE.format(details_status or log_status), **extra}))
@pytest.mark.parametrize("branch", sorted(BRANCHES))
@pytest.mark.parametrize("payload,expected_official,expected_pass", [
({"pass": True, "details": "All evaluation checks passed"}, "completed", True),
({"pass": False, "failure": SECRET}, "completed", False),
({"pass": None, "details": "no linked gate"}, "unknown", None),
("gate", "declined", None),
(None, "unreported", None),
])
def test_every_branch_keeps_the_exact_receipt_and_its_status(tmp_path, branch, payload, expected_official,
expected_pass):
summary, runner, status, reason = BRANCHES[branch]
if summary:
write_raw(tmp_path / "ouroboros_summary.json", dump_json(summary))
if payload == "gate":
raw = gate_fixture(tmp_path, summary.get("bench_status") or "failed")
elif payload is not None:
raw = write_raw(tmp_path / "eval_res.json", dump_json(payload))
before = {path.name: path.read_bytes() for path in tmp_path.iterdir() if path.is_file()}
row = launcher.ledger_row("task", tmp_path, runner)
assert (row["status"], row["reason_code"]) == (status, reason)
assert row["official_eval_status"] == expected_official != "not_run"
receipt = row["details"]["official_receipt"]
assert receipt["official_eval_status"] == expected_official
assert receipt["pass"] is expected_pass
assert receipt["path"] == str(tmp_path / "eval_res.json")
if payload is None:
assert (receipt["bytes"], receipt["sha256"], receipt["cause"]) == (None, None, "file_absent")
else:
assert receipt["bytes"] == len(raw)
assert receipt["sha256"] == hashlib.sha256(raw).hexdigest()
assert "eval" not in row["details"]
assert SECRET not in json.dumps(row, ensure_ascii=False)
assert {path.name: path.read_bytes() for path in tmp_path.iterdir() if path.is_file()} == before
@pytest.mark.parametrize("payload,status,reason,official", [
({"pass": True}, "passed", "passed", "completed"),
({"pass": False, "failure": "verifier output"}, "failed", "verifier_failed", "completed"),
({"pass": None}, "infra_failed", "missing_eval_result", "unknown"),
({}, "infra_failed", "missing_eval_result", "invalid"),
({"pass": "false"}, "infra_failed", "missing_eval_result", "invalid"),
({"pass": 1}, "infra_failed", "missing_eval_result", "invalid"),
({"pass": 0.0}, "infra_failed", "missing_eval_result", "invalid"),
(None, "infra_failed", "missing_eval_result", "unreported"),
])
def test_successful_agent_phase_scores_only_a_literal_boolean(tmp_path, payload, status, reason, official):
write_raw(tmp_path / "ouroboros_summary.json", dump_json({"bench_status": "success"}))
if payload is not None:
write_raw(tmp_path / "eval_res.json", dump_json(payload))
row = launcher.ledger_row("task", tmp_path, {"status": "success"})
assert (row["status"], row["reason_code"], row["official_eval_status"]) == (status, reason, official)
# The scored branch keeps its previous payload detail; every other branch carries only the receipt.
assert ("eval" in row["details"]) is (status in {"passed", "failed"})
if status in {"passed", "failed"}:
assert row["details"]["eval"] == payload
def test_receipt_classifies_every_malformed_file_without_coercion(tmp_path):
cases = {
"missing": None,
"directory": "dir",
"utf8": b'{"pass": true, "x": "\xff"}',
"truncated": b'{"pass": tr',
"not_object": b"[true]",
"deep": b"[" * 200_000,
"empty_object": b"{}",
"string": b'{"pass": "true"}',
"number": b'{"pass": 1}',
"array": b'{"pass": [true]}',
"nan_detail": b'{"pass":true,"details":NaN}',
"infinite_detail": b'{"pass":true,"details":Infinity}',
"duplicate_pass": b'{"pass":false,"pass":true}',
}
expected = {
"missing": ("unreported", "file_absent"),
"directory": ("unreadable", "os_error:"),
"utf8": ("unreadable", "utf8_error:byte_21"),
"truncated": ("unreadable", "json_error:line_1_col_10"),
"not_object": ("unreadable", "not_object:list"),
"deep": ("unreadable", "json_error:RecursionError"),
"empty_object": ("invalid", "pass_missing"),
"string": ("invalid", "pass_not_boolean:str"),
"number": ("invalid", "pass_not_boolean:int"),
"array": ("invalid", "pass_not_boolean:list"),
"nan_detail": ("unreadable", "json_error:ValueError"),
"infinite_detail": ("unreadable", "json_error:ValueError"),
"duplicate_pass": ("unreadable", "json_error:ValueError"),
}
for name, data in cases.items():
dump = tmp_path / name
dump.mkdir()
if data == "dir":
(dump / "eval_res.json").mkdir()
elif data is not None:
write_raw(dump / "eval_res.json", data)
receipt, payload = read_official_receipt(dump)
expected_status, expected_cause = expected[name]
assert receipt["official_eval_status"] == expected_status, name
if name == "directory":
# Windows can report PermissionError where POSIX reports IsADirectoryError.
assert receipt["cause"].startswith(expected_cause), name
else:
assert receipt["cause"] == expected_cause, name
assert receipt["pass"] is None and payload is None, name
assert len(receipt["cause"]) <= 200
if isinstance(data, bytes):
assert receipt["bytes"] == len(data)
assert receipt["sha256"] == hashlib.sha256(data).hexdigest()
assert (dump / "eval_res.json").read_bytes() == data
@pytest.mark.parametrize("log_status,details_status,extra,log_missing,expected", [
("max_turns_reached", None, {}, False, "declined"),
("failed", None, {}, False, "declined"),
("success", None, {}, False, "unknown"), # the gate never declines a successful log
("failed", "max_turns_reached", {}, False, "unknown"), # receipt names another status
("failed", None, {"failure": "x"}, False, "unknown"), # not the gate's exact payload
("failed", None, {}, True, "unknown"), # no linked log to prove the gate
])
def test_null_verdict_is_declined_only_with_a_linked_status_gate(tmp_path, log_status, details_status, extra,
log_missing, expected):
gate_fixture(tmp_path, log_status, details_status, **extra)
log_bytes = (tmp_path / "traj_log.json").read_bytes()
if log_missing:
(tmp_path / "traj_log.json").unlink()
receipt, _payload = read_official_receipt(tmp_path)
assert receipt["official_eval_status"] == expected
assert receipt["pass"] is None
if expected == "declined":
assert receipt["gate"] == {
"log_path": str(tmp_path / "traj_log.json"), "log_bytes": len(log_bytes),
"log_sha256": hashlib.sha256(log_bytes).hexdigest(), "log_status": log_status,
"rule": "upstream_status_gate",
}
else:
assert "gate" not in receipt
def test_valid_fixture_score_and_settlement_parity(tmp_path):
"""Statuses, counts and resume settlement on valid receipts equal the pre-receipt ledger."""
bench = tmp_path / "bench"
dumps = bench / "dumps" / launcher.dump_dir_name("m")
fixtures = {
"pass": ({"bench_status": "success"}, {"pass": True}, "passed"),
"fail": ({"bench_status": "success"}, {"pass": False, "failure": "x"}, "failed"),
"timeout": ({"bench_status": "max_turns_reached", "reason_code": "deadline_local"}, "gate", "agent_failed"),
"infra": ({"bench_status": "failed", "infra_failed": True, "reason_code": "llm_api_error"}, None,
"infra_failed"),
"unstarted": (None, None, "not_attempted"),
}
runner_rows = ["task,status,eval_pass,duration_s"]
for task, (summary, payload, _status) in fixtures.items():
dump = dumps / f"SingleUserTurn-{task}"
dump.mkdir(parents=True)
if summary is None:
continue
runner_rows.append(f"{task},{'success' if summary['bench_status'] == 'success' else 'failed'},null,1")
write_raw(dump / "ouroboros_summary.json", dump_json(summary))
if payload == "gate":
gate_fixture(dump, summary["bench_status"])
elif payload is not None:
write_raw(dump / "eval_res.json", dump_json(payload))
write_raw(bench / "benchmark_logs" / "fully_parallel_1" / "summary.csv",
("\n".join(runner_rows) + "\n").encode("utf-8"))
run = tmp_path / "run"
counts = launcher.write_ledger(run / "result_index.jsonl", bench, "m", list(fixtures))
assert counts == {"passed": 1, "failed": 1, "agent_failed": 1, "infra_failed": 1, "not_attempted": 1}
assert launcher.settled_tasks([run]) == {"pass", "fail", "timeout"}
rows = {row["instance_id"]: row for row in map(json.loads, (run / "result_index.jsonl").read_text().splitlines())}
assert {task: row["status"] for task, row in rows.items()} == {task: f[2] for task, f in fixtures.items()}
assert {task: row["official_eval_status"] for task, row in rows.items()} == {
"pass": "completed", "fail": "completed", "timeout": "declined",
"infra": "unreported", "unstarted": "unreported",
}
def test_audit_shows_the_literal_verdict_without_promoting_an_unscored_row(tmp_path):
dump = tmp_path / "dump"
write_raw(dump / "ouroboros_summary.json", dump_json({"bench_status": "failed", "reason_code": "wall_clock_timeout"}))
raw = write_raw(dump / "eval_res.json", dump_json({"pass": True, "details": SECRET}))
row = launcher.ledger_row("late", dump, {"status": "failed"})
historical = {"instance_id": "old", "status": "agent_failed", "official_eval_status": "not_run",
"output_paths": {"task_dump": str(tmp_path / "old")}}
legacy = {"instance_id": "bare", "status": "infra_failed", "output_paths": {"task_dump": str(tmp_path / "bare")}}
(tmp_path / "result_index.jsonl").write_text(
"".join(json.dumps(item) + "\n" for item in (row, historical, legacy)), encoding="utf-8")
report = audit_run(tmp_path)
late, old, bare = report["tasks"]
assert row["status"] == late["ledger_status"] == "agent_failed"
assert late["classification"] == "genuine_failure"
assert late["official_eval_status"] == "completed"
assert late["official_pass"] is True # evaluator fact, not promotion of execution/score
assert late["official_receipt"] == {"available": True, "pass": True, "bytes": len(raw),
"sha256": hashlib.sha256(raw).hexdigest()}
assert report["classifications"] == {"genuine_failure": 2, "infrastructure": 1}
# Historical rows are read as written: an explicit `not_run` stays; a missing field is unreported.
assert (old["official_eval_status"], old["official_receipt"]["available"]) == ("not_run", False)
assert (bare["official_eval_status"], bare["official_receipt"]["pass"]) == ("unreported", None)
assert SECRET not in json.dumps(report, ensure_ascii=False)
assert (dump / "eval_res.json").read_bytes() == raw
@pytest.mark.parametrize("given,expected", [
({}, "unreported"),
({"official_eval_status": ""}, "unreported"),
({"official_eval_status": None}, "unreported"),
({"official_eval_status": "not_run"}, "not_run"),
({"metadata": {"official_eval_status": "not_run"}}, "not_run"),
({"official_eval_status": "pending"}, "pending"),
])
def test_shared_default_is_unreported_and_explicit_not_run_survives(given, expected):
row = task_result_row(benchmark="unit", instance_id="x", status="failed", **given)
assert row["official_eval_status"] == expected
@pytest.mark.parametrize("branch", sorted(BRANCHES))
def test_runtime_disclosure_reads_only_exact_linked_task(tmp_path, branch):
summary, runner, _status, _reason = BRANCHES[branch]
# An unrelated result must never fill an unavailable link, including missing summary.
write_raw(tmp_path / "ouroboros" / "unrelated.json", dump_json({
"task_id": "unrelated", "status": "failed", "reason_code": "provider_unavailable"}))
if not summary:
row = launcher.ledger_row("task", tmp_path, runner)
assert row["runtime_outcome"] == {"available": False}
assert row["details"]["runtime_result_source"]["cause"] == "task_id_unavailable"
return
summary = {**summary, "ouroboros_task_id": "exact-task"}
write_raw(tmp_path / "ouroboros_summary.json", dump_json(summary))
target = tmp_path / "ouroboros" / "exact-task.json"
write_raw(target, dump_json({"task_id": "different", "reason_code": "provider_unavailable"}))
row = launcher.ledger_row("task", tmp_path, runner)
assert row["runtime_outcome"] == {"available": False}
assert row["details"]["runtime_result_source"]["cause"] == "task_id_mismatch"
raw = write_raw(target, dump_json({"task_id": "exact-task", "status": "completed",
"reason_code": "deadline_local"}))
row = launcher.ledger_row("task", tmp_path, runner)
assert row["runtime_outcome"]["reason_code"] == "deadline_local"
assert row["runtime_outcome"]["truncated"] is True
assert row["details"]["runtime_result_source"]["sha256"] == hashlib.sha256(raw).hexdigest()
def test_linked_runtime_failure_stays_gap_and_success_branch_has_disclosure(tmp_path):
write_raw(tmp_path / "ouroboros_summary.json", dump_json({
"bench_status": "success", "ouroboros_task_id": "exact"}))
write_raw(tmp_path / "eval_res.json", b'{"pass": false}')
target = tmp_path / "ouroboros" / "exact.json"
for raw in (None, b"\xff", b"{}", b'{"task_id":"foreign"}'):
if raw is not None:
write_raw(target, raw)
row = launcher.ledger_row("task", tmp_path, {})
assert row["status"] == "failed"
assert row["runtime_outcome"] == {"available": False}
write_raw(target, b'{"task_id":"exact","status":"completed","reason_code":"final_message"}')
row = launcher.ledger_row("task", tmp_path, {})
assert row["runtime_outcome"]["reason_code"] == "final_message"
assert row["status"] == "failed"
def test_dotted_runtime_id_links_exact_result_and_cannot_traverse(tmp_path):
write_raw(tmp_path / "ouroboros_summary.json", dump_json({
"bench_status": "success", "ouroboros_task_id": "exact.task"}))
write_raw(tmp_path / "eval_res.json", b'{"pass":false}')
write_raw(tmp_path / "ouroboros" / "exact.task.json", dump_json({
"task_id": "exact.task", "status": "completed", "reason_code": "deadline_local"}))
row = launcher.ledger_row("task", tmp_path, {})
assert row["runtime_outcome"]["reason_code"] == "deadline_local"
assert row["details"]["runtime_result_source"]["state"] == "linked"
write_raw(tmp_path / "ouroboros_summary.json", dump_json({
"bench_status": "success", "ouroboros_task_id": "../foreign"}))
row = launcher.ledger_row("task", tmp_path, {})
assert row["runtime_outcome"] == {"available": False}
assert row["details"]["runtime_result_source"]["cause"] == "task_id_unavailable"