ouroboros/tests/test_chat_zero_bench_neutrality.py
Ouroboros 7e0f9247a7 Type a plan review's outcome class and say every standing limitation of an answer in the owner's words, on both twins
The gate decision records why a plan review stayed open as a closed class
on the typed execution axis (only some reviewers answered, none answered,
answered but never closed), beside the existing awaiting case; the normal
delivery rail stamps terminal_plan_review_open on the result exactly as the
forced rails do. The owner cause table, in Python and in its browser twin,
gains the sentences for those classes, for a task held by a blocking
review, and for the standing limitations of an answer (a plan review still
open at delivery, deferred sub-task results, an unrecovered tool failure,
an internal error); phrases that named internal actors are reworded. The
completion verdict becomes an assembler on both twins: one clause per fact,
joined, never one fact chosen over another; the parity fixture carries the
new production-shaped cases. The durable task_summary row carries the same
clause as a reason_detail field for transports without a card.

Forced prompts hand their one model call the same limitations as typed
facts, priced by the existing reservation probe; a hold the owner's hurry
released reaches the model once before its last word.

Delivery no longer splits a terminal into an answer and a System notice:
the terminal_host_notice field keeps its bytes on the result for every
machine reader, and only the custody audit rides the send event as its own
typed row. A Telegram-only owner, who has no task card, receives one short
line on a non-clean root finish built from the stamped phase and the reason
sentence; clean finishes stay opt-in.

In the Reviews group a settled plan wave whose reviewers were too few for a
verdict reads no verdict with its counts in the neutral tone instead of the
stored DEGRADED word; reviewer rows name the model and quote the engine's
reported sentence, and the slot id and failure code stay in the task detail
and Logs. DESIGN.md and the architecture chapters state the same; two
chapter byte budgets rise with their reasons.
2026-09-22 01:52:59 +03:00

227 lines
11 KiB
Python

"""Closing the chat-id truthiness class must not move a benchmark's answer.
Terminal-Bench reads a run's final answer straight out of ``chat.jsonl``: the
LAST untyped row with ``direction == "out"`` (``atif._final_answer``). It ALSO
reads ``progress.jsonl`` and publishes those rows as the agent's own narration
(``atif.build_trajectory``). Headless benchmark roots run in the hidden partition
(chat 0), which is exactly the partition the class fix stops dropping — so a
notice that now reaches chat 0 must be a typed or system chat row (invisible to
the answer reader), and a live host TOAST must not be written there at all, or a
supervisor line would be published as something the model said.
"""
import json
import pathlib
import sys
import pytest
REPO = pathlib.Path(__file__).resolve().parents[1]
if str(REPO) not in sys.path:
sys.path.insert(0, str(REPO))
from devtools.benchmarks.terminal_bench.atif import _final_answer # noqa: E402
from ouroboros.contracts.chat_id_policy import HIDDEN_CHAT_ID # noqa: E402
def _agent_dir(tmp_path, chat_rows):
data = tmp_path / "ouroboros-data"
(data / "logs").mkdir(parents=True)
(data / "logs" / "chat.jsonl").write_text(
"".join(json.dumps(row, ensure_ascii=False) + "\n" for row in chat_rows),
encoding="utf-8",
)
return tmp_path
def test_progress_notices_never_enter_the_answer_stream(tmp_path):
"""The reaper/scheduler toasts the class fix un-drops are progress rows.
``send_with_budget(is_progress=True)`` appends to progress.jsonl, so a
grace toast for a hidden-partition root cannot be mistaken for its answer.
"""
from types import SimpleNamespace
from supervisor import message_bus
data = tmp_path / "data"
(data / "logs").mkdir(parents=True)
sent = []
monkey = pytest.MonkeyPatch()
try:
monkey.setattr(message_bus, "DATA_DIR", data)
monkey.setattr(
message_bus, "_BRIDGE",
SimpleNamespace(send_message=lambda *a, **k: sent.append((a, k))),
)
message_bus.send_with_budget(
HIDDEN_CHAT_ID, "⏱️ Task t1 has been running 600s", is_progress=True, task_id="t1",
)
finally:
monkey.undo()
assert sent, "the notice still reaches the live bridge"
assert (data / "logs" / "progress.jsonl").exists()
assert not (data / "logs" / "chat.jsonl").exists()
def test_a_system_incident_row_is_not_read_as_the_answer(tmp_path):
"""A provider-outage notice now reaches chat 0 instead of vanishing (P1).
It is persisted as a SYSTEM row, so the trajectory still reports the task's
own last answer rather than the incident sentence.
"""
agent_dir = _agent_dir(tmp_path, [
{"direction": "in", "chat_id": HIDDEN_CHAT_ID, "text": "solve it"},
{"direction": "out", "chat_id": HIDDEN_CHAT_ID, "text": "the real answer"},
{"direction": "system", "chat_id": HIDDEN_CHAT_ID, "type": "terminal_incident",
"text": "🔌 Task t1 was stopped by a model-provider outage"},
])
assert _final_answer(agent_dir) == "the real answer"
def test_typed_delivery_rows_after_the_answer_are_still_skipped(tmp_path):
agent_dir = _agent_dir(tmp_path, [
{"direction": "out", "chat_id": HIDDEN_CHAT_ID, "text": "the real answer"},
{"direction": "out", "chat_id": HIDDEN_CHAT_ID, "type": "task_summary",
"text": "Done with warnings"},
])
assert _final_answer(agent_dir) == "the real answer"
def test_the_extraction_rule_every_producer_must_respect(tmp_path):
"""States the rule the rest of this suite enforces, so it cannot be misread.
The reader takes the LAST untyped outbound row, whatever it says. That is
not a contract anyone should satisfy — it is the constraint every notice
producer has to route around, by being a progress row, a typed row or a
system row. The assertion below is what happens when a producer does NOT,
which is why the deep-review acknowledgement was typed rather than left
plain once the class fix let it reach this partition.
"""
agent_dir = _agent_dir(tmp_path, [
{"direction": "out", "chat_id": HIDDEN_CHAT_ID, "text": "the real answer"},
{"direction": "out", "chat_id": HIDDEN_CHAT_ID, "text": "⚠️ an untyped later notice"},
])
assert _final_answer(agent_dir) == "⚠️ an untyped later notice"
def test_the_deep_review_acknowledgement_is_typed_and_cannot_be_read_as_an_answer():
"""The one plain outbound producer the class fix newly pointed at chat 0.
`/review` sent with an explicit chat 0 is now answered in the hidden
partition instead of being re-routed to the owner. Its acknowledgement is a
SYSTEM row, so a run's recorded answer cannot be replaced by it.
"""
source = (REPO / "supervisor/queue.py").read_text(encoding="utf-8")
ack = next(line for line in source.splitlines() if "Deep self-review queued" in line)
assert 'role="system"' in ack and "system_type=" in ack, ack.strip()
def test_the_degraded_reason_change_moves_no_benchmark_classification():
"""C3 changes the VALUE of an existing key, never a bucket.
Two benchmark readers consume ``degraded_reason``: the shared result index,
and the Terminal-Bench installed agent, which writes it into
``ouroboros-run-summary.json`` AND prints that object to stdout on every
trial. Both must keep deciding ``infra_failed``/``truncated`` from their own
inputs, so a newly specific reason cannot silently re-bucket a run.
"""
index = (REPO / "devtools/benchmarks/common/result_index.py").read_text(encoding="utf-8")
harbor = (REPO / "devtools/benchmarks/terminal_bench/harbor_installed_agent.py").read_text(encoding="utf-8")
assert '"degraded_reason"' in index and '"degraded_reason"' in harbor
# The classification inputs are independent of the reason string.
assert 'infra_failed' in harbor and 'truncated' in harbor
for line in harbor.splitlines():
if ("infra_failed" in line or "truncated" in line) and "=" in line:
assert "degraded_reason" not in line, (
"a benchmark bucket must not be derived from the degraded reason: " + line.strip()
)
def test_the_two_toasts_this_sprint_touched_stay_out_of_a_headless_progress_log():
"""ATIF republishes progress rows as the agent's own narration.
The scheduled-subagent and subagent-rejection notices are host lines, not
the model's. They were dropped for chat 0 before this sprint and stay
dropped, now for a stated reason rather than by accident: the durable record
keeps the real address, the live toast needs a reader.
Scope, stated so this is not read as a general guarantee: OTHER host toasts
(the finalization-grace warning, for one) already reached the progress log
before this sprint and still do. They carry a HOST_NARRATION marker the
trajectory builder does not yet consult, which is a pre-existing gap in the
harness rather than something this change introduced — filed separately.
"""
# v7 split supervisor/events.py: _handle_schedule_task moved to
# events_schedule_task.py and the subagent-admission notice to
# events_subagent_admission.py. Same two lines, new owning leaves.
schedule = (REPO / "supervisor/events_schedule_task.py").read_text(encoding="utf-8")
assert "if _notice_chat is not None and _notice_chat != HIDDEN_CHAT_ID:" in schedule
admission = (REPO / "supervisor/events_subagent_admission.py").read_text(encoding="utf-8")
assert "if chat_id is None or chat_id == HIDDEN_CHAT_ID:" in admission
atif = (REPO / "devtools/benchmarks/terminal_bench/atif.py").read_text(encoding="utf-8")
assert 'progress.jsonl' in atif and 'narration_rows' in atif, (
"if ATIF stops reading progress.jsonl this guard can be revisited"
)
def test_the_real_terminal_producer_leaves_the_model_answer_as_the_hidden_partition_answer(
tmp_path, monkeypatch,
):
"""End to end, through the real producer and the real delivery seam, in chat 0.
The other tests in this file state the extraction rule over hand-written
rows. This one runs emit_task_results -> _handle_send_message ->
send_with_budget -> chat.jsonl -> atif._final_answer, so a change to WHICH
rows the host writes, or to HOW they are authored and typed, is caught here
instead of in a Terminal-Bench trajectory. Exactly one untyped outbound row
may exist and it must be the model's own answer; every host disclosure has
to be a system row, a typed row or no row at all.
"""
from collections import deque
from types import SimpleNamespace
from ouroboros import agent_task_pipeline as pipeline
from ouroboros.task_finalization import set_terminal_host_notice
from ouroboros.utils import append_jsonl
from supervisor import events_chat_delivery as delivery, message_bus
answer = "The model's own last word."
notice = "⚠️ Plan review is still open (DEGRADED).\n\n⚠️ DEFERRED CHILD RESULTS: child1."
data = tmp_path / "ouroboros-data"
(data / "logs").mkdir(parents=True)
monkeypatch.setattr(pipeline, "_run_post_task_processing_async", lambda *_a, **_kw: None)
usage = {"terminal_origin": "model_final"}
set_terminal_host_notice(usage, notice)
task = {"id": "bench-root", "type": "task", "chat_id": HIDDEN_CHAT_ID, "text": "solve it"}
pending = []
pipeline.emit_task_results(
SimpleNamespace(drive_root=data, repo_dir=data), None, None,
pending, task, answer, usage, {"tool_calls": [], "reasoning_notes": []},
start_time=0.0, drive_logs=data / "logs",
)
event = next(row for row in pending if row["type"] == "send_message")
assert event["chat_id"] == HIDDEN_CHAT_ID
bridge = message_bus.LocalChatBridge({})
bridge._broadcast_fn = lambda _frame: None
monkeypatch.setattr(message_bus, "DATA_DIR", data)
monkeypatch.setattr(message_bus, "get_bridge", lambda: bridge)
monkeypatch.setattr(message_bus, "load_state", lambda: {"owner_id": 7})
monkeypatch.setattr(message_bus, "_advance_project_visible_revision", lambda _chat: None)
monkeypatch.setattr(message_bus, "publish_event", lambda *_a, **_kw: None)
monkeypatch.setattr(delivery, "_DELIVERED_MESSAGE_IDS", deque(maxlen=256))
delivery._handle_send_message(event, SimpleNamespace(
DRIVE_ROOT=data, RUNNING={}, append_jsonl=append_jsonl,
send_with_budget=message_bus.send_with_budget,
))
rows = [json.loads(line) for line
in (data / "logs" / "chat.jsonl").read_text(encoding="utf-8").splitlines() if line.strip()]
untyped_outbound = [row["text"] for row in rows
if row["direction"] == "out" and not row.get("type")]
assert untyped_outbound == [answer]
assert _final_answer(tmp_path) == answer