mirror of
https://github.com/razzant/ouroboros.git
synced 2026-10-03 04:07:04 +00:00
The gate decision records why a plan review stayed open as a closed class on the typed execution axis (only some reviewers answered, none answered, answered but never closed), beside the existing awaiting case; the normal delivery rail stamps terminal_plan_review_open on the result exactly as the forced rails do. The owner cause table, in Python and in its browser twin, gains the sentences for those classes, for a task held by a blocking review, and for the standing limitations of an answer (a plan review still open at delivery, deferred sub-task results, an unrecovered tool failure, an internal error); phrases that named internal actors are reworded. The completion verdict becomes an assembler on both twins: one clause per fact, joined, never one fact chosen over another; the parity fixture carries the new production-shaped cases. The durable task_summary row carries the same clause as a reason_detail field for transports without a card. Forced prompts hand their one model call the same limitations as typed facts, priced by the existing reservation probe; a hold the owner's hurry released reaches the model once before its last word. Delivery no longer splits a terminal into an answer and a System notice: the terminal_host_notice field keeps its bytes on the result for every machine reader, and only the custody audit rides the send event as its own typed row. A Telegram-only owner, who has no task card, receives one short line on a non-clean root finish built from the stamped phase and the reason sentence; clean finishes stay opt-in. In the Reviews group a settled plan wave whose reviewers were too few for a verdict reads no verdict with its counts in the neutral tone instead of the stored DEGRADED word; reviewer rows name the model and quote the engine's reported sentence, and the slot id and failure code stay in the task detail and Logs. DESIGN.md and the architecture chapters state the same; two chapter byte budgets rise with their reasons.
227 lines
11 KiB
Python
227 lines
11 KiB
Python
"""Closing the chat-id truthiness class must not move a benchmark's answer.
|
|
|
|
Terminal-Bench reads a run's final answer straight out of ``chat.jsonl``: the
|
|
LAST untyped row with ``direction == "out"`` (``atif._final_answer``). It ALSO
|
|
reads ``progress.jsonl`` and publishes those rows as the agent's own narration
|
|
(``atif.build_trajectory``). Headless benchmark roots run in the hidden partition
|
|
(chat 0), which is exactly the partition the class fix stops dropping — so a
|
|
notice that now reaches chat 0 must be a typed or system chat row (invisible to
|
|
the answer reader), and a live host TOAST must not be written there at all, or a
|
|
supervisor line would be published as something the model said.
|
|
"""
|
|
|
|
import json
|
|
import pathlib
|
|
import sys
|
|
|
|
import pytest
|
|
|
|
REPO = pathlib.Path(__file__).resolve().parents[1]
|
|
if str(REPO) not in sys.path:
|
|
sys.path.insert(0, str(REPO))
|
|
|
|
from devtools.benchmarks.terminal_bench.atif import _final_answer # noqa: E402
|
|
from ouroboros.contracts.chat_id_policy import HIDDEN_CHAT_ID # noqa: E402
|
|
|
|
|
|
def _agent_dir(tmp_path, chat_rows):
|
|
data = tmp_path / "ouroboros-data"
|
|
(data / "logs").mkdir(parents=True)
|
|
(data / "logs" / "chat.jsonl").write_text(
|
|
"".join(json.dumps(row, ensure_ascii=False) + "\n" for row in chat_rows),
|
|
encoding="utf-8",
|
|
)
|
|
return tmp_path
|
|
|
|
|
|
def test_progress_notices_never_enter_the_answer_stream(tmp_path):
|
|
"""The reaper/scheduler toasts the class fix un-drops are progress rows.
|
|
|
|
``send_with_budget(is_progress=True)`` appends to progress.jsonl, so a
|
|
grace toast for a hidden-partition root cannot be mistaken for its answer.
|
|
"""
|
|
from types import SimpleNamespace
|
|
|
|
from supervisor import message_bus
|
|
|
|
data = tmp_path / "data"
|
|
(data / "logs").mkdir(parents=True)
|
|
sent = []
|
|
monkey = pytest.MonkeyPatch()
|
|
try:
|
|
monkey.setattr(message_bus, "DATA_DIR", data)
|
|
monkey.setattr(
|
|
message_bus, "_BRIDGE",
|
|
SimpleNamespace(send_message=lambda *a, **k: sent.append((a, k))),
|
|
)
|
|
message_bus.send_with_budget(
|
|
HIDDEN_CHAT_ID, "⏱️ Task t1 has been running 600s", is_progress=True, task_id="t1",
|
|
)
|
|
finally:
|
|
monkey.undo()
|
|
assert sent, "the notice still reaches the live bridge"
|
|
assert (data / "logs" / "progress.jsonl").exists()
|
|
assert not (data / "logs" / "chat.jsonl").exists()
|
|
|
|
|
|
def test_a_system_incident_row_is_not_read_as_the_answer(tmp_path):
|
|
"""A provider-outage notice now reaches chat 0 instead of vanishing (P1).
|
|
|
|
It is persisted as a SYSTEM row, so the trajectory still reports the task's
|
|
own last answer rather than the incident sentence.
|
|
"""
|
|
agent_dir = _agent_dir(tmp_path, [
|
|
{"direction": "in", "chat_id": HIDDEN_CHAT_ID, "text": "solve it"},
|
|
{"direction": "out", "chat_id": HIDDEN_CHAT_ID, "text": "the real answer"},
|
|
{"direction": "system", "chat_id": HIDDEN_CHAT_ID, "type": "terminal_incident",
|
|
"text": "🔌 Task t1 was stopped by a model-provider outage"},
|
|
])
|
|
assert _final_answer(agent_dir) == "the real answer"
|
|
|
|
|
|
def test_typed_delivery_rows_after_the_answer_are_still_skipped(tmp_path):
|
|
agent_dir = _agent_dir(tmp_path, [
|
|
{"direction": "out", "chat_id": HIDDEN_CHAT_ID, "text": "the real answer"},
|
|
{"direction": "out", "chat_id": HIDDEN_CHAT_ID, "type": "task_summary",
|
|
"text": "Done with warnings"},
|
|
])
|
|
assert _final_answer(agent_dir) == "the real answer"
|
|
|
|
|
|
def test_the_extraction_rule_every_producer_must_respect(tmp_path):
|
|
"""States the rule the rest of this suite enforces, so it cannot be misread.
|
|
|
|
The reader takes the LAST untyped outbound row, whatever it says. That is
|
|
not a contract anyone should satisfy — it is the constraint every notice
|
|
producer has to route around, by being a progress row, a typed row or a
|
|
system row. The assertion below is what happens when a producer does NOT,
|
|
which is why the deep-review acknowledgement was typed rather than left
|
|
plain once the class fix let it reach this partition.
|
|
"""
|
|
agent_dir = _agent_dir(tmp_path, [
|
|
{"direction": "out", "chat_id": HIDDEN_CHAT_ID, "text": "the real answer"},
|
|
{"direction": "out", "chat_id": HIDDEN_CHAT_ID, "text": "⚠️ an untyped later notice"},
|
|
])
|
|
assert _final_answer(agent_dir) == "⚠️ an untyped later notice"
|
|
|
|
|
|
def test_the_deep_review_acknowledgement_is_typed_and_cannot_be_read_as_an_answer():
|
|
"""The one plain outbound producer the class fix newly pointed at chat 0.
|
|
|
|
`/review` sent with an explicit chat 0 is now answered in the hidden
|
|
partition instead of being re-routed to the owner. Its acknowledgement is a
|
|
SYSTEM row, so a run's recorded answer cannot be replaced by it.
|
|
"""
|
|
source = (REPO / "supervisor/queue.py").read_text(encoding="utf-8")
|
|
ack = next(line for line in source.splitlines() if "Deep self-review queued" in line)
|
|
assert 'role="system"' in ack and "system_type=" in ack, ack.strip()
|
|
|
|
|
|
def test_the_degraded_reason_change_moves_no_benchmark_classification():
|
|
"""C3 changes the VALUE of an existing key, never a bucket.
|
|
|
|
Two benchmark readers consume ``degraded_reason``: the shared result index,
|
|
and the Terminal-Bench installed agent, which writes it into
|
|
``ouroboros-run-summary.json`` AND prints that object to stdout on every
|
|
trial. Both must keep deciding ``infra_failed``/``truncated`` from their own
|
|
inputs, so a newly specific reason cannot silently re-bucket a run.
|
|
"""
|
|
index = (REPO / "devtools/benchmarks/common/result_index.py").read_text(encoding="utf-8")
|
|
harbor = (REPO / "devtools/benchmarks/terminal_bench/harbor_installed_agent.py").read_text(encoding="utf-8")
|
|
|
|
assert '"degraded_reason"' in index and '"degraded_reason"' in harbor
|
|
# The classification inputs are independent of the reason string.
|
|
assert 'infra_failed' in harbor and 'truncated' in harbor
|
|
for line in harbor.splitlines():
|
|
if ("infra_failed" in line or "truncated" in line) and "=" in line:
|
|
assert "degraded_reason" not in line, (
|
|
"a benchmark bucket must not be derived from the degraded reason: " + line.strip()
|
|
)
|
|
|
|
|
|
def test_the_two_toasts_this_sprint_touched_stay_out_of_a_headless_progress_log():
|
|
"""ATIF republishes progress rows as the agent's own narration.
|
|
|
|
The scheduled-subagent and subagent-rejection notices are host lines, not
|
|
the model's. They were dropped for chat 0 before this sprint and stay
|
|
dropped, now for a stated reason rather than by accident: the durable record
|
|
keeps the real address, the live toast needs a reader.
|
|
|
|
Scope, stated so this is not read as a general guarantee: OTHER host toasts
|
|
(the finalization-grace warning, for one) already reached the progress log
|
|
before this sprint and still do. They carry a HOST_NARRATION marker the
|
|
trajectory builder does not yet consult, which is a pre-existing gap in the
|
|
harness rather than something this change introduced — filed separately.
|
|
"""
|
|
# v7 split supervisor/events.py: _handle_schedule_task moved to
|
|
# events_schedule_task.py and the subagent-admission notice to
|
|
# events_subagent_admission.py. Same two lines, new owning leaves.
|
|
schedule = (REPO / "supervisor/events_schedule_task.py").read_text(encoding="utf-8")
|
|
assert "if _notice_chat is not None and _notice_chat != HIDDEN_CHAT_ID:" in schedule
|
|
admission = (REPO / "supervisor/events_subagent_admission.py").read_text(encoding="utf-8")
|
|
assert "if chat_id is None or chat_id == HIDDEN_CHAT_ID:" in admission
|
|
|
|
atif = (REPO / "devtools/benchmarks/terminal_bench/atif.py").read_text(encoding="utf-8")
|
|
assert 'progress.jsonl' in atif and 'narration_rows' in atif, (
|
|
"if ATIF stops reading progress.jsonl this guard can be revisited"
|
|
)
|
|
|
|
|
|
def test_the_real_terminal_producer_leaves_the_model_answer_as_the_hidden_partition_answer(
|
|
tmp_path, monkeypatch,
|
|
):
|
|
"""End to end, through the real producer and the real delivery seam, in chat 0.
|
|
|
|
The other tests in this file state the extraction rule over hand-written
|
|
rows. This one runs emit_task_results -> _handle_send_message ->
|
|
send_with_budget -> chat.jsonl -> atif._final_answer, so a change to WHICH
|
|
rows the host writes, or to HOW they are authored and typed, is caught here
|
|
instead of in a Terminal-Bench trajectory. Exactly one untyped outbound row
|
|
may exist and it must be the model's own answer; every host disclosure has
|
|
to be a system row, a typed row or no row at all.
|
|
"""
|
|
from collections import deque
|
|
from types import SimpleNamespace
|
|
|
|
from ouroboros import agent_task_pipeline as pipeline
|
|
from ouroboros.task_finalization import set_terminal_host_notice
|
|
from ouroboros.utils import append_jsonl
|
|
from supervisor import events_chat_delivery as delivery, message_bus
|
|
|
|
answer = "The model's own last word."
|
|
notice = "⚠️ Plan review is still open (DEGRADED).\n\n⚠️ DEFERRED CHILD RESULTS: child1."
|
|
data = tmp_path / "ouroboros-data"
|
|
(data / "logs").mkdir(parents=True)
|
|
|
|
monkeypatch.setattr(pipeline, "_run_post_task_processing_async", lambda *_a, **_kw: None)
|
|
usage = {"terminal_origin": "model_final"}
|
|
set_terminal_host_notice(usage, notice)
|
|
task = {"id": "bench-root", "type": "task", "chat_id": HIDDEN_CHAT_ID, "text": "solve it"}
|
|
pending = []
|
|
pipeline.emit_task_results(
|
|
SimpleNamespace(drive_root=data, repo_dir=data), None, None,
|
|
pending, task, answer, usage, {"tool_calls": [], "reasoning_notes": []},
|
|
start_time=0.0, drive_logs=data / "logs",
|
|
)
|
|
event = next(row for row in pending if row["type"] == "send_message")
|
|
assert event["chat_id"] == HIDDEN_CHAT_ID
|
|
|
|
bridge = message_bus.LocalChatBridge({})
|
|
bridge._broadcast_fn = lambda _frame: None
|
|
monkeypatch.setattr(message_bus, "DATA_DIR", data)
|
|
monkeypatch.setattr(message_bus, "get_bridge", lambda: bridge)
|
|
monkeypatch.setattr(message_bus, "load_state", lambda: {"owner_id": 7})
|
|
monkeypatch.setattr(message_bus, "_advance_project_visible_revision", lambda _chat: None)
|
|
monkeypatch.setattr(message_bus, "publish_event", lambda *_a, **_kw: None)
|
|
monkeypatch.setattr(delivery, "_DELIVERED_MESSAGE_IDS", deque(maxlen=256))
|
|
delivery._handle_send_message(event, SimpleNamespace(
|
|
DRIVE_ROOT=data, RUNNING={}, append_jsonl=append_jsonl,
|
|
send_with_budget=message_bus.send_with_budget,
|
|
))
|
|
|
|
rows = [json.loads(line) for line
|
|
in (data / "logs" / "chat.jsonl").read_text(encoding="utf-8").splitlines() if line.strip()]
|
|
untyped_outbound = [row["text"] for row in rows
|
|
if row["direction"] == "out" and not row.get("type")]
|
|
assert untyped_outbound == [answer]
|
|
assert _final_answer(tmp_path) == answer
|