Prove Stop on a genuinely running root, and keep the marker off addressing frames

The two Stop browser tests replayed a progress row with the host-attested
marker and expected Stop with no live source, a premise retired on 09.09
(a replayed card offers Stop once the census or a durable record vouches
for the root); both were red on the base since then. The fixtures now seed a
queued root that the pool dispatches at boot and the mock model holds inside
its first completion (an interruptible hold released in the test's finally),
so the census vouches for a genuinely running root and the frozen dropdown
flows run against it.

The direct turn's Stop marker rides its work frames only: an addressing call
carries routing_action and no marker, so an addressing-only turn keeps no
block (owner 11.09), which the addressing browser lane proves unchanged.

Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com>
This commit is contained in:
Ouroboros 2026-09-15 23:42:13 +03:00
parent b840799b38
commit c4bdb6794a
6 changed files with 83 additions and 13 deletions

View file

@ -199,11 +199,7 @@ class ChatOutbound(TypedDict):
# endpoint uses, supervisor.workers.direct_chat_turn); never a subagent
# frame, never an ephemeral decision turn. Gates the UI "Cancel run" action.
cancelable: NotRequired[bool]
# The lane fact of a direct conversation turn, stamped by value on the
# turn's own frames (supervisor/log_addressing.py TurnEventQueue) so the
# chat block reads it before any census: present and true only for a
# direct turn's progress/tool frames and on every task_done.
_is_direct_chat: NotRequired[bool]
_is_direct_chat: NotRequired[bool] # lane fact stamped on a direct turn's own frames
# Monetary projections are nullable when the physical-attempt ledger cannot
# be read. ``None`` is deliberately distinct from a confirmed $0 result.
# C2 (owner 10=B) named these the HONEST names — accounted upper bounds,
@ -645,11 +641,10 @@ class FsDirsResponse(TypedDict):
class TaskNamedOutbound(TypedDict):
"""Outbound notice that a project name was coined for a task's card (the inline
naming of a turn-into-project conversion; direct turns are not named in the
background). The client sets the live card's title to ``suggested_name``. Not
chat-scoped — carries only ``task_id`` and is a no-op unless a thread already
holds that card."""
"""Outbound notice that a project name was coined for a task's card (inline naming of
a turn-into-project conversion; direct turns are not named in the background). The
client sets the live card's title to ``suggested_name``. Not chat-scoped — carries
only ``task_id`` and is a no-op unless a thread already holds that card."""
type: Literal["task_named"]
task_id: str

View file

@ -206,8 +206,13 @@ class TurnEventQueue:
# it rides its narration rows (events_chat_delivery stamps those
# through the same registry): a turn that only calls tools
# offers Stop on the block its rows already justify, and a turn
# that does neither keeps no block to hang a Stop on.
if data.get("type") in ("tool_call_started", "tool_call_finished"):
# that does neither keeps no block to hang a Stop on. An
# addressing call (``routing_action``) is a receipt, not work:
# it carries no marker, so an addressing-only turn keeps no block.
if (
data.get("type") in ("tool_call_started", "tool_call_finished")
and not data.get("routing_action")
):
data.setdefault("cancelable", True)
return item

View file

@ -2,8 +2,16 @@ from __future__ import annotations
import json
import threading
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
# Seconds the mock model holds every completion before answering (0 = answer at
# once). A browser fixture that needs a GENUINELY running root sets this so the
# dispatched task stays in its first model call for the test's duration; setting
# HOLD_RELEASE lets a held handler answer at once (fixture teardown joins it).
HOLD_SECONDS = 0.0
HOLD_RELEASE = threading.Event()
class MockLLMServer:
def __init__(self):
@ -28,6 +36,8 @@ class _Handler(BaseHTTPRequestHandler):
if self.path != "/v1/chat/completions":
self.send_error(404)
return
if HOLD_SECONDS > 0:
HOLD_RELEASE.wait(HOLD_SECONDS)
payload = {
"id": "mock-chat",
"object": "chat.completion",

View file

@ -471,6 +471,9 @@ def test_turn_event_queue_stamps_by_value_at_the_producer():
assert proxy.stamp({"type": "task_done", "task_id": "turn1", "_is_direct_chat": False})["_is_direct_chat"] is False
tool = proxy.stamp({"type": "log_event", "data": {"type": "tool_call_started", "task_id": "turn1", "tool": "read_file"}})
assert tool["data"]["cancelable"] is True and tool["data"]["_is_direct_chat"] is True
receipt = proxy.stamp({"type": "log_event", "data": {
"type": "tool_call_started", "task_id": "turn1", "tool": "promote_chat_to_task", "routing_action": "promote_chat_to_task"}})
assert "cancelable" not in receipt["data"] and receipt["data"]["_is_direct_chat"] is True
# Another task's event and an already-addressed event are left alone.
other = {"type": "log_event", "data": {"type": "x", "task_id": "other"}}

View file

@ -149,6 +149,49 @@ def _assert_menu_geometry(metrics: dict, *, placement: str | None = None) -> Non
@pytest.mark.ui_browser
def _hold_live_root(data_dir, task_id: str, *, hold_seconds: float = 90.0) -> None:
"""Make ``task_id`` a GENUINELY running managed root for the test's duration: a queued
row restored from the queue snapshot at boot, dispatched by the pool, and held inside
its first model call by the mock model (``fixtures_mock_llm.HOLD_SECONDS``). Since
54dfdc727 a replayed progress row's marker offers Stop only once the activity census
or a durable record vouches for the root, so the fixture must vouch, not just replay.
Reset the hold with ``_release_mock_model()`` in the test's ``finally``."""
import datetime as _dt
from tests import fixtures_mock_llm
from ouroboros.task_results import write_task_result
fixtures_mock_llm.HOLD_RELEASE.clear()
fixtures_mock_llm.HOLD_SECONDS = float(hold_seconds)
(data_dir / "state").mkdir(parents=True, exist_ok=True)
live_task = {
"id": task_id, "type": "task", "chat_id": 1, "priority": 0, "text": "the big thing",
"description": "the big thing", "objective": "the big thing", "title": "The big thing",
"root_task_id": task_id, "delegation_role": "root",
}
write_task_result(
data_dir, task_id, "scheduled", result="Task is queued.", description="the big thing",
objective="the big thing", chat_id=1, title="The big thing", delegation_role="root", root_task_id=task_id,
)
now = _dt.datetime.now(_dt.timezone.utc).isoformat()
(data_dir / "state" / "queue_snapshot.json").write_text(json.dumps({
"_schema_version": 1, "ts": now, "reason": "ui_smoke_seed",
"pending_count": 1, "running_count": 0, "reaping_count": 0,
"acceptance_fences": [], "budget_root_fences": [],
"pending": [{"id": task_id, "type": "task", "priority": 0, "attempt": 1, "queued_at": now,
"queue_seq": 1, "task": live_task}],
"running": [],
}), encoding="utf-8")
def _release_mock_model() -> None:
from tests import fixtures_mock_llm
fixtures_mock_llm.HOLD_SECONDS = 0.0
fixtures_mock_llm.HOLD_RELEASE.set()
def test_s3_chat_card_dropdown_hurry_and_soft_stop(direct_server_with_data):
"""Chat surface: frozen dropdown, no-chat hurry with idempotent request_id
retry, and the soft stop collapsing the pending menu to the escalation."""
@ -160,6 +203,8 @@ def test_s3_chat_card_dropdown_hurry_and_soft_stop(direct_server_with_data):
data_dir = direct_server_with_data["data_dir"]
logs_dir = data_dir / "logs"
logs_dir.mkdir(parents=True, exist_ok=True)
# Seed while no server runs (the live loop rewrites the queue snapshot every tick).
direct_server_with_data["stop_server"]()
(logs_dir / "chat.jsonl").write_text("", encoding="utf-8")
(logs_dir / "progress.jsonl").write_text(
json.dumps({
@ -168,6 +213,8 @@ def test_s3_chat_card_dropdown_hurry_and_soft_stop(direct_server_with_data):
}) + "\n",
encoding="utf-8",
)
_hold_live_root(data_dir, "live-root")
direct_server_with_data["start_server"]()
try:
with sync_playwright() as pw:
@ -480,6 +527,7 @@ def test_s3_chat_card_dropdown_hurry_and_soft_stop(direct_server_with_data):
)
finally:
browser.close()
_release_mock_model()
except PlaywrightError as exc:
if "Executable doesn't exist" in str(exc) or "playwright install" in str(exc).lower():
pytest.skip(str(exc))

View file

@ -3626,9 +3626,13 @@ def test_ui_smoke_cancel_run_button_eligibility_and_cancelled_state(direct_serve
data_dir = direct_server_with_data["data_dir"]
logs_dir = data_dir / "logs"
logs_dir.mkdir(parents=True, exist_ok=True)
# Seed while no server runs (the live loop rewrites the queue snapshot every tick).
direct_server_with_data["stop_server"]()
(logs_dir / "chat.jsonl").write_text("", encoding="utf-8")
rows = [
# Pooled live root: carries the supervisor's host-attested marker.
# Pooled live root: carries the supervisor's host-attested marker AND is
# genuinely running (dispatched at boot, held in its first model call) so
# the census vouches for it (the 09.09 rule).
{"ts": "2026-07-29T10:00:00+00:00", "chat_id": 1, "task_id": "live-root",
"content": "Working on the big thing", "cancelable": True},
# Direct-chat-turn shape: same card shape, NO marker -> no button.
@ -3663,6 +3667,10 @@ def test_ui_smoke_cancel_run_button_eligibility_and_cancelled_state(direct_serve
"execution": {"status": "cancelled"},
},
}) + "\n", encoding="utf-8")
from tests.test_s3_task_control_browser import _hold_live_root, _release_mock_model
_hold_live_root(data_dir, "live-root")
direct_server_with_data["start_server"]()
try:
with sync_playwright() as pw:
@ -3700,6 +3708,7 @@ def test_ui_smoke_cancel_run_button_eligibility_and_cancelled_state(direct_serve
page.screenshot(path=str(data_dir.parent / "cancel-run.png"), full_page=True)
finally:
browser.close()
_release_mock_model()
except PlaywrightError as exc:
if "Executable doesn't exist" in str(exc) or "playwright install" in str(exc).lower():
pytest.skip(str(exc))