mirror of
https://github.com/razzant/ouroboros.git
synced 2026-10-03 04:07:04 +00:00
Prove Stop on a genuinely running root, and keep the marker off addressing frames
The two Stop browser tests replayed a progress row with the host-attested marker and expected Stop with no live source, a premise retired on 09.09 (a replayed card offers Stop once the census or a durable record vouches for the root); both were red on the base since then. The fixtures now seed a queued root that the pool dispatches at boot and the mock model holds inside its first completion (an interruptible hold released in the test's finally), so the census vouches for a genuinely running root and the frozen dropdown flows run against it. The direct turn's Stop marker rides its work frames only: an addressing call carries routing_action and no marker, so an addressing-only turn keeps no block (owner 11.09), which the addressing browser lane proves unchanged. Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com>
This commit is contained in:
parent
b840799b38
commit
c4bdb6794a
6 changed files with 83 additions and 13 deletions
|
|
@ -199,11 +199,7 @@ class ChatOutbound(TypedDict):
|
|||
# endpoint uses, supervisor.workers.direct_chat_turn); never a subagent
|
||||
# frame, never an ephemeral decision turn. Gates the UI "Cancel run" action.
|
||||
cancelable: NotRequired[bool]
|
||||
# The lane fact of a direct conversation turn, stamped by value on the
|
||||
# turn's own frames (supervisor/log_addressing.py TurnEventQueue) so the
|
||||
# chat block reads it before any census: present and true only for a
|
||||
# direct turn's progress/tool frames and on every task_done.
|
||||
_is_direct_chat: NotRequired[bool]
|
||||
_is_direct_chat: NotRequired[bool] # lane fact stamped on a direct turn's own frames
|
||||
# Monetary projections are nullable when the physical-attempt ledger cannot
|
||||
# be read. ``None`` is deliberately distinct from a confirmed $0 result.
|
||||
# C2 (owner 10=B) named these the HONEST names — accounted upper bounds,
|
||||
|
|
@ -645,11 +641,10 @@ class FsDirsResponse(TypedDict):
|
|||
|
||||
|
||||
class TaskNamedOutbound(TypedDict):
|
||||
"""Outbound notice that a project name was coined for a task's card (the inline
|
||||
naming of a turn-into-project conversion; direct turns are not named in the
|
||||
background). The client sets the live card's title to ``suggested_name``. Not
|
||||
chat-scoped — carries only ``task_id`` and is a no-op unless a thread already
|
||||
holds that card."""
|
||||
"""Outbound notice that a project name was coined for a task's card (inline naming of
|
||||
a turn-into-project conversion; direct turns are not named in the background). The
|
||||
client sets the live card's title to ``suggested_name``. Not chat-scoped — carries
|
||||
only ``task_id`` and is a no-op unless a thread already holds that card."""
|
||||
|
||||
type: Literal["task_named"]
|
||||
task_id: str
|
||||
|
|
|
|||
|
|
@ -206,8 +206,13 @@ class TurnEventQueue:
|
|||
# it rides its narration rows (events_chat_delivery stamps those
|
||||
# through the same registry): a turn that only calls tools
|
||||
# offers Stop on the block its rows already justify, and a turn
|
||||
# that does neither keeps no block to hang a Stop on.
|
||||
if data.get("type") in ("tool_call_started", "tool_call_finished"):
|
||||
# that does neither keeps no block to hang a Stop on. An
|
||||
# addressing call (``routing_action``) is a receipt, not work:
|
||||
# it carries no marker, so an addressing-only turn keeps no block.
|
||||
if (
|
||||
data.get("type") in ("tool_call_started", "tool_call_finished")
|
||||
and not data.get("routing_action")
|
||||
):
|
||||
data.setdefault("cancelable", True)
|
||||
return item
|
||||
|
||||
|
|
|
|||
|
|
@ -2,8 +2,16 @@ from __future__ import annotations
|
|||
|
||||
import json
|
||||
import threading
|
||||
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
|
||||
# Seconds the mock model holds every completion before answering (0 = answer at
|
||||
# once). A browser fixture that needs a GENUINELY running root sets this so the
|
||||
# dispatched task stays in its first model call for the test's duration; setting
|
||||
# HOLD_RELEASE lets a held handler answer at once (fixture teardown joins it).
|
||||
HOLD_SECONDS = 0.0
|
||||
HOLD_RELEASE = threading.Event()
|
||||
|
||||
|
||||
class MockLLMServer:
|
||||
def __init__(self):
|
||||
|
|
@ -28,6 +36,8 @@ class _Handler(BaseHTTPRequestHandler):
|
|||
if self.path != "/v1/chat/completions":
|
||||
self.send_error(404)
|
||||
return
|
||||
if HOLD_SECONDS > 0:
|
||||
HOLD_RELEASE.wait(HOLD_SECONDS)
|
||||
payload = {
|
||||
"id": "mock-chat",
|
||||
"object": "chat.completion",
|
||||
|
|
|
|||
|
|
@ -471,6 +471,9 @@ def test_turn_event_queue_stamps_by_value_at_the_producer():
|
|||
assert proxy.stamp({"type": "task_done", "task_id": "turn1", "_is_direct_chat": False})["_is_direct_chat"] is False
|
||||
tool = proxy.stamp({"type": "log_event", "data": {"type": "tool_call_started", "task_id": "turn1", "tool": "read_file"}})
|
||||
assert tool["data"]["cancelable"] is True and tool["data"]["_is_direct_chat"] is True
|
||||
receipt = proxy.stamp({"type": "log_event", "data": {
|
||||
"type": "tool_call_started", "task_id": "turn1", "tool": "promote_chat_to_task", "routing_action": "promote_chat_to_task"}})
|
||||
assert "cancelable" not in receipt["data"] and receipt["data"]["_is_direct_chat"] is True
|
||||
|
||||
# Another task's event and an already-addressed event are left alone.
|
||||
other = {"type": "log_event", "data": {"type": "x", "task_id": "other"}}
|
||||
|
|
|
|||
|
|
@ -149,6 +149,49 @@ def _assert_menu_geometry(metrics: dict, *, placement: str | None = None) -> Non
|
|||
|
||||
|
||||
@pytest.mark.ui_browser
|
||||
|
||||
def _hold_live_root(data_dir, task_id: str, *, hold_seconds: float = 90.0) -> None:
|
||||
"""Make ``task_id`` a GENUINELY running managed root for the test's duration: a queued
|
||||
row restored from the queue snapshot at boot, dispatched by the pool, and held inside
|
||||
its first model call by the mock model (``fixtures_mock_llm.HOLD_SECONDS``). Since
|
||||
54dfdc727 a replayed progress row's marker offers Stop only once the activity census
|
||||
or a durable record vouches for the root, so the fixture must vouch, not just replay.
|
||||
Reset the hold with ``_release_mock_model()`` in the test's ``finally``."""
|
||||
import datetime as _dt
|
||||
|
||||
from tests import fixtures_mock_llm
|
||||
from ouroboros.task_results import write_task_result
|
||||
|
||||
fixtures_mock_llm.HOLD_RELEASE.clear()
|
||||
fixtures_mock_llm.HOLD_SECONDS = float(hold_seconds)
|
||||
(data_dir / "state").mkdir(parents=True, exist_ok=True)
|
||||
live_task = {
|
||||
"id": task_id, "type": "task", "chat_id": 1, "priority": 0, "text": "the big thing",
|
||||
"description": "the big thing", "objective": "the big thing", "title": "The big thing",
|
||||
"root_task_id": task_id, "delegation_role": "root",
|
||||
}
|
||||
write_task_result(
|
||||
data_dir, task_id, "scheduled", result="Task is queued.", description="the big thing",
|
||||
objective="the big thing", chat_id=1, title="The big thing", delegation_role="root", root_task_id=task_id,
|
||||
)
|
||||
now = _dt.datetime.now(_dt.timezone.utc).isoformat()
|
||||
(data_dir / "state" / "queue_snapshot.json").write_text(json.dumps({
|
||||
"_schema_version": 1, "ts": now, "reason": "ui_smoke_seed",
|
||||
"pending_count": 1, "running_count": 0, "reaping_count": 0,
|
||||
"acceptance_fences": [], "budget_root_fences": [],
|
||||
"pending": [{"id": task_id, "type": "task", "priority": 0, "attempt": 1, "queued_at": now,
|
||||
"queue_seq": 1, "task": live_task}],
|
||||
"running": [],
|
||||
}), encoding="utf-8")
|
||||
|
||||
|
||||
def _release_mock_model() -> None:
|
||||
from tests import fixtures_mock_llm
|
||||
|
||||
fixtures_mock_llm.HOLD_SECONDS = 0.0
|
||||
fixtures_mock_llm.HOLD_RELEASE.set()
|
||||
|
||||
|
||||
def test_s3_chat_card_dropdown_hurry_and_soft_stop(direct_server_with_data):
|
||||
"""Chat surface: frozen dropdown, no-chat hurry with idempotent request_id
|
||||
retry, and the soft stop collapsing the pending menu to the escalation."""
|
||||
|
|
@ -160,6 +203,8 @@ def test_s3_chat_card_dropdown_hurry_and_soft_stop(direct_server_with_data):
|
|||
data_dir = direct_server_with_data["data_dir"]
|
||||
logs_dir = data_dir / "logs"
|
||||
logs_dir.mkdir(parents=True, exist_ok=True)
|
||||
# Seed while no server runs (the live loop rewrites the queue snapshot every tick).
|
||||
direct_server_with_data["stop_server"]()
|
||||
(logs_dir / "chat.jsonl").write_text("", encoding="utf-8")
|
||||
(logs_dir / "progress.jsonl").write_text(
|
||||
json.dumps({
|
||||
|
|
@ -168,6 +213,8 @@ def test_s3_chat_card_dropdown_hurry_and_soft_stop(direct_server_with_data):
|
|||
}) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
_hold_live_root(data_dir, "live-root")
|
||||
direct_server_with_data["start_server"]()
|
||||
|
||||
try:
|
||||
with sync_playwright() as pw:
|
||||
|
|
@ -480,6 +527,7 @@ def test_s3_chat_card_dropdown_hurry_and_soft_stop(direct_server_with_data):
|
|||
)
|
||||
finally:
|
||||
browser.close()
|
||||
_release_mock_model()
|
||||
except PlaywrightError as exc:
|
||||
if "Executable doesn't exist" in str(exc) or "playwright install" in str(exc).lower():
|
||||
pytest.skip(str(exc))
|
||||
|
|
|
|||
|
|
@ -3626,9 +3626,13 @@ def test_ui_smoke_cancel_run_button_eligibility_and_cancelled_state(direct_serve
|
|||
data_dir = direct_server_with_data["data_dir"]
|
||||
logs_dir = data_dir / "logs"
|
||||
logs_dir.mkdir(parents=True, exist_ok=True)
|
||||
# Seed while no server runs (the live loop rewrites the queue snapshot every tick).
|
||||
direct_server_with_data["stop_server"]()
|
||||
(logs_dir / "chat.jsonl").write_text("", encoding="utf-8")
|
||||
rows = [
|
||||
# Pooled live root: carries the supervisor's host-attested marker.
|
||||
# Pooled live root: carries the supervisor's host-attested marker AND is
|
||||
# genuinely running (dispatched at boot, held in its first model call) so
|
||||
# the census vouches for it (the 09.09 rule).
|
||||
{"ts": "2026-07-29T10:00:00+00:00", "chat_id": 1, "task_id": "live-root",
|
||||
"content": "Working on the big thing", "cancelable": True},
|
||||
# Direct-chat-turn shape: same card shape, NO marker -> no button.
|
||||
|
|
@ -3663,6 +3667,10 @@ def test_ui_smoke_cancel_run_button_eligibility_and_cancelled_state(direct_serve
|
|||
"execution": {"status": "cancelled"},
|
||||
},
|
||||
}) + "\n", encoding="utf-8")
|
||||
from tests.test_s3_task_control_browser import _hold_live_root, _release_mock_model
|
||||
|
||||
_hold_live_root(data_dir, "live-root")
|
||||
direct_server_with_data["start_server"]()
|
||||
|
||||
try:
|
||||
with sync_playwright() as pw:
|
||||
|
|
@ -3700,6 +3708,7 @@ def test_ui_smoke_cancel_run_button_eligibility_and_cancelled_state(direct_serve
|
|||
page.screenshot(path=str(data_dir.parent / "cancel-run.png"), full_page=True)
|
||||
finally:
|
||||
browser.close()
|
||||
_release_mock_model()
|
||||
except PlaywrightError as exc:
|
||||
if "Executable doesn't exist" in str(exc) or "playwright install" in str(exc).lower():
|
||||
pytest.skip(str(exc))
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue