WIP: preserve reviewed consumer corrections; emergency-stop residual remains open

This commit is contained in:
Ouroboros 2026-09-27 03:25:09 +03:00
parent 81c71cfa27
commit e91d7e2b1e
38 changed files with 1532 additions and 201 deletions

View file

@ -623,6 +623,7 @@ def _gate_tool_trace(data_dir: Path, ouro_task_id: str, latest_status: Any = Non
log_path = data_dir / "state" / "headless_tasks" / ouro_task_id / "data" / "logs" / "tools.jsonl"
if not (ouro_task_id and log_path.is_file()):
return trace
rows = []
for line in log_path.read_text(encoding="utf-8", errors="replace").splitlines():
line = line.strip()
if not line:
@ -631,15 +632,24 @@ def _gate_tool_trace(data_dir: Path, ouro_task_id: str, latest_status: Any = Non
row = json.loads(line)
except Exception:
continue
if not isinstance(row, dict) or row.get("type") != "tool_call":
if not isinstance(row, dict) or row.get("type") not in {"tool_call", "tool_call_started", "tool_call_timeout"}:
continue
rows.append(row)
from ouroboros.tool_call_log import logical_calls
for call in logical_calls(rows):
row = call.get("settled") or call.get("started") or call.get("wait_ended") or {}
tool = str(row.get("tool") or "")
if not tool.startswith(prefix):
continue
trace.append({
"tool": tool[len(prefix):],
"args": row.get("args"),
"is_error": bool(row.get("is_error")),
"is_error": bool(row.get("is_error")) if call.get("settled") else None,
"state": call["state"],
"wait_ended": bool(call.get("wait_ended")),
"settled": bool(call.get("settled")),
"invocation_id": call.get("invocation_id"),
})
except Exception: # noqa: BLE001 - a sidecar must never change the flow
pass
@ -1105,6 +1115,7 @@ def _collect_budget_counters(data_dir: Path, latest: dict[str, Any], ouro_task_i
screenshots = gui = remote_exec = total = 0
src = log_path if log_path.is_file() else (fallback if fallback.is_file() else None)
if src is not None:
rows = []
for line in src.read_text(encoding="utf-8", errors="replace").splitlines():
line = line.strip()
if not line:
@ -1113,10 +1124,15 @@ def _collect_budget_counters(data_dir: Path, latest: dict[str, Any], ouro_task_i
row = json.loads(line)
except Exception:
continue
if not isinstance(row, dict) or row.get("type") != "tool_call":
if not isinstance(row, dict) or row.get("type") not in {"tool_call", "tool_call_started", "tool_call_timeout"}:
continue
if src is fallback and str(row.get("task_id") or "") != ouro_task_id:
continue
rows.append(row)
from ouroboros.tool_call_log import logical_calls
for call in logical_calls(rows):
row = call.get("settled") or call.get("started") or call.get("wait_ended") or {}
tool = str(row.get("tool") or "")
if not tool.startswith(prefix):
continue

View file

@ -762,22 +762,22 @@ patched in place; Expand shows the per-tool counts (`read_file ×3 ·
web_search`); its phase is `calling` while a tracked call is still running,
`warn` once a call failed, `result` otherwise, and the row says that phase in
ink rather than in extra words. A failed or timed-out call keeps
its own error row (content) and is counted in the evidence total.
`web/modules/chat_activity.js::toolEvidenceView` builds that row for the live
path and for the recorded metrics alike, so at rest the row carries the same
counts and names live, on reload and on reconnect; a cold reload mints it from
the metrics, so it carries the metrics' time and sits where the metrics
arrived, while a reconnect keeps the live position. Live it derives from the
observed call frames (once per call identity; identical repeats without an id
collapse into one), and the host's metrics replace those numbers as they
arrive, field by field: a fact that states a total says nothing about the
routing or error count, so it can neither erase one nor reclassify a receipt
row into content, and a call frame after the terminal changes nothing. Block presence is the same live, on
reload and on reconnect (a turn that moved itself into a Project with
`ensure_project_scope` is the exception: its block and answer live in the
Project room, and Main replays only the owner message and the Started
annotation); a child card reads the same voice rule for its own notes and folds
its calls live, but replays no evidence row.
its own diagnostic row and counts once. Wait end and operation settlement are
independent facts: late success retires the provisional timeout notice but keeps
“wait ended” in the evidence row; late failure keeps its operation error. Either
arrival order produces the same outcome. A historical start alone means outcome
unknown, never Running or Failed.
`web/modules/chat_activity.js::toolEvidenceView` builds the same row live and on
replay. Host invocation IDs join start/wait/settlement; legacy observations without
sufficient identity remain separate even when names and arguments match. Host
metrics fill absent counts field by field; canonical per-invocation evidence
reconstructs later settlements on history/reconnect without resurrecting frozen
wait errors. Reads are bounded and carry coverage; absent evidence is not proof
of success. Typed tool evidence after task terminal updates counts and diagnostics
on both root and child cards, preserving terminal task phase and controls.
The row keeps its live position; cold history places it with the summary. Block
presence is consistent across reload and reconnect. A turn moved into a Project
with `ensure_project_scope` lives there; Main retains its Started annotation.
`N notes` in the collapsed header counts timeline items, the evidence row
among them.

View file

@ -6,7 +6,7 @@ Ordinary window close preserves this installation's shared Claudexor daemon and
The owner's manual Restart (`/restart`, the chat Restart button) is a clean stop of everything the current server generation owns, then the re-exec. The bridge delegates to `server_restart._perform_owner_restart` (re-exported by `server`) with its transport notice callback. Before a supervisor publishes its bridge, HTTP admission binds these same Restart and Panic operations directly; deferred execution hands off only to a live supervisor with a published bridge. A thread still initializing, or a stale bridge left by a dead thread, cannot consume a command, so the direct owner executes it. That common operation is checkout-first: `_safe_restart_serialized` (update lock, strict managed-update transaction gate, then `safe_restart`'s checkout, dependency sync, import test and stable fallback) runs before anything is stopped, so every refusal — an assisted-update merge mid-resolution, a failed checkout, an unwritable no-resume flag — leaves the server intact and is answered with one "Restart cancelled" line. Past that point the restart always follows: the durable no-resume intent (`state/owner_restart_no_resume.flag` plus the `panic_stop.flag` compatibility pair, consumed by `auto_resume_after_restart`), then `server_restart._stop_owned_work` — one immediate cancel intent per RUNNING task, live direct activity and in-flight post-task synthesis (`cancel_intents.request_cancel`, source `owner_restart`), `kill_workers`, delegated-run cancellation through the owner-gone seam `reconcile_orphaned_runs(running_task_ids=set())` over `read_owned_gateway`, and the attested owned-daemon `stop_outcome()` exactly as Panic makes it. Everything between the cancel intents and the daemon stop is attach-only: an `ensure_owned_gateway` there would start a dead daemon. Nothing is a veto or a deferral: an unconfirmed or raising worker shutdown leaves a critical log, an unconfirmed or raising daemon stop the `process_stop_unconfirmed` supervisor row; custody stays retained and the restart proceeds. The next generation does its ordinary startup work: owner restart is a no-resume cause (`delegate_recovery.NO_RESUME_CAUSES`), so nothing is adopted; the startup custody sweep reconciles every open delegated run as owner-gone (a run the stopped daemon took down answers absent → `close_absent_run`, no invented spend; a journal-recovered terminal settles; a pending invocation replays under its own key), and an in-flight owned model operation stays a `dispatched`/`unresolved` physical-attempt row that already counts as spend upper bound. After an unconfirmed stop the new generation attaches to the still-live daemon rather than spawning a second one, and the lifespan's one background `warm_owned_daemon()` (provisioned homes only) makes the first delegation after a Restart find the daemon serving. The explicit stop is installation-wide — manual Restart ends runs served by the owned daemon, another client's included — which differs intentionally from ordinary window close; planned self-restart, managed-update handoff and Panic keep their separate contracts.
Panic issues native requests through existing owners before locks, disk or waits. Captured Popen/Job/multiprocessing handles prove ownership; spawn owners publish them before custody I/O and retire raced starts. Daemon admission pins measured custody between identity checks (Windows handle, Linux pidfd, Darwin exit/exec watch). Panic uses that identity, never guessed PIDs or successors. Missing proof and request errors stay explicit. Groups/Jobs cover their members; escaped descendants/backend-only processes need settlement. Independent helpers then settle custody and run existing port sweeps, followed by bounded flag/control writes. Helper launch and hard exit prove no child death. Local `/panic` precedes state/chat I/O; external ingress needs a known owner. Constructors/ticks/wakes respect the flag; boot consumes it after disabled controls persist. Ordinary close preserves the shared daemon.
Panic requests termination through existing owners before persistence, cleanup locks or exit waits. A pooled worker receives a nonblocking request over the private socket held by its multiprocessing owner; its existing parent lifeline requests each local process owner before killing the worker. This preserves the ownership of foreground commands in separate sessions; a shell spawn already inside Popen publishes its handle before worker exit. The request receipt proves channel delivery, not worker or child death. Captured Popen/Job handles retain the ordinary owned-child paths. An attached Linux target is signalled through its retained pidfd for that process only: a numeric group is not the same identity. A Darwin exit/exec watch cannot atomically signal its watched process, so an attachment without a signalable owned handle immediately reports unconfirmed; it never converts that watch into a numeric signal. Groups/Jobs cover their held members; escaped descendants/backend-only processes need settlement. Independent helpers settle the server's local owners and run existing port sweeps, followed by bounded flag/control writes. Pool queue/custody reconciliation belongs to the next boot, so cleanup cannot destroy a worker before its local requests. Helper launch and hard exit prove no child death. Local `/panic` precedes state/chat I/O; external ingress needs a known owner. Constructors/ticks/wakes respect the flag; boot consumes it after disabled controls persist. Ordinary close preserves the shared daemon.
`stop_outcome` returns `stopped`, `nothing_to_stop` or disclosed `unconfirmed`. With a valid owned marker and an authenticated same-home endpoint it first invokes the installed managed CLI `daemon stop --json` (read-only `resolve_cli_command(require_npm=False)`: operator shutdown needs the existing exact Node, not the npm installation toolchain); it never ensures a runtime, wakes a daemon or probes accounts. Exit 0 plus the typed terminal CLI receipt is required — an RPC acknowledgement or a clean lease release does not prove the captured OS processes have exited — and a surviving authenticated endpoint is disclosed without chasing a successor. The pre-request custody snapshot limits forced fallback to those original rows, and forced signalling still requires both measured birth and command identity or the manager's own Popen; legacy Windows empty-birth rows cannot authorize signals, though a healthy attached daemon can still stop cooperatively through the CLI. Token refusal, invalid discovery, malformed responses and a matching error-code string alone never provide stop authority. Only confirmed death permits signal-stop custody removal; concurrent rows survive.

View file

@ -41,8 +41,11 @@ P7 makes context fit a maintenance constraint, not a line-count aesthetic.
`ouroboros/review.py::MAX_TOTAL_FUNCTIONS` (the same runtime-only iterator;
the module gates include tests/devtools) — a high-water alarm with ample
headroom, raised only with a one-line campaign rationale in the same commit
(the current 10500 ceiling came with the exact budget-pause lifecycle,
owner-approved within 10%, after that change's own single-caller inlines).
(10500 came with the owner-approved exact budget-pause lifecycle; Batch1
proposes 10525, its exact product count after four redundant helper removals
and three required worker-ownership/history helpers, versus 10526 at `81c`).
The +25 adjustment requires independent review before publication; all module,
function, byte and debt-transition limits remain unchanged.
- Enforcement: the OFFICIAL repository's CI runs the dedicated `size_ratchet`
pytest lane as a blocking step (`OURO_SIZE_RATCHET_BASE_REF` names the event
base; lane placement and base fallback: ARCHITECTURE §8 "CI topology").

View file

@ -95,7 +95,8 @@ class BackgroundConsciousness:
state = self._read_state()
from supervisor.state import control_is
self._enabled = not panic_blocks_wake(self._drive_root) and control_is(state, "bg_consciousness_enabled", True) # unknown is not on (#1307)
self._stopped = panic_blocks_wake(self._drive_root) # unknown is suspension, Panic is a stop
self._enabled = not self._stopped and control_is(state, "bg_consciousness_enabled", True) # unknown is not on (#1307)
try:
persisted = float(state.get(NEXT_WAKE_STATE_KEY) or 0.0)
except (TypeError, ValueError):
@ -201,9 +202,10 @@ class BackgroundConsciousness:
from supervisor.state import control_is
if panic_blocks_wake(self._drive_root):
self._enabled = False
self._stopped, self._enabled = True, False
return "panic_stop"
if not self._enabled or not control_is(self._read_state(), "bg_consciousness_enabled", True):
self._enabled = not self._stopped and control_is(self._read_state(), "bg_consciousness_enabled", True)
if not self._enabled:
return "disabled"
wake, owner_live = self.live_turns()
if wake or owner_live:
@ -310,13 +312,14 @@ class BackgroundConsciousness:
return "Background consciousness stays disabled while Panic controls await persistence."
if self._enabled:
return "Background consciousness is already enabled."
self._enabled = True
self._stopped, self._enabled = False, True
# The clock did not advance while disabled: never announce a wake in the past.
self._set_next_wake(max(self._next_wake_at, time.time()))
return f"Background consciousness enabled; next wake-up at {time.strftime('%H:%M', time.localtime(self._next_wake_at))}."
def stop(self) -> str:
with self._lock:
self._stopped = True
was_enabled, self._enabled = self._enabled, False
wake, _owner = self.live_turns()
if wake:

View file

@ -1154,20 +1154,19 @@ def _drive_state_section(env: Any) -> str:
"evolution_owner_stopped", "evolution_cycle", "evolution_consecutive_failures",
"last_evolution_task_at", "bg_consciousness_enabled", "post_task_autostop",
"budget_drift_pct", "budget_drift_alert", "last_owner_message_at")
from supervisor.state import RECOVERY_KEY, read_state_copy
from supervisor.state import CONTROL_KEYS, RECOVERY_KEY, control_value, read_state
status, raw, detail = read_state_copy(env.drive_path("state/state.json"))
raw = raw or {}
projected = {k: raw[k] for k in keys if k in raw}
observed = read_state(env.drive_path("state/state.json").parent.parent)
raw = observed.values
authority = observed.projection()
projected = {k: (raw[k] if k not in CONTROL_KEYS or control_value(authority, k)[0]
else {"status": "unknown"}) for k in keys if k in raw or k in observed.unconfirmed}
omitted = sorted(set(raw) - set(projected) - {RECOVERY_KEY})
unconfirmed = (raw.get(RECOVERY_KEY) or {}).get("unconfirmed") if isinstance(raw.get(RECOVERY_KEY), dict) else None
note = ("Projection of state/state.json (spend/budget facts live in the Runtime "
"section, from the usage-accounting authority)."
+ (f" The file is {status}{f' ({detail})' if detail else ''}: its facts are UNKNOWN here, "
"not defaults." if status != "ok" else "")
+ (" Recovered from its backup: these controls are UNKNOWN until an owner decision "
f"confirms them: {', '.join(unconfirmed)}." if unconfirmed else "")
+ ((" Omitted keys: " + ", ".join(omitted) + ". Full file: "
+ (f" State authority is {observed.quality}: {observed.reason}. "
f"UNKNOWN controls: {', '.join(observed.unconfirmed)}." if observed.quality != "current" else "")
+ ((" Omitted keys: " + ", ".join(omitted) + ". Full raw source: "
"read_file(root='runtime_data', path='state/state.json').") if omitted else ""))
return ("## Drive state\n\n"
+ json.dumps(projected, ensure_ascii=False, indent=1, sort_keys=True, default=str)

View file

@ -453,6 +453,9 @@ def _annotate_terminal_task_truth(
and str(message.get("role") or "") in {"assistant", "system"}
and str(message.get("task_id") or "") not in progress_task_ids
}
from ouroboros.tool_call_log import replay_evidence
tool_evidence_by_task = {}
terminal_status_by_task: Dict[str, str] = {}
terminal_truth_by_task: Dict[str, Dict[str, Any]] = {}
terminal_receipt_by_task: Dict[str, Dict[str, Any]] = {}
@ -461,6 +464,7 @@ def _annotate_terminal_task_truth(
live_cost_by_task: Dict[str, Dict[str, Any]] = {}
finalizing_tasks: set = set()
for task_id in progress_task_ids | summary_task_ids | legacy_final_task_ids:
tool_evidence_by_task[task_id] = replay_evidence(data_dir, task_id)
result = _load_terminal_result(data_dir, task_id, cache)
child_meta = subagent_message_meta(result, task_id=task_id)
if child_meta:
@ -586,6 +590,7 @@ def _annotate_terminal_task_truth(
and latest_progress_by_task.get(task_id) is message
):
message.update(terminal_truth_by_task.get(task_id) or {})
message["tool_evidence"] = tool_evidence_by_task.get(task_id)
if (message.get("is_progress") or is_summary) and task_id in suggested_name_by_task:
message["suggested_name"] = suggested_name_by_task[task_id]
# Floor-symmetric closed lineage window for chat FINALS: strip runs

View file

@ -80,6 +80,11 @@ async def api_schedules_upsert(request: Request) -> JSONResponse:
"type": "task",
"text": str(body.get("description") or body.get("name") or "Scheduled task"),
}
# This owner door creates Main self-work or the named room's default.
# Existing followups retain their recorded explicit resource/none intent.
task = {**task, "metadata": dict(task.get("metadata") or {})}
project_id = str(task.get("project_id") or "").strip()
new_intent = {"kind": "room_default", "project_id": project_id} if project_id else {"kind": "system_repo"}
enabled = _enabled_value(body)
if isinstance(enabled, str):
return json_error(enabled, 400)
@ -101,7 +106,7 @@ async def api_schedules_upsert(request: Request) -> JSONResponse:
try:
stored = upsert_scheduled_task(
record, drive_root=request_drive_root(request), actor="owner:gateway",
reason=str(body.get("reason") or "").strip())
reason=str(body.get("reason") or "").strip(), new_resource_intent=new_intent)
except ScheduleRefused as refusal:
# 409 when the write could not be made SAFELY (its audit is down);
# 400 when the request itself asks for something this door cannot do.

View file

@ -727,6 +727,10 @@ def request_process_tree_kill(proc, *, job_handle=None) -> dict:
try:
if pid <= 0 or pid == os.getpid():
raise ValueError("refusing current/invalid process")
channel = None if pinned else getattr(proc, "_ouroboros_stop_socket", None)
if channel is not None:
channel.send(b"!") # private nonblocking channel; worker requests its held children first
return {**result, "requested": True, "scope": "worker_owners"}
if IS_WINDOWS:
if job_handle is not None:
result["scope"] = "job"
@ -739,21 +743,18 @@ def request_process_tree_kill(proc, *, job_handle=None) -> dict:
else:
if pinned:
if IS_MACOS:
if proc["handle"].control([], 1, 0):
proc["handle"].close() # an exit/exec refusal cannot be consumed then forgotten
raise ProcessLookupError("captured process exited or exec'd")
else:
signal.pidfd_send_signal(proc["handle"].fileno(), 0)
elif (proc.poll() if hasattr(proc, "poll") else proc.exitcode) is not None:
raise ProcessLookupError("owned child already exited")
pgid = os.getpgid(pid)
if pgid == pid and pgid != os.getpgrp() and (not pinned or pgid == proc["pgid"]):
os.killpg(pgid, signal.SIGKILL)
result["scope"] = "group"
elif pinned and not IS_MACOS:
proc["handle"].close()
raise RuntimeError("attached Darwin watch is not a signalable identity")
signal.pidfd_send_signal(proc["handle"].fileno(), signal.SIGKILL)
else:
os.kill(pid, signal.SIGKILL)
if (proc.poll() if hasattr(proc, "poll") else proc.exitcode) is not None:
raise ProcessLookupError("owned child already exited")
pgid = os.getpgid(pid)
if pgid == pid and pgid != os.getpgrp():
os.killpg(pgid, signal.SIGKILL)
result["scope"] = "group"
else:
os.kill(pid, signal.SIGKILL)
result["requested"] = True
except Exception as exc:
result["error"] = f"{type(exc).__name__}: {exc}"

View file

@ -440,7 +440,7 @@ def _multiprocessing_parent_sentinel() -> Optional[int]:
return None
def start_parent_lifeline(*, poll_sec: float = 5.0, label: str = "") -> None:
def start_parent_lifeline(*, poll_sec: float = 5.0, label: str = "", stop_socket=None, before_exit=None) -> None:
"""Daemon watchdog: group-suicide when the spawning parent dies (POSIX).
For OUR python entrypoints only (workers, extension runner, claude child):
@ -454,15 +454,18 @@ def start_parent_lifeline(*, poll_sec: float = 5.0, label: str = "") -> None:
worker's exit 255. Arbitrary-argv services and skills cannot get a watchdog
injected -- they are covered by the ledger + reaper instead.
"""
if os.name == "nt":
return # Windows children are covered by Job Objects
if os.name == "nt" and stop_socket is None:
return # ordinary Windows children are covered by Job Objects
import threading
import time as _time
from multiprocessing.connection import wait as _mp_wait
def _suicide() -> None:
log.warning("parent process died — lifeline group-suicide (%s)", label or "child")
if before_exit is not None:
before_exit() # held child requests precede logging and discovering-owner death
if stop_socket is None: # the emergency path cannot wait on a logging handler
log.warning("parent stopped — lifeline group-suicide (%s)", label or "child")
try:
from ouroboros.platform_layer import current_process_group_id
@ -477,7 +480,7 @@ def start_parent_lifeline(*, poll_sec: float = 5.0, label: str = "") -> None:
os._exit(1)
def _sentinel_hung_up(sentinel: int, timeout: float) -> bool:
return bool(_mp_wait([sentinel], timeout=timeout))
return bool(_mp_wait([sentinel, stop_socket] if stop_socket is not None else [sentinel], timeout=timeout))
initial_ppid = os.getppid()
sentinel = _multiprocessing_parent_sentinel()
@ -496,7 +499,10 @@ def start_parent_lifeline(*, poll_sec: float = 5.0, label: str = "") -> None:
nonlocal sentinel
while True:
if sentinel is None:
_time.sleep(poll_sec)
if stop_socket is not None and _mp_wait([stop_socket], timeout=poll_sec):
_suicide()
else:
_time.sleep(poll_sec)
else:
try:
hung_up = _sentinel_hung_up(sentinel, poll_sec)

View file

@ -104,7 +104,7 @@ def project_folder_basename(display_name: Any) -> str:
head = head[:-1]
head = head.rstrip(" .-")
name = f"{head}-{digest}" if head else digest
if name.split(".", 1)[0].casefold() in _RESERVED_NAMES:
if name.split(".", 1)[0].casefold() in (_RESERVED_NAMES | {f"{prefix}{digit}" for prefix in ("com", "lpt") for digit in "¹²³"}):
stem, dot, rest = name.partition(".")
name = f"{stem}_{dot}{rest}"
return name

View file

@ -29,7 +29,10 @@ MAX_FUNCTION_LINES = 300
# own single-caller inlines and an upstream base that grew ~42 functions in one
# day; the remaining delta is decomposition, not duplication, so buying the gap
# by merging load-bearing steps would read worse.
MAX_TOTAL_FUNCTIONS = 10500
# Batch1 exact candidate: 10525 product functions after four redundant helper
# removals and three required worker-ownership/history helpers (81c had 10526).
# Narrow +25 proposal for independent review; no other limit or ratchet changes.
MAX_TOTAL_FUNCTIONS = 10525
SIZE_RATCHET_MANIFEST_PATH = "ouroboros/size_ratchet_manifest.py"

View file

@ -178,8 +178,9 @@ def execute_panic_stop(
attempt("executors", lambda: kill_all_foreground(data_dir, wait=False), settle=True)
attempt("services", lambda: kill_all_services(data_dir, wait=False), settle=True)
attempt("companions", panic_kill_all, settle=True)
attempt("workers", lambda: kill_workers_fn(
force=True, archive_service_logs=False, reconcile_delegate_custody=False), settle=True)
# Workers received private lifeline requests above. Root-first tree cleanup
# here would destroy their local child ownership before those requests run.
# Queue/custody reconciliation belongs to the following supervisor boot.
attempt("main-port", lambda: kill_process_on_port(bound_port or 8765), settle=True)
attempt("host-port", lambda: kill_process_on_port(host_service_port()), settle=True)
@ -224,11 +225,6 @@ def _persist_panic_controls(data_dir: pathlib.Path) -> None:
from supervisor import state
from supervisor.evolution_lifecycle import complete_evolution_campaign, record_evolution_stop_intent
def _panic_controls(st: dict) -> None:
st.update(evolution_mode_enabled=False, bg_consciousness_enabled=False,
evolution_owner_stopped=True, post_task_autostop=False)
st.pop("evolution_stop_source", None) # an owner stop: no agent source may un-stick it
failures = []
for step in (
lambda: state.update_state(_panic_controls, confirm=PANIC_CONTROL_KEYS, lock_timeout_sec=0.5),
@ -246,6 +242,12 @@ def _persist_panic_controls(data_dir: pathlib.Path) -> None:
raise OSError(f"Panic controls unconfirmed: {failures}")
def _panic_controls(st: dict) -> None:
st.update(evolution_mode_enabled=False, bg_consciousness_enabled=False,
evolution_owner_stopped=True, post_task_autostop=False)
st.pop("evolution_stop_source", None) # an owner stop: no agent source may un-stick it
PANIC_CONTROL_KEYS = ("evolution_mode_enabled", "bg_consciousness_enabled", "evolution_owner_stopped",
"evolution_stop_source", "post_task_autostop")

View file

@ -111,7 +111,6 @@ BAND_PATHS = {
"ouroboros/budget_pause.py": "Entered the band from 990 lines (#1196): the direct-turn pause, the typed restore refusal beside the boolean gate and the authoritative Q10 refresh belong with the one pause/resume owner they extend; the supervisor-side grant lifecycle moved out to supervisor/budget_resume.py instead of growing here.",
"ouroboros/cancel_intents.py": "Entered the band from 929 lines: reciprocal timeout-retry lineage validation and physical-leaf/logical-root aliasing stay with the durable cancel-intent mutation authority so Stop-now hardens the same request across retry races.",
"ouroboros/capability_evidence.py": "Grew INTO the band by the #284 fix: a fresh exact-model density witness may honestly undercut the cold floor \u2014 evidence logic belongs beside the witness store it reads.",
"ouroboros/claudexor_daemon.py": "Installation daemon lifecycle owns marker and authenticated endpoint stop authority, confirmed self-started handles, and duplicate-start refusal; process signal and ledger mechanics remain in process_custody. No new lifecycle store or scheduler.",
"ouroboros/cli.py": "The existing command-line transport keeps task-event negotiation, bounded replay deduplication and result rendering together; the additive cursor does not introduce a second CLI or task engine.",
"ouroboros/context_compaction.py": "Existing compaction owns propagation of typed model outcomes; unchanged semantic compaction policy.",
"ouroboros/delegate_custody.py": "D07 DEL1 split brought the custody monolith DOWN from the 1600 hard cap into the band (1600->1305); reconcile family extracted to delegate_custody_reconcile.py, shrink-only direction",
@ -136,7 +135,6 @@ BAND_PATHS = {
"ouroboros/memory.py": "ibl-2b09abdadd25: scratchpad content-size cap added alongside the existing block-count cap in append_scratchpad_block's eviction loop",
"ouroboros/preflight_runner.py": None,
"ouroboros/presence_runner.py": "Presence turn admission, durable retry identity and transport custody remain one owner; separating them now would duplicate the gate and receipt seam.",
"ouroboros/projects_registry.py": "Entered the band from 999 lines: the stuck-Working liveness sprint homed the project-thread membership lens (mtime-cached) and its broadcast-choke marker here \u2014 registry semantics belong to the registry, not to message_bus.",
"ouroboros/reflection.py": "TZ-3 PR-1: reflection now stamps typed skip events, writer/route provenance and the project-vs-canonical reflection locator on its memory actions (948->1017); one owner for reflection generation and its memory-action application, no new subsystem",
"ouroboros/request_wire_receipts.py": "Wire candidates and semantic-success receipts share one exact serializer digest owner.",
"ouroboros/request_wire_recovery.py": "E4 (#447): typed CustomToolProjectionError fallback keeps the wire-recovery ladder alive; includes the one-site-sufficient decision record at both retry catch sites",
@ -161,13 +159,13 @@ BAND_PATHS = {
"ouroboros/tools/plan_review.py": "Entered the band from 999 lines: the required-affected_paths form (owner 9=A) added the schema field and the PLAN_RESOURCE_FORM_REQUIRED refusal, which must name the task's open wave and the $0 disposition exit \u2014 it belongs beside the one preamble both the paid and dry-run paths share, not in the pure plan_spec companion that owns no task state.",
"ouroboros/tools/registry_core.py": "ToolRegistry owns the registry class behind the protected facade; guard and dispatch implementations live in sibling leaves.",
"ouroboros/tools/review.py": "D06 F2.3a re-entry by extraction: the multi-model fan-out moved to review_multi_model.py (1550->1269); the remaining single-owner review cycle machinery lands in the 1001-1500 band with headroom",
"ouroboros/tools/services.py": "Existing service lifecycle publishes its held process before custody I/O and uses the same owner for nonblocking Emergency Stop requests.",
"ouroboros/tools/skill_exec.py": None,
"ouroboros/tools/skill_preflight.py": "Entered the band by absorbing the upstream classic-script validator, widget entry/grammar findings and the preflight schema description beside the v7 typed _run_check rework they attach to.",
"ouroboros/tools/skill_publish.py": "Entered the band from 952 lines: publish now writes the OuroborosHub publication receipt at pr_opened through the shared locked-update seam and maps the receipt from the validated serialized form (hubflow sprint, receipt-as-only-stored-fact design).",
"ouroboros/tools/subagent_integration.py": "Existing native result integration owner handles source patches and complete file artifacts under one disposition and target authority.",
"ouroboros/tools/tool_result.py": "Typed result composition owns producer payload and dispatch annotations together while preserving the status and metadata contract.",
"ouroboros/usage_compaction.py": "Entered the band from 971 lines: the C6 round-4 fixes homed here \u2014 dir-fd/O_NOFOLLOW anchoring of the archive writer and reader (a link planted after any path check cannot receive or serve monetary history) and the swap's last-instant snapshot re-proof inside the atomic replace \u2014 defenses that belong beside the compaction pass they defend.",
"ouroboros/workspace_executor.py": None,
"scripts/claudexor_platform_smoke.py": "The managed Claudexor platform smoke owns a multi-platform fixture, lifecycle receipt, and cleanup proof; keeping this runner in the documented band preserves the release gate without moving those checks into product runtime.",
"skills/telegram/plugin.py": None,
"skills/telegram/scripts/companion.py": None,

View file

@ -191,3 +191,30 @@ def logical_calls(rows: Iterable[Dict[str, Any]]) -> List[Dict[str, Any]]:
else "orphan" if "settled" in call
else "wait_ended" if "wait_ended" in call else "unknown")
return calls
def replay_evidence(drive_root: pathlib.Path, task_id: str, want: int = 200) -> Dict[str, Any]:
"""Bounded canonical invocation facts for history, independent of frozen wait metrics."""
from ouroboros.memory import Memory
from ouroboros.tool_capabilities import routing_action_for_tool
rows, coverage = Memory(drive_root).read_task_recent("tools.jsonl", task_id, want)
observations = []
legacy = {"calls": 0, "errors": 0, "wait_ended": False, "unknown": False}
for call in logical_calls(rows):
if not call.get("invocation_id"):
row = call.get("settled") or {}
legacy["calls"] += 1
legacy["errors"] += int(row.get("type") == CALL_SETTLED and bool(row.get("is_error")))
legacy["wait_ended"] |= row.get("type") == CALL_WAIT_ENDED
legacy["unknown"] |= row.get("type") == CALL_STARTED
continue # separate observations, never guessed joins to modern/live calls
base = {"key": f"tool:{task_id}:{call['invocation_id']}", "tool": call.get("tool"),
"receipt": bool(routing_action_for_tool(call.get("tool")))}
for slot, fact in (("started", "started"), ("wait_ended", "wait_ended"), ("settled", "settled")):
row = call.get(slot)
if row is not None:
observations.append({**base, "fact": fact, "live": False,
"status": ("error" if row.get("is_error") else "ok") if slot == "settled" else "unknown",
"hostError": row.get("status") == "host_error"})
return {"observations": observations, "legacy": legacy, "coverage": coverage}

View file

@ -29,6 +29,7 @@ from ouroboros.workspace_executor import map_host_path as executor_map_host_path
_active_subprocesses: set = set()
_subprocess_lock = threading.Lock()
_panic_requested = False
_spawning_subprocesses = 0
_RUN_SHELL_DEFAULT_TIMEOUT_SEC = 360
@ -38,16 +39,20 @@ def _tracked_subprocess_run(cmd, **kwargs):
interpreter, a DOOM framebuffer, raw bytes) surfaces as readable text instead
of raising UnicodeDecodeError and collapsing the whole call into a
shell_error."""
global _spawning_subprocesses
timeout = kwargs.pop("timeout", None)
if kwargs.get("text") or kwargs.get("universal_newlines"):
kwargs.setdefault("errors", "replace")
kwargs.setdefault("stdin", subprocess.DEVNULL)
kwargs.update(subprocess_new_group_kwargs())
if _panic_requested:
raise RuntimeError("Emergency Stop has retired command admission")
proc = subprocess.Popen(cmd, **kwargs)
# Publish the positive Popen identity before any helper lock or pipe I/O.
_active_subprocesses.add(proc)
_spawning_subprocesses += 1
try:
if _panic_requested:
raise RuntimeError("Emergency Stop has retired command admission")
proc = subprocess.Popen(cmd, **kwargs)
_active_subprocesses.add(proc) # publish before the spawning owner can exit
finally:
_spawning_subprocesses -= 1
try:
if _panic_requested:
request_process_tree_kill(proc)

View file

@ -547,6 +547,7 @@ def sync_skill_schedules(skills: List[Any], *, drive_root: pathlib.Path | None =
),
"metadata": {
"source": "skill_scheduled_task",
"resource_intent": {"kind": "system_repo"},
"skill": str(getattr(skill, "name", "")),
"scheduled_task": name,
},
@ -568,7 +569,12 @@ def sync_skill_schedules(skills: List[Any], *, drive_root: pathlib.Path | None =
and schedule_id not in touched
and not _is_suppressed(record)
):
by_id.pop(schedule_id, None)
from supervisor.schedule_occurrence import owed
if owed(record) is False:
by_id.pop(schedule_id, None)
else:
record.update(enabled=False, delete_requested_at=utc_now_iso())
changed = True
if changed:
data["tasks"] = list(by_id.values())

View file

@ -9,6 +9,8 @@ teardown it decides on is handed to the off-loop reaper.
from __future__ import annotations
import datetime
from supervisor.state import control_is
import logging
import pathlib
import time
@ -414,7 +416,7 @@ def _enforce_task_timeouts_locked(
)
# A stopped evolution campaign breaks the auto-retry chain. `st` is the live state
# loaded this tick, so this reflects the current owner decision.
if will_retry and task_type == "evolution" and not _evolution_known_enabled(st):
if will_retry and task_type == "evolution" and not control_is(st, "evolution_mode_enabled", True):
will_retry = False
# An unreadable projection/lineage cannot authorize a new dispatch.
# Readable active intents already yielded the timeout rail above.
@ -449,10 +451,3 @@ def _enforce_task_timeouts_locked(
"incident_toast_once": f"{task_id}:{terminal_reason}:{int(finalization_requested_at or now)}",
})
_queue().persist_queue_snapshot(reason="task_timeout_reap_queued")
def _evolution_known_enabled(st: Dict[str, Any]) -> bool:
"""A retry is new evolution work: it needs a KNOWN enabled control (#1307)."""
from supervisor.state import control_is
return control_is(st, "evolution_mode_enabled", True)

View file

@ -153,7 +153,8 @@ def _merge_onto_current(existing: Dict[str, Any], incoming: Dict[str, Any]) -> D
def upsert_scheduled_task(record: Dict[str, Any], *, drive_root: pathlib.Path | None = None,
actor: str = "", task_id: str = "", reason: str = "",
continuation_of: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
continuation_of: Optional[Dict[str, Any]] = None,
new_resource_intent: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
"""Create or replace a scheduled task record.
Returns the stored row plus an ``audit`` field: ``recorded`` when both audit
@ -174,6 +175,15 @@ def upsert_scheduled_task(record: Dict[str, Any], *, drive_root: pathlib.Path |
schedule_id = str(incoming.get("id") or "").strip() or uuid.uuid4().hex[:8]
incoming["id"] = schedule_id
existing = next((item for item in tasks if str(item.get("id") or "") == schedule_id), None)
if new_resource_intent is not None:
# The producer may describe a NEW row. Editing an old followup cannot
# turn absent intent into self-work; preserve its current evidence.
template = dict(incoming.get("task") or {})
metadata = dict(template.get("metadata") or {})
intent = ((existing.get("task") or {}).get("metadata") or {}).get("resource_intent") if existing else new_resource_intent
if intent is not None:
metadata.setdefault("resource_intent", dict(intent))
incoming["task"] = {**template, "metadata": metadata}
operation_id = uuid.uuid4().hex[:12]
action = "edit" if existing is not None else "create"
audit = {

View file

@ -353,22 +353,6 @@ def _bind(task: Dict[str, Any], root: str) -> None:
return None
def binding_moved(prepared: Dict[str, Any]) -> bool:
"""Cheap early rejection; the registry owner also fences actual queue publication."""
basis = prepared.get("binding_basis")
intent = (prepared.get("task") or {}).get("metadata", {}).get("resource_intent") or {}
if basis is None or intent.get("kind") != "room_default":
return False
try:
from ouroboros.projects_registry import get_reserved_project
project_id = str(prepared["task"].get("project_id") or intent.get("project_id") or "")
room = get_reserved_project(_queue().DRIVE_ROOT, project_id, strict=True) or {}
except Exception:
return True
return str(room.get("working_dir") or "").strip() != basis
# --- admit (queue + table locks) ---------------------------------------------
def admit(prepared: List[Dict[str, Any]]) -> None:
@ -379,7 +363,6 @@ def admit(prepared: List[Dict[str, Any]]) -> None:
ScheduleStoreUnreadable, _write_scheduled_tasks, load_schedule_store, schedule_transaction,
)
moved = {p["schedule_id"] for p in prepared if not p.get("stored_task") and binding_moved(p)}
windows = {p["schedule_id"]: _window(p["task"]) for p in prepared if not p.get("hold")}
with schedule_transaction(q.DRIVE_ROOT):
try:
@ -398,8 +381,7 @@ def admit(prepared: List[Dict[str, Any]]) -> None:
continue # deleted, or another pass already moved this row on
changed = True
fresh_claim = current.get("phase") == "claimed" and not item.get("stored_task")
if fresh_claim and (not record.get("enabled", True) or fingerprint(record) != item["fingerprint"]
or item["schedule_id"] in moved):
if fresh_claim and (not record.get("enabled", True) or fingerprint(record) != item["fingerprint"]):
record.pop("occurrence", None) # disabled/edited/rebound during prepare: future only
continue
if item.get("hold"):

View file

@ -249,7 +249,7 @@ def control_is(st: Dict[str, Any], key: str, expected: Any) -> bool:
return known and value == expected
def _backup_unconfirmed(backup: Dict[str, Any]) -> Tuple[str, ...]:
def _backup_unconfirmed(backup: Dict[str, Any], drive_root=None) -> Tuple[str, ...]:
"""The controls a backup copy cannot prove. A SET owner binding is proven when the
backup carries the completed initialization identity: its only writers fill a
known-empty slot and only an owner Reset (a new identity, both copies deleted)
@ -258,7 +258,7 @@ def _backup_unconfirmed(backup: Dict[str, Any]) -> Tuple[str, ...]:
from supervisor import state_initialization as witness
identity = str(backup.get("initialization_id") or "")
status, record = witness.read_witness(DRIVE_ROOT) if identity else ("missing", {})
status, record = witness.read_witness(drive_root or DRIVE_ROOT) if identity else ("missing", {})
same = status == "ok" and record.get("phase") == "complete" and record.get("initialization_id") == identity
return tuple(key for key in CONTROL_KEYS
if not (same and key.startswith("owner_") and backup.get(key) is not None
@ -270,23 +270,24 @@ def _recovery_unconfirmed(st: Dict[str, Any]) -> Tuple[str, ...]:
return tuple(recovery.get("unconfirmed") or ()) if isinstance(recovery, dict) else ()
def read_state() -> StateRead:
def read_state(drive_root=None) -> StateRead:
"""Classify both copies WITHOUT writing (a GET, a display, a boot probe)."""
from supervisor.state_initialization import authority_reason, read_witness
p_status, primary, _raw, p_detail = _read_json_file(STATE_PATH)
root = pathlib.Path(drive_root) if drive_root is not None else DRIVE_ROOT
p_status, primary, _raw, p_detail = _read_json_file(root / "state" / "state.json")
if p_status == "ok":
reason = authority_reason(DRIVE_ROOT, str(primary.get("initialization_id") or ""))
reason = authority_reason(root, str(primary.get("initialization_id") or ""))
if reason:
return StateRead("unavailable", "primary", dict(primary), CONTROL_KEYS, reason)
unconfirmed = tuple(set(_recovery_unconfirmed(primary)) | (set(CONTROL_KEYS) - primary.keys() - OPTIONAL_CONTROL_KEYS))
return StateRead("recovered" if unconfirmed else "current", "primary",
ensure_state_defaults(dict(primary)), unconfirmed)
b_status, backup, _braw, b_detail = _read_json_file(STATE_LAST_GOOD_PATH)
b_status, backup, _braw, b_detail = _read_json_file(root / "state" / "state.last_good.json")
reason = f"primary {p_status}{f' ({p_detail})' if p_detail else ''}; backup {b_status}"
if b_status == "ok":
return StateRead("recovered_transient", "backup", ensure_state_defaults(dict(backup)),
_backup_unconfirmed(backup), reason)
w_status, _witness = read_witness(DRIVE_ROOT)
_backup_unconfirmed(backup, root), reason)
w_status, _witness = read_witness(root)
quality = "uninitialized" if p_status == b_status == w_status == "missing" else "unavailable"
return StateRead(quality, "none", {}, CONTROL_KEYS, reason + (f" ({b_detail})" if b_detail else ""))
@ -304,8 +305,17 @@ def _preserve_corrupt_primary(raw: bytes) -> str:
except OSError as exc:
raise StateUnavailable("corrupt_primary_unpreserved", f"{type(exc).__name__}") from exc
try:
os.write(fd, raw)
offset = 0
while offset < len(raw):
written = os.write(fd, raw[offset:])
if written <= 0:
raise OSError("corrupt copy made no progress")
offset += written
os.fsync(fd)
if os.fstat(fd).st_size != len(raw):
raise OSError("corrupt copy length mismatch")
except OSError as exc:
raise StateUnavailable("corrupt_primary_unpreserved", type(exc).__name__) from exc
finally:
os.close(fd)
return target.name
@ -331,6 +341,8 @@ def _state_for_write(*, initializing: bool = False) -> Dict[str, Any]:
mark_unconfirmed(primary, key)
else:
raise StateUnavailable(reason)
for key in set(CONTROL_KEYS) - primary.keys() - OPTIONAL_CONTROL_KEYS:
mark_unconfirmed(primary, key)
return primary
if p_status == "unreadable":
raise StateUnavailable("primary_unreadable", p_detail)
@ -428,6 +440,9 @@ def save_state(st: Dict[str, Any]) -> None:
if p_status == "invalid":
primary = _state_for_write()
p_status = "ok"
if p_status == "ok":
for key in (set(CONTROL_KEYS) - primary.keys() - OPTIONAL_CONTROL_KEYS) | (set(CONTROL_KEYS) - st.keys() - OPTIONAL_CONTROL_KEYS):
mark_unconfirmed(primary, key)
if p_status == "ok" and isinstance(primary.get(RECOVERY_KEY), dict):
st[RECOVERY_KEY] = primary[RECOVERY_KEY]
if p_status == "ok":

View file

@ -28,6 +28,7 @@ is a disclosed residual, not proof of historylessness.
from __future__ import annotations
import hashlib
import json
import logging
import os
@ -160,7 +161,13 @@ def prepare_adoption(drive_root: Any, state: Dict[str, Any]) -> str:
"""
status, witness = read_witness(drive_root)
identity = str(state.get("initialization_id") or "")
source = pathlib.Path(drive_root) / "state" / "state.json"
source_digest = hashlib.sha256(source.read_bytes()).hexdigest()
if status == "ok":
if (not identity and witness.get("phase") == "pending"
and witness.get("origin") == "legacy_adopted"
and witness.get("legacy_source_sha256") == source_digest):
return str(witness["initialization_id"])
if identity != witness["initialization_id"]:
raise ValueError("initialization_identity_mismatch")
return identity
@ -170,7 +177,8 @@ def prepare_adoption(drive_root: Any, state: Dict[str, Any]) -> str:
raise ValueError("legacy_initialization_evidence_missing")
identity = identity or uuid.uuid4().hex
_write(drive_root, {"initialization_id": identity, "phase": "pending",
"origin": "legacy_adopted", "created_at": utc_now_iso()})
"origin": "legacy_adopted", "created_at": utc_now_iso(),
"legacy_source_sha256": source_digest})
return identity

View file

@ -684,14 +684,9 @@ def auto_resume_after_restart() -> None:
# write fails, the flag stays and every boot grant keeps reading it.
panic_flag = _pool().DRIVE_ROOT / "state" / "panic_stop.flag"
if panic_flag.exists():
from ouroboros.server_control import PANIC_CONTROL_KEYS
from ouroboros.server_control import PANIC_CONTROL_KEYS, _panic_controls
from supervisor.state import StateUnavailable, update_state
def _panic_controls(st: dict) -> None:
st.update(evolution_mode_enabled=False, bg_consciousness_enabled=False,
evolution_owner_stopped=True, post_task_autostop=False)
st.pop("evolution_stop_source", None)
try:
update_state(_panic_controls, confirm=PANIC_CONTROL_KEYS)
except StateUnavailable as exc:

View file

@ -19,6 +19,7 @@ from ouroboros.outcomes import (
)
from supervisor.log_addressing import resolve_project_chat
from supervisor.queue import _queue_lock
from supervisor.state import control_is
def _pool():
@ -513,7 +514,7 @@ def _recover_crashed_task_without_terminal(job: dict, queue: Any) -> None:
terminal_event = ("failed", reason_code)
from ouroboros.delegate_recovery import reconcile_unrecoverable_task
reconcile_unrecoverable_task(root, task_id)
elif task_type == "evolution" and not _evolution_known_enabled():
elif task_type == "evolution" and not control_is(_pool().load_state(), "evolution_mode_enabled", True):
# Evolution was stopped (or its control is unknown, #1307): do not resurrect
# a dead evolution worker into another cycle (mirrors the hard-timeout gate
# in queue.enforce_task_timeouts). A running task is never killed for this.
@ -751,9 +752,3 @@ def _recover_terminal_files(job: dict) -> None:
finally:
with q._queue_lock:
state["in_flight"] = False
def _evolution_known_enabled() -> bool:
from supervisor.state import control_is
return control_is(_pool().load_state(), "evolution_mode_enabled", True)

View file

@ -17,7 +17,6 @@ not pool state — nothing rebinds it — so the parent imports it back directly
from __future__ import annotations
import logging
from supervisor.worker_process import _current_custody_session_id, worker_main
import json
import os
import pathlib
@ -670,18 +669,13 @@ def _spawn_worker_slot(wid: int, old: Any = None, *, ready_attempt: int = 1) ->
ctx = _pool()._get_ctx()
in_q = ctx.Queue()
events_cursor, spawned_at = events_log_cursor(), time.time()
proc = ctx.Process(target=worker_main,
args=(wid, in_q, _pool().get_event_q(), str(_pool().REPO_DIR), str(_pool().DRIVE_ROOT),
_current_custody_session_id()))
proc.daemon = True
from supervisor.worker_process import spawn_worker_process
try:
proc.start()
proc = spawn_worker_process(ctx, wid, in_q, _pool().get_event_q(), _pool().REPO_DIR, _pool().DRIVE_ROOT)
except Exception:
try:
in_q.close()
in_q.cancel_join_thread()
except Exception:
pass
in_q.close()
in_q.cancel_join_thread()
raise
installed = False
with _queue_lock:

View file

@ -105,8 +105,61 @@ def _adopt_published_extensions(pool_drive_root: str) -> None:
log.debug("extension generation adoption failed", exc_info=True)
def spawn_worker_process(ctx, wid, in_q, out_q, repo_dir, drive_root):
"""Bind a private emergency channel to this exact worker before it starts."""
import socket
parent, child = socket.socketpair()
parent.setblocking(False)
proc = ctx.Process(target=worker_main, args=(wid, in_q, out_q, str(repo_dir), str(drive_root),
_current_custody_session_id(), child))
proc.daemon = True
try:
proc.start()
except BaseException:
parent.close()
raise
finally:
child.close()
proc._ouroboros_stop_socket = parent
return proc
def request_worker_owned_stops(drive_root):
"""Close admission and request every locally held owner before worker exit.
No custody/persistence/cleanup locks. A spawn already inside Popen must
publish its handle before this discovering owner may be destroyed.
"""
import time
from ouroboros.claudexor_daemon import get_owned_daemon
from ouroboros.extension_companion import panic_kill_all
from ouroboros.local_model import get_manager
from ouroboros.tools import shell_process
from ouroboros.tools.services import kill_all_services
from ouroboros.workspace_executor import kill_all_foreground
for request in (shell_process.kill_all_tracked_subprocesses,
lambda **kw: kill_all_foreground(pathlib.Path(drive_root), **kw),
lambda **kw: kill_all_services(pathlib.Path(drive_root), **kw), panic_kill_all):
try:
request(request_only=True)
except Exception:
pass # one owner failure must not skip other held identities
for owner in (get_owned_daemon(create=False), get_manager(create=False)):
if owner is not None:
try:
owner.panic_stop(request_only=True)
except Exception:
pass
while shell_process._spawning_subprocesses:
time.sleep(0.001)
shell_process.kill_all_tracked_subprocesses(request_only=True)
def worker_main(wid: int, in_q: Any, out_q: Any, repo_dir: str, drive_root: str,
custody_session_id: str = "") -> None:
custody_session_id: str = "", stop_socket=None) -> None:
import os as _os
# Mark this process as a worker BEFORE importing the agent/LLM stack so the
# central network-transport policy disables system proxy resolution
@ -149,7 +202,8 @@ def worker_main(wid: int, in_q: Any, out_q: Any, repo_dir: str, drive_root: str,
try:
from ouroboros.process_custody import start_parent_lifeline
start_parent_lifeline(label=f"worker-{wid}")
start_parent_lifeline(label=f"worker-{wid}", stop_socket=stop_socket,
before_exit=lambda: request_worker_owned_stops(drive_root))
except Exception:
pass
# Stream this worker's append_jsonl log lines to the dashboard Logs panel.

View file

@ -1134,11 +1134,9 @@ def spawn_workers(n: int = 0) -> None:
try:
for i in range(count):
in_q = _CTX.Queue()
proc = _CTX.Process(target=worker_main,
args=(i, in_q, event_q, str(REPO_DIR), str(DRIVE_ROOT),
_current_custody_session_id()))
proc.daemon = True
proc.start()
from supervisor.worker_process import spawn_worker_process
proc = spawn_worker_process(_CTX, i, in_q, event_q, REPO_DIR, DRIVE_ROOT)
# Unassignable until the readiness seam observes this child's worker_ready row.
new_workers[i] = Worker(wid=i, proc=proc, in_q=in_q, busy_task_id=None, reaping=True)
except Exception:

View file

@ -0,0 +1,765 @@
"""Batch1 (#1307/#1316) through the real candidate server, gateway, WS and SPA.
A byte-faithful candidate checkout serves its own Python and JS (origin proof).
Its isolated data root is initialized by the production state initializer and
one confirmed owner decision, then its primary copy is torn: the server's own
boot recovery must show every control as unknown, never as the backup's old
"on" and never as "off", while Chat keeps working.
Tool facts come from the real loop wrapper inside that server: two rounds reuse
one provider ``tool_call_id``; a ``journal_write`` appends to a FIFO, so its
15 s wait really ends and the call really settles once the test opens a reader;
a crash during a second blocked append leaves a start-only call. Only the
reversed arrival order (settlement row before the wait-end row) is produced
in-process by the same production wrapper: the server cannot make that race
deterministic. Its frames reach the served page over the page's own socket.
"""
from __future__ import annotations
import hashlib
import json
import os
import re
import signal
import subprocess
import threading
import time
import urllib.request
import uuid
from pathlib import Path
import pytest
from tests.candidate_checkout import candidate_checkout
from tests.system_e2e.harness import (
ArtifactOracle,
KeylessIsolatedServer,
ScriptedStubModel,
keyless_settings,
message_text,
wait_durable_result,
wait_until,
write_settings_file,
)
from tests.ui_chat_viewport_smoke import _CAPTURE_TEST_SOCKET
pytestmark = [pytest.mark.serial, pytest.mark.browser,
pytest.mark.skipif(os.name == "nt", reason="FIFO/crash qualification requires POSIX")]
REPO = Path(__file__).resolve().parents[1]
REPEATED_PROVIDER_ID = "call_repeated"
SERVED_MODULES = ("chat.js", "chat_activity.js", "log_events.js", "evolution.js", "logs.js", "activity.js")
def _launch(pw):
"""Playwright's own Chromium when installed, else the machine's Chrome channel."""
from playwright.sync_api import Error
channel = os.environ.get("OUROBOROS_BROWSER_CHANNEL", "")
if channel:
return pw.chromium.launch(channel=channel), channel
try:
return pw.chromium.launch(), "playwright-chromium"
except Error:
return pw.chromium.launch(channel="chrome"), "chrome"
@pytest.fixture
def batch1_clone(tmp_path):
pytest.importorskip("playwright.sync_api")
from playwright.sync_api import Error, sync_playwright
from ouroboros.tools.browser import _set_playwright_browsers_path_if_bundled
_set_playwright_browsers_path_if_bundled()
try:
with sync_playwright() as pw:
browser, _engine = _launch(pw)
browser.close()
except Error as exc:
if os.environ.get("OUROBOROS_EXPECT_BROWSER_ENGINES"):
pytest.fail(str(exc))
pytest.skip(str(exc))
with candidate_checkout(REPO, tmp_path / "clone", origin_proof=True) as candidate:
yield candidate
def _seed_recovering_state(root: Path) -> dict:
"""Production initializer plus one confirmed decision, then a torn primary."""
from supervisor import state as state_module
previous = state_module.DRIVE_ROOT
state_module.init(root)
try:
assert state_module.init_state(origin="first_boot").quality == "current"
state_module.update_state(
lambda st: st.update(evolution_mode_enabled=True, bg_consciousness_enabled=True),
confirm=("evolution_mode_enabled", "bg_consciousness_enabled"))
finally:
state_module.init(previous)
backup = json.loads((root / "state" / "state.last_good.json").read_text(encoding="utf-8"))
assert backup["evolution_mode_enabled"] is True and backup["bg_consciousness_enabled"] is True
torn = b'{"evolution_mode_enabled": true, "bg_consciousness_ena'
(root / "state" / "state.json").write_bytes(torn)
return {"torn": torn, "initialization_id": backup["initialization_id"]}
def _turn_of(body: dict, markers: dict) -> tuple:
"""(turn, round) of a chat-turn body: the LAST owner row naming a marker, and
the tool results the loop has fed back after it."""
messages = body.get("messages") or []
for index in range(len(messages) - 1, -1, -1):
if messages[index].get("role") != "user":
continue
text = message_text(messages[index])
for name, marker in markers.items():
if marker in text:
return name, sum(1 for row in messages[index + 1:] if row.get("role") == "tool")
return None, 0
class _Holds:
"""Event-gated holds of named (turn, round) model calls, outside the stub lock."""
def __init__(self, markers):
self.markers = markers
self.rules = {}
self._lock = threading.Lock()
def add(self, key):
self.rules[key] = {"arrived": threading.Event(), "release": threading.Event(), "taken": False}
return self.rules[key]
def __call__(self, body):
if not body.get("tools"):
return
rule = self.rules.get(_turn_of(body, self.markers))
if rule is None:
return
with self._lock:
if rule["taken"]:
return
rule["taken"] = True
rule["arrived"].set()
if not rule["release"].wait(300):
raise TimeoutError("scenario never released a held model round")
class _Batch1Model(ScriptedStubModel):
"""Scripted turns keyed by owner marker; a step may pin its provider tool_call_id."""
def __init__(self, respond, gate):
super().__init__([respond] * 400, gate=gate)
self._last_step = None
def _next_step(self, body):
step = super()._next_step(body)
self._last_step = step(body) if callable(step) else step
return self._last_step
def _answer(self, body, seq):
self._last_step = None
kind, message = super()._answer(body, seq)
step = self._last_step or {}
if step.get("provider_id") and message.get("tool_calls"):
message = {**message, "tool_calls": [{**message["tool_calls"][0], "id": step["provider_id"]}]}
return kind, message
def _get(url, timeout=15):
with urllib.request.urlopen(url, timeout=timeout) as response: # noqa: S310 - loopback test server
return response.read()
def _api(server, path):
return json.loads(_get(server.base_url + path))
def _tool_rows(server, task_id):
return [row for row in _api(server, f"/api/logs/tools?task_id={task_id}&limit=400")["entries"]
if row.get("task_id") == task_id]
def _git(*args):
return subprocess.run(["git", *args], cwd=REPO, capture_output=True, check=True).stdout
def _candidate_facts(server, candidate):
"""Bind served modules to this captured working candidate, without requiring a commit."""
served = {}
for name in SERVED_MODULES:
body = _get(f"{server.base_url}/static/modules/{name}")
expected = (REPO / "web/modules" / name).read_bytes()
assert body == expected, f"served {name} differs from captured source"
served[name] = hashlib.sha256(body).hexdigest()
return {"source_head": candidate.state.head.decode().strip(),
"candidate_identity": candidate.identity, "checkout_identity": candidate.checkout_identity,
"served_runtime_version": _api(server, "/api/health")["runtime_version"],
"served_module_sha256": served, "server_pid": server.proc.pid, "server_url": server.base_url}
def _open_fifo_reader(path: Path, timeout=30) -> bytes:
"""Become the blocked writer's reader, drain its one line, and return it."""
fd = os.open(str(path), os.O_RDONLY | os.O_NONBLOCK)
data = b""
deadline = time.monotonic() + timeout
try:
while time.monotonic() < deadline:
try:
chunk = os.read(fd, 65536)
except BlockingIOError:
chunk = None
if chunk:
data += chunk
elif chunk == b"" and data.endswith(b"\n"):
return data # the writer wrote its line and closed
time.sleep(0.05)
finally:
os.close(fd)
raise AssertionError(f"FIFO writer never completed: {data!r}")
def _order_b_rows(data_root: Path, task_id: str, monkeypatch) -> tuple:
"""Settlement BEFORE the wait end, through the production wrapper and writer.
The only instrumentation: the timeout path waits until the late worker's
settlement row is durable, making the reversed race deterministic."""
import ouroboros.loop_tool_execution as execution
from ouroboros.tools.tool_result import ToolResult
from tests.test_tool_call_log import _Registry
release = threading.Event()
def handler(_name, _args):
release.wait(timeout=20)
return ToolResult(status="ok", code="OK", text="Order-B late read result")
registry = _Registry(data_root, handler, round_id="orderb:round:1")
registry._ctx.task_metadata = {"budget_drive_root": str(data_root)}
logs = data_root / "logs"
original = execution._make_timeout_result
def settle_first(*args, **kwargs):
release.set()
invocation = kwargs["invocation"]["invocation_id"]
wait_until(lambda: any(row.get("invocation_id") == invocation and row.get("type") == "tool_call"
for row in _jsonl(logs / "tools.jsonl")), 20, 0.05)
return original(*args, **kwargs)
monkeypatch.setattr(execution, "_make_timeout_result", settle_first)
try:
tc = {"id": REPEATED_PROVIDER_ID, "function": {"name": "read_file", "arguments": json.dumps({"path": "VERSION"})}}
result = execution._execute_with_timeout(registry, tc, logs, 1, task_id)
finally:
monkeypatch.setattr(execution, "_make_timeout_result", original)
assert result["tool_result"].code == "TOOL_TIMEOUT"
frames = [frame for frame in registry.frames
if frame.get("type") in {"tool_call_started", "tool_call", "tool_call_timeout"}]
return frames
def _jsonl(path: Path) -> list:
try:
text = path.read_text(encoding="utf-8")
except FileNotFoundError:
return []
return [json.loads(line) for line in text.splitlines() if line.strip()]
class _Observer:
"""Every page error, console error, failed request and HTTP error, by scenario phase."""
def __init__(self, page):
self.phase = "boot1"
self.rows = []
self.statuses = {}
page.on("pageerror", lambda error: self._add("pageerror", str(error)))
page.on("console", lambda msg: msg.type == "error" and self._add("console", msg.text))
page.on("requestfailed", lambda req: self._add("requestfailed", f"{req.method} {req.url} {req.failure}", url=req.url))
page.on("response", self._response)
def _response(self, resp):
self.statuses.setdefault(resp.url, []).append(resp.status)
if resp.status >= 400:
self._add("http", f"{resp.status} {resp.url}")
def _add(self, kind, text, url=""):
self.rows.append({"phase": self.phase, "kind": kind, "text": text, "url": url})
def classified(self):
"""Chromium reports a body-less 204 fetch as ERR_ABORTED although the page got it."""
for row in self.rows:
if (row["kind"] == "requestfailed" and "ERR_ABORTED" in row["text"]
and 204 in self.statuses.get(row["url"], [])):
row["benign"] = "response 204 received for the same URL"
return self.rows
def outside(self, phases):
return [row for row in self.classified() if row["phase"] not in phases and not row.get("benign")]
def _card(page, task_id):
return page.locator(f'#page-chat .chat-live-card[data-task-id="{task_id}"]')
def _card_text(page, task_id):
return page.evaluate("id => document.querySelector(`#page-chat .chat-live-card[data-task-id=\"${id}\"]`)?.textContent || ''",
task_id)
def _card_lines(page, task_id):
"""Every rendered line of one live card (hidden ones included) with its phase class."""
return page.evaluate("""id => {
const card = document.querySelector(`#page-chat .chat-live-card[data-task-id="${id}"]`);
if (!card) return null;
return {phase: card.querySelector('[data-live-phase]')?.textContent?.trim() || '',
lines: [...card.querySelectorAll('.chat-live-line')].map(line => ({
cls: line.className, text: line.textContent.replace(/\\s+/g, ' ').trim()}))};
}""", task_id)
def _expand(page, task_id):
"""Open a live card's timeline, so a screenshot shows its rows, not only the summary."""
button = _card(page, task_id).locator("[data-live-summary-button]")
button.wait_for(state="attached", timeout=30000)
if button.get_attribute("aria-expanded") != "true":
button.click()
page.wait_for_function("id => document.querySelector(`#page-chat .chat-live-card[data-task-id=\"${id}\"] "
"[data-live-summary-button]`)?.getAttribute('aria-expanded') === 'true'",
arg=task_id, timeout=15000)
def _fold(view):
return next((line for line in (view or {}).get("lines", []) if " tool call" in line["text"]), None)
def _menu(page):
page.locator("#page-chat .chat-header-more").evaluate("details => { details.open = true; }")
return {cmd: page.locator(f'#page-chat [data-chat-command="{cmd}"]').evaluate(
"b => ({text: b.textContent, on: b.classList.contains('on'), tone: b.dataset.tone || '', title: b.title})")
for cmd in ("bg", "evolve")}
def _wait_menu(page, bg_text, evolve_text, timeout=45000):
page.wait_for_function(
"""([bg, evolve]) => {
const text = cmd => document.querySelector(`#page-chat [data-chat-command="${cmd}"]`)?.textContent;
return text('bg') === bg && text('evolve') === evolve;
}""", arg=[bg_text, evolve_text], timeout=timeout)
return _menu(page)
def _evolution_pills(page):
page.locator('[data-nav-page="dashboard"]').click()
page.locator('[data-dashboard-tab="evolution"]').click()
page.locator("#evo-mode-pill").wait_for(state="visible", timeout=30000)
page.wait_for_function("() => !/^Evolution$/.test(document.querySelector('#evo-mode-pill')?.textContent || '')",
timeout=30000)
return {"evolution": page.locator("#evo-mode-pill").inner_text(),
"consciousness": page.locator("#evo-bg-pill").inner_text(),
"evolution_class": page.locator("#evo-mode-pill").get_attribute("class"),
"consciousness_class": page.locator("#evo-bg-pill").get_attribute("class")}
def _to_chat(page):
page.locator('[data-nav-page="chat"]').click()
page.locator("#page-chat.active").wait_for(timeout=15000)
def _send(page, text):
page.locator("#chat-input").fill(text)
page.locator("#chat-send").click()
def _direct_task(oracle, marker, timeout=120):
return wait_until(lambda: next((row["task"] for row in oracle.events("task_received")
if marker in str(row.get("task", {}).get("text", ""))), None), timeout)
def test_batch1_unknown_controls_and_tool_history_reach_the_real_spa(batch1_clone, tmp_path, monkeypatch):
from playwright.sync_api import sync_playwright
from ouroboros.tool_call_log import logical_calls
evidence = Path(os.environ.get("OUROBOROS_BROWSER_EVIDENCE_OUT") or tmp_path / "evidence")
evidence.mkdir(parents=True, exist_ok=True)
receipt: dict = {"assertions": [], "red": []}
def passed(name, **facts):
receipt["assertions"].append({"name": name, **facts})
def red(name, **facts):
"""A product consumer defect: kept as evidence while the other branches still run."""
receipt["red"].append({"name": name, **facts})
root = tmp_path / "instance" / "data"
root.mkdir(parents=True)
seeded = _seed_recovering_state(root)
home = tmp_path / "home"
home.mkdir()
original_env = KeylessIsolatedServer._env
monkeypatch.setattr(KeylessIsolatedServer, "_env", lambda server: {
**original_env(server), "HOME": str(home), "USERPROFILE": str(home),
"XDG_CONFIG_HOME": str(home / ".config")})
markers = {name: f"B1_{name}_{uuid.uuid4().hex}" for name in ("A", "B", "C")}
fifo = {name: root / "projects" / f"batch1-fifo-{name.lower()}" / "journal.jsonl" for name in ("A", "B")}
scenario = {"crashed": False, "closed": False}
def journal(name):
fifo[name].parent.mkdir(parents=True, exist_ok=True)
if not fifo[name].exists():
os.mkfifo(fifo[name])
return {"tool": "journal_write", "arguments": {
"kind": "note", "text": f"Batch1 FIFO milestone {name}", "project_id": fifo[name].parent.name}}
steps = {
"A": [lambda: {"tool": "read_file", "arguments": {"root": "system_repo", "path": "VERSION"},
"provider_id": REPEATED_PROVIDER_ID},
lambda: {"tool": "read_file", "arguments": {"root": "system_repo", "path": "README.md"},
"provider_id": REPEATED_PROVIDER_ID},
lambda: journal("A"),
lambda: {"final": "Batch1 turn A is complete."}],
"B": [lambda: journal("B"), lambda: {"final": "Batch1 turn B is complete."}],
"C": [lambda: {"final": "Batch1 turn C is complete."}],
}
def respond(body):
turn, index = _turn_of(body, markers)
if scenario["closed"] or turn is None or (turn == "B" and scenario["crashed"]):
return {"final": "Nothing further is needed."}
return steps[turn][index]() if index < len(steps[turn]) else {"final": f"Batch1 turn {turn} is complete."}
holds = _Holds(markers)
hold_a = holds.add(("A", 3))
hold_c = holds.add(("C", 0))
shots = evidence
with _Batch1Model(respond, holds) as stub:
settings_path = root / "settings.json"
# The owner-settable global cap is only a floor over each tool's own
# timeout (max(setting, per-tool)); 1 s leaves journal_write its 15 s.
write_settings_file(settings_path, keyless_settings(stub, OUROBOROS_MAX_WORKERS=1,
OUROBOROS_TOOL_TIMEOUT_SEC=1))
server = KeylessIsolatedServer(batch1_clone, root, settings_path)
server.start(ready_timeout=300)
oracle = ArtifactOracle(root)
try:
receipt["candidate"] = _candidate_facts(server, batch1_clone)
# -- Boot recovery: the torn primary is preserved, the backup's "on" is not promoted.
api = _api(server, "/api/state")
assert api["evolution_enabled"] is None and api["bg_consciousness_enabled"] is None, api
assert api["state_quality"]["quality"] == "recovered", api["state_quality"]
assert {"evolution_mode_enabled", "bg_consciousness_enabled"} <= set(api["state_quality"]["unconfirmed"])
corrupt = sorted((root / "state").glob("state.corrupt-*.json"))
assert [path.read_bytes() for path in corrupt] == [seeded["torn"]]
passed("boot_recovery_unknown", api_state={k: api[k] for k in ("evolution_enabled", "bg_consciousness_enabled", "state_quality")},
evolution_status=api["evolution_state"].get("status"), bg_status=api["bg_consciousness_state"].get("status"),
corrupt_copy=corrupt[0].name)
with sync_playwright() as pw:
browser, engine = _launch(pw)
receipt["browser"] = {"engine": engine, "version": browser.version}
page = browser.new_page(viewport={"width": 1440, "height": 1000}, reduced_motion="reduce")
observer = _Observer(page)
page.add_init_script(f"({_CAPTURE_TEST_SOCKET})()")
try:
page.goto(server.base_url, wait_until="domcontentloaded")
page.wait_for_function("() => window.__testSockets?.some(s => s.readyState === WebSocket.OPEN)", timeout=60000)
menu = _wait_menu(page, "Consciousness · unknown", "Evolve · unknown")
assert not menu["bg"]["on"] and not menu["evolve"]["on"] and menu["bg"]["tone"] == "warn"
page.screenshot(path=str(shots / "01-boot1-unknown-controls-menu.png"), animations="disabled")
passed("chat_menu_unknown", menu=menu)
pills = _evolution_pills(page)
assert pills["evolution"] == "Evolution unknown" and pills["consciousness"] == "Consciousness unknown", pills
page.screenshot(path=str(shots / "02-boot1-unknown-evolution-panel.png"), animations="disabled")
passed("evolution_panel_unknown", pills=pills)
# A narrow viewport: the longer unknown labels must still fit the menu.
mobile = browser.new_page(viewport={"width": 390, "height": 844}, has_touch=True, is_mobile=True)
mobile_observer = _Observer(mobile)
try:
mobile.goto(server.base_url, wait_until="domcontentloaded")
mobile_menu = _wait_menu(mobile, "Consciousness · unknown", "Evolve · unknown")
fit = mobile.evaluate("""() => [...document.querySelectorAll('#page-chat [data-chat-command="bg"], '
+ '#page-chat [data-chat-command="evolve"]')].map(b => { const r = b.getBoundingClientRect();
return {left: r.left, right: r.right, width: r.width, visible: r.width > 0 && r.height > 0,
viewport: innerWidth}; })""")
mobile.screenshot(path=str(shots / "02m-boot1-unknown-controls-menu-390.png"), animations="disabled")
assert all(item["visible"] and item["left"] >= 0 and item["right"] <= item["viewport"] for item in fit), fit
passed("mobile_menu_unknown_fits", menu=mobile_menu, geometry=fit,
console_network=mobile_observer.outside(set()))
assert not mobile_observer.outside(set()), mobile_observer.rows
finally:
mobile.close()
_to_chat(page)
# -- Turn A: repeated provider id, a real wait end, then a real late success.
_send(page, f"{markers['A']} read two files and record a journal milestone")
task_a = _direct_task(oracle, markers["A"])
assert task_a, "the owner message never reached a turn under unknown controls"
a_id = task_a["id"]
assert hold_a["arrived"].wait(180), "turn A never reached the round after its wait end"
rows = _tool_rows(server, a_id)
calls = logical_calls(rows)
assert [call["tool"] for call in calls] == ["read_file", "read_file", "journal_write"], rows
assert [call["state"] for call in calls] == ["settled", "settled", "wait_ended"], calls
assert {call["started"]["tool_call_id"] for call in calls[:2]} == {REPEATED_PROVIDER_ID}
assert len({call["invocation_id"] for call in calls}) == 3
assert calls[2]["started"]["timeout_sec"] == 15 and calls[2]["wait_ended"]["waited_ms"] >= 14000
card = _card(page, a_id)
card.wait_for(state="attached", timeout=30000)
page.wait_for_function("id => (document.querySelector(`#page-chat .chat-live-card[data-task-id=\"${id}\"]`)"
"?.textContent || '').includes('3 tool calls · wait ended')", arg=a_id, timeout=30000)
held_text = _card_text(page, a_id)
assert "Tool wait ended; operation may still settle · journal_write" in held_text, held_text
held = _card_lines(page, a_id)
fold = _fold(held)
assert fold and "error" not in fold["text"] and " warn" not in fold["cls"] and "calling" not in fold["cls"], held
_expand(page, a_id)
card.scroll_into_view_if_needed()
page.screenshot(path=str(shots / "03-boot1-wait-ended-before-settlement.png"), animations="disabled")
passed("live_wait_end_is_not_failure", invocation_ids=[c["invocation_id"] for c in calls],
provider_ids=[c["started"]["tool_call_id"] for c in calls])
line = _open_fifo_reader(fifo["A"])
assert json.loads(line)["text"] == "Batch1 FIFO milestone A"
settled = wait_until(lambda: (lambda c: c if c[2]["state"] == "settled" else None)(
logical_calls(_tool_rows(server, a_id))), 30)
assert settled and settled[2]["settled"]["is_error"] is False
assert settled[2]["settled"]["result_preview"].startswith("OK: journal[batch1-fifo-a]")
assert settled[2]["wait_ended"]["invocation_id"] == settled[2]["invocation_id"]
page.wait_for_function("id => !(document.querySelector(`#page-chat .chat-live-card[data-task-id=\"${id}\"]`)"
"?.textContent || '').includes('operation may still settle')", arg=a_id, timeout=30000)
late = _card_lines(page, a_id)
assert "3 tool calls · wait ended" in _fold(late)["text"] and "error" not in _fold(late)["text"], late
assert not any(" warn" in line["cls"] or " error" in line["cls"] for line in late["lines"]), late
page.screenshot(path=str(shots / "04-boot1-late-settlement-while-turn-continues.png"), animations="disabled")
passed("late_success_retires_notice_while_task_continues",
settlement=settled[2]["settled"]["result_preview"][:80],
waited_ms=settled[2]["wait_ended"].get("waited_ms"), elapsed_ms=settled[2]["settled"].get("elapsed_ms"))
hold_a["release"].set()
result_a = wait_durable_result(oracle, a_id, timeout=120)
page.get_by_text("Batch1 turn A is complete.", exact=True).wait_for(timeout=60000)
page.wait_for_timeout(1500)
done = _card_lines(page, a_id)
assert "3 tool calls" in _fold(done)["text"] and "error" not in _fold(done)["text"], done
_expand(page, a_id)
fold = card.locator(".chat-live-line.expandable").filter(has_text="3 tool calls")
if fold.count() and fold.first.is_visible():
fold.first.click()
page.wait_for_timeout(500)
receipt["turn_a_expanded"] = _card_lines(page, a_id)
page.screenshot(path=str(shots / "05-boot1-turn-a-terminal-expanded.png"), animations="disabled")
execution_a = (result_a.get("outcome_axes") or {}).get("execution") or {}
passed("terminal_live_card_counts_one_invocation_each", status=result_a.get("status"),
card=done, held=held, late=late,
execution_axis={"status": execution_a.get("status"),
"unresolved_tool_errors": execution_a.get("unresolved_tool_errors")})
# -- An owner decision makes one control known; the other stays unknown.
observer.phase = "owner-commands"
_send(page, "/evolve stop")
wait_until(lambda: _api(server, "/api/state")["evolution_enabled"] is False, 60)
menu = _wait_menu(page, "Consciousness · unknown", "Evolve")
assert not menu["evolve"]["on"] and menu["evolve"]["tone"] == ""
page.screenshot(path=str(shots / "06-boot1-known-disabled-beside-unknown.png"), animations="disabled")
passed("known_disabled_distinct_from_unknown", menu=menu,
api={k: _api(server, "/api/state")[k] for k in ("evolution_enabled", "bg_consciousness_enabled")})
# -- A transiently unreadable primary: display-only backup, no known control.
observer.phase = "transient-unreadable"
primary = root / "state" / "state.json"
primary.chmod(0)
try:
wait_until(lambda: _api(server, "/api/state")["state_quality"]["quality"] == "recovered_transient", 30)
transient = _api(server, "/api/state")
assert transient["evolution_enabled"] is None and transient["bg_consciousness_enabled"] is None
menu = _wait_menu(page, "Consciousness · unknown", "Evolve · unknown")
page.screenshot(path=str(shots / "07-boot1-transient-unreadable-unknown.png"), animations="disabled")
passed("transient_unreadable_is_unknown", state_quality=transient["state_quality"], menu=menu)
finally:
primary.chmod(0o600)
wait_until(lambda: _api(server, "/api/state")["evolution_enabled"] is False, 30)
_wait_menu(page, "Consciousness · unknown", "Evolve")
passed("returning_primary_restores_known_control")
observer.phase = "boot1"
# -- Turn C, live: the reversed arrival order from the production wrapper.
_send(page, f"{markers['C']} answer after a short wait")
task_c = _direct_task(oracle, markers["C"])
c_id = task_c["id"]
assert hold_c["arrived"].wait(120)
frames = _order_b_rows(root, c_id, monkeypatch)
assert [frame["type"] for frame in frames] == ["tool_call_started", "tool_call", "tool_call_timeout"]
durable_c = logical_calls(_tool_rows(server, c_id))
assert len(durable_c) == 1 and durable_c[0]["state"] == "settled"
order = [row["type"] for row in _tool_rows(server, c_id)]
assert order == ["tool_call_started", "tool_call", "tool_call_timeout"], order
page.evaluate("""frames => {
const socket = window.__testSockets.find(s => s.readyState === WebSocket.OPEN);
for (const data of frames) socket.dispatchEvent(new MessageEvent('message',
{data: JSON.stringify({type: 'log', chat_id: 1, data: {...data, chat_id: 1}})}));
}""", frames)
page.wait_for_function("id => (document.querySelector(`#page-chat .chat-live-card[data-task-id=\"${id}\"]`)"
"?.textContent || '').includes('1 tool call · wait ended')", arg=c_id, timeout=30000)
c_text = _card_text(page, c_id)
c_view = _card_lines(page, c_id)
receipt["order_b_card"] = c_view
assert "1 tool call · wait ended" in _fold(c_view)["text"] and "error" not in _fold(c_view)["text"], c_view
_expand(page, c_id)
_card(page, c_id).scroll_into_view_if_needed()
page.screenshot(path=str(shots / "08-boot1-order-b-settled-before-wait-end.png"), animations="disabled")
passed("order_b_one_invocation_not_failed", durable_order=order,
invocation_id=durable_c[0]["invocation_id"], card=c_view)
if "operation may still settle" in c_text:
(shots / "08-order-b-card-dom.html").write_text(_card(page, c_id).evaluate("n => n.outerHTML"),
encoding="utf-8")
red("order_b_stale_wait_notice", card=c_view, frames=[f["type"] for f in frames],
detail="settlement arrived first; the later wait-end frame still adds the provisional "
"'operation may still settle' warn row, which nothing retires")
hold_c["release"].set()
wait_durable_result(oracle, c_id, timeout=120)
page.get_by_text("Batch1 turn C is complete.", exact=True).wait_for(timeout=60000)
page.wait_for_timeout(1500)
receipt["order_b_card_after_terminal"] = _card_lines(page, c_id)
# -- Turn B: a crash while its call is blocked leaves a start-only invocation.
_send(page, f"{markers['B']} record one more journal milestone")
task_b = _direct_task(oracle, markers["B"])
b_id = task_b["id"]
started = wait_until(lambda: [row for row in _tool_rows(server, b_id) if row["type"] == "tool_call_started"], 60, 0.2)
assert started and started[0]["tool"] == "journal_write" and started[0]["timeout_sec"] == 15, started
page.wait_for_function("id => (document.querySelector(`#page-chat .chat-live-card[data-task-id=\"${id}\"]`)"
"?.textContent || '').includes('1 tool call')", arg=b_id, timeout=10000)
_expand(page, b_id)
receipt["b_before_crash"] = _card_lines(page, b_id)
page.screenshot(path=str(shots / "09-boot1-turn-b-blocked-before-crash.png"), animations="disabled")
observer.phase = "crash-restart"
scenario["crashed"] = True
os.kill(server.proc.pid, signal.SIGKILL)
server.proc.wait(timeout=30)
rows_b = _jsonl(root / "logs" / "tools.jsonl")
assert [row["type"] for row in rows_b if row.get("task_id") == b_id] == ["tool_call_started"], rows_b
server.stop()
fifo["B"].unlink()
passed("crash_left_start_only", invocation_id=started[0]["invocation_id"])
# -- Boot 2 on the same root: reconnect, reload, Logs backfill.
server.start(ready_timeout=300)
receipt["candidate_boot2"] = _candidate_facts(server, batch1_clone)
page.wait_for_function("() => window.__testSockets?.some(s => s.readyState === WebSocket.OPEN)", timeout=120000)
api2 = _api(server, "/api/state")
assert api2["evolution_enabled"] is False and api2["bg_consciousness_enabled"] is None, api2
observer.phase = "boot2"
receipt["reconnected_b_initial"] = _card_lines(page, b_id)
# The restore hands the caught direct turn to the cancel-intent watchdog
# (20 s cadence, intents >= 10 s old): wait for ITS terminal, then read the card.
booted = time.monotonic()
stored_b = wait_until(lambda: (lambda r: r if r.get("status") not in ("running", "", None) else None)(
oracle.task_result(b_id)), 150, 1.0)
receipt["b_durable_after_restart"] = {
"status": (stored_b or oracle.task_result(b_id)).get("status"),
"seconds_after_boot2": round(time.monotonic() - booted, 1)}
assert stored_b, receipt["b_durable_after_restart"]
try:
page.wait_for_function("id => !['Working', 'Cancelling…'].includes((document.querySelector("
"`#page-chat .chat-live-card[data-task-id=\"${id}\"] [data-live-phase]`)"
"?.textContent || '').trim())", arg=b_id, timeout=60000)
except Exception:
pass
reconnected_b = _card_lines(page, b_id)
if _card(page, b_id).count():
_card(page, b_id).screenshot(path=str(shots / "10b-boot2-reconnected-turn-b-card.png"), animations="disabled")
if reconnected_b and (reconnected_b["phase"] in ("Working", "Cancelling…")
or any("calling" in line["cls"] for line in reconnected_b["lines"])):
red("kept_open_terminal_card_keeps_interrupted_call_calling", card=reconnected_b,
durable=receipt["b_durable_after_restart"])
page.screenshot(path=str(shots / "10-boot2-reconnected-same-page.png"), animations="disabled")
page.reload(wait_until="domcontentloaded")
page.get_by_text("Batch1 turn A is complete.", exact=True).wait_for(timeout=60000)
menu = _wait_menu(page, "Consciousness · unknown", "Evolve")
page.wait_for_timeout(3000)
for label, task in (("a", a_id), ("b", b_id), ("c", c_id)):
if _card(page, task).count():
_expand(page, task)
_card(page, task).scroll_into_view_if_needed()
_card(page, task).screenshot(path=str(shots / f"11{label}-boot2-reloaded-turn-{label}-card.png"),
animations="disabled")
reloaded = {task: _card_lines(page, task) for task in (a_id, b_id, c_id)}
page.screenshot(path=str(shots / "11-boot2-reloaded-chat.png"), full_page=True, animations="disabled")
receipt["reload"] = {"menu": menu, "cards": reloaded, "reconnected_b": reconnected_b}
a_reload_text = _card_text(page, a_id)
if re.search(r"\b\d+ errors?\b", a_reload_text) or (_fold(reloaded[a_id]) or {}).get("cls", "").find("warn") >= 0:
red("reload_turn_a_timeout_counted_as_error", card=reloaded[a_id],
detail="after reload the card carries only host round totals; the durable late success "
"in tools.jsonl does not reach Chat")
view_b = reloaded[b_id]
if view_b and (view_b["phase"] in ("Working", "Cancelling…")
or any("calling" in line["cls"] for line in view_b["lines"])):
red("reloaded_start_only_turn_still_live", card=view_b, durable=receipt["b_durable_after_restart"])
served_calls = {task: logical_calls(_tool_rows(server, task)) for task in (a_id, b_id, c_id)}
assert [call["state"] for call in served_calls[a_id]] == ["settled"] * 3
assert [call["state"] for call in served_calls[b_id]] == ["unknown"]
assert [call["state"] for call in served_calls[c_id]] == ["settled"]
passed("gateway_history_logical_calls",
states={task: [(c["tool"], c["state"], "wait_ended" in c) for c in calls] for task, calls in served_calls.items()})
page.locator('[data-nav-page="dashboard"]').click()
page.locator('[data-dashboard-tab="logs"]').click()
page.wait_for_function("ids => ids.every(id => document.querySelector(`[data-task-group=\"${id}\"]`))",
arg=[a_id, b_id, c_id], timeout=60000)
logs_view = page.evaluate("""ids => Object.fromEntries(ids.map(id => {
const card = document.querySelector(`[data-task-group="${id}"]`);
return [id, {headline: card.querySelector('[data-task-headline]').textContent,
phase: card.querySelector('[data-task-phase]').textContent,
timeline: [...card.querySelectorAll('.log-task-event .log-headline, .log-task-event .log-main')]
.map(node => node.textContent.replace(/\\s+/g, ' ').trim())}];
}))""", [a_id, b_id, c_id])
for task in (a_id, b_id):
group = page.locator(f'[data-task-group="{task}"]')
group.locator("details.log-task-details").evaluate("d => { d.open = true; }")
page.locator(f'[data-task-group="{b_id}"]').scroll_into_view_if_needed()
page.screenshot(path=str(shots / "12-boot2-logs-backfill.png"), full_page=True, animations="disabled")
receipt["logs_view"] = logs_view
_to_chat(page)
# -- Known enabled and known disabled after a real owner decision.
observer.phase = "owner-commands"
_send(page, "/bg start")
wait_until(lambda: _api(server, "/api/state")["bg_consciousness_enabled"] is True, 60)
scenario["closed"] = True
menu_on = _wait_menu(page, "Consciousness", "Evolve")
assert menu_on["bg"]["on"] and not menu_on["evolve"]["on"]
page.screenshot(path=str(shots / "13-boot2-known-enabled-and-disabled.png"), animations="disabled")
_send(page, "/bg stop")
wait_until(lambda: _api(server, "/api/state")["bg_consciousness_enabled"] is False, 60)
menu_off = _wait_menu(page, "Consciousness", "Evolve")
page.wait_for_function("() => !document.querySelector('#page-chat [data-chat-command=\"bg\"]').classList.contains('on')",
timeout=30000)
menu_off = _menu(page)
pills2 = _evolution_pills(page)
page.screenshot(path=str(shots / "14-boot2-known-controls-evolution-panel.png"), animations="disabled")
passed("known_enabled_and_disabled_distinct", menu_on=menu_on, menu_off=menu_off, pills=pills2)
observer.phase = "boot2"
receipt["observer"] = observer.classified()
unexpected = observer.outside({"crash-restart"})
receipt["unexpected_console_network"] = unexpected
assert not unexpected, unexpected
assert not receipt["red"], json.dumps(receipt["red"], ensure_ascii=False)[:4000]
except Exception:
page.screenshot(path=str(shots / "failure.png"), full_page=True, animations="disabled")
(shots / "failure-dom.html").write_text(page.content(), encoding="utf-8")
receipt["observer"] = observer.rows
raise
finally:
for rule in (hold_a, hold_c):
rule["release"].set()
browser.close()
finally:
for rule in (hold_a, hold_c):
rule["release"].set()
for path in fifo.values():
if path.exists() and not path.is_file():
try:
_open_fifo_reader(path, timeout=2)
except Exception:
pass
server.stop()
(evidence / "receipt.json").write_text(json.dumps(receipt, ensure_ascii=False, indent=2, default=str),
encoding="utf-8")

View file

@ -0,0 +1,351 @@
"""Narrow counterexamples from the frozen Batch1 review; isolated consumers only."""
from __future__ import annotations
import asyncio
import json
import os
import pathlib
import sys
import time
from types import SimpleNamespace
import pytest
from supervisor import state, state_initialization
from tests import test_schedule_occurrence as schedule_fixtures
from tests import test_state_authority as state_fixtures
root, _prior, _write = state_fixtures.root, state_fixtures._prior, state_fixtures._write
q, _rows = schedule_fixtures.q, schedule_fixtures._rows
pytestmark = pytest.mark.serial
@pytest.mark.parametrize("writer", ["update", "save", "init"])
def test_missing_controls_survive_all_existing_state_writers(root, writer):
state.save_state(_prior())
raw = json.loads(state.STATE_PATH.read_bytes())
for key in ("owner_external_id", "owner_external_chat_id", "evolution_owner_stopped"):
raw.pop(key)
_write(state.STATE_PATH, raw)
if writer == "update":
state.update_state(lambda live: live.update(last_owner_message_at="now"))
elif writer == "save":
state.save_state({**raw, "owner_external_id": None})
else:
state.init_state()
state.init(root) # cold read, not a process-local recovery flag
for key in ("owner_external_id", "owner_external_chat_id", "evolution_owner_stopped"):
assert state.control_value(state.load_state(), key) == (False, None)
state.update_state(lambda live: live.update(evolution_owner_stopped=True),
confirm=("evolution_owner_stopped",))
assert state.control_value(state.load_state(), "evolution_owner_stopped") == (True, True)
assert state.control_value(state.load_state(), "owner_external_id") == (False, None)
@pytest.mark.parametrize("changed", [False, True])
def test_interrupted_legacy_adoption_resumes_only_exact_source(root, monkeypatch, changed):
original = _write(state.STATE_PATH, _prior())
real = state.atomic_write_text
monkeypatch.setattr(state, "atomic_write_text", lambda *_: False)
assert state.init_state().quality == "unavailable"
identity = state_initialization.read_witness(root)[1]["initialization_id"]
assert state.STATE_PATH.read_bytes() == original
if changed:
_write(state.STATE_PATH, _prior(session_id="different"))
monkeypatch.setattr(state, "atomic_write_text", real)
observed = state.init_state()
assert (observed.quality == "unavailable") is changed
assert state_initialization.read_witness(root)[1]["initialization_id"] == identity
if not changed:
assert state.load_state()["session_id"] == "old-session"
assert state.control_value(state.load_state(), "owner_external_id") == (True, 99)
@pytest.mark.parametrize("zero_write", [False, True])
def test_corrupt_primary_is_not_replaced_until_full_copy(root, monkeypatch, zero_write):
state.save_state(_prior())
corrupt = b'{"truncated":' + b'x' * 4096
state.STATE_PATH.write_bytes(corrupt)
real = os.write
monkeypatch.setattr(os, "write", lambda fd, data: 0 if zero_write else real(fd, data[:17]))
if zero_write:
with pytest.raises(state.StateUnavailable, match="corrupt_primary_unpreserved"):
state.update_state(lambda st: st.update(message_offset=2))
assert state.STATE_PATH.read_bytes() == corrupt
else:
state.update_state(lambda st: st.update(message_offset=2))
[copy] = list((root / "state").glob("state.corrupt-*.json"))
assert copy.read_bytes() == corrupt
def test_context_uses_the_addressed_witness_authority(root):
from ouroboros.context import _drive_state_section
state.save_state(_prior())
witness = root / state_initialization.WITNESS_REL
raw = json.loads(witness.read_bytes())
raw["phase"] = "pending"
_write(witness, raw)
rendered = _drive_state_section(SimpleNamespace(drive_path=lambda p: root / p))
assert '"evolution_mode_enabled": {\n "status": "unknown"' in rendered
assert "initialization_incomplete" in rendered
raw["phase"] = "complete"
_write(witness, raw)
assert '"evolution_mode_enabled": true' in _drive_state_section(SimpleNamespace(drive_path=lambda p: root / p))
def test_consciousness_recovers_unknown_but_never_overrides_live_stop(root, monkeypatch):
from ouroboros.consciousness import BackgroundConsciousness
state.save_state(_prior())
primary, backup = state.STATE_PATH.read_bytes(), state.STATE_LAST_GOOD_PATH.read_bytes()
state.STATE_PATH.unlink()
state.STATE_LAST_GOOD_PATH.unlink()
clock = BackgroundConsciousness(root, root / "repo", lambda: 7, now=100)
monkeypatch.setattr(clock, "live_turns", lambda: ("", False))
assert clock.tick(now=100) == "disabled"
state.STATE_PATH.write_bytes(primary)
state.STATE_LAST_GOOD_PATH.write_bytes(backup)
assert clock.tick(now=100) == "not_due"
clock.stop()
assert clock.tick(now=100) == "disabled"
@pytest.mark.parametrize("project", [False, True])
@pytest.mark.parametrize("legacy", [False, True])
def test_real_gateway_producer_admits_without_injected_intent(q, monkeypatch, project, legacy):
from starlette.applications import Starlette
from starlette.routing import Route
from starlette.testclient import TestClient
from ouroboros.gateway import schedules
from ouroboros.gateway.schedules import api_schedules_upsert
from ouroboros.projects_registry import create_project
monkeypatch.setattr(schedules, "request_drive_root", lambda _: q.root)
if project:
create_project(q.root, "room", name="Folderless room")
if legacy:
from tests.test_schedule_occurrence import _row
_row(q, "api", project_id="room" if project else "")
app = Starlette(routes=[Route('/api/schedules', api_schedules_upsert, methods=['POST'])])
with TestClient(app) as client:
response = client.post('/api/schedules', json={"id": "api", "trigger": {
"type": "once", "run_at": "2000-01-01T00:00:00Z"}, "task": {
"type": "task", "text": "Do the scheduled work", "chat_id": 1,
**({"project_id": "room"} if project else {})}})
assert response.status_code == 200, response.text
q.queue.check_scheduled_tasks()
if legacy:
assert not q.pending
assert _rows(q)["api"]["hold"]["reason"] == "resource_intent_unknown"
return
[task] = q.pending
assert task["metadata"]["resource_intent"]["kind"] == ("room_default" if project else "system_repo")
assert not _rows(q)["api"].get("hold")
def test_real_skill_producer_removal_preserves_unpublished_admission(q, monkeypatch):
from supervisor import queue_schedules
from tests.test_consciousness_schedule_controls import _ready, _skill
_ready(monkeypatch)
queue_schedules.sync_skill_schedules([_skill()], drive_root=q.root)
record = _rows(q)["skill-demo-daily"]
record["next_run_at"] = "2000-01-01T00:00:00Z"
queue_schedules._write_scheduled_tasks({"tasks": [record]}, q.root)
real = q.queue.persist_queue_snapshot
monkeypatch.setattr(q.queue, "persist_queue_snapshot", lambda **_: False)
q.queue.check_scheduled_tasks()
assert not q.pending
accepted = _rows(q)["skill-demo-daily"]["occurrence"]["task_id"]
queue_schedules.sync_skill_schedules([], drive_root=q.root)
tombstone = _rows(q)["skill-demo-daily"]
assert tombstone["delete_requested_at"] and not tombstone["enabled"]
monkeypatch.setattr(q.queue, "persist_queue_snapshot", real)
q.queue.check_scheduled_tasks()
assert [task["id"] for task in q.pending] == [accepted]
assert q.pending[0]["metadata"]["resource_intent"] == {"kind": "system_repo"}
def test_cli_schedule_add_uses_the_gateway_producer(q, monkeypatch):
from starlette.applications import Starlette
from starlette.routing import Route
from starlette.testclient import TestClient
from ouroboros import cli
from ouroboros.gateway import schedules
from supervisor import queue_schedules
monkeypatch.setattr(schedules, "request_drive_root", lambda _: q.root)
app = Starlette(routes=[Route('/api/schedules', schedules.api_schedules_upsert, methods=['POST'])])
with TestClient(app) as client:
# Only HTTP transport is in-process: real parser, CLI body, gateway and store.
monkeypatch.setattr(cli, '_client', lambda _: SimpleNamespace(
request=lambda method, path, body: client.request(method, path, json=body).json()))
assert cli.main(['schedule', 'add', '--name', 'CLI work', '--cron', '* * * * *', 'Do work']) == 0
[record] = _rows(q).values()
record['next_run_at'] = '2000-01-01T00:00:00Z' # advance to the producer's due occurrence
queue_schedules._write_scheduled_tasks({'tasks': [record]}, q.root)
q.queue.check_scheduled_tasks()
[task] = q.pending
assert task['metadata']['resource_intent'] == {'kind': 'system_repo'}
def test_history_reloads_late_settlement_over_frozen_summary(tmp_path):
from ouroboros.gateway.history import make_chat_history_endpoint
from ouroboros.post_task_synthesis import task_tool_metrics
from ouroboros.tool_call_log import append_call_row
from ouroboros.utils import append_jsonl
metrics = task_tool_metrics({"tool_calls": [{"tool": "run_command", "is_error": True}]})
append_jsonl(tmp_path / "logs/chat.jsonl", {"type": "task_summary", "direction": "out",
"role": "assistant", "chat_id": 1, "task_id": "t", "text": "Done", "ts": "2026-09-27T01:00:00Z", **metrics})
for event in ("tool_call_started", "tool_call_timeout", "tool_call"):
assert append_call_row({}, tmp_path / "logs", {"type": event, "task_id": "t",
"invocation_id": "call", "tool": "run_command", "is_error": event == "tool_call_timeout",
"status": "ok" if event == "tool_call" else "timeout"})["task_log"]
response = asyncio.run(make_chat_history_endpoint(tmp_path)(SimpleNamespace(query_params={"limit": "10"})))
summary = next(row for row in json.loads(response.body)["messages"] if row.get("task_id") == "t")
observations = summary["tool_evidence"]["observations"]
assert [row["fact"] for row in observations] == ["started", "wait_ended", "settled"]
assert observations[-1]["status"] == "ok" and summary["tool_errors"] == 1 # frozen historical wait count
def test_osworld_readers_count_unfinished_calls_once_and_keep_legacy(tmp_path):
from devtools.benchmarks.osworld import run_cu_bridge_agent as bridge
from ouroboros.extension_loader import extension_name_prefix
prefix = extension_name_prefix(bridge.SKILL_NAME)
root = tmp_path / 'state/headless_tasks/t/data/logs'
root.mkdir(parents=True)
rows = [{"type": t, "task_id": "t", "tool": prefix + "screenshot", "invocation_id": "i"}
for t in ("tool_call_started", "tool_call_timeout")]
rows += [{"type": "tool_call", "tool": prefix + "remote_exec", "args": {"cmd": "pwd"}}] * 2
(root / 'tools.jsonl').write_text(''.join(json.dumps(row) + '\n' for row in rows))
counts = bridge._collect_budget_counters(tmp_path, {}, "t")
assert (counts["skill_tool_calls"], counts["screenshots"], counts["remote_exec_calls"]) == (3, 1, 2)
trace = bridge._gate_tool_trace(tmp_path, "t")
assert len(trace) == 3 and trace[0]["state"] == "wait_ended"
assert trace[0]["is_error"] is None and not trace[0]["settled"]
def _pooled_test_entry(*args):
"""Real worker_main and registry, with only model/extension setup replaced."""
from ouroboros import agent, extension_loader
from ouroboros.tools import shell_process
from ouroboros.tools.registry import ToolContext, ToolRegistry
from supervisor.worker_process import worker_main
class CommandAgent:
def handle_task(self, task):
import logging
from ouroboros import process_custody
root = pathlib.Path(args[4])
workspace = root / 'workspace'
workspace.mkdir(exist_ok=True)
ctx = ToolContext(repo_dir=root / 'repo', drive_root=root / 'data',
workspace_root=workspace, workspace_mode='external', task_id='pooled')
registry = ToolRegistry(repo_dir=ctx.repo_dir, drive_root=ctx.drive_root)
registry.set_context(ctx)
shell_process._subprocess_lock.acquire() # cleanup must not gate requests
handler = logging.StreamHandler()
handler.acquire() # persistence/logging must not gate the worker's own exit
process_custody.log.addHandler(handler)
process_custody.log.setLevel(logging.WARNING)
result = registry.execute('run_command', {'cmd': [sys.executable, '-c',
'import os,time,pathlib; pathlib.Path("child.pid").write_text(str(os.getpid())); time.sleep(60)']})
(root / 'command-result.txt').write_text(str(result))
return []
agent.make_agent = lambda **_: CommandAgent()
extension_loader.reload_all = lambda *_a, **_k: None
import ouroboros.safety
ouroboros.safety.check_safety = lambda *_a, **_k: (True, '')
worker_main(*args)
def test_actual_pooled_worker_requests_separate_session_command_before_owner_exit(tmp_path, monkeypatch):
import multiprocessing
from ouroboros.platform_layer import pid_is_alive
from ouroboros.process_containment import pid_is_zombie
from supervisor import worker_pool_lifecycle, worker_process
monkeypatch.setenv('OUROBOROS_RUNTIME_MODE', 'advanced')
monkeypatch.setattr(worker_process, 'worker_main', _pooled_test_entry)
ctx = multiprocessing.get_context('spawn')
incoming, outgoing = ctx.Queue(), ctx.Queue()
proc = worker_process.spawn_worker_process(ctx, 0, incoming, outgoing, tmp_path, tmp_path)
child_pid = 0
try:
incoming.put({'id': 'pooled', 'type': 'task'})
deadline = time.monotonic() + 20
while not (tmp_path / 'workspace/child.pid').exists() and time.monotonic() < deadline:
assert proc.is_alive(), proc.exitcode
time.sleep(.03)
assert (tmp_path / 'workspace/child.pid').exists(), (tmp_path / 'command-result.txt').read_text()
child_pid = int((tmp_path / 'workspace/child.pid').read_text())
if os.name != 'nt':
assert os.getpgid(child_pid) == child_pid != os.getpgid(proc.pid)
started = time.monotonic()
receipt = worker_pool_lifecycle.kill_worker_tree(proc.pid, panic_process=proc)
assert receipt['requested'] and receipt['scope'] == 'worker_owners'
assert time.monotonic() - started < .5
proc.join(timeout=5)
assert not proc.is_alive()
deadline = time.monotonic() + 3
while pid_is_alive(child_pid) and not pid_is_zombie(child_pid) and time.monotonic() < deadline:
time.sleep(.01)
assert not pid_is_alive(child_pid) or pid_is_zombie(child_pid)
finally:
if proc.is_alive():
proc.terminate()
proc.join(timeout=5)
if child_pid and pid_is_alive(child_pid) and not pid_is_zombie(child_pid):
# This test created and continuously observed the child; no unrelated PID search.
os.kill(child_pid, 9)
proc._ouroboros_stop_socket.close()
for channel in (incoming, outgoing):
channel.close()
channel.cancel_join_thread()
@pytest.mark.parametrize("platform_name", ["linux", "darwin"])
def test_attached_request_never_turns_held_identity_into_numeric_group(monkeypatch, platform_name):
from ouroboros import platform_layer as platform
calls = []
monkeypatch.setattr(platform, "IS_WINDOWS", False)
monkeypatch.setattr(platform, "IS_MACOS", platform_name == "darwin")
monkeypatch.setattr(platform.os, "getpgid", lambda *_: pytest.fail("numeric group lookup loses identity"))
monkeypatch.setattr(platform.os, "kill", lambda *_: pytest.fail("numeric PID cannot signal an attachment"))
monkeypatch.setattr(platform.os, "killpg", lambda *_: pytest.fail("numeric group cannot signal an attachment"))
monkeypatch.setattr(platform.signal, "pidfd_send_signal", lambda *a: calls.append(a), raising=False)
target = {"pid": 123, "handle": SimpleNamespace(fileno=lambda: 987, close=lambda: None), "pgid": 123}
receipt = platform.request_process_tree_kill(target)
assert receipt["requested"] is (platform_name == "linux")
assert receipt["scope"] == "process"
assert calls == ([(987, platform.signal.SIGKILL)] if platform_name == "linux" else [])
def test_partial_primary_cannot_register_a_stranger_after_bookkeeping(root, monkeypatch):
import server
from supervisor import message_bus
from tests.test_state_authority import _bridge, _ingress_ctx
state.save_state(_prior())
raw = json.loads(state.STATE_PATH.read_bytes())
raw.pop("owner_external_id")
raw.pop("owner_external_chat_id")
_write(state.STATE_PATH, raw)
state.update_state(lambda live: live.update(message_offset=99))
replies, panics = [], []
monkeypatch.setattr(message_bus, "log_chat", lambda *_a, **_k: None)
monkeypatch.setattr(message_bus, "record_inbound_message", lambda *_a, **_k: {})
monkeypatch.setattr(server, "_execute_panic_stop", lambda *_: panics.append(True))
server._process_bridge_updates(_bridge('/panic', user=5, chat=5), 0, _ingress_ctx(root, replies, panics))
assert not panics and any('unknown' in reply for reply in replies)
assert state.control_value(state.load_state(), 'owner_external_id') == (False, None)

View file

@ -54,10 +54,14 @@ def test_public_panic_requests_attached_and_local_children_before_blocked_settle
return result
def persist_after_requests(_root):
assert {r["pid"] for r in receipt_calls if r["requested"]} == {p.pid for p in procs}
expected = {procs[1].pid} if platform.IS_MACOS else {p.pid for p in procs}
assert {r["pid"] for r in receipt_calls if r["requested"]} == expected
if platform.IS_MACOS:
assert any(r["pid"] == procs[0].pid and not r["requested"] for r in receipt_calls)
assert manager._lock.locked() and model._lock.locked()
for proc in procs:
assert proc.wait(timeout=2) is not None
if proc.pid in expected:
assert proc.wait(timeout=2) is not None
flag_calls.append(True)
monkeypatch.setattr(platform, "request_process_tree_kill", observe_request)
@ -109,7 +113,13 @@ def test_attached_request_uses_pinned_identity_without_rereading_custody(
manager._lock.acquire()
try:
receipts = manager.panic_stop(request_only=True)
assert receipts[0]["requested"] and receipts[0]["pid"] == proc.pid
assert receipts[0]["pid"] == proc.pid
if platform.IS_MACOS:
assert not receipts[0]["requested"] and "signalable identity" in receipts[0]["error"]
assert proc.poll() is None
platform.request_process_tree_kill(proc) # ordinary owned child remains signalable
else:
assert receipts[0]["requested"]
assert proc.wait(timeout=2) is not None
finally:
manager._lock.release()
@ -193,7 +203,12 @@ def test_panic_settlement_does_not_request_a_successor_daemon(tmp_path, monkeypa
successor = None
try:
manager.ensure_running()
assert manager.panic_stop(request_only=True)[0]["requested"]
receipt = manager.panic_stop(request_only=True)[0]
if platform.IS_MACOS:
assert not receipt["requested"]
platform.request_process_tree_kill(original)
else:
assert receipt["requested"]
original.wait(timeout=2)
successor = _child(tmp_path, daemon.CUSTODY_PURPOSE)
monkeypatch.setattr(manager, "_classify_liveness", lambda **_: (object(), "running", ""))

View file

@ -6,6 +6,7 @@ import json
import pathlib
import types
import pytest
import ouroboros.config as config
import ouroboros.post_task_evolution as pte
@ -162,11 +163,11 @@ def test_v5_apply_pending_request_activates_gated_campaign(tmp_path, monkeypatch
"start_evolution_campaign",
lambda objective, source="": started.update(objective=objective, source=source) or {"id": "test"},
)
monkeypatch.setattr(st, "load_state", lambda: {"owner_chat_id": 7})
monkeypatch.setattr(st, "load_state", lambda: {"owner_chat_id": 7, "evolution_mode_enabled": False, "evolution_owner_stopped": False})
monkeypatch.setattr(st, "save_state", lambda s: saved.update(s))
def _fake_update_state(mutator, **_kwargs):
live = {"owner_chat_id": 7}
live = {"owner_chat_id": 7, "evolution_mode_enabled": False, "evolution_owner_stopped": False}
mutator(live)
saved.update(live)
return live
@ -373,7 +374,7 @@ def test_toggle_evolution_on_clears_owner_stop(tmp_path, monkeypatch):
calls = {"complete": [], "start": []}
def _fake_update_state(mutator, **_kwargs):
live = {"owner_chat_id": 7}
live = {"owner_chat_id": 7, "evolution_mode_enabled": False, "evolution_owner_stopped": False}
mutator(live)
captured.update(live)
return live
@ -386,7 +387,7 @@ def test_toggle_evolution_on_clears_owner_stop(tmp_path, monkeypatch):
lambda objective="", *, source="": calls["start"].append((objective, source)) or {"id": "test"})
ctx = types.SimpleNamespace(
load_state=lambda: {"owner_chat_id": 7},
load_state=lambda: {"owner_chat_id": 7, "evolution_mode_enabled": False, "evolution_owner_stopped": False},
send_with_budget=lambda cid, text, **kw: None,
)
_handle_toggle_evolution({"enabled": True, "objective": "improve X"}, ctx)
@ -404,7 +405,7 @@ def test_toggle_evolution_start_failure_sends_owner_correction(monkeypatch):
monkeypatch.setattr(evolution_lifecycle, "start_evolution_campaign", lambda *a, **k: {})
sent = []
ctx = types.SimpleNamespace(
load_state=lambda: {"owner_chat_id": 7, "evolution_mode_enabled": False},
load_state=lambda: {"owner_chat_id": 7, "evolution_mode_enabled": False, "evolution_owner_stopped": False},
send_with_budget=lambda chat_id, text, **kw: sent.append((chat_id, text)),
)
@ -489,7 +490,7 @@ def test_execute_panic_stop_wires_owner_stop(tmp_path, monkeypatch):
import ouroboros.tools.shell as _shell
import ouroboros.local_model as _lm
monkeypatch.setattr(_shell, "kill_all_tracked_subprocesses", lambda *a, **k: None)
monkeypatch.setattr(_lm, "get_manager", lambda: types.SimpleNamespace(stop_server=lambda: None))
monkeypatch.setattr(_lm, "get_manager", lambda **_: types.SimpleNamespace(stop_server=lambda: None, panic_stop=lambda **_: []))
class _StopPanic(Exception):
pass
@ -649,7 +650,7 @@ def _apply_with_request(tmp_path, monkeypatch, backlog_id):
)
monkeypatch.setattr(lifecycle, "_read_evolution_campaign", lambda: camp)
monkeypatch.setattr(lifecycle, "_write_evolution_campaign", lambda c: camp.update(c))
monkeypatch.setattr(stt, "load_state", lambda: {"owner_chat_id": 7})
monkeypatch.setattr(stt, "load_state", lambda: {"owner_chat_id": 7, "evolution_mode_enabled": False, "evolution_owner_stopped": False})
monkeypatch.setattr(stt, "save_state", lambda s: None)
def _fake_update_state(mutator, **_kwargs):
@ -657,7 +658,7 @@ def _apply_with_request(tmp_path, monkeypatch, backlog_id):
# unlocked loader, so the load_state patch above never reaches it — on a machine whose
# LIVE state carries evolution_owner_stopped=True the atomic re-check would then refuse
# the enable and apply would return False for reasons outside this test's control.
live = {"owner_chat_id": 7}
live = {"owner_chat_id": 7, "evolution_mode_enabled": False, "evolution_owner_stopped": False}
mutator(live)
return live
@ -724,3 +725,19 @@ def test_promotion_chooser_uses_main_model_slot(tmp_path, monkeypatch):
monkeypatch.setenv("OUROBOROS_MODEL", "")
pte._decide_promotion(env, {"id": "t2"}, {"reflection": "r"}, object(), force=False)
assert calls.get("model") == config.SETTINGS_DEFAULTS["OUROBOROS_MODEL"]
@pytest.mark.parametrize("missing", ["evolution_owner_stopped", "evolution_mode_enabled", "owner_chat_id"])
def test_apply_pending_preserves_request_while_control_is_unknown(tmp_path, monkeypatch, missing):
monkeypatch.setenv("OUROBOROS_POST_TASK_EVOLUTION", "true")
from supervisor import evolution_lifecycle, state
request = tmp_path / "state/post_task_evolution_request.json"
request.parent.mkdir()
request.write_text(json.dumps({"objective": "improve"}))
values = {"owner_chat_id": 7, "evolution_mode_enabled": False, "evolution_owner_stopped": False}
values.pop(missing)
monkeypatch.setattr(state, "load_state", lambda: values)
monkeypatch.setattr(evolution_lifecycle, "evolution_block_reason", lambda: "")
monkeypatch.setattr(evolution_lifecycle, "start_evolution_campaign", lambda *_a, **_k: pytest.fail("unknown cannot start"))
assert pte.apply_pending_request(tmp_path) is False
assert request.exists()

View file

@ -20,7 +20,7 @@ def test_readable_unicode_name_replaces_only_what_a_filesystem_cannot_hold():
assert project_folder_basename("x\ty\x00z") == "x_y_z"
assert project_folder_basename(" ..hidden.. ") == "hidden"
assert project_folder_basename("...") == project_folder_basename(" ") == ""
for reserved in ("CON", "con.md", "Lpt9", "nul.tar.gz"):
for reserved in ("CON", "con.md", "Lpt9", "nul.tar.gz", "COM¹", "com².txt", "COM³", "LPT¹.doc", "LPT²", "lpt³.tar.gz"):
stem = project_folder_basename(reserved).split(".", 1)[0]
assert stem.endswith("_") and stem[:-1].casefold() == reserved.split(".", 1)[0].casefold()
# Normalization happens only at creation: a decomposed spelling mints the NFC name.

View file

@ -223,7 +223,10 @@ def test_a_change_during_prepare_never_launches_the_old_choice(q, tmp_path, monk
if change == "delete":
assert "s1" not in _rows(q)
return
assert "occurrence" not in _rows(q)["s1"] # the claim was dropped, not held
if change == "rebind":
assert _rows(q)["s1"]["hold"]["reason"] == "project_routing_fence_changed"
else:
assert "occurrence" not in _rows(q)["s1"] # disabled/edited claims are dropped
monkeypatch.setattr(occurrences, "prepare", real_prepare)
q.queue.check_scheduled_tasks()
if change == "disable":

View file

@ -280,7 +280,7 @@ def test_panic_requests_every_physical_stop_before_any_persistence_and_never_wai
log=SimpleNamespace(critical=lambda *a, **k: release.wait(30)), bound_port=12345)
release.set()
assert time.monotonic() - started < 15 # every write bounded, never awaited
assert "workers" in order and ("port", 12345) in order
assert "workers" not in order and ("port", 12345) in order # lifelines own pooled child requests
assert order.index("flag") < order.index("state")
assert (root / "state" / "panic_stop.flag").read_text() == "panic"

View file

@ -814,19 +814,17 @@ export function createChatInstance({
|| record.toolErrors > 0;
}
// Tool accounting from a metrics or terminal fact: the meta counts and the
// block's one folded evidence row. A field the fact does not carry stays
// absent, so a partial snapshot cannot reclassify the row.
// Fold counters with canonical facts; absent fields stay absent.
function noteToolMetrics(taskId, metrics, rawTs, { suppressDomInsert = false } = {}) {
const known = (key) => (Number.isInteger(metrics?.[key]) ? metrics[key] : null);
const [calls, errors, routing] = ['tool_calls', 'tool_errors', 'routing_tool_calls'].map(known);
if ((!calls && !errors) || subagentChildParents.has(taskId)) return false;
if (!calls && !errors && !metrics.tool_evidence?.observations?.length && !metrics.tool_evidence?.legacy?.calls) return false;
return withStableViewport(() => {
const record = getLiveCardRecord(taskId);
const before = captureLiveCardProjection(record);
const duration = Number(metrics.duration_sec);
if (Number.isFinite(duration)) record.durationSec = duration;
const summary = noteToolHostMetrics(record, { calls, errors, routing, counts: metrics.tool_call_counts });
const summary = noteToolHostMetrics(record, { calls, errors, routing, counts: metrics.tool_call_counts, evidence: metrics.tool_evidence });
record.toolCalls = summary.calls;
record.toolErrors = summary.errors;
const { timelineUpdate } = upsertToolFoldRow(record, summary, normalizeLogTs(rawTs), rawTs);
@ -1702,6 +1700,9 @@ export function createChatInstance({
}
// One tool evidence row per block.
const foldView = summary.toolCall ? applyToolObservation(record, summary.toolCall) : null;
if (summary.toolCall?.fact === 'wait_ended' && record.toolFold.calls.get(summary.toolCall.key)?.settlement) {
summary = { ...summary, visible: false }; // wait history remains in the fold
}
if (record.finished && !isTerminalTaskPhase(nextPhase, summary.terminal)) {
if (foldView) {
upsertToolFoldRow(record, foldView, ts, rawTs);
@ -1782,7 +1783,6 @@ export function createChatInstance({
renderCollapsedActivity(record, activityText);
const shouldRenderLine = summary.visible !== false && Boolean(headline || summary.body);
// Parent-child replay and child lifecycle/progress update in place.
let timelineUpdate = 'none';
let patchIndex = -1;
if (_historyRow?.history_id) {
@ -1840,6 +1840,10 @@ export function createChatInstance({
// Every terminal route releases controls and subscriptions together.
function settleLiveCard(record, phase, wasFinished) {
record.root.dataset.finished = '1';
if (record.toolFold) {
upsertToolFoldRow(record, noteToolHostMetrics(record, {}), '', '');
renderLiveCardTimeline(record);
}
cancelableTaskIds.delete(record.groupId);
syncCancelRunButton(record);
modelWaits.finish(record.groupId);
@ -1852,8 +1856,7 @@ export function createChatInstance({
function finishLiveCardMutation(groupId = '', phase = '') {
const record = groupId ? liveCardRecords.get(groupId) : null;
if (!record) return false;
// A converted card is a terminal project chip now — ignore late terminal
// frames so they neither overwrite the chip nor touch its element refs (T4).
// Converted project chips ignore later task terminals (T4).
if (record.root?.dataset?.projectCreated === '1') return false;
const before = captureLiveCardProjection(record);
const typingBefore = typingEl.style.display;
@ -1989,10 +1992,10 @@ export function createChatInstance({
if (review !== undefined) return review;
if (!taskId) return false;
modelWaits.observe(taskId, msg);
let changed = false;
let changed = msg.tool_evidence ? noteToolMetrics(taskId, msg, rawTs) : false;
// Only host-attested progress grants Stop authority.
if (grantCancelAuthority && msg?.cancelable === true && msg?.task_id) {
changed = markTaskCancelable(String(msg.task_id));
changed = markTaskCancelable(String(msg.task_id)) || changed;
}
const lifecycleParent = taskKey(msg?.parent_task_id);
if (msg?.subagent_event && lifecycleParent) {
@ -2184,9 +2187,9 @@ export function createChatInstance({
}
const childInfo = subagentChildParents.get(taskId);
if (childInfo && eventType === 'task_done') return routeSubagentTerminalToCard(taskId, evt);
if (childInfo && subagentTerminalChildren.has(taskId)) return false;
// Tool accounting (metrics or the terminal) is the same fact for a
// child and its owner; the rows themselves come from the summarizer.
if (childInfo && subagentTerminalChildren.has(taskId)
&& !['tool_call_started', 'tool_call', 'tool_call_finished', 'tool_call_timeout', 'tool_timeout'].includes(eventType)) return false;
// Metrics and terminal facts share the root/child fold.
let changed = ['task_metrics_event', 'task_eval', 'task_done'].includes(eventType)
? noteToolMetrics(taskId, evt, rawTs) : false;
if (!childInfo) changed = attachTaskDetailReviews(taskId, evt) || changed;
@ -2205,10 +2208,7 @@ export function createChatInstance({
);
if (childInfo) return Boolean(changed || queued);
const subagentChanged = updateSubagentCardFromEvent(evt, rawTs);
// The host stamps the lane on the turn's own frames (task_done always,
// a direct turn's tool frames too), so the header pill never waits for
// a census; the host-attested Stop marker rides a direct turn's tool
// frames the way it rides its narration rows, so a tool-only turn offers Stop.
// Host-attested lane and Stop facts also travel on tool-only turns.
if (typeof evt._is_direct_chat === 'boolean') noteDirectTurn(liveCardRecords.get(taskId), evt._is_direct_chat);
if (evt.cancelable === true) markTaskCancelable(taskId);
if (eventType === 'task_done' && summary.terminal) {

View file

@ -153,7 +153,7 @@ export function noteToolCall(record, observation) {
}
} else if (fact === 'wait_ended') next.waitEnded = true;
else next.started = true;
next.live = next.live || observation.live === true || (!observation.fact && observation.status === 'calling');
next.live = !record.finished && (next.live || observation.live === true || (!observation.fact && observation.status === 'calling'));
next.status = next.settlement?.status || (next.waitEnded ? 'wait_ended' : next.live ? 'calling' : 'unknown');
calls.set(key, next);
return record;
@ -170,7 +170,13 @@ export function noteToolCall(record, observation) {
* is never explicitly emptied while the turn counts calls.
*/
export function noteToolHostMetrics(record, host) {
for (const observation of host?.evidence?.observations || []) applyToolObservation(record, observation);
const fold = ensureToolFold(record);
if (host?.evidence?.coverage) { fold.coverage = host.evidence.coverage; fold.legacy = host.evidence.legacy; }
if (record.finished) for (const call of fold.calls.values()) {
call.live = false;
call.status = call.settlement?.status || (call.waitEnded ? 'wait_ended' : 'unknown');
}
const known = fold.host || {};
const carry = (next, before) => (next === null || next === undefined ? (before ?? null) : next);
const counts = host?.counts && typeof host.counts === 'object' && Object.keys(host.counts).length > 0
@ -218,10 +224,12 @@ const perToolLine = (entries) => entries
export function toolEvidenceView(fold = null) {
const live = fold?.calls instanceof Map ? [...fold.calls.values()] : [];
const host = fold?.host || null;
const calls = Number.isInteger(host?.calls) ? host.calls : live.length;
// Host round totals include wait errors. Once every call has its own
// observation, use operation settlements rather than resurrecting a timeout.
const errors = live.length >= calls ? live.filter(call => call.status === 'error').length
const observed = live.length + (fold?.legacy?.calls || 0);
const calls = Math.max(Number.isInteger(host?.calls) ? host.calls : 0, observed);
// Frozen totals count model wait errors. Canonical evidence reports operation
// outcomes; a bounded partial read discloses its gap instead of reviving waits.
const partial = Boolean(fold?.coverage) && observed > 0 && observed < calls;
const errors = observed >= calls || partial ? live.filter(call => call.status === 'error').length + (fold?.legacy?.errors || 0)
: (Number.isInteger(host?.errors) ? host.errors : live.filter(call => call.status === 'error').length);
const liveCounts = new Map();
for (const call of live) liveCounts.set(call.tool, (liveCounts.get(call.tool) || 0) + 1);
@ -230,10 +238,10 @@ export function toolEvidenceView(fold = null) {
phase: errors > 0 ? 'warn'
: ((!host && live.some((call) => call.status === 'calling')) ? 'calling' : 'result'),
headline: `${plural(calls, 'tool call')}${errors > 0 ? ` · ${plural(errors, 'error')}` : ''}`
+ (live.some(call => call.waitEnded) ? ' · wait ended' : '')
+ (live.some(call => call.status === 'unknown') ? ' · outcome unknown' : ''),
+ (live.some(call => call.waitEnded) || fold?.legacy?.wait_ended ? ' · wait ended' : '')
+ (live.some(call => call.status === 'unknown') || fold?.legacy?.unknown || partial ? ' · outcome unknown' : ''),
body: '',
fullBody: perToolLine(host?.counts && typeof host.counts === 'object'
fullBody: (partial ? 'Invocation evidence is incomplete. ' : '') + perToolLine(host?.counts && typeof host.counts === 'object'
? Object.entries(host.counts) : [...liveCounts]),
visible: true,
// Addressing calls report themselves on the owner's message, so a block

View file

@ -351,18 +351,6 @@ function extractCommandText(args) {
return '';
}
// The compact row for one tool call: the command, else the first string
// argument (a path, a query, a url — whatever the tool names first), lexical
// only. The complete arguments stay behind the row's expand.
function toolCallTarget(args) {
const cmd = extractCommandText(args);
if (cmd) return cmd;
for (const value of Object.values(args && typeof args === 'object' ? args : {})) {
if (typeof value === 'string' && value.trim()) return value;
}
return '';
}
// Legacy observations lack host identity: retain them separately rather than
// guessing a call from a reused provider id, tool name or target.
const legacyToolObservations = new WeakMap();

View file

@ -745,10 +745,9 @@ test('a host note inside a child leaves the child\'s collapsed line and title al
} finally { f.close(); }
});
// The disclosed residual: a child folds its calls while they happen, and the
// host's at-rest metrics stay the owner's (`noteToolMetrics` skips children),
// so a reloaded child card carries no evidence row.
test('a child card folds its own tool calls live and takes no at-rest evidence row', () => {
// Child cards fold their own live calls and host counters. Canonical child
// history is exercised separately below.
test('a child folds its own live calls and host counters', () => {
const f = fixture();
try {
f.census(managed());
@ -764,8 +763,8 @@ test('a child card folds its own tool calls live and takes no at-rest evidence r
assert.match(folded()[0].innerHTML, /2 tool calls/);
f.log({ type: 'task_metrics_event', task_id: CHILD, tool_calls: 5, tool_errors: 0,
tool_call_counts: { read_file: 5 } });
assert.equal(folded().length, 1, 'the at-rest fact belongs to the owning turn: a child takes no row from it');
assert.doesNotMatch(folded()[0].innerHTML, /5 tool calls/);
assert.equal(folded().length, 1, 'host counters update the same folded row');
assert.match(f.meta(CHILD), /5 tool calls/);
} finally { f.close(); }
});
@ -805,3 +804,63 @@ test('real Chat clears a provisional wait notice after durable successful settle
assert.match(rows, /1 tool call/);
} finally { f.close(); }
});
for (const order of ['wait-first', 'settlement-first']) for (const failed of [false, true]) {
test(`terminal child keeps independent tool facts: ${order}, error=${failed}`, () => {
const f = fixture();
try {
f.emit('chat', { role: 'assistant', is_progress: true, content: 'Child working', task_id: TASK,
subagent_event: 'scheduled', subagent_task_id: 'kid', parent_task_id: TASK,
root_task_id: TASK, delegation_role: 'subagent', subagent_role: 'researcher' });
const identity = { task_id: 'kid', tool: 'read_file', invocation_id: 'kid-call' };
f.log({ type: 'tool_call_started', ...identity });
f.emit('chat', { role: 'assistant', is_progress: true, content: 'Child finished', task_id: TASK,
subagent_event: 'completed', subagent_task_id: 'kid', parent_task_id: TASK,
root_task_id: TASK, delegation_role: 'subagent', subagent_role: 'researcher' });
const wait = { type: 'tool_call_timeout', ...identity };
const result = { type: 'tool_call', ...identity, status: failed ? 'error' : 'ok', is_error: failed };
for (const row of order === 'wait-first' ? [wait, result] : [result, wait]) f.log(row);
f.card('kid').querySelector('[data-live-summary-button]').listeners.get('click')[0]({ detail: 0 });
const rows = f.rows('kid').map(row => row.innerHTML).join(' ');
assert.match(rows, /1 tool call/);
assert.match(rows, /wait ended/);
assert.doesNotMatch(rows, /operation may still settle/);
assert.equal(/1 error/.test(rows), failed);
assert.equal(f.card('kid').dataset.finished, '1');
assert.equal(Boolean(f.card('kid').querySelector('[data-cancel-run]')), false);
} finally { f.close(); }
});
}
for (const child of [false, true]) test(`cold Chat merges canonical settlement with persisted metrics: child=${child}`, async () => {
const id = child ? CHILD : TASK;
const base = { key: `tool:${id}:real-i`, tool: 'read_file', receipt: false, live: false };
const lineage = child ? { delegation_role: 'subagent', parent_task_id: TASK, root_task_id: TASK,
subagent_task_id: CHILD, subagent_role: 'scout' } : {};
const f = fixture([{ ...final, ...lineage, task_id: id, role: 'system', system_type: 'task_summary', text: 'Done',
ts: TS, chat_id: 1, tool_calls: 1, tool_errors: 1, tool_call_counts: { read_file: 1 },
tool_evidence: { observations: [
{ ...base, fact: 'started', status: 'unknown' },
{ ...base, fact: 'wait_ended', status: 'unknown' },
{ ...base, fact: 'settled', status: 'ok' },
] } }]);
try {
await f.instance.refreshHistory({ revision: 1 });
if (child) f.card(id).querySelector('[data-live-summary-button]').listeners.get('click')[0]({ detail: 0 });
const rows = f.rows(id).map(row => row.innerHTML).join(' ');
assert.match(rows, /1 tool call.*wait ended/);
assert.doesNotMatch(rows, /1 error|operation may still settle/);
assert.doesNotMatch(f.meta(id), /1 error/);
} finally { f.close(); }
});
test('terminal projection retires live calling without inventing settlement', () => {
const record = {};
noteToolCall(record, { key: 'interrupted', tool: 'read_file', fact: 'started', live: true });
record.finished = true;
const view = noteToolHostMetrics(record, {});
assert.match(view.headline, /outcome unknown/);
assert.equal(view.phase, 'result');
assert.equal(record.toolFold.calls.get('interrupted').settlement, undefined);
});