diff --git a/docs/DESIGN.md b/docs/DESIGN.md index de3769710..afdcf7c87 100644 --- a/docs/DESIGN.md +++ b/docs/DESIGN.md @@ -225,7 +225,11 @@ relabel the whole still-working task. A failed child keeps a compact factual authoritative status. Internal reason codes belong in details and diagnostics, not compact headlines. Where a card does show a cause, it says it in the owner's words while the record keeps the machine code; a cause with no sentence yet stays -raw rather than borrowing a wrong one. A terminal whose preserved output was +raw rather than borrowing a wrong one. The routing receipt under an owner +message is such a surface: a refused addressing act carries the host-composed +`cause` sentence (`project_dialogue.routing_refusal_cause` — one host table for +the receipt line, the System row and the picker toast), a landed act carries +none, and an unknown reason stays raw. A terminal whose preserved output was never reviewed shows that output labelled rather than hidden: a short labelled excerpt beside the pointer to the full copy, so a `Failed` card over applied work is never a bare headline and never names preserved bytes without a way to reach @@ -523,7 +527,17 @@ addressing calls, without error, is a receipt row too. So a turn that only addressed work («turn this into a project») draws no block, live or on reload: the annotation on the owner message and the managed root's own card are its whole record (owner decision 11.09). A failed addressing call is an error row -and therefore content, as is any recorded tool error. No client list of tool +and therefore content, as is any recorded tool error. A REFUSED addressing act +is told where the work lives and never in Ouroboros's voice (owner 16.09): the +receipt line states the cause in the owner's words, the failed call stays the +error row inside the block, and the tool result carries the cause with its +repair (`detail`) so the model narrates — no host bubble interrupts a narrating +turn. When the host itself issued the act (a Swarm message, a skill-card repair, +a picker click) no turn narrates, so the refusal lands as ONE typed System row — +`task_not_started`, or `task_start_unconfirmed` when admission could not be +confirmed — in the chat the owner wrote in, keyed to the never-started task, +beside the receipt; the Project start row is announced only once the task is +really queued. No client list of tool names decides presence (`docs/development/02-naming-and-boundaries.md`, "an open default behind a closed exception list"). diff --git a/docs/architecture/01-high-level-architecture.md b/docs/architecture/01-high-level-architecture.md index dc4609924..cd767b62f 100644 --- a/docs/architecture/01-high-level-architecture.md +++ b/docs/architecture/01-high-level-architecture.md @@ -32,7 +32,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de │ ├── active_activity.py ← Process-local owner of in-flight native chat actors, including preparation and post-task delivery (`DirectActivityRegistry`); private actor handles support controls and writer drain, public snapshots feed `/api/state` `active_direct_turns` and WS typing frames (`activity_id`, `client_message_id`, `phase`, `kind`); no queue records │ ├── message_bus.py ← Queue-based local message bus (Web UI + reviewed transport skills) │ ├── workers.py ← Multiprocessing worker pool (forkserver on Linux, spawn on macOS/Windows; never fork from the multi-threaded supervisor) - │ ├── worker_assignment.py, worker_chat_lane.py, worker_health.py, worker_pool_lifecycle.py, worker_process.py, worker_promotion.py ← The pool's leaves: handing a pending task to a free worker and refusing the ones that must not run; the direct chat lane and its resume after a restart (conversation stays admitted while the authorized assisted resolver holds the repository, and the owner-control path is preloaded before conflict markers land); health-owned crash detection and terminal-file recovery through the reaper queue; lifecycle-owned execution admission and pool population (readiness, exhausted-slot disablement, pid records, reaping, respawn) and `kill_worker_tree`, the ONE worker tree-kill every teardown and backstop uses — the installation's daemon roots spared always, kept services only on one task's cancel or timeout; what runs INSIDE a worker child process from entry to crash record; and turning a chat turn — or a project scope — into a queued task + │ ├── worker_assignment.py, worker_chat_lane.py, worker_health.py, worker_pool_lifecycle.py, worker_process.py, worker_promotion.py ← The pool's leaves: handing a pending task to a free worker and refusing the ones that must not run; the direct chat lane and its resume after a restart (conversation stays admitted while the authorized assisted resolver holds the repository, and the owner-control path is preloaded before conflict markers land); health-owned crash detection and terminal-file recovery through the reaper queue; lifecycle-owned execution admission and pool population (readiness, exhausted-slot disablement, pid records, reaping, respawn) and `kill_worker_tree`, the ONE worker tree-kill every teardown and backstop uses — the installation's daemon roots spared always, kept services only on one task's cancel or timeout; what runs INSIDE a worker child process from entry to crash record; and turning a chat turn — or a project scope — into a queued task, refusing typed with the cause and repair in `detail` (single durable writer `_persist_promote_rejection`) and announcing a Project start only once the task is really queued │ ├── worker_owner_wait.py ← Queue-owned active-capacity transfer for required owner waiting: the original task stays RUNNING and its process stays custodied, while another worker receives the active slot; correlated grant consumes the checkpoint before another model round │ ├── state.py ← Persistent state (state/state.json) with file locking │ ├── queue.py ← Task queue (PENDING/RUNNING lists) + activity-based timeout enforcement; the ONE task-state authority — the lifecycle/publication/transition modules below extend it without becoming second authorities; exact-attempt main-LLM in-flight state spares only the idle rail @@ -50,7 +50,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de │ ├── evolution_lifecycle.py ← Evolution campaign state + transaction lifecycle: campaign file IO, start/pause, begin/update transaction, cycle-outcome recording, deterministic worktree cleanup, owner cycle reports, idle dispatch over queue-owned state, supervisor auto-restart request │ ├── events.py ← Worker→supervisor event dispatcher with exact attempt/execution/round/call correlation for the process-local active main-LLM row; composes the frozen subagent task text (`_compose_subagent_text`; the acting `[WRITE SURFACE]` block states only the write-root authority boundary — actor identity comes from the immutable configured snapshot, startup/wake facts from bootstrap); an event type absent from `EVENT_HANDLERS` is DROPPED into a truncated `unknown_worker_event` row, and `tests/test_worker_event_registry.py` pins the registry by AST scan — an unexplained allowlist entry is exactly the silent blessing the scan exists to end; the scan is shape-bounded, and outside its reach the discipline is code review; holds shrink-only byte debt above the module byte ceiling │ ├── subagent_task_truth.py ← Delegation-truth enrichment of the subagent `task_done` transport frame (`enrich_task_done_event`) - │ ├── event_taxonomy.py, events_budget.py, events_chat_delivery.py, events_coop_checkpoint.py, events_evolution_done.py, events_project_routing.py, events_runtime_controls.py, events_schedule_task.py, events_subagent_admission.py, events_task_done.py, events_worker_reports.py ← The handler leaves `events.py` merges into `EVENT_HANDLERS`, one owner per family: the declared disposition of every event kind the runtime puts on `EVENT_Q` (`event_taxonomy.py`, the registry the AST scan reads); usage-accounting and budget-pause reports; owner-facing chat delivery (text, media, typing); cooperative repository checkpoints when a task tree goes quiescent; terminal handling of an evolution task and its campaign; where a chat turn becomes a task and where a project scope is bound; posture changes that are not one task's state; the `schedule_subagent` admission gates (chat target, depth, worker pool, active-child cap) and their refusals, with deliberately no semantic duplicate gate — exact task-id fencing, caps and cost ceilings are the floor and the parent decides what to spawn; admission facts for a requested subagent (census, caps, constraint); resolution of a terminal event into durable truth and delivery; and what a running worker reports about itself + │ ├── event_taxonomy.py, events_budget.py, events_chat_delivery.py, events_coop_checkpoint.py, events_evolution_done.py, events_project_routing.py, events_runtime_controls.py, events_schedule_task.py, events_subagent_admission.py, events_task_done.py, events_worker_reports.py ← The handler leaves `events.py` merges into `EVENT_HANDLERS`, one owner per family: the declared disposition of every event kind the runtime puts on `EVENT_Q` (`event_taxonomy.py`, the registry the AST scan reads); usage-accounting and budget-pause reports; owner-facing chat delivery (text, media, typing); cooperative repository checkpoints when a task tree goes quiescent; terminal handling of an evolution task and its campaign; where a chat turn becomes a task and where a project scope is bound (one publication boundary around every promote outcome: a host-issued act that is refused gets its one typed System row there, a tool-issued one only its receipt); posture changes that are not one task's state; the `schedule_subagent` admission gates (chat target, depth, worker pool, active-child cap) and their refusals, with deliberately no semantic duplicate gate — exact task-id fencing, caps and cost ceilings are the floor and the parent decides what to spawn; admission facts for a requested subagent (census, caps, constraint); resolution of a terminal event into durable truth and delivery; and what a running worker reports about itself │ ├── task_dispatch.py ← Pure admitted-event→worker-payload construction, including the identical top-level/metadata depth and configured-route projections consumed by workers │ ├── log_addressing.py ← Explicit audience for task-scoped live log events: `address_task_event` (lineage from the RUNNING row, project binding wins, explicit chat_id preserved — 0 is `HIDDEN_CHAT_ID`, the hidden partition of the Skill Review panel and headless runs without a registered project; A2A frames are suppressed at the `push_log` choke, not by dishonest addressing), `make_server_log_sink`, `address_handler_push` │ ├── steering.py ← Steering-message delivery to running tasks, keyed on the event's host-minted `issuer` fact: an OWNER turn writes owner text (generation bump, room veto from the registry lane, owner acknowledgement/notice), a TASK speaking for itself writes a `task_message` row with `independent_task` provenance to any host-listed active independent root (no veto, no chat, one `task_message_routed` Logs row naming author and target); mailbox routing to the drive the worker drains, plus a typed refusal while a cancel intent is pending (steering is fenced during a stop — what makes the owner-stop single-turn rail safe) @@ -177,7 +177,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de ├── project_facts.py ← project_id resolution (explicit `--project-id` or workspace-path hash); per-project knowledge dir `projects//knowledge` isolated from `memory/knowledge`; journal/workpad helpers ├── task_tree_ledger.py ← Append-only `data/task_trees//blackboard.jsonl`: EPHEMERAL swarm coordination in typed validated rows (prose is never authoritative), distinct from the durable project journal it is mirrored into by `tools/project_journal.py::mirror_tree_coordination_to_journal`; size-capped writes; exposed via `tree_note`/`tree_read` with the tail injected each turn; pruned on root terminal ├── projects_registry.py ← Durable `data/state/projects.json`: 80-char names, `active|deleting|tombstoned`; deletion preserves bindings/history/folder/memory; a tombstone blocks resurrection; reconcile registers existing stores and NEVER prunes - ├── project_dialogue.py ← Read-only chat lens + append-only `logs/chat_annotations.jsonl`; the sidecar never owns routing except the `needs_manual_target` decision card (token+options validate the click; first-wins closing rows); `build_owner_message_ref` mints refs at ingress + ├── project_dialogue.py ← Read-only chat lens + append-only `logs/chat_annotations.jsonl`; the sidecar never owns routing except the `needs_manual_target` decision card (token+options validate the click; first-wins closing rows); `build_owner_message_ref` mints refs at ingress; `routing_refusal_cause` composes the owner-facing `cause` sentence for every refused routing act from one host table, and `room_membership` keeps admission-notice rows in the chat they were sent to ├── project_lease.py ← One-writer-per-project lease in `assign_tasks`; same-project subagent swarms exempt; `""` is no lane ├── context.py ← Main agent context assembly; places ordered subagent-catalog JSON under `## Available subagents` in the semi-stable block; the knowledge index always carries each note's authored summary, and an unauthored shared overview renders as a visible gap line, never as silence ├── main_context_authority.py ← Deep-copies the context authority; replaces only oversized raw result strings with source-resolvable narrative or a typed gap @@ -226,7 +226,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de ├── workspace_preflight.py ← Read-only external-workspace git/manifest/toolchain snapshot used by gateway task creation ├── project_sources.py ← Folder attach validation (realpath, not the home root, no repo/data overlap); opt-in `init_git`, NEVER auto-init; atomic server-side clone with `GIT_TERMINAL_PROMPT=0` and typed `auth_required`; provenance + `clone_url` recorded; attaching IS the trust grant (`trusted_at` automatic) ├── promotion_source.py ← Promoted-task source admission off the event-drain loop, only after an executor/id reservation - ├── workspace_admission.py ← Shared admission for `/api/tasks` + promotion: disjoint git root, Project binding, `workspace="none"`, bounded preflight, loud refusal of a broken binding; empty-Project promotion provisions idempotent genesis, and a failure is typed `workspace_provisioning_failed` on the promotion path (supervisor/workers.py) — never a fallback onto the system repo + ├── workspace_admission.py ← Shared admission for `/api/tasks` + promotion: disjoint git root, Project binding, `workspace="none"`, bounded preflight, loud refusal of a broken binding; empty-Project promotion provisions idempotent genesis, and a failure is typed `workspace_provisioning_failed` on the promotion path (supervisor/workers.py) — never a fallback onto the system repo; `workspace_repair_hint` composes the model-facing cause + repair for every refused folder from its typed source ├── local_model.py ← Local LLM lifecycle (llama-cpp-python) ├── local_model_autostart.py ← Local model startup helper ├── deep_self_review.py ← Deep self-review of the whole system against BIBLE.md on the configured `deep_review` reviewer row (`reviewer_slot_config.deep_review_slot`; absent = the packed api row synthesized from `OUROBOROS_MODEL_DEEP_SELF_REVIEW`). THREE deliveries by the row's `retrieves` predicate. A direct api row is the PACKED review: deep-review atlas + the full memory whitelist on a ≥1M route → one `chat_observed` call, byte-identical on the wire to the pre-row review (golden digest test); a bounded in-prompt OMITTED section is reserved inside the fixed budget; `atlas_assembly_failed` (over hard budget, or a REQUIRED artifact omitted) ships no pack at all — the compact manifest is the atlas default, so there is no compact retry rung — and a final-shrink rebuild (tighter hard budget by the measured overage) precedes the gate, which remains the fail-closed last assertion; a required set that does not fit under a COLD density cap (no fresh exact-model witness — the refusal precedes every send, so the model could never record one) gets ONE bounded probe send on the exact model (the shared rung `capability_evidence.cold_start_density_probe`, the same one the commit gate runs: a slice of the real pack — `review_helpers.density_probe_sample` — `DENSITY_PROBE_MAX_TOKENS` output tokens, through `chat_observed`) that records the witness, then ONE rebuild under the recalibrated cap; a pack that still does not fit is the typed `deep_self_review_pack_unfit` refusal whose text asks the owner to switch the `deep_review` row to a retrieving delivery or a larger-window model — the host never falls back to another delivery on its own; centrality ranking (reverse in-degree) is deep-review-only. A configured-subagent api row is a NATIVE inspection episode and an `agent_session` row a delegated session, both through `review_execution._review_route_executor` exactly like the advisory: a hand-built `ReviewRequest(surface="deep_self_review")` whose route-owned task carries the role prompt, up to seven memory files inline byte-exact — EVERY whitelisted entry with its disposition (`inlined` / `missing` / `empty` / `oversized` / `read_error`) in the task's memory section, in the usage fact `deep_review_memory` and in the header (`memory=n/7`) — BIBLE.md as a MANDATORY full read and ARCHITECTURE/DEVELOPMENT/CHECKLISTS as `generate_doc_nav_map` navigation maps; `policy["output_contract"]` = the report contract plus the deep review's coverage-header sentence (the shape is `report` either way) and `policy["native_data_root"]` = the real runtime root (readable by the reviewer's tools; the memory whitelist itself arrives inline); a `ReviewSlot` with an explicit logical window (the task's absolute ceiling narrowed by the owner deadline); `record_reviewer_slot_executions("deep_self_review", …)` for «Выполняется как» — recorded on all three retrieving outcomes (responded, empty response, executor exception) from a usage that already carries `deep_review_memory`, which the returned usage keeps too (a typed failure included, together with the executor's failure custody); the durable execution projection itself persists route/model/status/capability_delta and typed failure facts only, never a deep-review-only field; prompt/response custody through `persist_call`. Every delivered report is prefixed by the host provenance header (`` + one human line; every comment value AND every external value on the human line sanitized and bounded (`_header_value`); `incomplete` is derived per delivery — the native episode's typed `native_incomplete`, the packed call's provider stop marker — the OpenAI-compatible `response_finish_reason == "length"` or the direct-Anthropic message `stop_reason == "max_tokens"` — as `output_reserve`, a session's completeness `unobserved`); after a native episode BIBLE.md coverage (`read` / `partial(fraction)` / `missing` / `unobserved`) is derived by `_native_read_coverage` from the executed repository-root `read_file` receipts on the reader's `opened_path` and `opened_root` (contract and edge rules: its docstring and the «Deep self-review» section) and anything but `read` is disclosed in the header and as a typed `capability_delta`, never refused; a session's coverage is `unobserved`. Availability is route-aware (`deep_review_route`): the packed row keeps the ≥1M floor (a window CONFIRMED below 1M by the shared `reviewer_window` resolver is a typed refusal — the pack is never shrunk to a smaller window; an unknown window keeps the full-window assumption, disclosed as `window=assumed_1000000`) and the `OPENAI_BASE_URL` trust rule, a native row needs its model's credentials, a session row the substrate's `route_health`. Every failure returns TYPED usage (`execution_status=infra_failed` + reason code) so `agent.py` never overwrites `memory/deep_review.md` with an error diff --git a/docs/architecture/03-web-ui-pages-and-buttons.md b/docs/architecture/03-web-ui-pages-and-buttons.md index 5cf24098d..aa0498e00 100644 --- a/docs/architecture/03-web-ui-pages-and-buttons.md +++ b/docs/architecture/03-web-ui-pages-and-buttons.md @@ -49,7 +49,7 @@ History reconciliation is one synchronous two-pass transaction over existing key `chat_history.js` retains three ordinary rendered pages plus any temporarily protected reading/focus/selection pages; its cache tracks page descriptors, while existing row/card owners retain the content. Distant page bodies and their media/listener resources leave the rendered window through their existing owners; lightweight page descriptors keep exact return navigation available. An empty scan is not a page in the reading window: the three retained pages count only pages that contributed rows, return navigation uses those exact handles, and no surface offers a newer control. The existing per-room scroll stash carries those descriptors and a physical reading anchor for reopen, never a second persistent history copy. Recent live refresh remains separate from the frozen archive range; reaching the old physical beginning does not claim every page is currently loaded. -The Main chat receives ordinary main-thread dialogue, required Project question pointers, and the two host-stamped Project lifecycle rows — `project_started` and the terminal `project_completion_summary`, each with the shared "Open Project" action; all other Project traffic stays in the Project thread. Lifecycle rows are plain dashboard text: the producer strips markdown once before durable write and live send, history normalizes older rows on read (`chat.jsonl` is never rewritten), and the renderer escapes any system row without `markdown: true` (except `skill_review`'s dedicated renderer). Finality metadata is row-specific. Durable `task_summary` rows carry the SAME phase the browser paints: `project_dialogue.outcome_phase` mirrors `log_events.js` — the terminality gate plus the severity fold over normalized axes — and stamps it as `outcome_phase` beside the status word taken from `OUTCOME_PHASE_HEADLINE`, so there is no second status-word family. The pre-finalization authored row carries the phase with `outcome_final=false` but no host verdict clause; only terminal `task_summary` rows append the host verdict clause (`_completion_verdict`, the host twin of `taskReasonDetail`'s acceptance branch) when one exists. The Main `project_completion_summary` instead puts the shared headline and any verdict directly in its plain text; it does not carry `outcome_phase`. The start row carries neither outcome finality nor a verdict. A verdict is the execution reason on ordinary terminal summaries, or — when the host's acceptance decision is anything but accepted — the decision status with its stored rationale, markdown-stripped and flattened like every other lifecycle-row text; the latter leads the Main completion row so a host-authored row never presents an unaccepted claim as the whole story. History keeps older status words unrewritten, and the Telegram skill reads the stamped task-summary phase instead of deriving one of its own. Project panels accept only their own registered `chat_id`; the complete Project chat-id set is separate from the sidebar's bounded summary, and a `projects_changed` frame adds a new chat id synchronously before the asynchronous state refresh, so an early frame cannot be misclassified as Main. +The Main chat receives ordinary main-thread dialogue, required Project question pointers, and the two host-stamped Project lifecycle rows — `project_started` and the terminal `project_completion_summary`, each with the shared "Open Project" action; all other Project traffic stays in the Project thread. Lifecycle rows are plain dashboard text: the producer strips markdown once before durable write and live send, history normalizes older rows on read (`chat.jsonl` is never rewritten), and the renderer escapes any system row without `markdown: true` (except `skill_review`'s dedicated renderer). An admission-notice row (`task_not_started`, `task_start_unconfirmed`) is origin-addressed: it stays in the chat it was sent to on replay even though its never-started task is bound to a Project (`project_dialogue.room_membership` ignores the task binding for those two types). Finality metadata is row-specific. Durable `task_summary` rows carry the SAME phase the browser paints: `project_dialogue.outcome_phase` mirrors `log_events.js` — the terminality gate plus the severity fold over normalized axes — and stamps it as `outcome_phase` beside the status word taken from `OUTCOME_PHASE_HEADLINE`, so there is no second status-word family. The pre-finalization authored row carries the phase with `outcome_final=false` but no host verdict clause; only terminal `task_summary` rows append the host verdict clause (`_completion_verdict`, the host twin of `taskReasonDetail`'s acceptance branch) when one exists. The Main `project_completion_summary` instead puts the shared headline and any verdict directly in its plain text; it does not carry `outcome_phase`. The start row carries neither outcome finality nor a verdict. A verdict is the execution reason on ordinary terminal summaries, or — when the host's acceptance decision is anything but accepted — the decision status with its stored rationale, markdown-stripped and flattened like every other lifecycle-row text; the latter leads the Main completion row so a host-authored row never presents an unaccepted claim as the whole story. History keeps older status words unrewritten, and the Telegram skill reads the stamped task-summary phase instead of deriving one of its own. Project panels accept only their own registered `chat_id`; the complete Project chat-id set is separate from the sidebar's bounded summary, and a `projects_changed` frame adds a new chat id synchronously before the asynchronous state refresh, so an early frame cannot be misclassified as Main. The Main composer exposes one-shot Swarm planning, the owner Nano/Low/Max context choice, file attachment, and Send. Swarm places a structural `force_plan` fact on the next ordinary message and disarms after that send — never inferred from keywords. The host admits one managed root in the sending room without a model call, carrying the verbatim owner objective, origin, surface and attachments; the root may choose its Project through `ensure_project_scope` (§6), and a promote it makes before any plan wave carries its planning obligation to the new root (§6, owner 3=A). Nano/Low/Max uses the dedicated owner endpoint; a derived Low fit never masquerades as an owner-selected posture, and selecting Max may invoke the exact-route capability acknowledgement when provider metadata cannot prove the window. Project panels omit global Restart, Panic, evolution, consciousness, review, and budget controls — those belong to the one Ouroboros process, not a Project thread. @@ -57,7 +57,7 @@ Attachments stage via paperclip, paste, or drag-and-drop with no count or size r Simple text messages may appear immediately as pending local bubbles and reconcile against their echoed `client_message_id`. Delivered documents capture immutable task-owned bytes and carry a verified `file_ref` plus canonical download URL (with a compatible Files URL when available), rebuilding from history without persisting base64. Files above the 50 MiB inline boundary are file-backed; browser downloads use HEAD followed by a native download request and desktop saves stream through the existing launcher helper; photos and videos keep base64 only in the live frame — supported media is stored under the canonical task artifact root with a content-addressed URL so replay and rebuilds restore the same bubble. Structured `links` rows render at most twelve independently revalidated HTTP(S) buttons, identically live and replayed. The supervisor transport owns durable media copies so all producers converge on the canonical data root; failed persistence stays an honest caption row and cannot finalize a task card. -Ordinary Main and Project conversations use complete native execution even while other work runs. Each turn owns its agent and creates no supervisor queue record. The thread-safe process-local `supervisor.active_activity.DirectActivityRegistry` retains its private actor handle from preparation through result delivery; Stop, Hurry, quiz and steering resolve that exact actor, while `active_direct_turns` and typing frames expose only its presentation fields. The update admission lock covers check-and-registration only; writer drain waits the registered executions, including post-task work. Custody maintenance and restart census include those same live IDs, and Background Consciousness stays paused while any foreground execution remains. The producer carries `_is_direct_chat` into the durable result and terminal event so late/replayed ordinary Project completions stay in their room and do not emit Main task-completion summaries; the supervisor's rebuilt `task_done` (from the worker frame or the durable result), the authored `task_summary` row, the history terminal-truth annotation (`gateway/history.py`) and the activity census `kind` all carry that one fact, so a chat block reads the same host truth live, on reload and on reconnect, and a direct turn waiting for a model after its answer is never relabelled `managed_task`. The authored summary row also carries the turn's `typed_routing_action`, replayed as `addressing_only`. A turn's activity block is in the transcript iff its record holds content, open attention or launched work — an always-shown kind, a model wait, a pending or host-offered Stop, a child, a review group, a content row, or a terminal outcome other than Done (`web/modules/chat.js::blockVisible`, one predicate re-read at every mutation, no sticky flag); the completion note is not content, so a greeting keeps no block. Successful tool calls are visible compact rows, so a tool-using turn shows its block from its first call; a direct conversation turn (host fact `_is_direct_chat`) renders the same component as a compact activity block without a `Task` title, status chip or conversion, while a managed or Swarm root keeps the full task card. An addressing call (`promote_chat_to_task`, `route_to_project`, `steer_task`, `ensure_project_scope`) is stamped by the host on its live tool-call frames (`routing_action`) and counted in the task metrics (`routing_tool_calls`), both from the one routing-verb table in `tool_capabilities.py`; the typed annotation on the owner message is its receipt, so the call is a receipt row that never makes the block exist, and a turn that ran only such calls without error draws no block live or on reload — no client tool-name list exists. The existing metrics and authored task-summary rows carry exact per-tool invocation counts and error counts from the same execution trace; history passes those observations through, and replay shows one summary row from them (per-tool rows are live-only), so block presence is the same live, on reload and on reconnect (except a turn that moved itself into a Project, whose block and answer replay in the Project room). Original totals, cost, outcomes and authored answers remain intact; no extra request or store is introduced. Historical rows missing those facts remain legacy, without an inferred cost or recovered outcome. `GET /api/state` also exposes `active_chat_activities`: those rows united with ROOT managed queue tasks as `kind="managed_task"` with phase `queued`, `budget_paused` (PENDING fenced by a budget-pause fact; nothing dispatches it until an explicit resume, so it must not masquerade as queued), `working`, or `finalizing` (RUNNING with an open post-task checkpoint), so a chat instance created after the task started hydrates from the queue authority rather than transient typing frames. `active_chat_activities_complete` authorizes absence-based reconciliation only when every required live-identity source was read successfully; partial snapshots still carry positive activity. Optional question-detail failure does not invalidate the live-id set. That census is the only inserter into the client's live-activity set (finals and the census itself delete): a complete supervisor-ready snapshot deletes every id it does not list, whatever the entry's `kind`, and an incomplete one deletes nothing, so an unready supervisor loop or a failed live-identity source pauses deletion until the next complete snapshot. Hydration is a plain projection of that census through the page-wide snapshot sequencer — re-read every 3 seconds while the chat page is open and every 20 seconds otherwise — with no wall-clock request barrier and no per-entry generation; the barrier survives only for the card scan's `lastLiveObservedAt`. A typing frame is a submission receipt, not liveness: it retires the matching local `Sending...` by `client_message_id` and asks for a census read (the pending row retires once that read has answered: a listing census steps the header straight from `Sending...` to `Thinking...`, an omitting or failed read settles it at `Online` until the next census; a read that never settles leaves `Sending...` to the routing receipt, the next Main poll or reconnect replay), and never inserts, revives or extends a live-set entry. `kind` is presentation — it selects the label (`Thinking` for a direct turn, `Working`/`Queued`/`Paused` for a managed root) and exempts nothing from deletion; the server still sends child typing without a kind, which the Telegram transport's native typing consumer ignores. The client status reducer derives the chat header only from connection state, that census, the owner's own unconfirmed sends, and live managed task cards (a direct turn's block is not one: the header keeps the census verdict beside it, and any mounted unfinished block hosts its own running indicator in place of the typing bubble); a terminal failure stays a factual task result, never a reasonless header `Attention`, and a local `Sending...` submission is retired only by authoritative evidence (typing receipt, snapshot turn, durable routing receipt, replayed user row, turn conclusion, or offline-queue eviction) — never by the live user-row echo or a socket write. The census enumerates direct turns and managed ROOTS; descendants are carried by their own cards, so a running child that has typed but not yet emitted a progress row is absent from the header until that row mints its card. The same holds for any root the census does not enumerate: a Presence turn is never registered in the direct registry, so its header presence comes from its card alone. +Ordinary Main and Project conversations use complete native execution even while other work runs. Each turn owns its agent and creates no supervisor queue record. The thread-safe process-local `supervisor.active_activity.DirectActivityRegistry` retains its private actor handle from preparation through result delivery; Stop, Hurry, quiz and steering resolve that exact actor, while `active_direct_turns` and typing frames expose only its presentation fields. The update admission lock covers check-and-registration only; writer drain waits the registered executions, including post-task work. Custody maintenance and restart census include those same live IDs, and Background Consciousness stays paused while any foreground execution remains. The producer carries `_is_direct_chat` into the durable result and terminal event so late/replayed ordinary Project completions stay in their room and do not emit Main task-completion summaries; the supervisor's rebuilt `task_done` (from the worker frame or the durable result), the authored `task_summary` row, the history terminal-truth annotation (`gateway/history.py`) and the activity census `kind` all carry that one fact, so a chat block reads the same host truth live, on reload and on reconnect, and a direct turn waiting for a model after its answer is never relabelled `managed_task`. The authored summary row also carries the turn's `typed_routing_action`, replayed as `addressing_only`. A turn's activity block is in the transcript iff its record holds content, open attention or launched work — an always-shown kind, a model wait, a pending or host-offered Stop, a child, a review group, a content row, or a terminal outcome other than Done (`web/modules/chat.js::blockVisible`, one predicate re-read at every mutation, no sticky flag); the completion note is not content, so a greeting keeps no block. Successful tool calls are visible compact rows, so a tool-using turn shows its block from its first call; a direct conversation turn (host fact `_is_direct_chat`) renders the same component as a compact activity block without a `Task` title, status chip or conversion, while a managed or Swarm root keeps the full task card. An addressing call (`promote_chat_to_task`, `route_to_project`, `steer_task`, `ensure_project_scope`) is stamped by the host on its live tool-call frames (`routing_action`) and counted in the task metrics (`routing_tool_calls`), both from the one routing-verb table in `tool_capabilities.py`; the typed annotation on the owner message is its receipt, so the call is a receipt row that never makes the block exist, and a turn that ran only such calls without error draws no block live or on reload — no client tool-name list exists. A refused addressing call keeps the block by its error row, and its receipt line says the cause in the owner's words: the host's `cause` sentence rides the `message_annotation` frame and the replayed annotation alike, `routingAnnotationText` prints it verbatim, and «Choose a target» is shown only with real options; an act the host itself issued and then refused adds one plain `📋 System` row (`task_not_started` / `task_start_unconfirmed`) in the owner's chat, keyed to the never-started task, which replay neither mints nor finishes a card for. The existing metrics and authored task-summary rows carry exact per-tool invocation counts and error counts from the same execution trace; history passes those observations through, and replay shows one summary row from them (per-tool rows are live-only), so block presence is the same live, on reload and on reconnect (except a turn that moved itself into a Project, whose block and answer replay in the Project room). Original totals, cost, outcomes and authored answers remain intact; no extra request or store is introduced. Historical rows missing those facts remain legacy, without an inferred cost or recovered outcome. `GET /api/state` also exposes `active_chat_activities`: those rows united with ROOT managed queue tasks as `kind="managed_task"` with phase `queued`, `budget_paused` (PENDING fenced by a budget-pause fact; nothing dispatches it until an explicit resume, so it must not masquerade as queued), `working`, or `finalizing` (RUNNING with an open post-task checkpoint), so a chat instance created after the task started hydrates from the queue authority rather than transient typing frames. `active_chat_activities_complete` authorizes absence-based reconciliation only when every required live-identity source was read successfully; partial snapshots still carry positive activity. Optional question-detail failure does not invalidate the live-id set. That census is the only inserter into the client's live-activity set (finals and the census itself delete): a complete supervisor-ready snapshot deletes every id it does not list, whatever the entry's `kind`, and an incomplete one deletes nothing, so an unready supervisor loop or a failed live-identity source pauses deletion until the next complete snapshot. Hydration is a plain projection of that census through the page-wide snapshot sequencer — re-read every 3 seconds while the chat page is open and every 20 seconds otherwise — with no wall-clock request barrier and no per-entry generation; the barrier survives only for the card scan's `lastLiveObservedAt`. A typing frame is a submission receipt, not liveness: it retires the matching local `Sending...` by `client_message_id` and asks for a census read (the pending row retires once that read has answered: a listing census steps the header straight from `Sending...` to `Thinking...`, an omitting or failed read settles it at `Online` until the next census; a read that never settles leaves `Sending...` to the routing receipt, the next Main poll or reconnect replay), and never inserts, revives or extends a live-set entry. `kind` is presentation — it selects the label (`Thinking` for a direct turn, `Working`/`Queued`/`Paused` for a managed root) and exempts nothing from deletion; the server still sends child typing without a kind, which the Telegram transport's native typing consumer ignores. The client status reducer derives the chat header only from connection state, that census, the owner's own unconfirmed sends, and live managed task cards (a direct turn's block is not one: the header keeps the census verdict beside it, and any mounted unfinished block hosts its own running indicator in place of the typing bubble); a terminal failure stays a factual task result, never a reasonless header `Attention`, and a local `Sending...` submission is retired only by authoritative evidence (typing receipt, snapshot turn, durable routing receipt, replayed user row, turn conclusion, or offline-queue eviction) — never by the live user-row echo or a socket write. The census enumerates direct turns and managed ROOTS; descendants are carried by their own cards, so a running child that has typed but not yet emitted a progress row is absent from the header until that row mints its card. The same holds for any root the census does not enumerate: a Presence turn is never registered in the direct registry, so its header presence comes from its card alone. Owner-message continuity is journal-backed: each locally sent owner row is kept (bounded, with its routing annotation) until a fetched history response returns the same `client_message_id`, and a full rebuild re-renders unconfirmed journal rows after the server response, so a stale history snapshot cannot erase a message the owner just sent. Task finalization is honest: a root's early final answer carries `task_phase="finalizing"` (the same fact replay derives from the open post-task checkpoint), the card holds a sticky `Finalizing…` phase, and `task_cost_finalized` is bookkeeping that never resolves a card. If an observed managed root disappears from a queue-authoritative snapshot whose request began after the observation, the state-refresh fan-out removes activity/cancel authority and starts one single-flight durable task-detail read; request generations prevent an older response from undoing a newer projection. Two liveness invariants close the class of liveness the server cannot name — typing-fed header phantoms and cards shielded by a client-held live-set: the history window is lineage-closed — subagent lineage older than the window's progress recency floor is kept only while the child still runs or finalizes, or its parent is REPRESENTED by the response (a row that proves the task's card: telemetry, its own message or summary, a folded review group or plan reference it owns — a typed photo/document/quiz delivery does NOT count), or the parent is alive, and an anchored child represents its own children in turn, so a nested swarm is kept or dropped whole; a FINAL row failing that test is emitted without lineage fields (`ouroboros/gateway/history.py`, one anchor set computed in the quota pass), so replay cannot mint an unfinishable parent card while a visible or live parent keeps its completed children and their executor receipts — and liveness reconciles over the CARD SET, not only the activity census: a card is shielded by an id that census actually lists and by nothing else, an unvouched connected root card is finished ONLY by proven durable terminal detail, and a card without a current result stays unconfirmed until a complete fresh activity snapshot excludes it; it then uses the retained typed terminal fact or a neutral `Outcome unavailable` anchor, without task phase or controls; a root proven terminal (its `task_done` or its terminal detail) settles the descendant cards a lost child terminal left open from each child's OWN durable result — one single-flight read per child through the ordinary child-terminal path, never a cascade from the parent's outcome, never from the card-set scan. Until one of those reads settles it, a child whose terminal frames were all lost keeps a `Working` chip without live dots — a card-level residual the header no longer masks. The header badge has exactly one writer, the status reducer, with the panel-boot "Online" seed as the sole exception. diff --git a/docs/architecture/04-server-api-endpoints.md b/docs/architecture/04-server-api-endpoints.md index b41a43018..abca95661 100644 --- a/docs/architecture/04-server-api-endpoints.md +++ b/docs/architecture/04-server-api-endpoints.md @@ -154,7 +154,7 @@ A built-in `command` frame carries a slash command and enters the same bridge wi Built-in outbound envelopes: `chat`, `photo`, `video`, `document`, `typing`, `log`, `heartbeat`, `extension_lifecycle`, `message_annotation`, `projects_changed`, `task_named`, `update_status_ready`, and `update_progress_changed`. The latter only invalidates the process-local update observation; it is never boot-completion proof. Chat progress may carry task lineage, role, requested/effective model lane, delegated route, terminal execution evidence, review projection, cancellation eligibility, outcome axes, artifact references, and nullable cost/finality fields — additive presentation facts; consumers must not infer a missing execution receipt, cost, or task result from the absence of one optional field. -Thread routing is explicit. Project chat, typing, media, and log frames carry `chat_id`; a Project panel consumes its own thread, while Main admits required Project question pointers and the two host-stamped Project lifecycle rows (`project_started`, `project_completion_summary`). `projects_changed` carries a new chat id so every tab can extend its fan-out set before fetching the registry; when even that ordering loses the race, the server-stamped `project_thread` marker on the frame itself keeps Main from adopting it — set once at the message-bus broadcast choke from the registry (a membership lens, never a numeric range, so external transport ids such as Telegram stay unstamped) and enforced by Main's fan-out gate (`chat_activity.mainThreadAccepts`). Task-scoped LOG events acquire their final chat id at supervisor ingress: worker diagnostics carry only their own `task_id`, and `supervisor/log_addressing.py::address_task_event` stamps the audience from host-attested truth (the precedence chain lives in its docstring; an explicit event chat_id of 0 is the hidden partition, `HIDDEN_CHAT_ID`, never "missing"); direct turns carry their chat BY VALUE, stamped at the producer, because the registry entry dies with the turn while queued events drain later. Addressing is honest — an A2A row keeps its true audience, suppressed only at the broadcast choke (`push_log`) so machine traffic never reaches the browser; the same addressing runs in the server-process append sink and at every supervisor handler owning a suppressed type's explicit push, and a genuinely unaddressable event keeps the legacy chat-0 frame. `message_annotation` updates one canonical owner message without creating another bubble; `task_named` updates a card only where that task already exists. Media/document consumers validate MIME, base64, and download-route shapes before building browser URLs. +Thread routing is explicit. Project chat, typing, media, and log frames carry `chat_id`; a Project panel consumes its own thread, while Main admits required Project question pointers and the two host-stamped Project lifecycle rows (`project_started`, `project_completion_summary`). `projects_changed` carries a new chat id so every tab can extend its fan-out set before fetching the registry; when even that ordering loses the race, the server-stamped `project_thread` marker on the frame itself keeps Main from adopting it — set once at the message-bus broadcast choke from the registry (a membership lens, never a numeric range, so external transport ids such as Telegram stay unstamped) and enforced by Main's fan-out gate (`chat_activity.mainThreadAccepts`). Task-scoped LOG events acquire their final chat id at supervisor ingress: worker diagnostics carry only their own `task_id`, and `supervisor/log_addressing.py::address_task_event` stamps the audience from host-attested truth (the precedence chain lives in its docstring; an explicit event chat_id of 0 is the hidden partition, `HIDDEN_CHAT_ID`, never "missing"); direct turns carry their chat BY VALUE, stamped at the producer, because the registry entry dies with the turn while queued events drain later. Addressing is honest — an A2A row keeps its true audience, suppressed only at the broadcast choke (`push_log`) so machine traffic never reaches the browser; the same addressing runs in the server-process append sink and at every supervisor handler owning a suppressed type's explicit push, and a genuinely unaddressable event keeps the legacy chat-0 frame. `message_annotation` updates one canonical owner message without creating another bubble — a refused routing act's frame, its replayed annotation and the picker's 409 `dispatch_rejected` body all carry the host's `cause` sentence beside the machine `reason`; `task_named` updates a card only where that task already exists. Media/document consumers validate MIME, base64, and download-route shapes before building browser URLs. Extension WebSocket traffic is structurally namespaced by `extension_loader.extension_surface_name()` so an extension cannot shadow a built-in type. On each incoming extension frame the gateway resolves the owning skill and reconciles whether its extension is still desired, reviewed, granted, enabled, and live. A missing or failed handler returns a visible log frame. Out-of-process handlers execute in their extension child off the event loop; in-process handlers first record the required execution/cost disclosure. A non-`None` result returns as `.reply`; exceptions become typed error log frames rather than terminating the socket loop. diff --git a/docs/architecture/05-supervisor-loop.md b/docs/architecture/05-supervisor-loop.md index f9cbb1c1e..1276fc548 100644 --- a/docs/architecture/05-supervisor-loop.md +++ b/docs/architecture/05-supervisor-loop.md @@ -4,7 +4,7 @@ This chapter owns the single scheduler for pooled work: what a healthy tick does `server.py::_run_supervisor()` is the single scheduler for pooled tasks. A healthy tick publishes liveness, rotates the runtime logs, checks worker health, drains worker, direct-chat, and consciousness events, accepts owner bridge input, enforces deadlines and schedules, runs throttled reconciliation and evolution admission, assigns eligible work, and persists `state/queue_snapshot.json`. Bridge intake precedes timeout, maintenance, evolution, and assignment work so a slow control-plane step cannot make a new owner message invisible. Three consecutive loop failures clear supervisor readiness, stop its watchdog generation, and notify the owner instead of leaving a healthy-looking server that no longer assigns work; a failure raised while a shutdown or restart is already in progress (the lifespan teardown sets a process-local stop event first and joins the loop for a bounded window before workers, bridge and event bus go down) is not a crash — the loop exits quietly, without the counter, the error, or the alarm — and the crash backoff waits on that stop event so a shutdown is never held by it. -`PENDING` and `RUNNING`, guarded by `supervisor.queue._queue_lock`, are the live task-lifecycle authority. Admission reserves identity before project, workspace, attachment, or routing side effects can create a duplicate; refuses a disabled pool, duplicate task, project deletion, accepted or sealed root, or exhausted root budget; attaches the task contract; and preserves stable priority order. Assignment runs against the same locked state and skips reaping slots, budget-paused work, closed project roots, conflicting project writers, and tasks exceeding the root's subagent capacity or depth reservation; evolution tasks are dropped there when `evolution_block_reason()` is set (Light runtime mode, `supervisor/workers.py`). That is the last of three evolution-only runtime-mode fences: owner and post-task entry points refuse a campaign start, `enqueue_evolution_task_if_needed()` independently pauses and disables a carried campaign before queueing it, and assignment drops what still slipped through (`supervisor/evolution_lifecycle.py`). Generic `supervisor.queue.enqueue_task()` has no runtime-mode predicate at all. Configured worker count is therefore not available capacity: the truthful value is the currently assignable idle count after custody, reaping, and admission fences. +`PENDING` and `RUNNING`, guarded by `supervisor.queue._queue_lock`, are the live task-lifecycle authority. Admission reserves identity before project, workspace, attachment, or routing side effects can create a duplicate; refuses a disabled pool, duplicate task, project deletion, accepted or sealed root, exhausted root budget, or an unusable or unprovisionable workspace (typed, with the repair in `detail`); attaches the task contract; and preserves stable priority order. Assignment runs against the same locked state and skips reaping slots, budget-paused work, closed project roots, conflicting project writers, and tasks exceeding the root's subagent capacity or depth reservation; evolution tasks are dropped there when `evolution_block_reason()` is set (Light runtime mode, `supervisor/workers.py`). That is the last of three evolution-only runtime-mode fences: owner and post-task entry points refuse a campaign start, `enqueue_evolution_task_if_needed()` independently pauses and disables a carried campaign before queueing it, and assignment drops what still slipped through (`supervisor/evolution_lifecycle.py`). Generic `supervisor.queue.enqueue_task()` has no runtime-mode predicate at all. Configured worker count is therefore not available capacity: the truthful value is the currently assignable idle count after custody, reaping, and admission fences. A headless task is ADDRESSED when it is admitted, not when it is displayed (`log_addressing.ingress_chat_id`). A registered project's run has exactly ONE destination: an explicit `chat_id` may only agree with that thread, and any other value — the hidden partition included — is refused with a typed 400 rather than honoured or silently overridden, because a run addressed away from its room is the one shape that puts a card in Main whose project holds none of its work. Without a Project, ordinary API tasks default to `HIDDEN_CHAT_ID` (0). The confirmed browser Publish flow explicitly carries `source="web"` and `WEB_UI_CHAT_ID` to request Main; source is caller-declared addressing on the existing owner API, not a new authentication proof. Other non-Project conversation addresses remain refused. A run scoped to a REGISTERED, active project is admitted into that project's thread (dialogue, children, attachments and answer in the room the owner already has; Main still receives the one host-stamped completion row), and Main is told it finished only when its work is actually in that room — addressed there at admission or BOUND to the project; Registration alone does not qualify. Every other run stays in the hidden partition, silent in every chat, read back through the terminal, `--result-json-out`, the chat-blind Logs panel and `GET /api/tasks/`; a reserved but inactive project keeps its chat acceptable so the queue's lifecycle fence refuses with its own typed reason, and a project deleted mid-run keeps its reserved chat. The run is also NAMED at admission and chat promotion, without a new model call: a caller-supplied `title` (`ouroboros run --title` or the top-level contract field; `metadata.title` is refused with a 400 like `metadata.project_id`) is authorship and fills both `title` and `suggested_name`; otherwise the request's first line, stripped of markdown and capped at the project-name length, fills `suggested_name` ALONE, so a truncated prompt never outranks a real name coined later, and a `task_named` frame is broadcast on admission so the live card is never born showing its status phrase as a title — the client buffers a `task_named` that arrives before the card's record exists (`web/modules/chat.js`), so frame order does not matter. diff --git a/docs/architecture/06-agent-core.md b/docs/architecture/06-agent-core.md index 57a15fa36..6eb44711b 100644 --- a/docs/architecture/06-agent-core.md +++ b/docs/architecture/06-agent-core.md @@ -86,7 +86,7 @@ Disclosed cancel-lifecycle residuals (deliberate): a cascade over a tree with no ### Tool capability and execution -`tool_capabilities.py` is the SSOT for core, meta, parallel-safe, stateful-browser, untruncated, capped-result, and reviewed-mutative tool classes. `tool_policy.py` chooses the initial capability set; `ToolRegistry` remains the execution authority; `loop_tool_execution.py` owns timeouts, concurrency, live evidence, result handling, and mutative ceilings. Ordinary top-level presets share one built-in name surface: project focus changes the default target, while root policy, runtime mode, task-contract disables, credentials, resources, selected-resource bindings, and delegated-child profiles narrow independently — a tool being registered or discoverable is therefore not the same as being callable for a particular target. Lazy capability discovery returns an explicit capability omission or `CAPABILITY_UNAVAILABLE` fact, never a silent disappearance. `enable_tools`/discovery answer a REGISTERED tool filtered by real policy with a typed "hidden by policy: " (`ToolRegistry.policy_hidden_reason`), never the same "Not found" as a nonexistent name; the contract-disabled check precedes registration so a disabled extension/MCP name also reports its reason. Child allowlists remain deliberate narrower principals. Review output and cognitive artifacts remain outside ordinary result truncation. Outcome classification keeps policy refusal separate from execution failure: `user_files_path_blocked`, `cwd_blocked`, and `artifact_output_undeclared` are typed non-failure surfaces, while a declared output that cannot be registered remains the genuine `artifact_output_error` — an expected authority boundary never falsely becomes the task's headline failure. The same separation runs through the identifier register (`tool_result._EXACT_IDENTIFIER_CODES`): a typed refusal is recorded as a refusal, never as `ok`. The register keys on the CODE, so the plain-string refusals (a disposition the ledger refused to record, a steer the host rejected or could not confirm) and the stored traces of past ones are all reached by one table rather than a producer cutover. The registered refusal identifiers cover the whole routing family — `STEER_REJECTED`/`STEER_UNCONFIRMED` beside `PROMOTE_REJECTED`, `PROMOTE_UNCONFIRMED`, `ROUTE_REJECTED`, `ROUTE_UNCONFIRMED`, `ROUTING_UNCONFIRMED` and `NEEDS_MANUAL_TARGET`: a promote or route that scheduled nothing was read as a SUCCESSFUL call while any of them was missing from the register. A refused steer, promote or route is visible to the agent (`is_error`, policy-denial telemetry) and does not degrade the execution axis: the host refused, the agent did not fail. Discovery then has three outcomes rather than two: a confinement refusal is an error, a listing that failed is a failure, and a path holding nothing is a MISS (`LIST_FILES_NOT_FOUND`, a warning beside `DATA_NOT_YET_CREATED`) that leaves the rest of the outcome uncoloured, because the recovery scan credits only a later success on the same target signature and a differently spelled path can never match it. The external-executor family reaches the same register from the producer side: `delegate_start`, `delegate_wait`, `delegate_cancel` and `delegate_answer` speak a native `ToolResult` among themselves, and `delegate_shared._fail` writes the envelope (`ok: false` plus `host_code`) BESIDE the domain payload, so an unreachable daemon, a run this task does not own, a containment fault, a refused cancel and a spent subscription window are RECORDED as refusals instead of counted as successful calls. The domain `reason` keeps its own vocabulary and is never renamed into the host code; substrate refusals map to `TOOL_REPORTED_FAILURE` (recorded, never degrading) and malformed calls to `TOOL_ARG_ERROR` (degrades, feeds reflection), never to a timeout or a generic tool error. A terminal run whose state is `failed` or `cancelled` stays a SUCCESSFUL observation of an unsuccessful leaf. +`tool_capabilities.py` is the SSOT for core, meta, parallel-safe, stateful-browser, untruncated, capped-result, and reviewed-mutative tool classes. `tool_policy.py` chooses the initial capability set; `ToolRegistry` remains the execution authority; `loop_tool_execution.py` owns timeouts, concurrency, live evidence, result handling, and mutative ceilings. Ordinary top-level presets share one built-in name surface: project focus changes the default target, while root policy, runtime mode, task-contract disables, credentials, resources, selected-resource bindings, and delegated-child profiles narrow independently — a tool being registered or discoverable is therefore not the same as being callable for a particular target. Lazy capability discovery returns an explicit capability omission or `CAPABILITY_UNAVAILABLE` fact, never a silent disappearance. `enable_tools`/discovery answer a REGISTERED tool filtered by real policy with a typed "hidden by policy: " (`ToolRegistry.policy_hidden_reason`), never the same "Not found" as a nonexistent name; the contract-disabled check precedes registration so a disabled extension/MCP name also reports its reason. Child allowlists remain deliberate narrower principals. Review output and cognitive artifacts remain outside ordinary result truncation. Outcome classification keeps policy refusal separate from execution failure: `user_files_path_blocked`, `cwd_blocked`, and `artifact_output_undeclared` are typed non-failure surfaces, while a declared output that cannot be registered remains the genuine `artifact_output_error` — an expected authority boundary never falsely becomes the task's headline failure. The same separation runs through the identifier register (`tool_result._EXACT_IDENTIFIER_CODES`): a typed refusal is recorded as a refusal, never as `ok`. The register keys on the CODE, so the plain-string refusals (a disposition the ledger refused to record, a steer the host rejected or could not confirm) and the stored traces of past ones are all reached by one table rather than a producer cutover. The registered refusal identifiers cover the whole routing family — `STEER_REJECTED`/`STEER_UNCONFIRMED` beside `PROMOTE_REJECTED`, `PROMOTE_UNCONFIRMED`, `ROUTE_REJECTED`, `ROUTE_UNCONFIRMED`, `ROUTING_UNCONFIRMED` and `NEEDS_MANUAL_TARGET`: a promote or route that scheduled nothing was read as a SUCCESSFUL call while any of them was missing from the register. Each refusal identifier carries the host's cause and repair in its `(reason: detail)` clause. A refused steer, promote or route is visible to the agent (`is_error`, policy-denial telemetry) and does not degrade the execution axis: the host refused, the agent did not fail. Discovery then has three outcomes rather than two: a confinement refusal is an error, a listing that failed is a failure, and a path holding nothing is a MISS (`LIST_FILES_NOT_FOUND`, a warning beside `DATA_NOT_YET_CREATED`) that leaves the rest of the outcome uncoloured, because the recovery scan credits only a later success on the same target signature and a differently spelled path can never match it. The external-executor family reaches the same register from the producer side: `delegate_start`, `delegate_wait`, `delegate_cancel` and `delegate_answer` speak a native `ToolResult` among themselves, and `delegate_shared._fail` writes the envelope (`ok: false` plus `host_code`) BESIDE the domain payload, so an unreachable daemon, a run this task does not own, a containment fault, a refused cancel and a spent subscription window are RECORDED as refusals instead of counted as successful calls. The domain `reason` keeps its own vocabulary and is never renamed into the host code; substrate refusals map to `TOOL_REPORTED_FAILURE` (recorded, never degrading) and malformed calls to `TOOL_ARG_ERROR` (degrades, feeds reflection), never to a timeout or a generic tool error. A terminal run whose state is `failed` or `cancelled` stays a SUCCESSFUL observation of an unsuccessful leaf. Tool API v2 exposes neutral canonical names directly (`read_file`, `list_files`, `search_code`, `write_file`, `edit_text`, `edit_batch`, `apply_patch`, `run_command`, `run_script`, `verify_and_record`, service tools, `commit_reviewed`, `vcs_*`, `schedule_subagent`, `schedule_followup`, `wait_task`, `wait_tasks`); legacy public names are neither exposed nor translated. The file tools share a path-based public ABI, and because payload-borne paths (`edit_batch` entries, `apply_patch` targets) miss the dispatch seam that rewrites a `path` ARG, both ends canonicalize explicitly through `tool_access.canonical_repo_relative_path` — one normalization contract keeps a guard from judging `repo/BIBLE.md` while the write lands on `BIBLE.md`; `_ROOT_ARG_REPO_WRITE_TOOLS` is the single set every repo-write fence keys on. @@ -733,13 +733,13 @@ The projects registry owns immutable project identity, canonical chat id, option Project `journal.jsonl` records curated milestones and `workpad.md` retains active working context; focused context includes the workpad in full and recent journal rows with a visible pointer to older entries. On root completion, only high-signal blockers, questions, and interface contracts are mirrored once from the ephemeral task-tree ledger into the durable journal, and a finished root whose effective working tree is not the registered `working_dir` writes one typed "work lives at @ " journal row from facts the task record already holds. A project digest gives consciousness a concise completion signal without pretending to be the raw project memory. -The four routing verbs — `promote_chat_to_task`, `route_to_project`, `steer_task` and `ensure_project_scope` — ride one receipt rail and become successful only after their token-matched supervisor facts are durable in the existing task result, queue snapshot, annotation, or mailbox authority; with several possible tasks the LLM chooses, code auto-delivers only the unambiguous one-target case, and an unconfirmed or stale receipt fails visibly rather than launching a second root. WHO is speaking is ONE host-minted fact on the event (`control_routing._routing_issuer`, by value; the model has no argument): an OWNER TURN — a direct chat turn, whose ingress stamped `client_message_id`, or a task relaying the owner message it just drained — or a TASK speaking for itself (a pooled, Swarm, project or headless root). An owner turn's steer travels as owner text (`[Message from my human]`, the owner corpus, the generation bump that supersedes a reviewed answer, the room veto computed from the registry lane of the issuing chat — a Project room reaches its own roots, Main reaches every host-listed root — and the owner acknowledgement). A task's own words NEVER travel as owner text: they are written through the one task-message writer `forward_to_worker` also uses, as `independent_task` provenance, to any host-listed active independent root (owner 6C; hidden-partition roots included, no room veto), render as `[Message from independent task ]`, enter no owner corpus (so `owner_source_sha256`, the post-drain growth check and the acceptance premises stay the owner's — owner 4=A), notify no chat, carry no attachments, and are confirmed to the issuer as WRITTEN, or refused with the host's reason, under the act's own synthetic receipt id; the true author and target ride the receipt and one `task_message_routed` Logs row, and the receiver's `task_message_injected` row names the sender. Independent roots learn the roster from a `[INDEPENDENT_ROOTS]` TAIL note (`peer_roster.py`) appended only when it changed and never merged into a row already sent. Routing receipts are retained per `(owner message, routing token)`: an earlier act's receipt under a message that later carried another act stays readable by its token (`chat_annotation_receipt`), while the message's latest row remains the UI projection and the picker's liveness test. A promote by a root whose Swarm planning obligation is still UNMET (`force_plan` and no plan wave recorded — `owner_hurry.unmet_force_plan_obligation`) carries the obligation onto the new root and releases the promoter in the admission transaction (`supervisor/plan_obligation.py`; owner 3=A): the promote result names the move and says that work the promoter keeps doing itself is unplanned; a met obligation moves nothing, and `ensure_project_scope` keeps it. A promote that lands in a project other than the one the request's work already has says so (owner 5A: promote stays a free choice, disclosed). A KNOWN rejection is returned as `rejected` with its reason, never as a timeout: the wait's accept-set holds every settled outcome the producers write, so `UNCONFIRMED` keeps its one meaning — no matching receipt exists — and a task-authored act that belongs to no owner message is keyed by its synthetic `agent-steer:` id rather than left with no receipt to poll at all. An admitted promote reports the destination the admission receipt returns, not the `project_id`/`project_name` the caller asked for: an implicit scope is re-resolved under the origin claim lock, so the requested room and the effective one can differ. A routing/promote decision turn receives host-built ground truth (each project's registry `working_dir` plus bounded typed projections of recent task results — never raw result text), read through the registry row's durable `last_task_result_id` pointer with a bounded scan fallback; only the absent-pointer case writes the pointer back, because a non-empty pointer that failed to resolve is typically a split-drive result whose canonical copy-back has not landed. Those recent results are the owner's ROOT results only: a swarm child is not an addressable predecessor (it is reachable through its root) and is counted as its own omission rather than evicting a root from the window, while a project room's direct-chat root stamps the same pointer, so the room's own work is offerable without acquiring letters home. The turn is also shown any routing receipt that already exists for the SAME owner message, a fact on the contract it already receives and never a ban: whether one message becomes a second root stays the model's judgment. `steer_task` transports the owner's exact ingress bytes on the turn's FIRST routing act, while it is still acting on the message that started it. Two typed facts end that window: the latest owner message the turn actually DRAINED from its mailbox (`ToolContext.last_owner_delivery`, stamped at the loop's mailbox drain — a message merely written to the mailbox has not reached the turn and ends nothing), or a landed `promote_chat_to_task`/`route_to_project`/`steer_task` receipt already on the origin message (a refused or unconfirmed act carried nothing, so the message stays unrouted and the next act still relays it). Past either one the turn is RELAYING its own words, so it delivers the message the model was given, and its receipt is written against the owner message actually relayed — the drained delivery's own client id when it carries one, else the synthetic `agent-steer:` id — never again on an origin message this turn has already routed. An agent-authored steer that belongs to no owner message earns its receipt under a synthetic `agent-steer:` id: delivery and typed refusal stay confirmable through the same `routing_wait` poll, while no chat row carries that id, so no owner message is labelled by the agent's act and, for a steer that belongs to no owner message, the owner message's own routing receipt (what a later decision turn reads) is left standing; a steer that relays a drained owner message keys its receipt on that message, replacing whatever receipt it carried. Each steer's mailbox entry is keyed by its own routing token beside the owner message id and target, so several instructions relayed under one origin are several deliveries while a retried emit of the same steer stays one. Canonical owner routing and project UI are projections over those same authorities: a routing receipt proves admission, not completion; an unread indicator proves a visible revision, not memory isolation. +The four routing verbs — `promote_chat_to_task`, `route_to_project`, `steer_task` and `ensure_project_scope` — ride one receipt rail and become successful only after their token-matched supervisor facts are durable in the existing task result, queue snapshot, annotation, or mailbox authority; with several possible tasks the LLM chooses, code auto-delivers only the unambiguous one-target case, and an unconfirmed or stale receipt fails visibly rather than launching a second root. WHO is speaking is ONE host-minted fact on the event (`control_routing._routing_issuer`, by value; the model has no argument): an OWNER TURN — a direct chat turn, whose ingress stamped `client_message_id`, or a task relaying the owner message it just drained — or a TASK speaking for itself (a pooled, Swarm, project or headless root). An owner turn's steer travels as owner text (`[Message from my human]`, the owner corpus, the generation bump that supersedes a reviewed answer, the room veto computed from the registry lane of the issuing chat — a Project room reaches its own roots, Main reaches every host-listed root — and the owner acknowledgement). A task's own words NEVER travel as owner text: they are written through the one task-message writer `forward_to_worker` also uses, as `independent_task` provenance, to any host-listed active independent root (owner 6C; hidden-partition roots included, no room veto), render as `[Message from independent task ]`, enter no owner corpus (so `owner_source_sha256`, the post-drain growth check and the acceptance premises stay the owner's — owner 4=A), notify no chat, carry no attachments, and are confirmed to the issuer as WRITTEN, or refused with the host's reason, under the act's own synthetic receipt id; the true author and target ride the receipt and one `task_message_routed` Logs row, and the receiver's `task_message_injected` row names the sender. Independent roots learn the roster from a `[INDEPENDENT_ROOTS]` TAIL note (`peer_roster.py`) appended only when it changed and never merged into a row already sent. Routing receipts are retained per `(owner message, routing token)`: an earlier act's receipt under a message that later carried another act stays readable by its token (`chat_annotation_receipt`), while the message's latest row remains the UI projection and the picker's liveness test. A promote by a root whose Swarm planning obligation is still UNMET (`force_plan` and no plan wave recorded — `owner_hurry.unmet_force_plan_obligation`) carries the obligation onto the new root and releases the promoter in the admission transaction (`supervisor/plan_obligation.py`; owner 3=A): the promote result names the move and says that work the promoter keeps doing itself is unplanned; a met obligation moves nothing, and `ensure_project_scope` keeps it. A promote that lands in a project other than the one the request's work already has says so (owner 5A: promote stays a free choice, disclosed). A KNOWN rejection is returned as `rejected` with its reason, never as a timeout: the wait's accept-set holds every settled outcome the producers write, so `UNCONFIRMED` keeps its one meaning — no matching receipt exists — and a task-authored act that belongs to no owner message is keyed by its synthetic `agent-steer:` id rather than left with no receipt to poll at all. Every promote/route refusal carries `detail` — the cause with its repair; `workspace_admission.workspace_repair_hint` composes the workspace family's from the typed source of the refused folder (a Presence profile folder is fixed in the profile, a path the request named is re-promoted against the project folder or with `workspace='none'`, a project folder in Projects) — and `_persist_promote_rejection` is its single durable writer. The exact Ouroboros repository root named as `workspace_root` is the documented default spelled out, so `_promote_chat_to_task` maps it to the `workspace="none"` sentinel and says so in the result; a subfolder of the repository, the data drive and every `/api/tasks`, Presence and `source` caller are still refused, typed. Who is told of a refusal follows who can narrate it: a tool-issued act gets its receipt (with the host's `cause` sentence), its error row and `detail`, and the model narrates; an act the host issued — `host_initiated`, stamped by the skill-card, Swarm and picker producers — gets ONE typed System row (`task_not_started`, or `task_start_unconfirmed` for an unconfirmed admission) in the chat the owner wrote in, from the one publication boundary `_handle_promote_chat_to_task` wraps around every outcome (a replayed receipt sends nothing), never a bubble in Ouroboros's voice; the Project start row is announced only after the task is really enqueued. An admitted promote reports the destination the admission receipt returns, not the `project_id`/`project_name` the caller asked for: an implicit scope is re-resolved under the origin claim lock, so the requested room and the effective one can differ. A routing/promote decision turn receives host-built ground truth (each project's registry `working_dir` plus bounded typed projections of recent task results — never raw result text), read through the registry row's durable `last_task_result_id` pointer with a bounded scan fallback; only the absent-pointer case writes the pointer back, because a non-empty pointer that failed to resolve is typically a split-drive result whose canonical copy-back has not landed. Those recent results are the owner's ROOT results only: a swarm child is not an addressable predecessor (it is reachable through its root) and is counted as its own omission rather than evicting a root from the window, while a project room's direct-chat root stamps the same pointer, so the room's own work is offerable without acquiring letters home. The turn is also shown any routing receipt that already exists for the SAME owner message, a fact on the contract it already receives and never a ban: whether one message becomes a second root stays the model's judgment. `steer_task` transports the owner's exact ingress bytes on the turn's FIRST routing act, while it is still acting on the message that started it. Two typed facts end that window: the latest owner message the turn actually DRAINED from its mailbox (`ToolContext.last_owner_delivery`, stamped at the loop's mailbox drain — a message merely written to the mailbox has not reached the turn and ends nothing), or a landed `promote_chat_to_task`/`route_to_project`/`steer_task` receipt already on the origin message (a refused or unconfirmed act carried nothing, so the message stays unrouted and the next act still relays it). Past either one the turn is RELAYING its own words, so it delivers the message the model was given, and its receipt is written against the owner message actually relayed — the drained delivery's own client id when it carries one, else the synthetic `agent-steer:` id — never again on an origin message this turn has already routed. An agent-authored steer that belongs to no owner message earns its receipt under a synthetic `agent-steer:` id: delivery and typed refusal stay confirmable through the same `routing_wait` poll, while no chat row carries that id, so no owner message is labelled by the agent's act and, for a steer that belongs to no owner message, the owner message's own routing receipt (what a later decision turn reads) is left standing; a steer that relays a drained owner message keys its receipt on that message, replacing whatever receipt it carried. Each steer's mailbox entry is keyed by its own routing token beside the owner message id and target, so several instructions relayed under one origin are several deliveries while a retried emit of the same steer stays one. Canonical owner routing and project UI are projections over those same authorities: a routing receipt proves admission, not completion; an unread indicator proves a visible revision, not memory isolation. ### Skills and extensions Skill capability grows through distinct gates: discovery and manifest parsing (`skill_loader.py`), content-hash-bound review (`skill_review.py` / `skill_review_runner.py`), owner grants, dependency reconciliation (`marketplace/isolated_deps.py`), enablement, readiness (`skill_readiness.py`), and execution (`tools/skill_exec.py`). Discovery establishes identity, source, provenance, conflicts, and hash; it does not confer trust. Review status, grants, enablement, and dependency health remain independent durable facts under `data/state/skills//`. The model-facing catalogue (`list_skills` and the per-turn Installed Skills section) projects the same live extension facts as `/api/extensions`; each enablement change best-effort appends a typed `skill_enabled_changed` disclosure row to `logs/events.jsonl` naming the actor, empty for writers not yet labelled, and an append failure is logged and never blocks the enablement change. Execution uses `skill_readiness_for_execution()` and its mode-aware review projection, not a raw verdict string. Cyber review failures remain visible advice; visibility, successful execution and review PASS are still different facts. -A skill may declare a reviewed `presence:` behavior profile: instructions, knowledge topics, runtime defaults, and portable capability requests that installation-local selections resolve to exact targets. Admission requires the bound behavior skill installed, enabled, freshly reviewed, and complete for every required request, then compiles one immutable positive capability ceiling copied through `task_contract` — so mutable skill/Settings state cannot broaden a live turn. An optional owner-local `workspace_root`, selected through `configure_presence` or the existing Skills settings form, is validated by `workspace_admission.validate_workspace_root` and frozen into the task's workspace contract. It changes the file/process target while retaining canonical shared memory; it neither derives Project scope nor creates a forked data drive. Runtime/capability edits preserve this selection, and promotion/follow-up tasks inherit the admitted folder rather than rereading mutable settings. Bundled native skills and editable marketplace/user payloads occupy separate payload-plane buckets with owner and review state outside the payload; declared conflicts are symmetric and never delete a payload. `extension_loader.py` and the isolated-dependency layer load only a ready, hash-matching extension; a review PASS alone does not prove its dependencies installed or its widget loadable. +A skill may declare a reviewed `presence:` behavior profile: instructions, knowledge topics, runtime defaults, and portable capability requests that installation-local selections resolve to exact targets. Admission requires the bound behavior skill installed, enabled, freshly reviewed, and complete for every required request, then compiles one immutable positive capability ceiling copied through `task_contract` — so mutable skill/Settings state cannot broaden a live turn. An optional owner-local `workspace_root`, selected through `configure_presence` or the existing Skills settings form, is validated by `workspace_admission.validate_workspace_root` and frozen into the task's workspace contract. It changes the file/process target while retaining canonical shared memory; it neither derives Project scope nor creates a forked data drive. Runtime/capability edits preserve this selection, and promotion/follow-up tasks inherit the admitted folder rather than rereading mutable settings. An admitted folder that has become unusable refuses the promotion typed (`workspace_unusable`; the repair in `detail` names the profile) and writes no chat row — the public conversation's synthetic chat id is never spoken to. Bundled native skills and editable marketplace/user payloads occupy separate payload-plane buckets with owner and review state outside the payload; declared conflicts are symmetric and never delete a payload. `extension_loader.py` and the isolated-dependency layer load only a ready, hash-matching extension; a review PASS alone does not prove its dependencies installed or its widget loadable. `skill_lifecycle_queue.py` serializes install, update, review, grant, enable, and removal work and exposes queued/running/succeeded/failed plus stale metadata; stale is recovery evidence, not a fake unlock of a still-running thread. Scheduled work is reconciled by `resync_skill_schedules()` and can run only after `skill_readiness_for_execution()`. Schedule evaluation is a DST-aware system using the shared cron/timezone contract. Evolution remains hard-blocked in `light` runtime mode. These separations let skills expand capability without turning discovery, a UI toggle, or old review state into execution authority. diff --git a/docs/architecture/12-host-service-companions-and-chat-ids.md b/docs/architecture/12-host-service-companions-and-chat-ids.md index d63baa1ab..b039942c1 100644 --- a/docs/architecture/12-host-service-companions-and-chat-ids.md +++ b/docs/architecture/12-host-service-companions-and-chat-ids.md @@ -36,7 +36,7 @@ Operation correlation (#667): a named injected message has `operation_ref= None` or the producer's success. One host warning reports the aggregate; diagnostics contain no message bodies, URLs or credentials. A child that dies before its final envelope may lose this aggregate; the existing measured death/timeout facts remain authoritative. The separate streaming route implementation carries the same aggregate in completion X after body/background work. -Presence flow: `POST /presence/turn` requires the content-hash-bound `presence` permission, one `binding_id`, one exact transport event, and optionally staged files confined to the skill's state root; the binding resolves from `state/presence_bindings.json` with exact provider/account/conversation/thread origin verification; cross-process locks enforce the installation-wide cap and serialize one `conversation_key`; a stable event-derived task id makes transport retries idempotent; input and output join ordinary dialogue history with full transport/actor provenance. The one typed outcome is message/silent/tool_delivered/deferred — `deferred` only with a correlated `work_ref`, because an unanchored "deferred" would be an unanchored promise — and `GET /presence/work/{work_ref}` polls the bound late result without exposing the general task API. Promotion of Presence work into a managed task clears requested Project/workspace/source widening: a public conversation may promote long work but cannot choose new authority, and the cost ceiling plus return destination follow the promoted root by value. `presence_cancel_work` acts only on a `work_ref` whose stored binding and conversation match the current turn; owner chat and Background Consciousness may `initiate_presence` on an existing enabled binding. Every admitted turn's ceiling also carries the cognitive-memory baseline (knowledge read/write/list, scratchpad, identity, chat history): one mind keeps one memory in every channel, so an external message can prompt an in-turn memory or identity revision on the profile's model slot rather than only a post-task knowledge write; the correspondent gains no tool, and public text still carries no owner-command authority; on the read side, an unselected baseline `chat_history` grant is unbound, so — like `knowledge_read` over the notes kept about people — a reply can surface the owner's own words to the correspondent through the model's judgment. Ceilings frozen before the baseline existed verify their own digest and carry no baseline until their profile is recompiled. +Presence flow: `POST /presence/turn` requires the content-hash-bound `presence` permission, one `binding_id`, one exact transport event, and optionally staged files confined to the skill's state root; the binding resolves from `state/presence_bindings.json` with exact provider/account/conversation/thread origin verification; cross-process locks enforce the installation-wide cap and serialize one `conversation_key`; a stable event-derived task id makes transport retries idempotent; input and output join ordinary dialogue history with full transport/actor provenance. The one typed outcome is message/silent/tool_delivered/deferred — `deferred` only with a correlated `work_ref`, because an unanchored "deferred" would be an unanchored promise — and `GET /presence/work/{work_ref}` polls the bound late result without exposing the general task API. Promotion of Presence work into a managed task clears requested Project/workspace/source widening: a public conversation may promote long work but cannot choose new authority, and the cost ceiling plus return destination follow the promoted root by value. An unusable profile folder refuses that promotion typed (`workspace_unusable`, repair in `detail`) and sends no chat text. `presence_cancel_work` acts only on a `work_ref` whose stored binding and conversation match the current turn; owner chat and Background Consciousness may `initiate_presence` on an existing enabled binding. Every admitted turn's ceiling also carries the cognitive-memory baseline (knowledge read/write/list, scratchpad, identity, chat history): one mind keeps one memory in every channel, so an external message can prompt an in-turn memory or identity revision on the profile's model slot rather than only a post-task knowledge write; the correspondent gains no tool, and public text still carries no owner-command authority; on the read side, an unselected baseline `chat_history` grant is unbound, so — like `knowledge_read` over the notes kept about people — a reply can surface the owner's own words to the correspondent through the model's judgment. Ceilings frozen before the baseline existed verify their own digest and carry no baseline until their profile is recompiled. Companion processes are host-supervised: reviewed manifest-declared descriptors enter durable custody, reconcile after lifecycle changes and restart, and stop on disable/unload/panic; `state/extension_generation.json` carries the opposite direction — the server's published live set, which a running task worker adopts at a task's start or at a dispatch miss, so an enable after boot is not invisible until the pool respawns. Worker-side changes write durable reconcile requests (`state/extension_reconcile/`) rather than spawning server-owned children; every reconcile state names the process that answered and whether that marker request was written, and the tool and review receipts pass both facts through. Health observations are process-qualified: aggregate and Skills UI health use the server observation as authority and expose the worker observation with its handoff outcome only as a qualifier, so failed handoff success cannot advance `last_known_good`; restart-budget exhaustion persists a terminal reason in that health state, cleared only by a later successful start. A companion's cwd is the reviewed payload directory, so a payload edit stales review before reload instead of silently mutating a live process. The live projection is `state/extension_companions.json`. diff --git a/docs/development/02-naming-and-boundaries.md b/docs/development/02-naming-and-boundaries.md index 8bdbcbf26..ca7b473dc 100644 --- a/docs/development/02-naming-and-boundaries.md +++ b/docs/development/02-naming-and-boundaries.md @@ -155,10 +155,13 @@ task-specific auto-retry, fallback, cleanup, resume, or terminal-flow state machines. Explicitly naming a documented default is never a different request. An argument -whose value is what omitting it already means (`directory_strategy="direct"` with -no `scope_paths`) takes the omitted path on a shape that cannot serve the argument -at all; only values that genuinely ask for something are refused there, typed, at -the earliest layer holding the authority to judge them, with the repair named. +whose value is what omitting it already means — `directory_strategy="direct"` with +no `scope_paths` on a shape that cannot serve the argument at all, or a value +whose meaning equals the omitted path's documented meaning (`workspace_root` +naming the Ouroboros repository itself, where a workspace-less task already +works) — takes the omitted path, disclosed in the result; only values that +genuinely ask for something are refused there, typed, at the earliest layer +holding the authority to judge them, with the repair named. A producer that already knows its call failed publishes that fact typed: a `ToolResult` through `tool_result._publish_tool_result`, or a first-line @@ -354,7 +357,11 @@ from the one table it owns (`ouroboros/tool_capabilities.py::ROUTING_VERBS`: `routing_action` on the live tool-call frames, `routing_tool_calls` in the task metrics, `typed_routing_action` on the terminal event), never a client-side exception list. The same shape hides in "hide unless kind ∈ {…}" and "count -unless name ∈ {…}": when the list is the rule, the rule is missing. +unless name ∈ {…}": when the list is the rule, the rule is missing. The owner +sentence on a routing receipt follows the same rule: the host composes it from +the typed reason (`ouroboros/project_dialogue.py::routing_refusal_cause`) and +ships it as `cause`; the browser prints it verbatim and keeps no reason→sentence +map of its own. ### Task-authored messages are never owner text diff --git a/docs/v7next/DATA_LAYOUT_INVENTORY.md b/docs/v7next/DATA_LAYOUT_INVENTORY.md index be3c5c0ae..c19635da1 100644 --- a/docs/v7next/DATA_LAYOUT_INVENTORY.md +++ b/docs/v7next/DATA_LAYOUT_INVENTORY.md @@ -2,7 +2,7 @@ Machine extraction of the `docs/ARCHITECTURE.md` "Data layout (`~/Ouroboros/`)" tree — the durable-file orientation carrier (this tree's counterpart of the reference PERSISTENCE_OWNERS derivation checklist) — regenerated by `python scripts/regenerate_inventories.py`. Do not edit. Every entry is probed against reality: repo entries must exist as tracked paths; data-plane entries must appear as a literal in the runtime sources that construct them. A durable file renamed or removed in code while its tree row survives = red (`tests/test_generated_inventories.py`). -Source: `docs/architecture/01-high-level-architecture.md`, physical LF lines 564-654; UTF-8 SHA-256 `ef1a039cc0bee57ea2a096f49dc64ce970b09aa98b1877473e51249abc36dd77`. +Source: `docs/architecture/01-high-level-architecture.md`, physical LF lines 564-654; UTF-8 SHA-256 `9477aa40239b63ec896cc68ee6da9859027603e65c9328aea7642abffd1596bf`. - entries: **79** (code-ref: 72, repo-dir: 6, repo-path: 1) diff --git a/docs/v7next/FACADE_INVENTORY.md b/docs/v7next/FACADE_INVENTORY.md index a4276b465..dfe7e5429 100644 --- a/docs/v7next/FACADE_INVENTORY.md +++ b/docs/v7next/FACADE_INVENTORY.md @@ -56,11 +56,11 @@ AST-derived inventory of compatibility facades, regenerated by `python scripts/r | `ouroboros/tools/subagent_integration.py` | D07 | 13 | `ouroboros/headless.py` (2 ✗D17)
`ouroboros/tools/subagent_integration_delegated.py` (11) | | `ouroboros/usage_accounting.py` | D16 | 42 | `ouroboros/_usage_cache_splits.py` (4)
`ouroboros/_usage_rows.py` (6)
`ouroboros/_usage_rows_memo.py` (6)
`ouroboros/usage_ledger.py` (18)
`ouroboros/usage_legacy_import.py` (5)
`ouroboros/utils.py` (3 ✗D18) | | `server.py` | D11 | 49 | `ouroboros/server_liveness.py` (4)
`ouroboros/server_maintenance.py` (14)
`ouroboros/server_owner_routing.py` (5)
`ouroboros/server_process.py` (6)
`ouroboros/server_restart.py` (7)
`ouroboros/server_routing_context.py` (13) | -| `supervisor/events.py` | D08 | 93 | `ouroboros/config.py` (1 ✗D12)
`ouroboros/contracts/task_constraint.py` (1 ✗D19)
`ouroboros/cost_projection.py` (3 ✗D16)
`ouroboros/subagent_messages.py` (1 ✗D07)
`ouroboros/task_results.py` (2 ✗D17)
`ouroboros/tool_capabilities.py` (2 ✗D04)
`ouroboros/utils.py` (4 ✗D18)
`supervisor/cognitive_operations.py` (2)
`supervisor/events_budget.py` (4)
`supervisor/events_chat_delivery.py` (9)
`supervisor/events_coop_checkpoint.py` (6)
`supervisor/events_evolution_done.py` (1)
`supervisor/events_project_routing.py` (9)
`supervisor/events_runtime_controls.py` (6)
`supervisor/events_schedule_task.py` (4)
`supervisor/events_subagent_admission.py` (15)
`supervisor/events_task_done.py` (8)
`supervisor/events_worker_reports.py` (7)
`supervisor/log_addressing.py` (4)
`supervisor/queue_transitions.py` (1)
`supervisor/steering.py` (2 ✗D09)
`supervisor/task_dispatch.py` (1) | +| `supervisor/events.py` | D08 | 94 | `ouroboros/config.py` (1 ✗D12)
`ouroboros/contracts/task_constraint.py` (1 ✗D19)
`ouroboros/cost_projection.py` (3 ✗D16)
`ouroboros/subagent_messages.py` (1 ✗D07)
`ouroboros/task_results.py` (2 ✗D17)
`ouroboros/tool_capabilities.py` (2 ✗D04)
`ouroboros/utils.py` (4 ✗D18)
`supervisor/cognitive_operations.py` (2)
`supervisor/events_budget.py` (4)
`supervisor/events_chat_delivery.py` (9)
`supervisor/events_coop_checkpoint.py` (6)
`supervisor/events_evolution_done.py` (1)
`supervisor/events_project_routing.py` (10)
`supervisor/events_runtime_controls.py` (6)
`supervisor/events_schedule_task.py` (4)
`supervisor/events_subagent_admission.py` (15)
`supervisor/events_task_done.py` (8)
`supervisor/events_worker_reports.py` (7)
`supervisor/log_addressing.py` (4)
`supervisor/queue_transitions.py` (1)
`supervisor/steering.py` (2 ✗D09)
`supervisor/task_dispatch.py` (1) | | `supervisor/git_ops.py` | D10 | 37 | `ouroboros/utils.py` (1 ✗D18)
`supervisor/git_ops_remotes.py` (4)
`supervisor/git_ops_rescue.py` (8)
`supervisor/git_ops_reset.py` (10)
`supervisor/git_ops_updates.py` (8)
`supervisor/state.py` (4 ✗D08)
`supervisor/update_recovery.py` (2) | | `supervisor/queue.py` | D08 | 102 | `ouroboros/config.py` (6 ✗D12)
`ouroboros/contracts/task_contract.py` (3 ✗D19)
`ouroboros/schedule_contract.py` (2)
`ouroboros/skill_loader.py` (1 ✗D14)
`ouroboros/utils.py` (3 ✗D18)
`supervisor/evolution_lifecycle.py` (12 ✗D15)
`supervisor/message_bus.py` (3)
`supervisor/queue_schedules.py` (12)
`supervisor/queue_snapshot.py` (5)
`supervisor/queue_timeouts.py` (8)
`supervisor/queue_transitions.py` (3)
`supervisor/schedule_time.py` (7)
`supervisor/state.py` (6)
`supervisor/task_admission.py` (9)
`supervisor/task_lifecycle.py` (17 ✗D09)
`supervisor/task_reaper.py` (5 ✗D09) | | `supervisor/state.py` | D08 | 4 | `ouroboros/utils.py` (4 ✗D18) | | `supervisor/task_lifecycle.py` | D09 | 32 | `supervisor/cancel_publication.py` (20)
`supervisor/queue_transitions.py` (11 ✗D08)
`supervisor/task_admission.py` (1 ✗D08) | | `supervisor/task_model_wait.py` | D08 | 1 | `ouroboros/model_wait.py` (1 ✗D09) | | `supervisor/update_merge.py` | D10 | 26 | `supervisor/update_candidate.py` (23)
`supervisor/update_merge_plan.py` (3) | -| `supervisor/workers.py` | D08 | 45 | `supervisor/log_addressing.py` (1)
`supervisor/message_bus.py` (1)
`supervisor/worker_assignment.py` (3)
`supervisor/worker_chat_lane.py` (5)
`supervisor/worker_health.py` (4)
`supervisor/worker_pool_lifecycle.py` (15)
`supervisor/worker_process.py` (6)
`supervisor/worker_promotion.py` (10) | +| `supervisor/workers.py` | D08 | 44 | `supervisor/log_addressing.py` (1)
`supervisor/message_bus.py` (1)
`supervisor/worker_assignment.py` (3)
`supervisor/worker_chat_lane.py` (5)
`supervisor/worker_health.py` (4)
`supervisor/worker_pool_lifecycle.py` (15)
`supervisor/worker_process.py` (6)
`supervisor/worker_promotion.py` (9) | diff --git a/ouroboros/gateway/contracts.py b/ouroboros/gateway/contracts.py index 0858cb256..1fb498e38 100644 --- a/ouroboros/gateway/contracts.py +++ b/ouroboros/gateway/contracts.py @@ -479,7 +479,7 @@ class ProjectsChangedOutbound(TypedDict): class MessageAnnotationOutbound(TypedDict): - """Bubble-free presentation update for one canonical owner message.""" + """Bubble-free presentation update for one owner message; ``cause``: the host's sentence for a refused act.""" type: Literal["message_annotation"] annotation_type: Literal["routing_ack"] @@ -494,10 +494,10 @@ class MessageAnnotationOutbound(TypedDict): project_chat_id: NotRequired[int] options: NotRequired[List[Dict[str, Any]]] attachment_manifest: NotRequired[List[AttachmentManifestEntry]] - # #198: the exact refusal-attempt identity — the picker card composes its - # decision_id (routing:{client_message_id}:{routing_token}) from it; a - # presentation frame without it renders text, never a clickable card. + # #198: the exact refusal-attempt identity — the picker card composes its decision_id + # (routing:{client_message_id}:{routing_token}) from it; a frame without it renders text, never a card. routing_token: NotRequired[str] + cause: NotRequired[str] ts: NotRequired[str] diff --git a/ouroboros/gateway/decision_contracts.py b/ouroboros/gateway/decision_contracts.py index 5b9c9075c..c6f45d48d 100644 --- a/ouroboros/gateway/decision_contracts.py +++ b/ouroboros/gateway/decision_contracts.py @@ -47,7 +47,8 @@ class DecisionResponse(TypedDict, total=False): a settled task is 409 with the true state (expired_terminal/answered), so the card settles instead of inviting retries. Routing adds dispatched (confirmed durable receipt), task_id (derived promoted id), latest_status - (superseding status on 409), and reason/detail diagnostics. + (superseding status on 409), reason/detail diagnostics, and cause (the + owner-facing sentence for a refused routing act). Model waits distinguish accepted (202, applied false) from the worker's applied_request_id. wait carries the current projection and revision; @@ -67,6 +68,7 @@ class DecisionResponse(TypedDict, total=False): latest_status: str reason: str detail: str + cause: str request_id: str applied: bool saved: Optional[bool] diff --git a/ouroboros/gateway/history.py b/ouroboros/gateway/history.py index 279f1e718..2d0f1b7b0 100644 --- a/ouroboros/gateway/history.py +++ b/ouroboros/gateway/history.py @@ -191,7 +191,7 @@ def _user_annotation( return { key: annotation.get(key) for key in ( - "action", "target", "target_label", "status", "detail", "options", + "action", "target", "target_label", "status", "detail", "cause", "options", "attachment_manifest", "routing_token", "project_id", "project_chat_id", ) if key in annotation diff --git a/ouroboros/gateway/routing_decision.py b/ouroboros/gateway/routing_decision.py index 4c520bd0f..9d5d92569 100644 --- a/ouroboros/gateway/routing_decision.py +++ b/ouroboros/gateway/routing_decision.py @@ -94,7 +94,11 @@ def handle_routing_decision( if not client_message_id or not token: return 400, {"ok": False, "error": "malformed_decision_id", "decision_id": decision_id} - from ouroboros.project_dialogue import append_chat_annotation, latest_chat_annotations + from ouroboros.project_dialogue import ( + append_chat_annotation, + latest_chat_annotations, + routing_refusal_cause, + ) # The card is live only while its token is the message's LATEST act: # receipts are kept per token, but a newer routing attempt on the same @@ -214,6 +218,9 @@ def handle_routing_decision( # routing decision on a refused message, so it wears the # route_to_project receipt label regardless of source chat. "routed_from_main": True, + # The owner's click, not a model turn, issued this promote: a + # refusal is told by the handler's typed System row (no narrator). + "host_initiated": True, "client_message_id": client_message_id, "attachment_uploads": attachment_uploads, **provenance, @@ -313,9 +320,18 @@ def handle_routing_decision( # the latest row — re-assert the refusal under the ORIGINAL token so # the card the UI re-opens still validates and replays cleanly. _reopen_refusal() + reason = str(outcome.get("reason") or outcome_status) + # The same act the receipt wore, so the sentence carries its prefix. + receipt_action = ( + "steer_task" if action == "steer_task" + else ("route_to_project" if evt.get("routed_from_main") else "promote_chat_to_task") + ) return 409, {"ok": False, "error": "dispatch_rejected", "decision_id": decision_id, "state": "open", - "reason": str(outcome.get("reason") or outcome_status)} + "reason": reason, + # The owner-facing sentence from the host's own table (the + # toast shows it instead of the raw code). + "cause": routing_refusal_cause(receipt_action, "needs_manual_target", reason, None)} # Unconfirmed: honestly retriable — the derived identities make a replay # of the SAME request byte-identical, so the supervisor dedupes it. return 503, {"ok": False, "error": "dispatch_unconfirmed", diff --git a/ouroboros/project_dialogue.py b/ouroboros/project_dialogue.py index 09c290f2a..599ff29c9 100644 --- a/ouroboros/project_dialogue.py +++ b/ouroboros/project_dialogue.py @@ -206,7 +206,10 @@ def room_membership(chat_id: int, project_chat_ids: set, source_refs: list, row = entry if isinstance(entry, dict) else {} if is_a2a_chat_id(entry_chat): return False - bound = bound_room_chat(bindings, row) + # An admission notice (" · Not started: …") is addressed to the + # chat the OWNER wrote in; its task id is bound to the destination + # project it never started in, so binding lineage must not move it. + bound = 0 if row.get("type") in ADMISSION_NOTICE_TYPES else bound_room_chat(bindings, row) lifecycle = row.get("type") in {"project_started", "project_completion_summary"} if chat_id in project_chat_ids: return not lifecycle and (bound == chat_id or entry_chat == chat_id @@ -398,6 +401,7 @@ def append_chat_annotation( routing_token: str = "", reason: str = "", detail: str = "", + cause: str = "", options: Any = None, attachment_manifest: Any = None, require_latest_status: Any = None, @@ -438,6 +442,10 @@ def append_chat_annotation( row["reason"] = str(reason)[:200] if str(detail or ""): row["detail"] = str(detail)[:1000] + if str(cause or ""): + # Q3=A: the host-owned owner-facing sentence for a refused act; the + # browser renders it verbatim on the receipt line (reason stays a code). + row["cause"] = str(cause)[:200] if isinstance(options, list): row["options"] = [dict(item) for item in options[:100] if isinstance(item, dict)] if isinstance(attachment_manifest, list): @@ -498,6 +506,106 @@ def routing_target_label( return "Task" +# Q3=A: the HOST owns the owner-facing sentence for a REFUSED routing act. One +# factual PHRASE per typed reason (what the producer branch observed: ≤ 60 +# chars, lower-case start, no codes, no trailing period); the prefix comes from +# the ACT and its outcome in routing_refusal_cause, so one reason reads +# "Not started: …" on a promote and "Not moved: …" on a scope bind. The receipt +# under the owner's message, the host-initiated System row and the picker's +# 409 toast all read this table; the browser renders the sentence verbatim (no +# client table). A reason without a row stays raw ("Not started (<reason>)") +# so a new refusal is visible before it has words. +ROUTING_REFUSAL_CAUSES: Dict[str, str] = { + "workspace_unusable": "the working folder can't be used", + "workspace_provisioning_failed": "no working folder could be created for the project", + "worker_pool_unavailable": "no worker is available right now", + "worker_pool_state_unavailable": "the worker pool could not be checked", + "duplicate_task_id": "this task already exists", + "admission_reservation_owned": "another request already owns this task id", + "admission_reservation_lost": "the task lost its place in the queue", + "admission_reservation_failed": "the task could not be admitted", + "admission_fence": "the task could not be admitted", + "admission_rejected": "the task could not be admitted", + "invalid_admission_reservation": "the task id or its token was missing", + "task_id_lookup_failed": "the task record could not be read", + "empty_objective": "the request was empty", + "project_routing_fence": "the project no longer accepts new work", + "project_routing_fence_lookup_failed": "the project state could not be checked", + "project_binding_failed": "the project could not be set up", + "project_registration_failed": "the project could not be set up", + "ensure_project_scope_failed": "the project could not be set up", + "project_source_error": "the project folder could not be attached", + "attachment_admission_rejected": "the attachments could not be staged", + "staging_unavailable": "the attachments could not be staged", + "queue_snapshot_persist_unavailable": "the task queue could not be saved", + "queue_snapshot_persist_failed": "the task queue could not be saved", + "invalid_skill_repair_constraint": "the skill repair request was invalid", + "skill_repair_payload_missing": "the skill's files are missing", + "skill_repair_payload_unreadable": "the skill's files could not be read", + "skill_repair_admission_unwritable": "the skill repair request could not be recorded", + "repair_promotion_failed": "the skill repair request could not be started", + "task_acceptance_fence": "the task tree is already being accepted", + "invalid_task_depth": "the task depth was invalid", + "promotion_persistence_failed": "the task record could not be saved", + "routing_receipt_persist_failed": "the receipt could not be saved", + "routing_annotation_persist_failed": "the receipt could not be saved", + "source_continuation_publish_failed": "the source hand-off could not be published", + "confirmation_timeout": "no confirmation arrived in time", + "target_unknown": "that task is no longer running", + "direct_chat_turn": "that reply has already been given", + "subagent_target": "that task is a helper of another task", + "chat_mismatch": "that task belongs to another chat", + "cancel_pending": "that task is being stopped", + "target_closed": "that task has already finished", + "target_finished": "that task has already finished", + "acceptance_fence_sealed": "that task has already finished", + "mailbox_write_failed": "the message could not be saved", + "project_scope_conflict": "the task already belongs to another project", + "missing_task_or_project": "no task or project was named", +} + +# The typed System rows a host-initiated admission refusal sends (Q2=A): plain +# system bubbles addressed to the chat the OWNER wrote in, never moved by the +# refused task's project binding (room_membership) and never a terminal fact. +ADMISSION_NOTICE_TYPES = frozenset({"task_not_started", "task_start_unconfirmed"}) + +# Statuses of an act that LANDED (or is still in flight): no cause sentence. +_LANDED_ROUTING_STATUSES = frozenset({"scheduled", "delivered", "pending", "dispatch_pending", "accepted"}) + + +def routing_refusal_cause(action: str, status: str, reason: str, options: Any = None) -> str: + """The owner-facing sentence for one routing receipt; "" when the act landed + or when the picker keeps «Choose a target» (a refusal WITH options). + + The prefix states only what the act's outcome proves: an UNCONFIRMED act + reads "Not confirmed" whatever it was; a refused steer "Not delivered"; a + refused scope bind "Not moved"; every other refused act (promote, route, + skill repair) "Not started". An unconfirmed act with an unknown reason + reads as the honest "may or may not have started"; any other unknown reason + stays raw (``Not started (<reason>)`` — DESIGN sanctions raw over invented).""" + status_text = str(status or "").strip() + if status_text in _LANDED_ROUTING_STATUSES: + return "" + if status_text == "needs_manual_target" and isinstance(options, list) and options: + return "" + action_text = str(action or "").strip() + if status_text == "unconfirmed": + prefix = "Not confirmed" + elif action_text == "steer_task": + prefix = "Not delivered" + elif action_text == "ensure_project_scope": + prefix = "Not moved" + else: + prefix = "Not started" + reason_text = str(reason or "").strip() + phrase = ROUTING_REFUSAL_CAUSES.get(reason_text, "") + if phrase: + return f"{prefix}: {phrase}" + if status_text == "unconfirmed": + return "Not confirmed: the task may or may not have started" + return f"{prefix} ({reason_text})" if reason_text else prefix + + def routing_options_with_labels(drive_root: Any, options: Any) -> List[Dict[str, Any]]: """Stamp human labels on manual task choices while retaining their raw ids.""" rows: List[Dict[str, Any]] = [] diff --git a/ouroboros/server_owner_routing.py b/ouroboros/server_owner_routing.py index 733296949..f4bd6099d 100644 --- a/ouroboros/server_owner_routing.py +++ b/ouroboros/server_owner_routing.py @@ -447,7 +447,7 @@ def _route_owner_message(bridge: Any, ctx: Any, incoming: Dict[str, Any]) -> Non # An explicit selected-skill development request already asks for a # managed task. Preserve its source/caller facts while ordinary promotion # validates the payload and records the admitted revision. - from supervisor.events import _handle_promote_chat_to_task + from supervisor.events import _handle_promote_chat_to_task, _notify_host_initiated_refusal ctx.consciousness.inject_observation( f"Message from my human: {incoming.get('log_text') or ''}" @@ -462,6 +462,10 @@ def _route_owner_message(bridge: Any, ctx: Any, incoming: Dict[str, Any]) -> Non "client_message_id": client_message_id, "task_constraint": task_constraint, "routed_from_main": True, + # The host issued this promote (skill card), so no model turn waits on + # the receipt: a refusal is told to the owner by ONE typed System row + # from the promote handler, in the chat the owner wrote in. + "host_initiated": True, } metadata = task_metadata if isinstance(task_metadata, dict) else {} if isinstance(metadata.get("client_surface"), dict): @@ -484,6 +488,8 @@ def _route_owner_message(bridge: Any, ctx: Any, incoming: Dict[str, Any]) -> Non "reason": "repair_promotion_failed", "task_id": task_id, } + # Minted OUTSIDE the handler, so its publication boundary never saw it. + _notify_host_initiated_refusal(ctx, event, outcome) outcome = outcome if isinstance(outcome, dict) else {"status": "scheduled", "task_id": task_id} outcome_status = str(outcome.get("status") or "needs_manual_target") if outcome_status == "scheduled": @@ -494,15 +500,8 @@ def _route_owner_message(bridge: Any, ctx: Any, incoming: Dict[str, Any]) -> Non ) except Exception: log.debug("Repair promotion success notification failed", exc_info=True) - else: - reason = str(outcome.get("reason") or outcome_status) - try: - ctx.send_with_budget( - chat_id, - f"⚠️ Repair task was not started ({reason}). Please retry from the skill card.", - ) - except Exception: - log.debug("Repair promotion refusal notification failed", exc_info=True) + # A refusal is already told by the promote handler's typed System row + # (host_initiated) plus the receipt under the owner's message. return reserved_project = _reserved_project_for_chat(ctx, chat_id) project_id = ( @@ -584,6 +583,9 @@ def _route_owner_message(bridge: Any, ctx: Any, incoming: Dict[str, Any]) -> Non "client_message_id": client_message_id, "task_constraint": task_constraint, "force_plan": True, "force_plan_source": task_metadata.get("force_plan_source"), "attachment_uploads": list(task_metadata.get("chat_attachment_uploads") or []), + # Swarm: the host promotes with no model turn waiting on the receipt, + # so a refusal reaches the owner as the handler's typed System row. + "host_initiated": True, } if isinstance(task_metadata.get("client_surface"), dict): event["client_surface"] = dict(task_metadata["client_surface"]) diff --git a/ouroboros/tools/control.py b/ouroboros/tools/control.py index ff79af492..47f09dfb8 100644 --- a/ouroboros/tools/control.py +++ b/ouroboros/tools/control.py @@ -162,7 +162,7 @@ def get_tools() -> List[ToolEntry]: "project_name": {"type": "string", "description": "Set ONLY to create a brand-new NAMED project now and start a NEW independent task in it (e.g. 'airi research'); to move THIS task into a project use ensure_project_scope. The display name; a filesystem id is derived from it.", "default": ""}, "expected_output": {"type": "string", "description": "What done looks like.", "default": ""}, "project_id": {"type": "string", "description": "Optional EXISTING project scope (filesystem-clean id).", "default": ""}, - "workspace_root": {"type": "string", "description": "Optional absolute working-folder path (validated at admission as an ordinary folder or Git worktree root outside the Ouroboros repo/data). Git-specific operations require a Git worktree; ordinary file and process work is supported directly in a validated folder. When omitted for a project-scoped task, the project's registered working_dir is used by default.", "default": ""}, + "workspace_root": {"type": "string", "description": "Optional absolute working-folder path (validated at admission as an ordinary folder or Git worktree root outside the Ouroboros repo/data). Git-specific operations require a Git worktree; ordinary file and process work is supported directly in a validated folder. When omitted for a project-scoped task, the project's registered working_dir is used by default. Leave empty to work in Ouroboros's own repository (the Main default).", "default": ""}, "workspace": {"type": "string", "description": "Pass 'none' to opt OUT of the project room's default working folder (a folder-less task in a folder-ful project). Leave empty otherwise.", "default": ""}, "source": {"type": "string", "description": "Attach or clone the project's working folder in ONE move: a git URL (https://... or git@host:path — cloned server-side into the projects root; private repos fail typed auth_required) or an existing folder path (validated attach). The folder is registered on the project (provenance + trusted_at) and becomes this task's active workspace. Use for 'help me debug this GitHub repo / this folder' asks.", "default": ""}, "predecessor_task_id": {"type": "string", "description": "Required explicit selector: pass an empty string for fresh work, or the completed result id shown by the host routing manifest to continue it."}, diff --git a/ouroboros/tools/control_routing.py b/ouroboros/tools/control_routing.py index b32faa92a..a5ac2ce68 100644 --- a/ouroboros/tools/control_routing.py +++ b/ouroboros/tools/control_routing.py @@ -303,6 +303,36 @@ def _promote_chat_to_task( current_chat_id = int(getattr(ctx, "current_chat_id", None) or 0) except (TypeError, ValueError): current_chat_id = 0 + requested_root = str(workspace_root or "").strip() + workspace_sentinel = str(workspace or "").strip().lower() + repo_root_note = "" + if requested_root: + # Q4=A: naming the Ouroboros repository ITSELF names the documented + # default (no separate workspace — the ordinary self-modification task), + # so the EXACT root maps onto the existing "none" sentinel, in Main and + # in every project room alike. A subfolder, the data drive and every + # other path pass through unchanged; admission refuses them with the + # repair hint. + from ouroboros.tool_access import paths_overlap_casefold + from ouroboros.workspace_admission import WORKSPACE_NONE + + system_repo = getattr(ctx, "system_repo_dir", None) or getattr(ctx, "repo_dir", None) + try: + requested_path = Path(requested_root).expanduser().resolve(strict=False) + system_repo_path = Path(str(system_repo)).resolve(strict=False) if system_repo else None + same_root = ( + system_repo_path is not None + and len(requested_path.parts) == len(system_repo_path.parts) + and paths_overlap_casefold(requested_path, system_repo_path) + ) + except (OSError, ValueError, RuntimeError): + same_root = False + if same_root: + requested_root, workspace_sentinel = "", WORKSPACE_NONE + repo_root_note = ( + " (workspace_root named the Ouroboros repository itself; started as an " + "ordinary task over it — no separate workspace)" + ) tid = uuid.uuid4().hex[:16] routing_token = uuid.uuid4().hex disabled_reason = _promotion_pool_disabled_from_snapshot(ctx) @@ -328,13 +358,13 @@ def _promote_chat_to_task( "project_id": pid, "project_name": display_name, "title": str(title or "").strip()[:80], - "workspace_root": str(workspace_root or "").strip(), + "workspace_root": requested_root, # Source admission is intentionally supervisor-side, after the # authoritative worker-pool and duplicate-id gates. "source": str(source or "").strip(), # v6.58.0: "none" opts a project-room task OUT of the room's working_dir # default (a folder-less task in a folder-ful project stays possible). - "workspace": str(workspace or "").strip().lower(), + "workspace": workspace_sentinel, "chat_id": current_chat_id, "client_message_id": str( ((getattr(ctx, "task_metadata", {}) or {}).get("client_message_id") or "") @@ -361,6 +391,7 @@ def _promote_chat_to_task( "presence": dict(presence), "task_contract": dict(getattr(ctx, "task_contract", {}) or {}), }) + repo_root_note = "" # Presence runs in its admitted folder, never over the repo _attach_origin_from_metadata(ctx, evt) predecessor_error = _attach_predecessor_authority_from_metadata( ctx, evt, predecessor_task_id, @@ -383,7 +414,8 @@ def _promote_chat_to_task( effective_pid = str(confirmation.get("effective_project_id") or "") scope_note = _effective_scope_note(ctx, effective_pid) response = ( - f"OK: task {tid}{scope_note} accepted and durably scheduled ({mode}).{source_confirmation} " + f"OK: task {tid}{scope_note} accepted and durably scheduled ({mode}){repo_root_note}." + f"{source_confirmation} " "The task now runs independently, and follow-up chat can steer it. " "Use wait_task/get_task_result if its result " "is needed in this conversation." diff --git a/ouroboros/workspace_admission.py b/ouroboros/workspace_admission.py index b23e6ec0b..f76cf355a 100644 --- a/ouroboros/workspace_admission.py +++ b/ouroboros/workspace_admission.py @@ -175,6 +175,85 @@ def resolve_room_workspace( return (str(resolved) if resolved else ""), "" +def workspace_repair_hint( + *, + ws_error: str, + explicit_workspace: str = "", + project_id: str = "", + project_folder: str = "", + presence: bool = False, + retired_worktree: bool = False, + drive_root: Any = None, + system_repo_dir: Any = None, +) -> str: + """The MODEL-facing repair for one refused workspace: the typed cause plus + the one move that fixes it, following the SOURCE of the refused folder. + + ``resolve_room_workspace`` already types the source, so the repair follows + it instead of sending the caller to a Projects setting the failure never + read: a Presence profile's folder is fixed in the profile; a path the + REQUEST named is re-promoted against the project's folder or with + ``workspace='none'`` (a subfolder of the Ouroboros repository can never be a + workspace, and a delegated-run worktree is gone once its run ends); a + project ``working_dir`` — or a failed auto-provision — is fixed in Projects. + The project folder is named only when the registry can be read; the + worktree/repo facts are derived here unless the caller already knows them. + Never raises; this text rides ``detail`` into the typed refusal. + """ + cause = str(ws_error or "").strip().rstrip(".") + if presence: + return ( + f"The folder configured in the Presence profile is unusable: {cause}. " + "Fix the Presence profile's workspace_root or clear it." + ) + explicit = str(explicit_workspace or "").strip() + if not explicit: + return ( + f"{cause}. Fix the project's working folder (Projects → this project) " + "or re-promote with workspace='none' for a folder-less task." + ) + requested = pathlib.Path(explicit).expanduser() + if system_repo_dir is not None: + from ouroboros.tool_access import paths_overlap_casefold + + try: + under_repo = paths_overlap_casefold(requested, pathlib.Path(system_repo_dir)) + except Exception: + under_repo = False + log.debug("workspace repair hint: repo-overlap check failed for %r", explicit, exc_info=True) + if under_repo: + return ( + f"{cause}. workspace_root must be a folder outside the Ouroboros repository, " + "or empty (or workspace='none') to work in the repository itself." + ) + if not retired_worktree: + try: + from ouroboros.config import get_subagent_worktree_root + from ouroboros.tool_access_paths import path_is_relative_to + + retired_worktree = path_is_relative_to(requested, pathlib.Path(get_subagent_worktree_root())) + except Exception: + log.debug("workspace repair hint: worktree-root check failed for %r", explicit, exc_info=True) + folder = str(project_folder or "").strip() + if not folder and str(project_id or "").strip() and drive_root is not None: + try: + from ouroboros.projects_registry import get_project + + folder = str((get_project(drive_root, project_id) or {}).get("working_dir") or "").strip() + except Exception: + log.debug("workspace repair hint: project working_dir unreadable for %s", project_id, exc_info=True) + named_folder = f" ({folder})" if folder else "" + if retired_worktree: + return ( + f"{cause}. That path is inside a delegated-run worktree, which is removed when " + f"its run ends; re-promote with the project's folder{named_folder} or with workspace='none'." + ) + return ( + f"{cause}. This task asked for {explicit} explicitly; re-promote it against the " + f"project folder{named_folder} or with workspace='none' for a folder-less task." + ) + + def room_chat_lens_dir(drive_root: Any, project_id: str) -> tuple[str, str]: """The selected folder for a direct conversation, with an availability note. diff --git a/supervisor/events.py b/supervisor/events.py index 206f6f629..be88d6776 100644 --- a/supervisor/events.py +++ b/supervisor/events.py @@ -190,6 +190,7 @@ from supervisor.events_project_routing import ( # noqa: E402, F401 -- intention _handle_project_digest, _handle_promote_chat_to_task, _handle_routing_manual_target, + _notify_host_initiated_refusal, _persist_promote_rejection, _prepare_promote_source_off_loop, _publish_routing_ack, diff --git a/supervisor/events_project_routing.py b/supervisor/events_project_routing.py index 0b0c7f82d..63330ada0 100644 --- a/supervisor/events_project_routing.py +++ b/supervisor/events_project_routing.py @@ -61,6 +61,8 @@ def _emit_routing_receipt( publish: bool = True, ) -> Dict[str, Any]: """Persist and publish one token-bound routing annotation receipt.""" + from ouroboros.project_dialogue import routing_refusal_cause + if target and not str(target_label or "").strip(): from ouroboros.project_dialogue import routing_target_label @@ -69,6 +71,9 @@ def _emit_routing_receipt( routing_token = str(evt.get("routing_token") or "").strip() annotation_status = "not_applicable" project_address = _routing_project_address(ctx, target, status) + # Q3=A: the owner-facing sentence for a REFUSED act (host table; "" for a + # landed row or the picker), computed once for the durable row and the ack. + cause = routing_refusal_cause(action, status, reason, options) if client_message_id: try: from ouroboros.project_dialogue import append_chat_annotation @@ -85,6 +90,7 @@ def _emit_routing_receipt( routing_token=routing_token, reason=reason, detail=detail, + cause=cause, options=options, attachment_manifest=attachment_manifest, **project_address, @@ -129,6 +135,7 @@ def _emit_routing_receipt( status=effective_status, options=options, attachment_manifest=attachment_manifest, + cause=cause, ) return receipt @@ -143,6 +150,7 @@ def _publish_routing_ack( status: str, options: Optional[list] = None, attachment_manifest: Optional[list] = None, + cause: str = "", ) -> None: """Publish a live non-bubble acknowledgement after durable authority exists.""" try: @@ -170,6 +178,8 @@ def _publish_routing_ack( ack_kwargs["options"] = options if attachment_manifest is not None: ack_kwargs["attachment_manifest"] = attachment_manifest + if str(cause or ""): + ack_kwargs["cause"] = str(cause) if str(evt.get("routing_token") or ""): ack_kwargs["routing_token"] = str(evt.get("routing_token")) ack( @@ -316,6 +326,8 @@ def _prepare_promote_source_off_loop(evt: Dict[str, Any], ctx: Any) -> None: task_id = str(evt.get("task_id") or "") routing_token = str(evt.get("routing_token") or "") supervisor_queue.release_task_admission(task_id, routing_token) + # No host producer (skill card, Swarm, picker click) stamps `source`, so + # this exit never bypasses the wrapper's host-initiated refusal notice. failed = { "status": "unconfirmed", "reason": "source_continuation_publish_failed", @@ -341,12 +353,14 @@ def _prepare_promote_source_off_loop(evt: Dict[str, Any], ctx: Any) -> None: log.exception("Failed to persist promote source continuation failure") -def _handle_promote_chat_to_task(evt: Dict[str, Any], ctx: Any) -> Dict[str, Any]: +def _promote_chat_to_task_outcome(evt: Dict[str, Any], ctx: Any) -> Dict[str, Any]: """Spawn a first-class pooled owner task from a conversation-lane promote. Unlike ``schedule_subagent`` the child is NOT a subagent: it is a normal owner task (live card, canonical drive, project lease participation). The - conversation lane that emitted the event stays free. + conversation lane that emitted the event stays free. Every exit returns the + typed outcome to ``_handle_promote_chat_to_task``, the one publication + boundary that tells the owner about a host-initiated refusal. """ from supervisor.workers import ( _broadcast_task_named, @@ -380,6 +394,9 @@ def _handle_promote_chat_to_task(evt: Dict[str, Any], ctx: Any) -> Dict[str, Any "status": str((admission or {}).get("status") or "unconfirmed"), "task_id": task_id, "reason": str((admission or {}).get("reason") or ""), + # A replay of an already-settled admission: the owner was + # told once, so the refusal notice stays silent. + "replayed": True, } if reservation_status == "already_reserved": return {"status": "preparing", "task_id": task_id} @@ -617,6 +634,49 @@ def _handle_promote_chat_to_task(evt: Dict[str, Any], ctx: Any) -> Dict[str, Any return failed_outcome +def _notify_host_initiated_refusal(ctx: Any, evt: Dict[str, Any], outcome: Any) -> None: + """The ONE place a HOST-issued promote (skill card, Swarm, picker click) + tells the owner it did not start (Q2=A). No model turn narrates such a + refusal, so exactly one typed System row lands in the chat the OWNER wrote + in — never ``task["chat_id"]``, which project admission may have rewritten + to a new room — bound to the task that never started. A tool-issued promote + gets nothing here (its receipt plus the failed call are the record and the + model narrates); preparing, scheduled and replayed outcomes send nothing. + """ + if not evt.get("host_initiated") or not isinstance(outcome, dict) or outcome.get("replayed"): + return + status = str(outcome.get("status") or "") + if status not in {"needs_manual_target", "unconfirmed"}: + return + try: + from ouroboros.project_dialogue import routing_refusal_cause + from supervisor.message_bus import notification_chat_route + + chat = notification_chat_route(evt.get("chat_id")) + if chat is None: + return + first_line = next(iter(str(evt.get("objective") or "").strip().splitlines()), "") + title = str(evt.get("title") or evt.get("suggested_name") or "").strip() or first_line[:60] or "Task" + action = "route_to_project" if bool(evt.get("routed_from_main")) else "promote_chat_to_task" + reason = str(outcome.get("reason") or ("admission_rejected" if status == "needs_manual_target" else "")) + ctx.send_with_budget( + chat, f"{title} · {routing_refusal_cause(action, status, reason, None)}", role="system", + system_type="task_start_unconfirmed" if status == "unconfirmed" else "task_not_started", + task_id=str(outcome.get("task_id") or evt.get("task_id") or ""), + ) + except Exception: + log.debug("host-initiated refusal row failed for %s", evt.get("task_id"), exc_info=True) + + +def _handle_promote_chat_to_task(evt: Dict[str, Any], ctx: Any) -> Dict[str, Any]: + """The one publication boundary of a promote: every outcome of the handler + body — the reservation-blocked and unconfirmed early returns included — + passes the host-initiated refusal notice before it is returned.""" + outcome = _promote_chat_to_task_outcome(evt, ctx) + _notify_host_initiated_refusal(ctx, evt, outcome) + return outcome + + def _record_obligation_transfer(ctx: Any, transfer: Dict[str, Any]) -> None: """The promoter's task details name where its planning obligation went.""" promoter = str(transfer.get("from") or "") diff --git a/supervisor/message_bus.py b/supervisor/message_bus.py index ce3256622..3617fefe1 100644 --- a/supervisor/message_bus.py +++ b/supervisor/message_bus.py @@ -566,6 +566,7 @@ class LocalChatBridge: options: Optional[List[Dict[str, Any]]] = None, attachment_manifest: Optional[List[Dict[str, Any]]] = None, routing_token: str = "", + cause: str = "", ) -> None: """Emit a typed routing receipt without creating an assistant bubble. @@ -593,6 +594,9 @@ class LocalChatBridge: # #198: the picker card's click identity; presentation-only frames # without it stay text lines. payload["routing_token"] = str(routing_token) + if str(cause or ""): + # Q3=A: the host-owned owner-facing sentence for a refused act. + payload["cause"] = str(cause) if options is not None: payload["options"] = [dict(row) for row in options if isinstance(row, dict)] if attachment_manifest is not None: diff --git a/supervisor/steering.py b/supervisor/steering.py index e7ebcea20..40178447f 100644 --- a/supervisor/steering.py +++ b/supervisor/steering.py @@ -106,6 +106,10 @@ def _refuse_steering_while_cancelling( ) # A cancel-pending refusal is a live notice to the OWNER who asked; a task # that spoke for itself reads its typed refusal in the tool result instead. + # An act that wears an owner message already shows its cause on that + # message's receipt line; only an UNLABELLED owner act (a synthetic receipt + # id) needs the standalone sentence — the same rule as the other refusals. + notify = notify and not _relayed_owner_message(str(evt.get("client_message_id") or "").strip()) if notify and not _task_issued(evt) and chat_id: try: ctx.send_with_budget(chat_id, _cancel_pending_notice(target_label)) @@ -455,10 +459,13 @@ def _handle_steer_task(evt: Dict[str, Any], ctx: Any) -> None: # linger in the dying task's artifact store. The notice is the owner's: # a task that spoke for itself has its typed refusal and Logs row. if not task_issued and chat_id: - try: - ctx.send_with_budget(chat_id, _cancel_pending_notice(target_label)) - except Exception: - log.debug("steer_task cancel-pending notice failed", exc_info=True) + # Same owner-labelled rule as the up-front check: a relayed owner + # message already carries the cause on its receipt line. + if not relayed_owner_message_id: + try: + ctx.send_with_budget(chat_id, _cancel_pending_notice(target_label)) + except Exception: + log.debug("steer_task cancel-pending notice failed", exc_info=True) if delivered: if fence_generation_changed: ctx.persist_queue_snapshot(reason="acceptance_fence_owner_message") diff --git a/supervisor/worker_promotion.py b/supervisor/worker_promotion.py index 638ee317d..c5a017a33 100644 --- a/supervisor/worker_promotion.py +++ b/supervisor/worker_promotion.py @@ -232,6 +232,7 @@ def _admit_project_scope( "status": "needs_manual_target", "reason": "project_routing_fence", "project_lifecycle": existing_lifecycle, + "detail": f"project lifecycle: {existing_lifecycle}", "task_id": tid, }, attachment_manifest) except Exception: @@ -310,7 +311,10 @@ def _admit_project_scope( # project off-loop (_prepare_promote_source_off_loop) — same # agent-initiated creation, so the announce gate honors it. project = {**(project or {}), "created": True} - _pool()._announce_created_project(project, tid, task=task) + # The "Project · Started" row is owed only once the task is REALLY in the + # queue: promote_chat_to_task announces it after enqueue_task succeeded, + # so a workspace refusal after project creation announces nothing. + task["_announce_project"] = project except Exception: log.warning("promote: project registration failed for %s", pid, exc_info=True) return _pool()._reject_promoted_after_attachment_stage({ @@ -555,6 +559,8 @@ def promote_chat_to_task(evt: dict, ctx: Any) -> dict: task["attachment_images"] = [row for row in attachment_manifest if row.get("is_image")] if isinstance(task.get("task_contract"), dict): task["task_contract"].update(authority) + # Popped BEFORE the row is built/queued so the announce fact never rides it. + announce_project = task.pop("_announce_project", None) attach_task_contract(task) admitted = ctx.enqueue_task(task) if isinstance(admitted, dict) and admitted.get("_admission_blocked"): @@ -564,6 +570,8 @@ def promote_chat_to_task(evt: dict, ctx: Any) -> dict: "project_lifecycle": str(admitted.get("_project_lifecycle") or ""), "task_id": tid, }, attachment_manifest) + if announce_project is not None: + _pool()._announce_created_project(announce_project, tid, task=task) # Owner 3=A: the promoter's unmet planning obligation now belongs to this root # (stamped by _promoted_force_plan_metadata above); release it on the promoter's # live row BEFORE the snapshot persist below, so one persist shows both facts. @@ -635,6 +643,7 @@ def _admit_promoted_workspace(evt: dict, ctx: Any, task: dict, *, pid: str, tid: bounded_workspace_preflight, compose_workspace_block, resolve_room_workspace, + workspace_repair_hint, ) if task.get("_presence_origin"): @@ -646,8 +655,10 @@ def _admit_promoted_workspace(evt: dict, ctx: Any, task: dict, *, pid: str, tid: project_id="", explicit_workspace=str(workspace.get("root") or ""), ) if ws_error: - _fail_promoted_task_loudly(ctx, task, ws_error) - return {"status": "needs_manual_target", "reason": "workspace_unusable", "task_id": tid} + return { + "status": "needs_manual_target", "reason": "workspace_unusable", "task_id": tid, + "detail": workspace_repair_hint(ws_error=ws_error, presence=True), + } if resolved_ws: task.update(workspace_root=resolved_ws, workspace_mode="external", memory_mode="shared") return None @@ -693,15 +704,14 @@ def _admit_promoted_workspace(evt: dict, ctx: Any, task: dict, *, pid: str, tid: # Bind-or-fail (v6.58.0): falling through to a workspace-less # self_modification-profile task over the system repo is exactly # the silent degradation the admission SSOT exists to kill. - _fail_promoted_task_loudly( - ctx, task, - f"project {pid!r} has no working folder and auto-provisioning one failed; " - "see the supervisor log (ensure_project_workspace)", - ) return { "status": "needs_manual_target", "reason": "workspace_provisioning_failed", "task_id": tid, + "detail": workspace_repair_hint( + ws_error=f"project {pid!r} has no working folder and auto-provisioning " + "one failed; see the supervisor log (ensure_project_workspace)", + ), } task.setdefault("metadata", {})["workspace_autoprovisioned"] = True @@ -713,11 +723,13 @@ def _admit_promoted_workspace(evt: dict, ctx: Any, task: dict, *, pid: str, tid: workspace_sentinel=str(evt.get("workspace") or ""), ) if ws_error: - _fail_promoted_task_loudly( - ctx, task, ws_error, - explicit_workspace=str(evt.get("workspace_root") or "").strip(), project_id=pid, - ) - return {"status": "needs_manual_target", "reason": "workspace_unusable", "task_id": tid} + return { + "status": "needs_manual_target", "reason": "workspace_unusable", "task_id": tid, + "detail": workspace_repair_hint( + ws_error=ws_error, explicit_workspace=str(evt.get("workspace_root") or "").strip(), + project_id=pid, drive_root=_pool().DRIVE_ROOT, system_repo_dir=_pool().REPO_DIR, + ), + } if resolved_ws: task["workspace_root"] = resolved_ws task["workspace_mode"] = "external" @@ -767,85 +779,6 @@ def _admit_promoted_workspace(evt: dict, ctx: Any, task: dict, *, pid: str, tid: return None -def _explicit_workspace_remedy(explicit: str, project_id: str) -> str: - """What to do about a folder the REQUEST named, not the project's own. - - ``resolve_room_workspace`` already types the source, so the remedy follows - it instead of sending the owner to a Projects setting the failure never - touched. The project's folder is named only when the registry can be read, - and a path under the delegated-run worktree root gets the one clause that - explains why it is gone.""" - import pathlib - - folder = "" - try: - from ouroboros.projects_registry import get_project - - folder = str((get_project(_pool().DRIVE_ROOT, project_id) or {}).get("working_dir") or "").strip() - except Exception: - log.debug("promote loud-fail: project working_dir unreadable for %s", project_id, exc_info=True) - retired = False - try: - from ouroboros.config import get_subagent_worktree_root - from ouroboros.tool_access_paths import path_is_relative_to - - retired = path_is_relative_to(pathlib.Path(explicit), pathlib.Path(get_subagent_worktree_root())) - except Exception: - log.debug("promote loud-fail: worktree-root check failed for %r", explicit, exc_info=True) - return ( - f"This task asked for {explicit} explicitly, so the project's working folder was never used." - + (" That path is inside a delegated-run worktree, which is removed when its run ends." if retired else "") - + " Re-promote it against" - + (f" the project folder ({folder})" if folder else " the project folder") - + " or with workspace='none' for a folder-less task." - ) - - -def _fail_promoted_task_loudly( - ctx: Any, task: dict, ws_error: str, *, - explicit_workspace: str = "", project_id: str = "", -) -> None: - """v6.58.0 loud-fail invariant: a room task whose workspace is SET-but-unusable - is terminally FAILED at admission with a visible card + chat message — never - silently admitted workspace-less (which would run the self_modification profile - over the system repo). Never raises. - - The remedy follows the SOURCE of the refused folder: a request that named its - own path is not fixed in Projects, and saying so is the difference between an - actionable message and one that points at a setting the failure never read.""" - tid = str(task.get("id") or "") - chat_id = 0 - try: - chat_id = int(task.get("chat_id") or 0) - except (TypeError, ValueError): - chat_id = 0 - explicit = str(explicit_workspace or "").strip() - remedy = ( - _explicit_workspace_remedy(explicit, str(project_id or "")) if explicit else - "Fix the project's working folder (Projects → this project) or re-promote with " - "workspace='none' for a folder-less task." - ) - message = f"⚠️ WORKSPACE_UNUSABLE: task {tid} was NOT started — {ws_error} {remedy}" - try: - from ouroboros.task_results import STATUS_FAILED, write_task_result - - write_task_result( - _pool().DRIVE_ROOT, tid, STATUS_FAILED, - reason_code="workspace_unusable", - result=message, - description=str(task.get("description") or ""), - chat_id=chat_id, - project_id=str(task.get("project_id") or ""), - ) - except Exception: - log.warning("promote loud-fail: task_result write failed for %s", tid, exc_info=True) - try: - if chat_id: - ctx.send_with_budget(chat_id, message) - except Exception: - log.debug("promote loud-fail: chat message failed for %s", tid, exc_info=True) - - def ensure_project_scope(evt: dict, ctx: Any) -> dict: """Create/attach the registry project for an in-task ensure_project_scope call and bind the CURRENT (already-running) task to it, then broadcast so the UI moves diff --git a/supervisor/workers.py b/supervisor/workers.py index 6081d70be..15e998aa6 100644 --- a/supervisor/workers.py +++ b/supervisor/workers.py @@ -2169,7 +2169,6 @@ from supervisor.worker_process import ( # noqa: E402, F401 -- intentional publi from supervisor.worker_promotion import ( # noqa: E402, F401 -- intentional public re-exports _admit_promoted_workspace, _canonical_promoted_repair_constraint, - _fail_promoted_task_loudly, _origin_from_mapping, _origin_from_task_record, _promote_duplicate_reason, diff --git a/tests/test_cancel_protocol_inventory_s6.py b/tests/test_cancel_protocol_inventory_s6.py index 878bade96..dc4ca418e 100644 --- a/tests/test_cancel_protocol_inventory_s6.py +++ b/tests/test_cancel_protocol_inventory_s6.py @@ -109,7 +109,6 @@ TERMINAL_WRITERS = { ('supervisor/worker_health.py::_recover_crashed_task_without_terminal', 'STATUS_CANCELLED'): 'terminal', ('supervisor/worker_health.py::_recover_crashed_task_without_terminal', 'STATUS_FAILED'): 'terminal', ('supervisor/worker_pool_lifecycle.py::_write_failure_result', 'final_status'): 'dynamic', - ('supervisor/worker_promotion.py::_fail_promoted_task_loudly', 'STATUS_FAILED'): 'terminal', ('supervisor/workers.py::_settle_cancelled_pending_row', 'status_cancelled'): 'dynamic', ('supervisor/workers.py::_terminalize_invalid_pending_depth', 'STATUS_FAILED'): 'terminal', } diff --git a/tests/test_chat_id_truthiness_guard.py b/tests/test_chat_id_truthiness_guard.py index b19c13a43..d0bb624c6 100644 --- a/tests/test_chat_id_truthiness_guard.py +++ b/tests/test_chat_id_truthiness_guard.py @@ -60,11 +60,6 @@ ALLOWED = { "chat_id means 'the event carried no chat' and the owner chat is the " "fallback address, not the hidden partition.", ), - ("supervisor/worker_promotion.py", "if chat_id:"): ( - 1, - "Same promote lane: the loud-fail notice needs a reader, and the hidden " - "partition has none.", - ), ("supervisor/worker_chat_lane.py", "if not chat_id:"): ( 1, "Auto-resume gate, where owner_chat_id 0 means 'no owner chat " diff --git a/tests/test_gateway_history.py b/tests/test_gateway_history.py index 7dfb908f8..ccc8a2d8f 100644 --- a/tests/test_gateway_history.py +++ b/tests/test_gateway_history.py @@ -1137,3 +1137,19 @@ def test_chat_history_replays_the_live_subtree_ceiling_for_a_running_root(tmp_pa # Only ROOT lineage reaches the ledger; the finished root is served by its # durable terminal truth, not by a live read. assert sorted(seen_roots) == ["root-empty", "root-live"] + + +def test_user_annotation_projects_the_host_cause_sentence(): + """Q3=A: the owner-facing sentence for a refused routing act replays with the + owner message; the reason code stays a model artefact off the live frame.""" + from ouroboros.gateway.history import _user_annotation + + projected = _user_annotation("user", "cm-1", {"cm-1": { + "action": "promote_chat_to_task", "status": "needs_manual_target", + "reason": "workspace_unusable", "cause": "Not started: the working folder can't be used", + "detail": "explicit workspace_root is unusable: …", + }}) + + assert projected["cause"] == "Not started: the working folder can't be used" + assert projected["status"] == "needs_manual_target" + assert "reason" not in projected diff --git a/tests/test_gateway_parity.py b/tests/test_gateway_parity.py index 881741862..2900ee53e 100644 --- a/tests/test_gateway_parity.py +++ b/tests/test_gateway_parity.py @@ -457,6 +457,7 @@ def test_gateway_contract_endpoint_index_matches_router_and_types(tmp_path): "project_id", "project_chat_id", "routing_token", + "cause", "status", "options", "attachment_manifest", diff --git a/tests/test_message_bus.py b/tests/test_message_bus.py index 56758bf1d..f9ca24847 100644 --- a/tests/test_message_bus.py +++ b/tests/test_message_bus.py @@ -698,3 +698,26 @@ def test_project_thread_stamp_survives_meta_and_covers_typing_and_echo(monkeypat echoes = [f for f in frames if f.get("role") == "user"] assert echoes[0]["project_thread"] is True assert "project_thread" not in echoes[1] + + +def test_send_routing_ack_carries_the_cause_only_when_present(monkeypatch): + """Q3=A: the host's owner-facing sentence rides the live routing_ack frame + and the outbound transport copy; a landed act carries no cause at all.""" + bridge = _make_bridge(monkeypatch) + frames, events = [], [] + bridge._broadcast_fn = frames.append + monkeypatch.setattr(message_bus, "publish_event", lambda topic, data: events.append((topic, data))) + + bridge.send_routing_ack( + 1, client_message_id="cm-1", action="promote_chat_to_task", target="t1", + status="needs_manual_target", routing_token="tok-1", + cause="Not started: the working folder can't be used", + ) + bridge.send_routing_ack( + 1, client_message_id="cm-2", action="promote_chat_to_task", target="t2", + status="scheduled", routing_token="tok-2", + ) + + assert frames[0]["cause"] == "Not started: the working folder can't be used" + assert events[0][1]["cause"] == frames[0]["cause"] + assert "cause" not in frames[1] and "cause" not in events[1][1] diff --git a/tests/test_presence_workspace.py b/tests/test_presence_workspace.py index a170e515a..451edd34f 100644 --- a/tests/test_presence_workspace.py +++ b/tests/test_presence_workspace.py @@ -247,3 +247,41 @@ def test_explicit_project_choice_still_wins_over_presence_folder(installed): assert resolve_project_id({**task, "project_id": "chosen-project"}) == "chosen-project" ordinary = {"workspace_root": str(installed.workspace)} assert resolve_project_id(ordinary).startswith("proj_") + + +def test_promotion_with_an_unusable_presence_folder_returns_the_typed_repair_and_sends_nothing( + installed, monkeypatch, +): + """The Presence branch of workspace admission: the admitted folder vanished + before the promote was admitted, so the typed refusal names the Presence + profile as the place to fix it (the cause rides `detail` to the model) and + admission sends no chat text — nothing fans out into a public conversation.""" + import shutil + + from ouroboros.tools.control import _promote_chat_to_task + from supervisor import workers + from tests.test_promote_chat_flow import _confirm_promote + + task = _build_task(_admit(installed), _event(), drive_root=installed.data, staged_files=()) + ctx = _context(installed, task) + _confirm_promote(monkeypatch) + monkeypatch.setattr(workers, "DRIVE_ROOT", installed.data) + monkeypatch.setattr(workers, "REPO_DIR", installed.repo) + result = _promote_chat_to_task(ctx, "Finish the report", predecessor_task_id="") + assert result.startswith("OK: task"), result + event = ctx.pending_events[0] + shutil.rmtree(installed.workspace) # the admitted folder disappears before admission + + sent, enqueued = [], [] + outcome = workers.promote_chat_to_task(event, SimpleNamespace( + enqueue_task=lambda row: enqueued.append(row) or row, + persist_queue_snapshot=lambda **_kwargs: True, + load_state=lambda: {"owner_chat_id": 1}, + send_with_budget=lambda *args, **kwargs: sent.append((args, kwargs)), + )) + + assert (outcome["status"], outcome["reason"]) == ("needs_manual_target", "workspace_unusable") + assert outcome["detail"].startswith("The folder configured in the Presence profile is unusable: ") + assert str(installed.workspace) in outcome["detail"] + assert outcome["detail"].endswith("Fix the Presence profile's workspace_root or clear it.") + assert sent == [] and enqueued == [] diff --git a/tests/test_promote_chat_flow.py b/tests/test_promote_chat_flow.py index b76719608..c36e1dab6 100644 --- a/tests/test_promote_chat_flow.py +++ b/tests/test_promote_chat_flow.py @@ -2796,11 +2796,13 @@ def test_decision_turn_metadata_injects_running_tasks_and_client_id(tmp_path): # --- Q10=A (owner, 2026-08-08): file-less project promotes auto-provision ----- -def _promote_ctx(enqueued): +def _promote_ctx(enqueued, sent=None): return types.SimpleNamespace( enqueue_task=lambda task: enqueued.append(task), persist_queue_snapshot=lambda **_kwargs: True, load_state=lambda: {"owner_chat_id": 1}, + # Records every chat send, so a refusal can prove it sent NOTHING. + send_with_budget=lambda *args, **kwargs: (sent if sent is not None else []).append((args, kwargs)), ) @@ -2887,17 +2889,23 @@ def test_promote_broken_working_dir_loud_fails_never_blind_ensures(tmp_path, mon gone = tmp_path / "gone-folder" update_project(tmp_path, "brokenp", working_dir=str(gone)) # never existed - enqueued = [] + enqueued, sent = [], [] outcome = workers.promote_chat_to_task({ "type": "promote_chat_to_task", "task_id": "broken1", "objective": "Continue", "project_id": "brokenp", "chat_id": 1, - }, _promote_ctx(enqueued)) + }, _promote_ctx(enqueued, sent)) assert outcome["status"] == "needs_manual_target" assert outcome["reason"] == "workspace_unusable" + # The repair rides `detail` to the MODEL (PROMOTE_REJECTED renders it); + # admission itself sends no chat text (Q1=A) and writes no task result — + # the handler's single rejection writer does. + assert outcome["detail"].startswith("project 'brokenp' working_dir is unusable") + assert "Projects → this project" in outcome["detail"] and "workspace='none'" in outcome["detail"] + assert sent == [] assert enqueued == [] # The broken value is preserved for the owner to fix — not overwritten. assert get_project(tmp_path, "brokenp")["working_dir"] == str(gone) @@ -2911,17 +2919,20 @@ def test_promote_provisioning_failure_loud_fails_not_silent_fileless(tmp_path, m monkeypatch.setattr(workers, "DRIVE_ROOT", tmp_path) monkeypatch.setattr(projects_registry, "ensure_project_workspace", lambda *a, **k: "") - enqueued = [] + enqueued, sent = [], [] outcome = workers.promote_chat_to_task({ "type": "promote_chat_to_task", "task_id": "provfail1", "objective": "Build", "project_id": "provfail-proj", "chat_id": 1, - }, _promote_ctx(enqueued)) + }, _promote_ctx(enqueued, sent)) assert outcome["status"] == "needs_manual_target" assert outcome["reason"] == "workspace_provisioning_failed" + assert outcome["detail"].startswith("project 'provfail-proj' has no working folder and auto-provisioning") + assert "Projects → this project" in outcome["detail"] and "workspace='none'" in outcome["detail"] + assert sent == [] assert enqueued == [] @@ -3048,78 +3059,96 @@ def test_the_steer_tool_renders_the_typed_reason_and_still_defaults_without_one( assert "(target_not_steerable)" in control_routing._steer_task(ctx, "t1", "go") -def _loud_workspace_failure(tmp_path, monkeypatch, ws_error: str, **kwargs): - """Run the loud-fail writer directly and return (chat message, stored row).""" - import supervisor.workers as workers - from ouroboros.task_results import load_task_result - from supervisor import worker_promotion +def _repair_hint(**kwargs): + """The MODEL-facing repair for one refused workspace: one line, never empty.""" + from ouroboros.workspace_admission import workspace_repair_hint - monkeypatch.setattr(workers, "DRIVE_ROOT", tmp_path) - sent: list = [] - ctx = types.SimpleNamespace(send_with_budget=lambda chat_id, text: sent.append(text)) - worker_promotion._fail_promoted_task_loudly( - ctx, {"id": "wsfail", "chat_id": 3}, ws_error, **kwargs, - ) - assert len(sent) == 1 - return sent[0], load_task_result(tmp_path, "wsfail") + hint = workspace_repair_hint(**kwargs) + assert hint and "\n" not in hint + return hint -def test_loud_workspace_failure_remedy_follows_the_source_of_the_refused_folder( - tmp_path, monkeypatch): +def test_workspace_repair_hint_follows_the_source_of_the_refused_folder(tmp_path): """The request named its own folder, so the project's working folder was never - read: sending the owner to Projects points at a setting the failure never + read: sending the caller to Projects points at a setting the failure never touched. The project folder is named as the way back when it is readable.""" from ouroboros.projects_registry import create_project create_project(tmp_path, "roomp", name="RoomP", working_dir=str(tmp_path / "room-tree")) - message, stored = _loud_workspace_failure( - tmp_path, monkeypatch, - "explicit workspace_root is unusable: not a git checkout.", + hint = _repair_hint( + ws_error="explicit workspace_root is unusable: not a git checkout.", explicit_workspace=str(tmp_path / "asked-for"), project_id="roomp", + drive_root=tmp_path, system_repo_dir=tmp_path / "system-repo", ) - assert "asked for" in message and str(tmp_path / "asked-for") in message - assert "Projects → this project" not in message - assert str(tmp_path / "room-tree") in message - assert "workspace='none'" in message - assert stored["status"] == "failed" and stored["reason_code"] == "workspace_unusable" + assert hint.startswith("explicit workspace_root is unusable: not a git checkout. ") + assert "asked for" in hint and str(tmp_path / "asked-for") in hint + assert "Projects → this project" not in hint + assert str(tmp_path / "room-tree") in hint + assert "workspace='none'" in hint -def test_loud_workspace_failure_keeps_the_projects_remedy_for_a_project_folder( - tmp_path, monkeypatch): - """A project working_dir failure — and an unreadable registry entry — is fixed - exactly where today's message says, so that text is unchanged.""" +def test_workspace_repair_hint_keeps_the_projects_remedy_for_a_project_folder(): + """A project working_dir failure, an unreadable registry entry and a failed + auto-provision are fixed exactly where the sentence says: in Projects.""" for ws_error in ( "project 'roomp' working_dir is unusable: not a git checkout.", "project 'roomp' registry entry is unreadable (OSError: boom) — cannot determine " "the task's workspace", + "project 'roomp' has no working folder and auto-provisioning one failed; " + "see the supervisor log (ensure_project_workspace)", ): - message, stored = _loud_workspace_failure(tmp_path, monkeypatch, ws_error) - assert ws_error in message - assert message.endswith( + hint = _repair_hint(ws_error=ws_error) + assert hint.startswith(ws_error.rstrip(".")) + assert hint.endswith( "Fix the project's working folder (Projects → this project) or re-promote with " "workspace='none' for a folder-less task." ) - assert "asked for" not in message - assert stored["reason_code"] == "workspace_unusable" + assert "asked for" not in hint -def test_loud_workspace_failure_names_a_retired_delegated_run_worktree(tmp_path, monkeypatch): +def test_workspace_repair_hint_names_a_retired_delegated_run_worktree(tmp_path, monkeypatch): """The path the agent passed twice in one minute was a delegated-run worktree its own run had already retired; nothing in the old message said so.""" from ouroboros import config worktrees = tmp_path / "subagent_worktrees" monkeypatch.setattr(config, "get_subagent_worktree_root", lambda: str(worktrees)) - message, stored = _loud_workspace_failure( - tmp_path, monkeypatch, - "explicit workspace_root is unusable: path does not exist.", + hint = _repair_hint( + ws_error="explicit workspace_root is unusable: path does not exist.", explicit_workspace=str(worktrees / "dlg_060055d5_x"), project_id="", ) - assert "delegated-run worktree" in message and "removed when its run ends" in message - assert "Projects → this project" not in message - assert stored["reason_code"] == "workspace_unusable" + assert "delegated-run worktree" in hint and "removed when its run ends" in hint + assert "Projects → this project" not in hint and "workspace='none'" in hint + + +def test_workspace_repair_hint_points_a_repo_subfolder_at_the_empty_default(tmp_path): + """A SUBFOLDER of the Ouroboros repository can never be a workspace (the exact + root never reaches admission: the promote tool maps it onto no workspace); + the repair names the argument shape that works instead.""" + repo = tmp_path / "Ouroboros" / "repo" + (repo / "ouroboros").mkdir(parents=True) + hint = _repair_hint( + ws_error="explicit workspace_root is unusable: workspace_root must not overlap the " + "Ouroboros system repo", + explicit_workspace=str(repo / "ouroboros"), system_repo_dir=repo, + ) + + assert "workspace_root must be a folder outside the Ouroboros repository" in hint + assert "empty (or workspace='none') to work in the repository itself" in hint + assert "asked for" not in hint + + +def test_workspace_repair_hint_names_the_presence_profile(): + hint = _repair_hint( + ws_error="explicit workspace_root is unusable: workspace_root is not a directory: /gone", + presence=True, + ) + + assert hint.startswith("The folder configured in the Presence profile is unusable: ") + assert "workspace_root is not a directory: /gone" in hint + assert hint.endswith("Fix the Presence profile's workspace_root or clear it.") def test_promote_without_an_explicit_target_inherits_the_binding_then_the_origin( @@ -3306,3 +3335,186 @@ def test_the_implicit_promote_claim_creates_and_binds_under_the_claim_lock( assert enqueued[0]["project_id"] == "the-work" assert (registry.project_binding_for_task(tmp_path, "root01") or {}).get( "project_id") == "the-work" + + +# --- promote refusals: placement (Q1/Q2=A), the host cause (Q3=A), the repo root (Q4=A) --- + +def _host_promote(host, monkeypatch, **overrides): + """Drive the REAL promote handler (reservation, admission, receipts, the + publication boundary) for a Main event; ``host.notices`` records every chat + send with its kwargs and ``host.bridge.acks`` every live receipt.""" + from supervisor.events import _handle_promote_chat_to_task + + evt = { + "type": "promote_chat_to_task", "task_id": "refused0001", "routing_token": "tok-refused-1", + "objective": "Audit the GitHub tool", "chat_id": 1, "client_message_id": "cm-refused-1", + "workspace": "none", + } + evt.update(overrides) + return evt, _handle_promote_chat_to_task(evt, host.ctx) + + +def test_host_initiated_workspace_refusal_sends_one_typed_row_to_the_owner_chat(swarm_host, monkeypatch): + """Q2=A: no model turn narrates a host-issued promote, so the owner is told + ONCE — a typed System row in the chat the owner wrote in (never the project + room admission just created), bound to the never-started task — and the + receipt under the message carries the same host sentence (Q3=A).""" + from tests.test_swarm_host_admission import rows + + host = swarm_host + announced = [] + monkeypatch.setattr( + "supervisor.terminal_delivery.enqueue_terminal_delivery", + lambda _root, event, **_k: announced.append(dict(event)) or True, + ) + evt, outcome = _host_promote( + host, monkeypatch, host_initiated=True, title="Аудит GitHub-инструмента", + project_id="refused-room", project_name="Refused Room", + workspace="", workspace_root=str(host.root / "missing-folder"), + ) + + assert (outcome["status"], outcome["reason"]) == ("needs_manual_target", "workspace_unusable") + assert outcome["detail"].startswith("explicit workspace_root is unusable") + assert host.pending == [] + [notice] = host.notices + assert notice["chat_id"] == 1 # the OWNER's chat, not the new project room + assert notice["text"] == "Аудит GitHub-инструмента · Not started: the working folder can't be used" + assert notice["role"] == "system" and notice["system_type"] == "task_not_started" + assert notice["task_id"] == evt["task_id"] + # R9: the project was created, but its "Started" row is owed only to real work. + assert announced == [] + row = rows(host.root / "logs/chat_annotations.jsonl")[-1] + assert (row["status"], row["reason"]) == ("needs_manual_target", "workspace_unusable") + assert row["cause"] == "Not started: the working folder can't be used" + assert host.bridge.acks[-1]["status"] == "needs_manual_target" + assert host.bridge.acks[-1]["cause"] == "Not started: the working folder can't be used" + + +def test_tool_issued_workspace_refusal_sends_nothing(swarm_host, monkeypatch): + """Q1=A: a model turn waits on the receipt and narrates; the receipt (with + its cause) and the failed call are the record — no host chat text at all.""" + host = swarm_host + evt, outcome = _host_promote( + host, monkeypatch, workspace="", workspace_root=str(host.root / "missing-folder"), + ) + + assert (outcome["status"], outcome["reason"]) == ("needs_manual_target", "workspace_unusable") + assert host.notices == [] + assert host.bridge.acks[-1]["cause"] == "Not started: the working folder can't be used" + + +def test_host_initiated_reservation_refusal_sends_one_row_and_a_replay_sends_nothing( + swarm_host, monkeypatch, +): + """R13: the reservation-blocked exit (a disabled worker pool) passes the same + publication boundary; a replay of the settled admission (same token) is + silent — the owner was told once. The picker's route (`routed_from_main`) + wears the same «Not started» prefix as a promote (R14).""" + import supervisor.workers as workers + from supervisor.events import _handle_promote_chat_to_task + + host = swarm_host + monkeypatch.setattr( + workers, "_worker_pool_execution_state", + lambda *_a, **_k: {"available": False, "disabled_reason": "maintenance"}, + ) + evt, outcome = _host_promote(host, monkeypatch, host_initiated=True, routed_from_main=True) + + assert (outcome["status"], outcome["reason"]) == ("needs_manual_target", "worker_pool_unavailable") + [notice] = host.notices + assert notice["text"] == "Audit the GitHub tool · Not started: no worker is available right now" + assert notice["system_type"] == "task_not_started" and notice["task_id"] == evt["task_id"] + + replay = _handle_promote_chat_to_task(dict(evt), host.ctx) + assert replay["replayed"] is True and replay["reason"] == "worker_pool_unavailable" + assert len(host.notices) == 1 + + +def test_host_initiated_unconfirmed_admission_sends_the_unconfirmed_row(swarm_host, monkeypatch): + """R14: an admission whose receipt could not be persisted is UNCONFIRMED — the + row says exactly that (`task_start_unconfirmed`, «Not confirmed»), never + «Not started».""" + host = swarm_host + monkeypatch.setattr("ouroboros.project_dialogue.append_chat_annotation", lambda *_a, **_k: False) + evt, outcome = _host_promote(host, monkeypatch, host_initiated=True) + + assert (outcome["status"], outcome["reason"]) == ("unconfirmed", "routing_annotation_persist_failed") + [notice] = host.notices + assert notice["text"] == "Audit the GitHub tool · Not confirmed: the receipt could not be saved" + assert notice["role"] == "system" and notice["system_type"] == "task_start_unconfirmed" + assert notice["task_id"] == evt["task_id"] + + +def test_promote_maps_the_exact_repo_root_onto_no_workspace(tmp_path, monkeypatch): + """Q4=A: naming the Ouroboros repository ITSELF names the documented default + (the ordinary self-modification task): the event carries the existing + "none" sentinel — in a project room too — and the tool result discloses it.""" + from ouroboros.tools.control import _promote_chat_to_task + + _confirm_promote(monkeypatch) + repo = tmp_path / "Ouroboros" / "repo" + repo.mkdir(parents=True) + events = [] + ctx = types.SimpleNamespace( + pending_events=events, event_queue=None, current_chat_id=1, drive_root=tmp_path / "data", + repo_dir=repo, system_repo_dir=repo, + ) + out = _promote_chat_to_task( + ctx, "Audit the GitHub tool", project_id="racer", workspace_root=f"{repo}/", + predecessor_task_id="", + ) + + assert out.startswith("OK: task") + assert ( + "(workspace_root named the Ouroboros repository itself; started as an ordinary task " + "over it — no separate workspace)" + ) in out + [evt] = events + assert evt["workspace_root"] == "" and evt["workspace"] == "none" + assert evt["project_id"] == "racer" + + +def test_promote_passes_a_repo_subfolder_through_to_admission_unchanged(tmp_path, monkeypatch): + """A SUBFOLDER is a different ask the shape cannot serve: it reaches admission + unchanged and is refused there with the repair hint.""" + from ouroboros.tools.control import _promote_chat_to_task + + _confirm_promote(monkeypatch) + repo = tmp_path / "Ouroboros" / "repo" + sub = repo / "ouroboros" + sub.mkdir(parents=True) + events = [] + ctx = types.SimpleNamespace( + pending_events=events, event_queue=None, current_chat_id=1, drive_root=tmp_path / "data", + repo_dir=repo, system_repo_dir=repo, + ) + out = _promote_chat_to_task(ctx, "Audit the GitHub tool", workspace_root=str(sub), predecessor_task_id="") + + assert out.startswith("OK: task") and "named the Ouroboros repository itself" not in out + [evt] = events + assert evt["workspace_root"] == str(sub) and evt["workspace"] == "" + + +def test_project_started_is_not_announced_when_workspace_admission_refuses(tmp_path, monkeypatch): + """R9: the «Project · Started» row is owed only once the task is REALLY in the + queue — a promote that creates a project and then fails workspace admission + announces nothing (the project itself stays; the refusal names it).""" + import supervisor.workers as workers + from ouroboros.projects_registry import get_project + + monkeypatch.setattr(workers, "DRIVE_ROOT", tmp_path) + queued = [] + monkeypatch.setattr( + "supervisor.terminal_delivery.enqueue_terminal_delivery", + lambda _root, event, **_k: queued.append(dict(event)) or True, + ) + enqueued, sent = [], [] + outcome = workers.promote_chat_to_task({ + "type": "promote_chat_to_task", "task_id": "doomed0001", "objective": "Build", + "project_id": "doomed-room", "project_name": "Doomed Room", + "workspace_root": str(tmp_path / "never-existed"), "chat_id": 1, + }, _promote_ctx(enqueued, sent)) + + assert (outcome["status"], outcome["reason"]) == ("needs_manual_target", "workspace_unusable") + assert get_project(tmp_path, "doomed-room")["name"] == "Doomed Room" # the project stays + assert queued == [] and enqueued == [] and sent == [] diff --git a/tests/test_routing_decision.py b/tests/test_routing_decision.py index 0e7e6f057..91433935a 100644 --- a/tests/test_routing_decision.py +++ b/tests/test_routing_decision.py @@ -150,6 +150,9 @@ def test_promote_click_confirms_from_the_admission_record(tmp_path, monkeypatch) def _supervisor_schedules(evt): assert evt["task_id"] == derived_task_id + # The owner's click issued this promote: the handler's publication + # boundary owns any refusal notice (no model turn narrates it). + assert evt["host_initiated"] is True and evt["routed_from_main"] is True def _mut(current): from ouroboros.contracts.schema_versions import SCHEMA_VERSION_KEY @@ -327,6 +330,9 @@ def test_rejected_dispatch_reopens_the_original_card(tmp_path, monkeypatch): status, body = handle_routing_decision( tmp_path, request_id="r1", decision_id="routing:cm-1:tok-1", option_index=0) assert (status, body["state"]) == (409, "open") + # R5/R16: the toast shows the host's sentence for the refused act, not the code. + assert body["reason"] == "target_closed" + assert body["cause"] == "Not delivered: that task has already finished" reopened = chat_annotation_receipt(tmp_path, "cm-1", "tok-1") assert reopened["status"] == "needs_manual_target" assert [row["action"] for row in reopened["options"]] == [ diff --git a/tests/test_routing_receipts_wave1.py b/tests/test_routing_receipts_wave1.py index 1330c7c71..9d25ce431 100644 --- a/tests/test_routing_receipts_wave1.py +++ b/tests/test_routing_receipts_wave1.py @@ -514,3 +514,40 @@ def test_two_roots_under_one_owner_message_keep_two_readable_receipts_after_comp assert latest_chat_annotations(tmp_path)["cm-1"]["routing_token"] == "tok-steer-2" assert "msg-expired" not in latest_chat_annotations(tmp_path) + + +# --- (e) a relayed owner steer refused by a pending cancel: receipt, no bubble --- + +def test_a_relayed_owner_steer_refused_by_a_pending_cancel_gets_no_standalone_bubble( + tmp_path, monkeypatch, +): + """R15: an act that wears an owner message shows its cause on that message's + receipt line; only an UNLABELLED owner act (a synthetic receipt id, so no + chat row can show the refusal) still gets the standalone cancel-pending + notice.""" + import ouroboros.cancel_intents as cancel_intents + from ouroboros.project_dialogue import AGENT_RECEIPT_ID_PREFIX, latest_chat_annotations + from supervisor.events import _handle_steer_task + + monkeypatch.setattr( + cancel_intents, "cancel_pending", lambda _root, task_id, **_k: task_id == "t-target", + ) + notices = [] + supervisor = _supervisor_ctx( + tmp_path, notices, running={"t-target": {"task": {"id": "t-target", "chat_id": 1}}}, + ) + base = {"type": "steer_task", "target_task_id": "t-target", "message": "stop after this file", + "chat_id": 1} + + _handle_steer_task({**base, "client_message_id": "cm-owner-1", "routing_token": "tok-1"}, supervisor) + + row = latest_chat_annotations(tmp_path)["cm-owner-1"] + assert (row["status"], row["reason"]) == ("rejected", "cancel_pending") + assert row["cause"] == "Not delivered: that task is being stopped" + assert notices == [] + + unlabelled = f"{AGENT_RECEIPT_ID_PREFIX}tok-2" + _handle_steer_task({**base, "client_message_id": unlabelled, "routing_token": "tok-2"}, supervisor) + + assert latest_chat_annotations(tmp_path)[unlabelled]["cause"] == "Not delivered: that task is being stopped" + assert len(notices) == 1 diff --git a/tests/test_routing_refusal_causes.py b/tests/test_routing_refusal_causes.py new file mode 100644 index 000000000..38b72b83c --- /dev/null +++ b/tests/test_routing_refusal_causes.py @@ -0,0 +1,177 @@ +"""Host-owned cause sentences for refused routing acts (Q3=A; R6/R7/R11/R12/R14). + +The receipt line under the owner's message, the host-initiated System row and +the picker's 409 toast all read ONE host table; the browser renders the +sentence verbatim. These pins keep the table complete against every producer +of a refusal reason, keep the phrases factual and short, keep the prefix an +honest statement of the act's outcome, and keep the admission-notice rows in +the chat the owner wrote in. +""" + +from __future__ import annotations + +import pathlib +import re + +import pytest + +from ouroboros.project_dialogue import ( + ADMISSION_NOTICE_TYPES, + ROUTING_REFUSAL_CAUSES, + room_membership, + routing_refusal_cause, +) + +REPO = pathlib.Path(__file__).resolve().parents[1] + +# Every typed refusal reason that reaches `_emit_routing_receipt`, the promote +# outcome (the host-initiated notice) or the steer receipt — one row each. +REFUSAL_REASONS = ( + # promote / route admission (worker_promotion, events_project_routing, + # queue.enqueue_task `_admission_blocked`, task_admission.reserve_task_admission) + "workspace_unusable", "workspace_provisioning_failed", "worker_pool_unavailable", + "worker_pool_state_unavailable", "duplicate_task_id", "admission_reservation_owned", + "admission_reservation_lost", "admission_reservation_failed", "admission_fence", + "admission_rejected", "invalid_admission_reservation", "task_id_lookup_failed", + "empty_objective", "project_routing_fence", "project_routing_fence_lookup_failed", + "project_binding_failed", "project_registration_failed", "project_source_error", + "attachment_admission_rejected", "staging_unavailable", + "queue_snapshot_persist_unavailable", "queue_snapshot_persist_failed", + "invalid_skill_repair_constraint", "skill_repair_payload_missing", + "skill_repair_payload_unreadable", "skill_repair_admission_unwritable", + "repair_promotion_failed", "task_acceptance_fence", "invalid_task_depth", + # the unconfirmed family (handler crash, receipt persistence, routing_wait) + "promotion_persistence_failed", "routing_receipt_persist_failed", + "routing_annotation_persist_failed", "source_continuation_publish_failed", + "confirmation_timeout", + # steer (supervisor/steering.py) + "target_unknown", "direct_chat_turn", "subagent_target", "chat_mismatch", + "cancel_pending", "target_closed", "target_finished", "acceptance_fence_sealed", + "mailbox_write_failed", + # ensure_project_scope (worker_promotion, events_project_routing) + "project_scope_conflict", "missing_task_or_project", "ensure_project_scope_failed", +) + +PRODUCERS = ( + "supervisor/worker_promotion.py", + "supervisor/events_project_routing.py", + "supervisor/steering.py", + "supervisor/queue.py", + "supervisor/task_admission.py", + "ouroboros/server_owner_routing.py", +) +_REASON_LITERAL = re.compile(r'(?:"reason":\s*|\breason=)"([a-z][a-z0-9_]*)"') + +# Literals the producer scan finds that are NOT owner-facing refusal reasons. +EXEMPT = { + # snapshot-persist / rollback reasons (a persist call's audit label, never a receipt) + "promote_chat_to_task_rejected", "promote_chat_to_task_failed", "promote_chat_to_task", + "drain_all_pending", "acceptance_fence_owner_message", "evolve_off", "deep_self_review_enqueued", + # the bind-failure events.jsonl row's reason (project_binding_unreadable), never a receipt + "project_binding_unreadable", + # success / pseudo reasons on a delivered ensure_project_scope receipt + "created", "adopted", "attached", "delivered", "renamed", "name_unchanged", "rename_failed", + # the picker's claim/closing pseudo-reasons (`claimed_option:N` / `answered_option:N`) + # and its with-options refusal, which keeps «Choose a target» (no cause) + "answered_option", "claimed_option", "target_unspecified", + # internal wait/transport outcomes that never reach an owner surface + "client_message_id_missing", "event_serialization_failed", "handler_returned_no_outcome", +} + + +def test_every_refusal_reason_has_a_short_factual_phrase(): + assert set(REFUSAL_REASONS) == set(ROUTING_REFUSAL_CAUSES) + for reason, phrase in ROUTING_REFUSAL_CAUSES.items(): + assert phrase == phrase.strip() and len(phrase) <= 60, (reason, phrase) + assert "_" not in phrase and not phrase.endswith("."), (reason, phrase) + assert phrase[0].islower() and not phrase.startswith("Not "), (reason, phrase) + + +@pytest.mark.parametrize( + "action, status, expected", + [ + ("promote_chat_to_task", "needs_manual_target", "Not started"), + ("route_to_project", "needs_manual_target", "Not started"), + ("promote_chat_to_task", "rejected", "Not started"), + ("steer_task", "needs_manual_target", "Not delivered"), + ("steer_task", "rejected", "Not delivered"), + ("ensure_project_scope", "rejected", "Not moved"), + ("promote_chat_to_task", "unconfirmed", "Not confirmed"), + ("steer_task", "unconfirmed", "Not confirmed"), + ("ensure_project_scope", "unconfirmed", "Not confirmed"), + ], +) +def test_prefix_follows_the_act_and_its_outcome(action, status, expected): + """The prefix states only what the outcome proves: an UNCONFIRMED act reads + «Not confirmed» whatever it was; a refused steer «Not delivered»; a refused + scope bind «Not moved»; every other refused act «Not started».""" + sentence = routing_refusal_cause(action, status, "project_binding_failed") + assert sentence == f"{expected}: the project could not be set up" + + +def test_landed_rows_and_the_picker_carry_no_cause(): + for status in ("scheduled", "delivered", "pending", "dispatch_pending", "accepted"): + assert routing_refusal_cause("promote_chat_to_task", status, "workspace_unusable") == "" + # A refusal WITH options is the picker: «Choose a target · A / B» stays. + assert routing_refusal_cause("route_decision", "needs_manual_target", "target_unspecified", + [{"action": "steer_task", "task_id": "t1"}]) == "" + assert routing_refusal_cause("route_decision", "needs_manual_target", "target_unspecified", []) == ( + "Not started (target_unspecified)" + ) + + +def test_unknown_reasons_stay_raw_and_unconfirmed_stays_honest(): + """DESIGN sanctions raw over invented: a reason with no row is shown as its + code (the raw fallback is exempt from the no-underscore rule), while an + unconfirmed act with an unknown reason reads as the honest uncertainty.""" + assert routing_refusal_cause("promote_chat_to_task", "needs_manual_target", "x_y") == "Not started (x_y)" + assert routing_refusal_cause("steer_task", "rejected", "x_y") == "Not delivered (x_y)" + assert routing_refusal_cause("promote_chat_to_task", "unconfirmed", "x_y") == ( + "Not confirmed: the task may or may not have started" + ) + assert routing_refusal_cause("promote_chat_to_task", "needs_manual_target", "") == "Not started" + assert routing_refusal_cause("promote_chat_to_task", "needs_manual_target", "workspace_unusable") == ( + "Not started: the working folder can't be used" + ) + + +def test_producer_literals_are_all_covered_or_exempt(): + """Completeness against the code: every `"reason": "<literal>"` / + `reason="<literal>"` a producer writes has a sentence or an explicit exemption.""" + found = {} + for relative in PRODUCERS: + source = (REPO / relative).read_text(encoding="utf-8") + for match in _REASON_LITERAL.finditer(source): + found.setdefault(match.group(1), set()).add(relative) + assert found, "the producer scan found nothing — the regex or the paths drifted" + missing = { + literal: sorted(files) for literal, files in found.items() + if literal not in ROUTING_REFUSAL_CAUSES and literal not in EXEMPT + } + assert missing == {}, f"refusal reasons without a host sentence: {missing}" + assert not (set(EXEMPT) & set(ROUTING_REFUSAL_CAUSES)), "a reason cannot be both exempt and worded" + + +def test_admission_notice_types_are_the_two_host_rows(): + """R3/R14: a refusal is `task_not_started`; an admission whose receipt could + not be confirmed is `task_start_unconfirmed` — never «not started».""" + assert ADMISSION_NOTICE_TYPES == frozenset({"task_not_started", "task_start_unconfirmed"}) + + +@pytest.mark.parametrize("notice_type", sorted(ADMISSION_NOTICE_TYPES)) +def test_admission_notice_rows_stay_in_the_owners_chat(notice_type): + """R12: the refused task id is bound to the project it never started in, so + binding lineage would move the notice out of Main into that project on + reload. The notice is addressed to the chat the OWNER wrote in.""" + bindings = {"refused-task": 77} + main = room_membership(1, {77}, [], bindings) + project = room_membership(77, {77}, [], bindings) + notice = {"direction": "system", "type": notice_type, "task_id": "refused-task", "chat_id": 1} + + assert main(1, notice) is True + assert project(1, notice) is False + assert project(77, {**notice, "chat_id": 77}) is True + # An ordinary bound row keeps today's behaviour: lineage moves it to its room. + ordinary = {"direction": "out", "type": "chat", "task_id": "refused-task", "chat_id": 1} + assert main(1, ordinary) is False + assert project(1, ordinary) is True diff --git a/tests/test_task_constraint_server_routing.py b/tests/test_task_constraint_server_routing.py index a0e6c2b15..2057c35fd 100644 --- a/tests/test_task_constraint_server_routing.py +++ b/tests/test_task_constraint_server_routing.py @@ -69,44 +69,71 @@ def test_constrained_repair_promotes_managed_task_before_busy_direct_lane(monkey "payload_root": "skills/external/alpha", } assert event["origin_suppressed"] is True + # The skill card issued this promote: the handler's publication boundary + # owns any refusal notice, so the event says so. + assert event["host_initiated"] is True assert len(calls["sent"]) == 1 assert calls["sent"][0][0] == 1 assert "accepted and durably scheduled" in calls["sent"][0][1] -def test_constrained_repair_refusal_is_reported_to_owner(monkeypatch): - sent = [] +_REPAIR_INCOMING = { + "chat_id": 1, + "text": "repair skill", + "client_message_id": "repair-1", + "task_constraint": { + "mode": "skill_repair", + "skill_name": "alpha", + "payload_root": "skills/external/alpha", + }, +} + + +def test_constrained_repair_refusal_sends_no_untyped_bubble(monkeypatch): + """The handler's own publication boundary tells the owner (one typed System + row); the routing lane adds no «⚠️ Repair task was not started» bubble.""" + sent, events = [], [] ctx = SimpleNamespace( consciousness=SimpleNamespace(inject_observation=lambda *_: None), - send_with_budget=lambda chat_id, text: sent.append((chat_id, text)), + send_with_budget=lambda chat_id, text, **kwargs: sent.append((chat_id, text, kwargs)), ) monkeypatch.setattr( "supervisor.events._handle_promote_chat_to_task", - lambda event, _ctx: { + lambda event, _ctx: events.append(event) or { "status": "needs_manual_target", "reason": "skill_repair_payload_missing", "task_id": event["task_id"], }, ) - server._route_owner_message( - FakeBridge(), - ctx, - { - "chat_id": 1, - "text": "repair skill", - "client_message_id": "repair-1", - "task_constraint": { - "mode": "skill_repair", - "skill_name": "alpha", - "payload_root": "skills/external/alpha", - }, - }, + server._route_owner_message(FakeBridge(), ctx, dict(_REPAIR_INCOMING)) + + assert sent == [] + assert events[0]["host_initiated"] is True + + +def test_constrained_repair_outer_failure_sends_one_typed_not_started_row(monkeypatch): + """R13: `repair_promotion_failed` is minted OUTSIDE the handler, so the lane + calls the same notice helper — one typed row in the owner's chat, bound to + the task id that never started, with the host's sentence.""" + sent = [] + ctx = SimpleNamespace( + consciousness=SimpleNamespace(inject_observation=lambda *_: None), + send_with_budget=lambda chat_id, text, **kwargs: sent.append((chat_id, text, kwargs)), ) - assert len(sent) == 1 - assert sent[0][0] == 1 - assert "skill_repair_payload_missing" in sent[0][1] + def _crash(event, _ctx): + raise RuntimeError("handler crashed") + + monkeypatch.setattr("supervisor.events._handle_promote_chat_to_task", _crash) + + server._route_owner_message(FakeBridge(), ctx, dict(_REPAIR_INCOMING)) + + [(chat_id, text, kwargs)] = sent + assert chat_id == 1 + assert text == "repair skill · Not started: the skill repair request could not be started" + assert kwargs["role"] == "system" and kwargs["system_type"] == "task_not_started" + assert kwargs["task_id"] def test_repair_ui_copy_does_not_promise_a_removed_decision_round(): diff --git a/web/modules/api_types.js b/web/modules/api_types.js index 82cf011c1..0de4d9874 100644 --- a/web/modules/api_types.js +++ b/web/modules/api_types.js @@ -551,6 +551,7 @@ * @property {string=} latest_status * @property {string=} reason * @property {string=} detail + * @property {string=} cause // the owner-facing sentence for a refused routing act (409 dispatch_rejected) */ /** @@ -608,6 +609,7 @@ * @property {Array<Object>=} options * @property {AttachmentManifestEntry[]=} attachment_manifest * @property {string=} routing_token + * @property {string=} cause // host-authored owner sentence for a REFUSED act; absent on scheduled/delivered/pending and on the picker frame * @property {boolean} suppress_bubble * @property {string=} ts */ diff --git a/web/modules/chat_activity.js b/web/modules/chat_activity.js index b19b8b476..94a5f3dfb 100644 --- a/web/modules/chat_activity.js +++ b/web/modules/chat_activity.js @@ -839,6 +839,11 @@ export function routingOptionLabel(option) { /** Human text for a typed routing annotation ('' hides the line). */ export function routingAnnotationText(annotation) { if (!annotation || typeof annotation !== 'object') return ''; + // A refused act carries the host's own owner-facing sentence (`cause`); + // it outranks the status matrix below. Absent on scheduled/delivered/ + // pending rows and on the picker frame, so those labels are unchanged. + const cause = String(annotation.cause || '').trim(); + if (cause) return cause; const action = String(annotation.action || ''); const status = String(annotation.status || ''); const target = String(annotation.target || ''); diff --git a/web/modules/chat_decision.js b/web/modules/chat_decision.js index 5b058dd01..e12a1a60b 100644 --- a/web/modules/chat_decision.js +++ b/web/modules/chat_decision.js @@ -586,7 +586,7 @@ export function createChatDecision({ body.state === 'open' ? 'open' : body.state, Number.isInteger(body.answered_index) ? body.answered_index : null); showToast(body.state === 'open' - ? `Not routed: ${body.reason || 'the destination refused this message'} — pick again.` + ? `Not routed: ${body.cause || body.reason || 'the destination refused this message'} — pick again.` : body.state === 'pending' ? 'Another choice is already being routed.' : 'This message was already routed.', 'error'); diff --git a/web/tests/chat_activity_block.test.js b/web/tests/chat_activity_block.test.js index 78e26c860..bcc75e4ad 100644 --- a/web/tests/chat_activity_block.test.js +++ b/web/tests/chat_activity_block.test.js @@ -404,3 +404,56 @@ test('Stop stays reachable on a census-vouched root, managed card and direct blo } finally { f.close(); } } }); + +// Promote-refusal placement (Q1=A): a refused addressing call inside a live +// turn is a FAILED call — an error row, content the block stands on — while +// the owner message carries the host's sentence (`cause`) as its receipt. +// Live and on reload the block holds exactly one error row. +test('a refused promote keeps a block with exactly one error row live and on reload; the owner message carries the cause', async () => { + const cause = 'Not started: the working folder can\'t be used'; + const refused = { annotation_type: 'routing_ack', client_message_id: 'owner-1', action: 'promote_chat_to_task', + status: 'needs_manual_target', target: 'never-started', target_label: 'Requested work', cause }; + const f = fixture(); + try { + f.census(direct()); + f.emit('chat', ownerRow); + f.log({ type: 'tool_call_started', ...promote({ args: { objective: 'the project' } }) }); + assert.equal(f.card(), null, 'still a receipt while the call runs'); + f.emit('message_annotation', refused); + f.log({ type: 'tool_call_finished', ...promote({ is_error: true, error: 'workspace_unusable', duration_sec: 0.4 }) }); + assert.ok(f.card(), 'the failed call is content'); + assert.equal(f.rows().length, 1, 'the failure lands on the call\'s own row'); + assert.match(f.rows()[0].innerHTML, /promote_chat_to_task/); + // The stub cannot repaint an in-place patch; the producer states the failed row. + const started = summarizeChatLiveEvent({ type: 'tool_call_started', task_id: TASK, ...promote({ args: { objective: 'the project' } }) }); + const failure = summarizeChatLiveEvent({ type: 'tool_call_finished', task_id: TASK, ...promote({ is_error: true, error: 'workspace_unusable', duration_sec: 0.4 }) }); + assert.equal(failure.dedupeKey, started.dedupeKey); + assert.deepEqual([started.receipt, failure.phase, failure.visible, Boolean(failure.receipt)], [true, 'error', true, false]); + const owner = f.messages.children.find((n) => n.dataset.clientMessageId === 'owner-1'); + assert.equal(owner?.querySelector('.msg-routing-annotation')?.textContent, cause); + f.emit('chat', { ...final, tool_calls: 1, tool_errors: 1 }); + f.log({ ...final, type: 'task_done', status: 'completed', tool_calls: 1, tool_errors: 1, _is_direct_chat: true }); + f.log({ type: 'task_metrics_event', tool_calls: 1, tool_errors: 1, routing_tool_calls: 1, tool_call_counts: { promote_chat_to_task: 1 } }); + assert.ok(f.card()); + assert.equal(f.rows().length, 2, 'the call\'s own error row and the terminal note, no summary row'); + assert.equal(f.rows().filter((n) => /promote_chat_to_task/.test(n.innerHTML)).length, 1); + assert.match(f.meta(), /1 error/); + } finally { f.close(); } + const g = fixture([ + { ...ownerRow, chat_annotation: refused }, + { ...final, tool_calls: 1, ts: '2026-09-15T12:00:05Z', chat_id: 1, _is_direct_chat: true }, + { ...final, role: 'system', system_type: 'task_summary', text: 'The working folder could not be used.', rounds: 2, + tool_calls: 1, tool_errors: 1, routing_tool_calls: 1, tool_call_counts: { promote_chat_to_task: 1 }, + addressing_only: 'promote_chat_to_task', _is_direct_chat: true, ts: '2026-09-15T12:00:06Z', chat_id: 1 }, + ]); + try { + await g.instance.refreshHistory({ revision: 1 }); + assert.ok(g.card(), 'a counted error is content, so the replay summary is not a receipt'); + const errorRows = g.rows().filter((n) => /1 tool call · 1 error/.test(n.innerHTML)); + assert.equal(errorRows.length, 1); + assert.ok(errorRows[0].classList.contains('warn')); + assert.match(g.meta(), /1 error/); + const owner = g.messages.children.find((n) => n.dataset.clientMessageId === 'owner-1'); + assert.equal(owner?.querySelector('.msg-routing-annotation')?.textContent, cause); + } finally { g.close(); } +}); diff --git a/web/tests/chat_decision.test.js b/web/tests/chat_decision.test.js index 2fb50809b..26012c89d 100644 --- a/web/tests/chat_decision.test.js +++ b/web/tests/chat_decision.test.js @@ -641,6 +641,48 @@ test('an actionable refusal renders the picker card; other statuses fall back to } finally { fx.restore(); } }); +test('a refusal with a token but no options renders the host cause as the plain line, never a card', () => { + // The incident shape (client_message_id + routing_token, NO options): the + // host sends `cause`, the owner reads a sentence, and nothing is clickable. + const cause = 'Not started: the working folder can\'t be used'; + const fx = fixture(); + try { + const bubble = routingBubble('cm-refused'); + const refused = { + action: 'promote_chat_to_task', status: 'needs_manual_target', routing_token: 'tok-r', + target: 'never-started', target_label: 'Аудит', cause, + }; + assert.equal(fx.decision.renderRoutingDecision(bubble, refused), true); + assert.equal(bubble.querySelector('.chat-routing-card'), null); + const note = bubble.querySelector('.msg-routing-annotation'); + assert.equal(note.textContent, cause); + assert.equal(note.dataset.annotationStatus, 'needs_manual_target'); + assert.equal(bubble.dataset.chatAnnotationStatus, 'needs_manual_target'); + assert.equal(fx.decision.renderRoutingDecision(bubble, refused), false); + } finally { fx.restore(); } +}); + +test('a routing 409 that reopens the card names the host cause before the raw reason', async () => { + const cause = 'Not started: the working folder can\'t be used'; + const fx = fixture({ + fetchImpl: async () => ({ + ok: false, status: 409, + json: async () => ({ state: 'open', reason: 'workspace_unusable', cause }), + }), + }); + try { + const bubble = routingBubble('cm-5'); + fx.decision.renderRoutingDecision(bubble, ROUTING_ANNOTATION); + const card = bubble.querySelector('.chat-routing-card'); + card.querySelectorAll('.chat-quiz-option')[0].click(); + await new Promise((resolve) => setTimeout(resolve, 0)); + assert.equal(card.dataset.state, 'open'); + assert.equal(fx.toasts.length, 1); + assert.equal(fx.toasts[0].text, `Not routed: ${cause} — pick again.`); + assert.equal(fx.toasts[0].text.includes('workspace_unusable'), false); + } finally { fx.restore(); } +}); + test('a routing click posts the routing decision id with a STABLE request id', async () => { const fx = fixture(); try { diff --git a/web/tests/chat_history_integration.test.js b/web/tests/chat_history_integration.test.js index 7078582e3..77caa6433 100644 --- a/web/tests/chat_history_integration.test.js +++ b/web/tests/chat_history_integration.test.js @@ -175,6 +175,32 @@ test('repeated physical history identity updates the routing annotation on the s assert.match(note.textContent, /Investigation/); }); +test('a replayed refusal receipt shows the host cause and a later scheduled receipt patches the same note', async (t) => { + const cause = 'Not started: the working folder can\'t be used'; + const message = row('chat:250', 'Audit the GitHub tool', { + role: 'user', client_message_id: 'owner-refused', + chat_annotation: { + action: 'promote_chat_to_task', status: 'needs_manual_target', + target: 'never-started', target_label: 'Аудит', cause, + }, + }); + const f = fixture(t, page([message])); + await f.refresh(); + const bubble = f.bubbles().find((node) => node.dataset.historyId === 'chat:250'); + assert.ok(bubble); + assert.equal(bubble.dataset.chatAnnotationStatus, 'needs_manual_target'); + const note = bubble.querySelector('.msg-routing-annotation'); + assert.equal(note.textContent, cause); + await f.refresh(page([{ ...message, chat_annotation: { + action: 'promote_chat_to_task', status: 'scheduled', target: 'task-started', target_label: 'Investigation', + } }])); + assert.equal(f.bubbles().filter((node) => node.dataset.historyId === 'chat:250').length, 1); + assert.equal(f.bubbles().find((node) => node.dataset.historyId === 'chat:250'), bubble); + assert.equal(bubble.querySelector('.msg-routing-annotation'), note); + assert.equal(bubble.dataset.chatAnnotationStatus, 'scheduled'); + assert.equal(note.textContent, 'Started task · Investigation'); +}); + test('two physical rows with identical timestamp and body remain two messages across refresh', async (t) => { const rows = [row('chat:300', 'Same words'), row('chat:400', 'Same words')]; const f = fixture(t, page(rows)); diff --git a/web/tests/chat_inflight_indicator.test.js b/web/tests/chat_inflight_indicator.test.js index 358a192df..f40ed9535 100644 --- a/web/tests/chat_inflight_indicator.test.js +++ b/web/tests/chat_inflight_indicator.test.js @@ -68,6 +68,24 @@ test('legacy routing receipts and manual choices use neutral labels, never raw i assert.equal(manual.includes('opaque-'), false); }); +test('a host cause on a refused routing receipt outranks the status matrix; an empty cause changes nothing', () => { + // The host authors the owner sentence for a REFUSED act (`cause`); the + // client never derives one from `reason`. The field is absent/empty on + // scheduled, delivered and pending rows and on the picker frame, so those + // keep their labels (the options case above stays «Choose a target · …»). + const cause = 'Not started: the working folder can\'t be used'; + const refused = routingAnnotationText({ + action: 'promote_chat_to_task', status: 'needs_manual_target', + target: 'x', target_label: 'Аудит', cause, + }); + assert.equal(refused, cause); + assert.equal(refused.includes('Choose a target'), false); + assert.equal(routingAnnotationText({ + action: 'steer_task', status: 'delivered', target: 'opaque-task-id', + target_label: 'Launch', cause: '', + }), 'Steered task · Launch'); +}); + test('state snapshots and failure authority stay monotonic across reversed completion', () => { const applied = []; const requestTimes = [100, 200]; diff --git a/web/tests/chat_plain_system_rows.test.js b/web/tests/chat_plain_system_rows.test.js index a9b50d874..10dbb5bca 100644 --- a/web/tests/chat_plain_system_rows.test.js +++ b/web/tests/chat_plain_system_rows.test.js @@ -390,6 +390,108 @@ test('system row without a markdown flag renders plain (cancel_receipt class)', } }); +// Promote-refusal placement (R3/R4/R14): an owner-initiated refusal is ONE typed +// system row bound to the never-started task's id, with plain `<title> · <cause>` +// text — `task_not_started` for a confirmed refusal («Not started: …») and +// `task_start_unconfirmed` when the host cannot tell («Not confirmed: …»). +// A typed keyed row is neither a terminal fact nor a plain untyped final, so it +// renders as an ordinary plain system bubble and mints/finishes no live card — +// live or on replay. Both types get the same assertions. +const ADMISSION_NOTICE_ROWS = [ + { + chat_id: 2, + role: 'system', + system_type: 'task_not_started', + task_id: 'abc123', + content: 'Аудит · Not started: the working folder can\'t be used', + ts: '2026-09-16T00:00:04Z', + }, + { + chat_id: 2, + role: 'system', + system_type: 'task_start_unconfirmed', + task_id: 'def456', + content: 'Аудит · Not confirmed: the task may or may not have started', + ts: '2026-09-16T00:00:05Z', + }, +]; + +function findCard(node, taskId) { + if (node?.dataset?.taskId === taskId && node.classList?.contains('chat-live-card')) return node; + for (const child of node?.children || []) { + const hit = findCard(child, taskId); + if (hit) return hit; + } + return null; +} + +function liveCards() { + const messages = globalThis.document.byId.get('chat-messages'); + return messages.children.filter((node) => node.classList.contains('chat-live-card')); +} + +for (const noticeRow of ADMISSION_NOTICE_ROWS) { + test(`${noticeRow.system_type} renders as a plain system bubble and mints no card, live and after reload`, async () => { + const expectedText = new RegExp(noticeRow.content.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')); + let liveHtml = ''; + { + const { prior, mount } = installDom(); + let instance; + try { + const made = makeInstance(mount); + instance = made.instance; + made.handlers.get('chat')(noticeRow); + const bubble = findBubble('system'); + assert.ok(bubble, 'the notice row rendered a system bubble'); + assert.match(bubble.innerHTML, /📋 System/); + assert.match(bubble.innerHTML, expectedText); + assert.doesNotMatch(bubble.innerHTML, /<br>|<h1|<h2|md-h1|md-h2|<strong/); + assert.equal(bubble.getAttribute('data-chat-markdown-enhanced'), ''); + assert.equal(bubble.dataset.taskId, noticeRow.task_id, 'the row stays bound to the never-started task'); + const messages = globalThis.document.byId.get('chat-messages'); + assert.equal(findCard(messages, noticeRow.task_id), null, 'no live card is minted for the never-started task'); + assert.equal(liveCards().length, 0); + liveHtml = bubble.innerHTML; + } finally { + instance?.destroy(); + restoreDom(prior); + } + } + const historyRow = { + text: noticeRow.content, + role: 'system', + ts: noticeRow.ts, + is_progress: false, + system_type: noticeRow.system_type, + task_id: noticeRow.task_id, + markdown: false, + }; + const { prior, mount } = installDom(async (url) => { + if (String(url).startsWith('/api/chat/history')) { + return { ok: true, json: async () => ({ messages: [historyRow] }) }; + } + return { ok: true, json: async () => ({ active_direct_turns: [] }) }; + }); + let instance; + try { + ({ instance } = makeInstance(mount)); + await settle(); + await settle(); + const bubble = findBubble('system'); + assert.ok(bubble, 'history replay rendered the notice row'); + assert.equal(bubble.innerHTML, liveHtml, + 'live DOM and reload DOM are byte-identical for the notice row'); + const messages = globalThis.document.byId.get('chat-messages'); + assert.equal(findCard(messages, noticeRow.task_id), null, + 'replay neither mints nor finishes a card for the never-started task'); + assert.equal(liveCards().length, 0); + } finally { + instance?.destroy(); + restoreDom(prior); + } + }); +} + test('render arm order and enhancement guard are pinned in source', () => { // The plain-system arm sits between the dedicated skill_review renderer // (bug report #8) and the byte-pinned final markdown arm