diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index d25df7fb6..f65532956 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -31,7 +31,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de │ └── modules/widgets.js + widget_module.js + widget_frame.js + widget_card.js + widget_reorder.js + widget_chart.js + widget_list.js + masonry.js ← Widgets page host (`mountTab` dispatcher, card registry, declarative renderer) + the two framed mounts (extension-route iframe; module `srcdoc` iframe with its CSP/sandbox constants, parent fetch/resize bridge and disposer) + the child-side bootstrap served into the module frame (bridge grammar, `Response` rebuilt over a stream, resize reports, dispose acknowledgement) + the framed card chrome (launch policy incl. `retain`, Start/Stop, policy menu, facade) + key-order reorder handles + declarative chart/table helpers + pure list helpers (per-card and order-independent change signatures, keyed patch plan) + the key-ordered masonry that writes only `--masonry-*` custom properties │ ├── supervisor/ ← Background thread inside server.py - │ ├── active_activity.py ← In-memory registry of in-flight direct/ephemeral chat turns (`DirectActivityRegistry`); feeds `/api/state` `active_direct_turns` and WS typing frames (`activity_id`, `client_message_id`, `phase`, `kind`); no queue records + │ ├── active_activity.py ← Process-local owner of in-flight native chat actors, including preparation and post-task delivery (`DirectActivityRegistry`); private actor handles support controls and writer drain, public snapshots feed `/api/state` `active_direct_turns` and WS typing frames (`activity_id`, `client_message_id`, `phase`, `kind`); no queue records │ ├── message_bus.py ← Queue-based local message bus (Web UI + reviewed transport skills) │ ├── workers.py ← Multiprocessing worker pool (forkserver on Linux, spawn on macOS/Windows; never fork from the multi-threaded supervisor) │ ├── worker_assignment.py, worker_chat_lane.py, worker_health.py, worker_pool_lifecycle.py, worker_process.py, worker_promotion.py ← The pool's leaves: handing a pending task to a free worker and refusing the ones that must not run; the direct and ephemeral chat lanes and their resume after a restart (conversation stays admitted while the authorized assisted resolver holds the repository, and the owner-control path is preloaded before conflict markers land); crash detection and the terminal a host-side teardown publishes; keeping the pool populated (the readiness gate every spawned or respawned slot passes before it is assignable, pid records, reaping, respawn) and `kill_worker_tree`, the ONE worker tree-kill every teardown and backstop uses — the installation's daemon roots spared always, kept services only on one task's cancel or timeout; what runs INSIDE a worker child process from entry to crash record; and turning a chat turn — or a project scope — into a queued task @@ -200,10 +200,10 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de ├── configured_subagents.py ← Canonical `OUROBOROS_SUBAGENTS` parser/serializer: strict validation, stable ids, fingerprinting; owner free text is never host-parsed ├── subagent_runtime.py ← Immutable task-start subagent snapshots, exact `subagent_id` selection, typed alternatives, bounded deterministic legacy-input seam ├── subagent_route_health.py ← Route health: the ONE manifest reader behind every delegated dispatch - ├── subagent_work_order.py ← One complete work-order compiler under a single 250,000-char total wire budget (a serializer bound, not a context claim); over-budget yields a full-SHA source-selector partial lens, never a prefix; generic manifest capability read — no harness name interpreted + ├── subagent_work_order.py ← Complete chosen work-order compiler and normalized host authority without arbitrary admission cuts; exact-source readers and legacy source-response validation share the unchanged renderer ├── subagent_bootstrap.py ← Host pre-start of the exact snapshotted leaf BEFORE the first metered round, through the same wrapper as `delegate_start(prompt="")`; branch order recovery → fences → blocked → pre-start, and a fence-wake outranks every terminal because a fence may hide a live run; the host never waits (`configured_session_started` receipt); only a definite typed refusal ends unrun at $0 — everything ambiguous wakes the model, because a false "spent nothing" terminal over a possibly-live run is the one direction classification must never fail toward ├── delegate_supervision.py ← Event-only sleeping-nanny loop: quiet windows renew without a model call; terminal/interaction/fault/addressed/control (or one reasoned checkpoint) triggers a durable wake with fresh coordination context (parent intent, time, tree spend, host-visible descendants and root review capacity — every fact observed READ-ONLY, so a metadata-poor task reports `time.state = "not_set"` rather than latching an anchor from a poll; polling writes nothing of its own and inherits only the canonical usage-ledger reader's bounded maintenance — the torn-tail quarantine after a SINGLE crash mid-append, which every reader performs identically (a crash inside that repair itself, a torn quarantine sink, is a known residual: issue #586), the empty `state/` directory the reader's lock lives in on a never-initialized root, and removal of a stale `usage_attempts.lock` older than the reader's 90 s stale window (`usage_ledger._locked` → `platform_layer.acquire_exclusive_file_lock`, whose stale-age branch unlinks the lock file and retries) — each pinned by a regression; an absent ledger answers known-zero settled spend through that same canonical reader); replay returns the stored snapshot - ├── delegate_start_instructions.py ← Stable host start instructions + a bounded separately-hashed appendix; host pre-start sends no appendix; over-budget refuses before provisioning + ├── delegate_start_instructions.py ← Stable host start instructions + a complete separately-hashed coordination appendix; host pre-start sends no appendix ├── delegate_recovery.py ← Narrow exact-leaf recovery for proven crash + planned self-restart; validates bindings; vetoes every no-resume cause ├── delegate_registration_policy.py ← `persistent_registration` + the STARTED-row field tables ├── delegate_pending.py ← Durable pending-invocation replay preserving the original idempotency key + canonical start body @@ -393,7 +393,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de │ ├── tool_result.py, tool_catalog.py, tool_resolution.py ← Dispatch-side vocabulary: the typed internal tool result with its byte-compatible legacy text adapter (the producer stamps outcome facts; a note append degrades rather than rewrites); the intrinsic tool descriptors shared by tool modules and registry dispatch; and argument normalization with physical target binding │ ├── review_response.py ← Pure response-envelope projection for multi-model review rows │ ├── shell_process.py, shell_effects.py, shell_outputs.py ← The command-running substrate: the process execution shared by every command tool; what a command did to its working tree and which of it was throwaway; and declared process outputs — resolution, fingerprints, artifact registration, and the per-path export-eligibility rules that reuse the workspace-patch credential-shape SSOT - │ ├── plan_review_artifacts.py ← Exact plan-review waves, bounded successor authority core, and reviewer-continuation inputs; reconstructs the API transcript + │ ├── plan_review_artifacts.py ← Exact plan-review waves and full operative specs in existing source handles, bounded successor index, and reviewer-continuation inputs; reconstructs the API transcript │ ├── evolution_stats.py ← Generates evolution.json metrics from sampled git history │ ├── owner_delivery.py ← Owner event delivery: live queue XOR `pending_events` fallback with sticky deferral, preserving narrative order after the first live failure; lineage stamped; background-consciousness frames unstamped and deferred │ ├── deliverables_shell.py ← cp/mv/ln into deliverables with symlink checks @@ -497,7 +497,7 @@ Completion compares against the captured preflight base (task-local commits stay Forked/empty task state lives under `data/state/headless_tasks//data`: a forked drive copies `identity.md`, `WORLD.md`, `registry.md` (a project fork carries `memory/knowledge/patterns.md` and omits global knowledge; an empty drive starts blank); dialogue, scratchpad, mailbox, and history never cross. The child drive is execution state: the result copies back to the canonical root, declared artifacts rebase to `data/task_results/artifacts//` (missing source = copy failure; collisions get a deterministic suffix), and verification-receipt replicas union with exact-row de-dup. Once the canonical result is terminal, late copy-back and effective reads pass through the same pure field-custody projection — the parent-owned terminal marker and cost/round/token fields cannot be overwritten. `memory_export.json` is an explicit artifact, never merged automatically. -Complete input sets are preserved in the existing artifact store. `task_contract.attachment_manifest` is a preview of at most 25 rows; an additive `attachment_manifest_ref` names the full immutable JSON under `source_handles/context_checkpoints` with count, size and SHA. Pure contract normalization preserves this shape; inheritance, owner-mailbox reads, physical retries and copy-back resolve the complete file closure through `artifacts`, verifying captured bytes before reuse. Legacy inline lists remain readable. Ordinary artifact and input copy failures share `child_ref_promotion` pending custody, retry and cleanup protection for both headless and direct-task drives. A preview never substitutes for an unreadable full reference. Native-image bounds, the 250000-character work order and transport-specific Telegram limits keep their separate meanings. +Complete input sets are preserved in the existing artifact store. `task_contract.attachment_manifest` is a preview of at most 25 rows; an additive `attachment_manifest_ref` names the full immutable JSON under `source_handles/context_checkpoints` with count, size and SHA. Pure contract normalization preserves this shape; inheritance, owner-mailbox reads, physical retries and copy-back resolve the complete file closure through `artifacts`, verifying captured bytes before reuse. Legacy inline lists remain readable. Ordinary artifact and input copy failures share `child_ref_promotion` pending custody, retry and cleanup protection for both headless and direct-task drives. A preview never substitutes for an unreadable full reference. Native-image bounds and transport-specific Telegram limits remain separate from complete work-order delivery. Automatic genesis output listing is discovery, separate from artifact custody. Unreadable directories and changing files remain explicit rows; `complete` and @@ -705,7 +705,7 @@ Attachments stage via paperclip, paste, or drag-and-drop without the former coun Simple text messages may appear immediately as pending local bubbles and reconcile against their echoed `client_message_id`. Delivered documents capture immutable task-owned bytes and carry a verified `file_ref` plus canonical download URL (with a compatible Files URL when available), rebuilding from history without persisting base64. Files above the former 50 MiB inline boundary remain file-backed; browser downloads use HEAD followed by a native download request and desktop saves stream through the existing launcher helper; photos and videos keep base64 only in the live frame — supported media is stored under the canonical task artifact root with a content-addressed URL so replay and rebuilds restore the same bubble. Structured `links` rows render at most twelve independently revalidated HTTP(S) buttons, identically live and replayed. The supervisor transport owns durable media copies so all producers converge on the canonical data root; failed persistence stays an honest caption row and cannot finalize a task card. -Direct in-process chat turns and ephemeral decision turns create no supervisor queue records; their live state sits in the thread-safe process-local `supervisor.active_activity.DirectActivityRegistry`, exposed via `active_direct_turns` in `GET /api/state` and typed fields on `typing` frames. Both render their progress, tool telemetry and typed conclusion on the ordinary live card; an ephemeral decision turn's `ephemeral_decision` marker only withholds the task claims a managed root would get — no `Turn into project`, and no Cancel because no host `cancelable` marker ever rides its frames — it never hides the work, its blank-status `task_done` concludes the activity like the final frame, while the already-computed outcome axes, reason and accounting decide success, warning or failure. The final chat row retains those same facts and its ephemeral marker so history can replay the conclusion without a durable task result; a ledger-final amount stays final because ephemeral turns run no post-task synthesis. Historical rows missing those facts remain legacy, without an inferred cost or recovered outcome. `GET /api/state` also exposes `active_chat_activities`: those rows united with ROOT managed queue tasks as `kind="managed_task"` with phase `queued`, `budget_paused` (PENDING fenced by a budget-pause fact; nothing dispatches it until an explicit resume, so it must not masquerade as queued), `working`, or `finalizing` (RUNNING with an open post-task checkpoint), so a chat instance created after the task started hydrates from the queue authority rather than transient typing frames. The client status reducer derives the chat header only from connection and authoritative activity state; a terminal failure stays a factual task result, never a reasonless header `Attention`, and a local `Sending...` submission is retired only by authoritative evidence (typing frame, snapshot turn, durable routing receipt, replayed user row, turn conclusion, or offline-queue eviction) — never by the live user-row echo or a socket write. +Ordinary Main and Project conversations use complete native execution even while other work runs; Project presence and routing metadata do not select the transient ephemeral contract. Each turn owns its agent and creates no supervisor queue record. The thread-safe process-local `supervisor.active_activity.DirectActivityRegistry` retains its private actor handle from preparation through result delivery; Stop, Hurry, quiz and steering resolve that exact actor, while `active_direct_turns` and typing frames expose only its presentation fields. The update admission lock covers check-and-registration only; writer drain waits the registered executions, including post-task work. Custody maintenance and restart census include those same live IDs, and Background Consciousness stays paused while any foreground execution remains. Explicit Swarm routing retains its ephemeral contract. The producer carries `_is_direct_chat` into the durable result and terminal event so late/replayed ordinary Project completions stay in their room and do not emit Main task-completion summaries. Both render their progress, tool telemetry and typed conclusion on the ordinary live card; an ephemeral decision turn's `ephemeral_decision` marker only withholds the task claims a managed root would get — no `Turn into project`, and no Cancel because no host `cancelable` marker ever rides its frames — it never hides the work, its blank-status `task_done` concludes the activity like the final frame, while the already-computed outcome axes, reason and accounting decide success, warning or failure. The final chat row retains those same facts and its ephemeral marker so history can replay the conclusion without a durable task result; a ledger-final amount stays final because ephemeral turns run no post-task synthesis. Historical rows missing those facts remain legacy, without an inferred cost or recovered outcome. `GET /api/state` also exposes `active_chat_activities`: those rows united with ROOT managed queue tasks as `kind="managed_task"` with phase `queued`, `budget_paused` (PENDING fenced by a budget-pause fact; nothing dispatches it until an explicit resume, so it must not masquerade as queued), `working`, or `finalizing` (RUNNING with an open post-task checkpoint), so a chat instance created after the task started hydrates from the queue authority rather than transient typing frames. The client status reducer derives the chat header only from connection and authoritative activity state; a terminal failure stays a factual task result, never a reasonless header `Attention`, and a local `Sending...` submission is retired only by authoritative evidence (typing frame, snapshot turn, durable routing receipt, replayed user row, turn conclusion, or offline-queue eviction) — never by the live user-row echo or a socket write. Owner-message continuity is journal-backed: each locally sent owner row is kept (bounded, with its routing annotation) until a fetched history response returns the same `client_message_id`, and a full rebuild re-renders unconfirmed journal rows after the server response, so a stale history snapshot cannot erase a message the owner just sent. Task finalization is honest: a root's early final answer carries `task_phase="finalizing"` (the same fact replay derives from the open post-task checkpoint), the card holds a sticky `Finalizing…` phase, and `task_cost_finalized` is bookkeeping that never resolves a card. If an observed managed root disappears from a queue-authoritative snapshot whose request began after the observation, the state-refresh fan-out removes activity/cancel authority and starts one single-flight durable task-detail read; request generations prevent an older response from undoing a newer projection. Two liveness invariants close the stuck-`Working...` class: the history window is lineage-closed — subagent lineage older than the window's progress recency floor is kept only while the child still runs or finalizes, or its parent is REPRESENTED by the response (a row that proves the task's card: telemetry, its own message or summary, a folded review group or plan reference it owns — a typed photo/document/quiz delivery does NOT count), or the parent is alive, and an anchored child represents its own children in turn, so a nested swarm is kept or dropped whole; a FINAL row failing that test is emitted without lineage fields (`ouroboros/gateway/history.py`, one anchor set computed in the quota pass), so replay cannot mint an unfinishable parent card while a visible or live parent keeps its completed children and their executor receipts — and liveness reconciles over the CARD SET, not only the activity registry: an unvouched connected root card is finished ONLY by proven durable terminal detail, and a card with no durable result honestly keeps its state; a root proven terminal (its `task_done` or its terminal detail) settles the descendant cards a lost child terminal left open from each child's OWN durable result — one single-flight read per child through the ordinary child-terminal path, never a cascade from the parent's outcome, never from the card-set scan. The header badge has exactly one writer, the status reducer, with the panel-boot "Online" seed as the sole exception. @@ -717,7 +717,7 @@ The task card's `Reviews` section is a read-only projection over independent dom Card cost is sticky task-scope evidence. Only frames carrying task accounting status, finality, subtree, reservation, or unknown-cost fields may update it; an unrelated per-call `cost_usd` delta is never relabelled as the task total. Compact cards render one amount — the accounted upper bound worded `up to` while the ledger is open, plain once final — preferring the complete subtree projection; a running root's heartbeat may carry a non-final aggregate from the physical-attempt ledger, so the amount advances without a second timer, endpoint, or client-side sum, and a history window replays the same non-final projection (`live_root_cost_projection`, the one cost owner) on a running or finalizing root's latest in-window progress row so a reload between heartbeats shows the same ceiling — a subtree with no attributable rows stays absent, never zero. Precedence: unavailable, then pending, then final, newer evidence winning within a class; costless frames cannot erase a known value. Dashboard and task-detail accounting derive from the physical-attempt ledger, not the card; compact review rows copy or sum no money, and Skill wave attempt/slot money appears only inside the lazy exact-job detail, joined from the same ledger via `physical_attempt_v1`. -A Cancel action appears only on unfinished, unconverted, pooled root cards carrying the supervisor's host-attested `cancelable=true` — card shape alone is insufficient because direct in-process turns look identical but have no queue entry. The stop control is the shared dropdown above, sent with `cascade:true` (root plus live descendant subtree): the hard path answers only after teardown or a typed refusal; the graceful "Wrap up" path records a durable finalize intent and answers immediately with a typed 202 pending acknowledgement. Natural completion wins a race, a missing live task reconciles from the durable record, and a process that cannot be proven dead remains a visible refusal rather than being painted Cancelled. +A Cancel action appears only on unfinished, unconverted root cards carrying the host-attested `cancelable=true`; both pooled tasks and addressable native direct turns use the existing custody owner, while card shape alone grants no control. The stop control is the shared dropdown above, sent with `cascade:true` (root plus live descendant subtree): the hard path answers only after teardown or a typed refusal; the graceful "Wrap up" path records a durable finalize intent and answers immediately with a typed 202 pending acknowledgement. Natural completion wins a race, a missing live task reconciles from the durable record, and a process that cannot be proven dead remains a visible refusal rather than being painted Cancelled. A Main root card may be turned into a Project: conversion creates or reuses the Project, names it owner-facing, binds the task and its canonical origin message, moves live work onto the Project lane, and replaces the Main action with a calm Project pointer; tasks already bound to a Project get no second conversion button. The pointer is the one shared project chip (`ui_helpers.js::renderProjectChip`): cards inside a Project panel and nested subagent cards never receive it, and clicking opens the panel or does nothing when already open — a pointer never closes what it points at. Naming reuses an already coined model title or falls back through the server naming path; the UI invents no second name authority. Project-SCOPED is not project-BOUND: a headless/CLI run carries a `project_id` for lease and memory without a durable binding — addressed to the Project thread at admission, it never mints a Main card and needs no conversion button. @@ -1007,7 +1007,7 @@ A spawned or respawned slot is not capacity until its child confirms it. Both sp Unexpected worker death is a three-way decision. An already-terminal durable result wins and is projected idempotently; a negative process exit code is terminal for every task, because replaying the same infrastructure signal usually repeats the failure and burns budget; only an otherwise-eligible non-signal crash retries within `QUEUE_MAX_RETRIES`. Repeated busy-worker or all-workers-dead failures trip the crash-storm fence, disable pooled admission, and surface recovery instead of cycling workers indefinitely. Direct chat stays available because it is not owned by the pooled scheduler. -Startup and throttled maintenance reconcile three residue classes. Process custody checks strict PID, start-time, command fingerprint, owner task, session, and generation evidence before reaping an owned process. Delegated-run reconciliation applies the same owner-gone reasoning to external harness rows; the write-side healing seams (boot-backfill reverse join, cursor pass, sweep refresh) and the counters-as-snapshot rule are specified once under §6 Delegated subagents. Task, review, and project reconciliation repair durable records whose producer no longer exists. None of these are command-line-class kill sweeps, and one development or runtime instance never reaps another. The dedicated watchdog separately observes supervisor-loop liveness and a stuck in-process direct turn; it alerts and recommends `/restart` but cannot safely unlock another thread's lock or kill work whose custody it does not own. Ephemeral owner turns remain a separate responsive lane, not a second scheduler. +Startup and throttled maintenance reconcile three residue classes. Process custody checks strict PID, start-time, command fingerprint, owner task, session, and generation evidence before reaping an owned process. Delegated-run reconciliation applies the same owner-gone reasoning to external harness rows; the write-side healing seams (boot-backfill reverse join, cursor pass, sweep refresh) and the counters-as-snapshot rule are specified once under §6 Delegated subagents. Task, review, and project reconciliation repair durable records whose producer no longer exists. None of these are command-line-class kill sweeps, and one development or runtime instance never reaps another. The dedicated watchdog separately observes supervisor-loop liveness and every registered native actor; it alerts and recommends `/restart` but cannot kill an individual in-process thread. Other owner conversations run on independent native actors, without a second scheduler. Cooperative project checkpointing has two equivalent quiescence triggers: a host-minted genesis or cooperative tree is checked when its root settles with no live descendants, and again when the last child settles beneath an already-terminal root. The second trigger exists because a root-scope budget stop terminalizes the root before its children reach their own dispatch boundaries; a root-only trigger would see a live tree once and never return. Event dispatch detects the condition only after removing the finishing task from RUNNING. The bounded git chain runs on a daemon thread, revalidates quiescence under the queue lock immediately before mutation, and uses a per-root latch that replays a trigger arriving during an in-flight check. Only host-minted project roots are eligible; owner-attached folders are never auto-committed, credential-shaped files stay excluded and disclosed, and every material success, skip, or error receives a durable receipt. @@ -1017,7 +1017,7 @@ The bridge recognizes `/panic`, `/restart`, `/review`, `/evolve [on|off]`, `/bg ### Task lifecycle -A queued user task enters through a reviewed transport, is admitted by the supervisor queue, and runs in `OuroborosAgent`; a direct-chat turn runs on the shared in-process chat agent and an ephemeral decision turn builds its own; both are tracked by the process-local `DirectActivityRegistry` and create no `PENDING`/`RUNNING` queue record (§3). The root pipeline captures the task contract and immutable context core, executes the LLM/tool loop, preserves a delivery candidate, stores the result and artifacts, emits lifecycle and usage evidence, performs the root-only post-task work, and publishes the typed outcome. Queue admission proves only that asynchronous work was durably accepted; completion, objective satisfaction, artifact finality, verification, and review acceptance remain separate facts. +A queued user task enters through a reviewed transport, is admitted by the supervisor queue, and runs in `OuroborosAgent`; each direct-chat turn runs on its own in-process agent, while an explicit Swarm routing turn retains the ephemeral contract; both are tracked by the process-local `DirectActivityRegistry` and create no `PENDING`/`RUNNING` queue record (§3). The root pipeline captures the task contract and immutable context core, executes the LLM/tool loop, preserves a delivery candidate, stores the result and artifacts, emits lifecycle and usage evidence, performs the root-only post-task work, and publishes the typed outcome. Queue admission proves only that asynchronous work was durably accepted; completion, objective satisfaction, artifact finality, verification, and review acceptance remain separate facts. `DeliveryCandidate` is retained before verification or review so a later notice, reviewer failure, deadline, or provider outage cannot erase a useful answer. `outcomes.py` combines execution, objective, review, artifact, and child-absorption axes without converting one axis into another — the terminal custody overlay (`outcomes.custody_debt_axes`) is an instance of that rule, not an exception to it; verify-before-done receipts and exact artifact references are host-attested evidence — declarations and answer prose are not substitutes. A forced exit may publish the best current candidate only with its typed rail and evidence-freshness disclosure, and lifecycle may remain `completed` while the objective or review axis records a best-effort or unaccepted result. @@ -1056,7 +1056,7 @@ Disclosed cancel-lifecycle residuals (deliberate): a cascade over a tree with no Tool API v2 exposes neutral canonical names directly (`read_file`, `list_files`, `search_code`, `write_file`, `edit_text`, `edit_batch`, `apply_patch`, `run_command`, `run_script`, `verify_and_record`, service tools, `commit_reviewed`, `vcs_*`, `schedule_subagent`, `schedule_followup`, `wait_task`, `wait_tasks`); legacy public names are neither exposed nor translated. The file tools share a path-based public ABI, and because payload-borne paths (`edit_batch` entries, `apply_patch` targets) miss the dispatch seam that rewrites a `path` ARG, both ends canonicalize explicitly through `tool_access.canonical_repo_relative_path` — one normalization contract keeps a guard from judging `repo/BIBLE.md` while the write lands on `BIBLE.md`; `_ROOT_ARG_REPO_WRITE_TOOLS` is the single set every repo-write fence keys on. -Filesystem tool output is self-locating: results use canonical `root:path` labels and `run_command`/`run_script` echo the resolved `cwd`. File bindings preserve physical identity: absolute paths inside any selected base normalize to that base; an outside absolute path is refused before `safe_relpath` can turn it into a similarly named file. Repo basename-prefix and canonical delegated-artifact read redirects retain their existing contracts. `ToolContext.repo_path`/`drive_path` use the same physical resolver, and unsupported roots of repo-only batch/patch editing reach the existing typed handler refusal before payload selectors are resolved. `user_files` is the first-class root for user-visible files under the owner's home (the Ouroboros repo and runtime control-plane are rejected); `task_drive` is task-scoped scratch; `artifact_store` is task-scoped under `data/task_results/artifacts//`, and external deliverables written through `user_files` or declared process `outputs` are copied there for audit — declared directory outputs as complete manifest+zip pairs written and hashed in chunks, and a rewritten user-visible file retains its previous copy under `task_results/artifact_versions//` (see the §1 tree). Two READ-ONLY orchestrator roots complete the set: `subagent_projects` and `deliverables` grant `read`/`list`/`search` only to orchestrator profiles, so a parent can inspect a child's tree or a finished deliverable when synthesizing; a top-level task may still write the physical Deliverables container through the existing `user_files`-authorized paths. For argv-visible targets the shell guard checks the lexical Deliverables origin before generic roots, then the symlink-resolved destination, so hidden, credential-like, protected, and symlink-escaping descendants do not inherit a broader root's admission; the same target-first rule applies to declared-output custody and Presence ceilings, whose logical `user_files`-relative prefix survives a remapped physical binding. Pre-execution target extraction is segment-aware but conservative: shell `-c` recursion is bounded at three levels; a visible heredoc is interpreter program text only when there is no inline program or script-file operand, and shell stdin bodies recurse like `-c`. Python UNKNOWN remains unprovable even beside recovered targets. `cd`/`pushd` and `env -C`/`--chdir` update a sequential, symlink-resolved effective cwd for later/wrapped relative writes; find/xargs replacement words are templates, not concrete targets. Uncertainty widens only the owning row's tokens and inline/heredoc body, never independent provably read-only segments. The raw mention view stays separate for Windows drive/UNC spellings, and the light fence keeps its unfiltered inline-body view. This is not shell interpretation: computed destinations and unsupported wrapper grammars remain fail-closed only when write shape or uncertainty is visible. Runtime Light applies the same per-row writer targets and effective cwd to runtime-data access, expanding the existing known runtime/home spellings only on those targets and preserving separate secret/project-store read boundaries; a second substring write scan cannot turn comparisons or prose into writes. Root process writes may select any already-authorized `user_files` target independently of cwd, while acting children retain their own write root. `shell_parse.local_shell_subject` removes SSH remote command arguments only from filesystem writer-target inspection, preserving options, the local `-E` log sink and outer input/output redirects through the shared grammar; every other guard and execution receives the original argv. Child, external-task and Light read policies consume the same physical paths from `shell_guards.shell_inspection_paths`, preserving sequential and wrapper cwd without giving source names credential authority. Positional GitHub policy still classifies direct `gh` and shell-wrapper segments only; remote `ssh ... gh auth` remains an inherited residual. A remote body can still reach local files through an existing SSH trust relationship, including loopback; local deterministic inspection does not interpret remote effects or provide an SSH sandbox. Unknown option/heredoc forms keep the existing conservative inspection. Ordinary `.config`, `Library`, `settings.json` and exact `~/.ssh/config` mutations use the existing resource authority; the enumerated owner locations in `credential_shapes.owner_credential_locations`, credential-leaf rules and VCS control directories stay protected. This is not blanket protection for all credential stores: unlisted locations such as `.cargo/credentials.toml`, `.terraform.d/credentials.tfrc.json` and `.kaggle/kaggle.json` retain ordinary root configuration access. Child read/list/search/query share source visibility for ordinary `auth/` and `tokens/` paths and public PEM certificates. Task and artifact files are not repository credential stores. `core_secret_paths.restricted_data_roots` anchors owner/control checks on the child, canonical parent and configured admission roots, shared with vision/media. Restricted file reads mask complete private-key blocks before selecting line/character windows, preserving positions and line breaks; known credential bytes remain masked on delivered views. The post-execution shell audit is still best-effort and cannot reconstruct post-`cd` relative writes, variable/indirect destinations, inline-code path construction, unwalked recursive copies, or inode aliases (`shell_guards.py`). Additional disclosed residuals: a cp/mv/ln invocation carrying an unknown value-taking long option keeps its operands mention-only; a Deliverables path used as a cp SOURCE is a read and takes no Deliverables target policy; `write_shape`'s safe-stdio redirect set is matched by exact string, so a glued `2>/dev/null;` still classifies a read-only line write-shaped — no block, but the refusal a protected-root read still gets is worded as a write; a quoted standalone `'>'` token is stripped as a redirect; and the redirect grammar is duplicated in `shell_audit.py` and spelled a third way inside `light_shell_repo_mutation`. +Filesystem tool output is self-locating: results use canonical `root:path` labels and `run_command`/`run_script` echo the resolved `cwd`. A direct Project room selects one active physical folder for reads, writes, editing, process cwd, VCS and delegation; governance remains at `system_repo`. A selected missing folder keeps its address and warning rather than falling back to Ouroboros source. Plain folders support ordinary file/process work without Git; mutating delegation uses the existing snapshot capability and returns its typed target-specific failure when Git is unavailable. File bindings preserve physical identity: absolute paths inside any selected base normalize to that base; an outside absolute path is refused before `safe_relpath` can turn it into a similarly named file. Repo basename-prefix and canonical delegated-artifact read redirects retain their existing contracts. `ToolContext.repo_path`/`drive_path` use the same physical resolver, and unsupported roots of repo-only batch/patch editing reach the existing typed handler refusal before payload selectors are resolved. `user_files` is the first-class root for user-visible files under the owner's home (the Ouroboros repo and runtime control-plane are rejected); `task_drive` is task-scoped scratch; `artifact_store` is task-scoped under `data/task_results/artifacts//`, and external deliverables written through `user_files` or declared process `outputs` are copied there for audit — declared directory outputs as complete manifest+zip pairs written and hashed in chunks, and a rewritten user-visible file retains its previous copy under `task_results/artifact_versions//` (see the §1 tree). Two READ-ONLY orchestrator roots complete the set: `subagent_projects` and `deliverables` grant `read`/`list`/`search` only to orchestrator profiles, so a parent can inspect a child's tree or a finished deliverable when synthesizing; a top-level task may still write the physical Deliverables container through the existing `user_files`-authorized paths. For argv-visible targets the shell guard checks the lexical Deliverables origin before generic roots, then the symlink-resolved destination, so hidden, credential-like, protected, and symlink-escaping descendants do not inherit a broader root's admission; the same target-first rule applies to declared-output custody and Presence ceilings, whose logical `user_files`-relative prefix survives a remapped physical binding. Pre-execution target extraction is segment-aware but conservative: shell `-c` recursion is bounded at three levels; a visible heredoc is interpreter program text only when there is no inline program or script-file operand, and shell stdin bodies recurse like `-c`. Python UNKNOWN remains unprovable even beside recovered targets. `cd`/`pushd` and `env -C`/`--chdir` update a sequential, symlink-resolved effective cwd for later/wrapped relative writes; find/xargs replacement words are templates, not concrete targets. Uncertainty widens only the owning row's tokens and inline/heredoc body, never independent provably read-only segments. The raw mention view stays separate for Windows drive/UNC spellings, and the light fence keeps its unfiltered inline-body view. This is not shell interpretation: computed destinations and unsupported wrapper grammars remain fail-closed only when write shape or uncertainty is visible. Runtime Light applies the same per-row writer targets and effective cwd to runtime-data access, expanding the existing known runtime/home spellings only on those targets and preserving separate secret/project-store read boundaries; a second substring write scan cannot turn comparisons or prose into writes. Root process writes may select any already-authorized `user_files` target independently of cwd, while acting children retain their own write root. `shell_parse.local_shell_subject` removes SSH remote command arguments only from filesystem writer-target inspection, preserving options, the local `-E` log sink and outer input/output redirects through the shared grammar; every other guard and execution receives the original argv. Child, external-task and Light read policies consume the same physical paths from `shell_guards.shell_inspection_paths`, preserving sequential and wrapper cwd without giving source names credential authority. Positional GitHub policy still classifies direct `gh` and shell-wrapper segments only; remote `ssh ... gh auth` remains an inherited residual. A remote body can still reach local files through an existing SSH trust relationship, including loopback; local deterministic inspection does not interpret remote effects or provide an SSH sandbox. Unknown option/heredoc forms keep the existing conservative inspection. Ordinary `.config`, `Library`, `settings.json` and exact `~/.ssh/config` mutations use the existing resource authority; the enumerated owner locations in `credential_shapes.owner_credential_locations`, credential-leaf rules and VCS control directories stay protected. This is not blanket protection for all credential stores: unlisted locations such as `.cargo/credentials.toml`, `.terraform.d/credentials.tfrc.json` and `.kaggle/kaggle.json` retain ordinary root configuration access. Child read/list/search/query share source visibility for ordinary `auth/` and `tokens/` paths and public PEM certificates. Task and artifact files are not repository credential stores. `core_secret_paths.restricted_data_roots` anchors owner/control checks on the child, canonical parent and configured admission roots, shared with vision/media. Restricted file reads mask complete private-key blocks before selecting line/character windows, preserving positions and line breaks; known credential bytes remain masked on delivered views. The post-execution shell audit is still best-effort and cannot reconstruct post-`cd` relative writes, variable/indirect destinations, inline-code path construction, unwalked recursive copies, or inode aliases (`shell_guards.py`). Additional disclosed residuals: a cp/mv/ln invocation carrying an unknown value-taking long option keeps its operands mention-only; a Deliverables path used as a cp SOURCE is a read and takes no Deliverables target policy; `write_shape`'s safe-stdio redirect set is matched by exact string, so a glued `2>/dev/null;` still classifies a read-only line write-shaped — no block, but the refusal a protected-root read still gets is worded as a write; a quoted standalone `'>'` token is stripped as a redirect; and the redirect grammar is duplicated in `shell_audit.py` and spelled a third way inside `light_shell_repo_mutation`. Structured tools also accept a Docker workspace's explicitly mapped backend absolute address through `workspace_executor.map_backend_path`; the resolved @@ -1401,7 +1401,7 @@ An undisclosed spend contributes `0.0` to `accounted_usd` — inventing a conser **Four nanny verbs** (`tools/delegate.py`): `delegate_start`, `delegate_wait`, `delegate_cancel`, `delegate_answer`. There is deliberately no fake `hurry`: Claudexor truthfully exposes cancel and answers, not in-place steering. `delegate_start` takes an exact `agent_session` `subagent_id` (or recovery-only `retry_of`); API actor ids are refused. A scheduled configured nanny's bootstrap (`subagent_bootstrap`, mechanics in the §1 row) starts the exact snapshotted leaf before the first model round — after recovery adoption first, then the durable zero-run/unknown-evidence fences, because a fence may hide a live prior run and a fence-wake outranks every terminal — through the same `delegate_start(prompt="")` wrapper the model itself uses: one start path, one set of refusal shapes. `delegate_start(subagent_id=..., prompt=..., root="skill_payload", bucket=..., skill_name=...)` is the orthogonal exact-resource selector: the id chooses transport; the selector NAMES a resource, authorized through a fresh `ResolvedResourceBinding` for `skill_payload.write` — it never grants one. A DEFINITE start refusal (typed, no custody handle, closed `_DEFINITE_UNRUN_REASONS` set) ends the child UNRUN at $0; everything ambiguous wakes the model, because a false "spent nothing" terminal over a possibly-live run is the one direction classification must never fail toward. An unregistered project root is one such typed refusal: the nanny registers first, and a registration WE created is retired at settlement (`delegate_registration_policy`). -**Work orders.** The compiler has one total 250,000-character wire budget — an Ouroboros serializer-integrity bound, not a model context limit. Fitting orders are byte-complete and the host never sends a prefix, which the child would believe is the whole contract: a larger order starts only a route whose live manifest declares an interactive question channel (exact source character ranges, answered via `delegate_answer(source_response=...)` and verified against selector, digest, bounds, and exact bytes). The manifest observation is a preflight, not a lease: only durable verified source-range coverage authorizes completion, and crash recovery replays the durable request body and fingerprint rather than recompiling. +**Work orders.** The compiler sends the entire chosen assignment, preserving context and instruction roles without an arbitrary host-size cutoff. Direct starts carry the normalized host contract once in instructions and the chosen assignment separately in prompt; coordination context remains complete. Real native/HTTP limits return their actual failure with the original input and any pending invocation retained. Exact-source readers remain optional capabilities. Legacy partial starts still verify their original renderer digest, selector and source intervals, and recover by replaying the recorded request body; removing new partial-start production never certifies an incomplete old run or launches a duplicate after ambiguous dispatch. **Supervision.** `delegate_wait` is model-visible as an event-only sleep, not a caller-sized poll: `delegate_supervision.supervised_wait` renews bounded transport windows in host code at zero LLM calls, and only a meaningful event (settlement, interaction, fault, addressed message, child signal, control, recovery judgment) becomes a coalesced durable wake, replayed across worker interruption until acknowledged; deadline, ceiling, budget, and cancellation remain outer bounds. Every receipt and wake carries one host-rendered `coordination_context` (intent note, deadline remaining, root-tree spend, active descendants, remaining paid acceptance capacity) — facts for LLM judgment, never thresholds. A requested future inspection (`checkpoint_after_sec`) wakes once and is consumed by any earlier real event — no cadence, stall classifier, or hidden polling. On a wake the nanny holds its full tool surface and the parent's captured model/effort. Nanny economics are structurally quiet: the burn baseline resets only on real acts of delegation, and coordination verbs never buy metered silence (`nanny_pacing.py`). @@ -1418,7 +1418,7 @@ An undisclosed spend contributes `0.0` to `accounted_usd` — inventing a conser | task authority | access | mode | isolation | `execution.delegated` | |---|---|---|---|---| | acting subagent (valid write surface) | `workspace_write` | `agent` | `live` | `true` | -| ROOT of an external-workspace task (validated active workspace) | `workspace_write` | `agent` | `live` | `true` | +| Ordinary root with a validated external workspace or a selected Project room | `workspace_write` | `agent` | `live` | `true` | | top-level task selecting an exact skill payload (`root="skill_payload"`) | `workspace_write` | `agent` | `live` | `true` | | anything else, including a fail-closed subagent | `readonly` | `ask` | envelope (default) | not sent | @@ -1427,7 +1427,7 @@ WHERE a mutating run's changes are destined is the second, separate record — t | source | `target_root` derivation | `capture_mode` | |---|---|---| | `acting_constraint` | the child's own `task_constraint.write_root`, required to equal the genuinely ACTIVE workspace root | `delegated_snapshot` | -| `external_workspace_root` | the root task's validated active external workspace | `delegated_snapshot` | +| `external_workspace_root` | the root task's validated external workspace or selected Project room | `delegated_snapshot` | | `skill_payload` | the exact payload the fresh `skill_payload.write` binding resolved | `delegated_snapshot` | | `readonly` | ordinary active root (nothing to write) | `none` | @@ -1441,7 +1441,7 @@ At terminal, `delegate_wait` captures the run's diff against the baseline durabl **The stored `delegated_runs_unreconciled` projection is healed only from the write side, at four seams.** `delegated_custody_unreconciled` is the disclosure a task carries when it wrote its terminal result while one of its OWN delegated runs was neither applied nor rejected. The overlay is ADDITIVE (`outcomes.custody_debt_axes`): it merges an objective WARNING and sets the top-level reason code, and it never rewrites the derived execution, review, objective or artifacts axes — so the paid verdicts the task actually earned survive. A rail truncation code from `BEST_EFFORT_REASON_CODES` keeps the single Reason line, since a round-limit or budget-exhausted victim is more usefully labelled by its truncation. Such a task reads Done with warnings with `Reason: delegated_custody_unreconciled` rather than Failed; the debt list itself lives on `delegated_runs_unreconciled` plus the `delegate_terminal_reconciliation` envelope, and sweep refresh/backfill heals it once anyone disposes, while the Reason line stays historical. Two disclosed residuals: a provider-death terminal carrying custody debt still shows the custody code on its one Reason line while remaining Failed (the devtools-only infra codes are not in the runtime best-effort set), and a task with a CLEAN derived outcome plus custody debt no longer increments `evolution_consecutive_failures`, because a warning is not a failure. `project_dialogue` derives failed/degraded from axis STATUSES only and never reads `objective.warning`, so a custody-debt child reads as a plain success in project dialogue even though its own card and the Main/Project rows say Done with warnings. Readers serve the stored projection (projection-over-replay, no live custody join), so a run settled after its task's terminal write stays stale until: the periodic sweep refreshes tasks named in its own reconcile outcomes; the boot backfill (`delegate_terminal.backfill_terminal_reconciliations`, once per generation) re-audits stored terminal rows still carrying a disclosure — the generation-crossing residual; the cursor pass (`delegate_terminal.refresh_recently_settled_terminals`, a durable byte-offset cursor over the append-only custody log, bounded per tick) catches terminal-boundary settlements no outcome-driven refresh can reach; or a kill path clears a stale list. Every refresh is audit-only — it never rewrites `reason_code` or recomputes the frozen `delegated_runs_*` counters, so a healed row may honestly read `unreconciled: []` beside `settled: 0`; current liveness lives in the `delegate_terminal_reconciliation` envelope, and patch debt survives every refresh as `patch:`, never a blind clear. -Disclosed delegated-isolation residuals (deliberate): a run whose owner's terminality cannot be proven (its task result is missing, unreadable, or still an unreaped `running` row) keeps its target locked and cannot be disposed at all — there is no time-based release; direct chat (`operator_control`) is not a top-level principal, so an orphan is disposed from a task turn rather than from Main chat; a live top-level task with a different active root may reject and release another dead task's snapshot, and the disposition row records who did it; an orphan disposed by a non-owner writes its verdict artifact and `delegate_run_patch_verdict` row under the DISPOSER's task while the capture directory and the `PATCH_DISPOSED` row keep the OWNER's task id — readers key verdicts by `run_` so nothing breaks, but the owner's acceptance packet will not list that verdict; a GC-lost snapshot or permanently failing capture discloses in a typed refusal that its obligation can never be satisfied; the baseline is worktree-primary; the git lock is task-drive scoped, so two nannies integrating into the SAME external tree can interleave apply+stage (the drift check makes the loser's apply a typed conflict); and a credential-shaped file the CHILD creates fails the whole patch rather than shipping a partial diff. +Disclosed delegated-isolation residuals (deliberate): a run whose owner's terminality cannot be proven (its task result is missing, unreadable, or still an unreaped `running` row) keeps its target locked and cannot be disposed at all — there is no time-based release; a live top-level task with a different active root may reject and release another dead task's snapshot, and the disposition row records who did it; an orphan disposed by a non-owner writes its verdict artifact and `delegate_run_patch_verdict` row under the DISPOSER's task while the capture directory and the `PATCH_DISPOSED` row keep the OWNER's task id — readers key verdicts by `run_` so nothing breaks, but the owner's acceptance packet will not list that verdict; a GC-lost snapshot or permanently failing capture discloses in a typed refusal that its obligation can never be satisfied; the baseline is worktree-primary; the git lock is task-drive scoped, so two nannies integrating into the SAME external tree can interleave apply+stage (the drift check makes the loser's apply a typed conflict); and a credential-shaped file the CHILD creates fails the whole patch rather than shipping a partial diff. **A mutating run asks for containment, reads back what it got, and DISCLOSES the gap instead of refusing the work.** In place because Claudexor otherwise hands the harness the operator's real `$HOME` — which holds the daemon token, so a compromised child could start its own runs at any access level. Four facts, one mechanism. (1) The `execution.delegated: true` marker rides in the same record as `isolation: live`, built from `delegated_run_shape` in one place, so one cannot be sent without the other. (2) The version floor is about the SCHEMA: `config.CLAUDEXOR_DELEGATED_MARKER_MIN_VERSION` (3.3.0) is the oldest engine whose strict `RunExecution` accepts `delegated`; below it the start is a 400 and no run exists, and `route_health` gives dispatcher and nanny the identical typed blocker before a token is spent. Read-only delegation sends no marker and keeps the 3.2.0 transport floor; an engine between the two floors serves read-only and refuses mutating. The floor is a schema fact, never a containment fact — the OS boundary is platform-dependent while a build declares one version everywhere; threat model and measured bands: `docs/DELEGATED_ADMISSION.md`. (3) What was APPLIED is asked of the attempt, never of the OS: `delegate_wait` reads `harness_home_isolated`, `confinement_mechanism`, and the proven denied path from the attempt record (`gateways.claudexor.attempt_containment`); `sys.platform` appears nowhere in the decision. (4) A missing boundary is disclosed (durable `delegate_run_unconfined` event, the child's instructions, the parent's terminal payload), not refused — the child already holds a shell in this worktree, and cutting the lane on every boundary-less host costs more than the marginal step it prevents. A recorded FALSE is still a fault: `harness_home_isolated: false`, or a scoped home that IS the operator's own, cancels as a typed containment fault, exactly like a widened access profile — those two exact facts are the WHOLE breach rule; a MISSING home fact is neither breach nor proof, so unproven is REPORTED. @@ -1520,15 +1520,15 @@ Plan review, task acceptance, commit review, and deep self-review answer differe #### Plan construction and review -`plan_task` reviews an INTENTION before the work starts — the same organ for code, research, deliverables, and actions (BIBLE P3). The submitted envelope carries the goal, the plan prose, and a typed domain-neutral SPEC (`in_scope`, `non_goals`, `acceptance_claims`, `invariants`, `decisions` with rejected alternatives, `deferred`, `affected_resources`, `evidence`); `ouroboros/tools/plan_spec.py` normalizes it, mints the ids that are the only valid `breaks` targets (`goal`, `claim_N`, `invariant_N`, `decision_N`, `deferred_N`) — host-minted positionally, because a caller-chosen id could shadow another target and corrupt what a blocking finding `breaks` — and hashes it. Governance documents always come from the system repository; declared targets and evidence resolve against `active_repo_dir_for(ctx)`, and a path escaping the active subject or an unreadable root is a named omission, never a silent gap. +`plan_task` reviews an INTENTION before the work starts — the same organ for code, research, deliverables, and actions (BIBLE P3). The submitted envelope carries the goal, the plan prose, and a typed domain-neutral SPEC (`in_scope`, `non_goals`, `acceptance_claims`, `invariants`, `decisions` with rejected alternatives, `deferred`, `affected_resources`, `evidence`); `ouroboros/tools/plan_spec.py` validates and normalizes its full operative content without shortening strings or dropping excess items, then mints the ids that are the only valid `breaks` targets (`goal`, `claim_N`, `invariant_N`, `decision_N`, `deferred_N`) — host-minted positionally, because a caller-chosen id could shadow another target and corrupt what a blocking finding `breaks` — and hashes it. Governance documents always come from the system repository; declared targets and evidence resolve against `active_repo_dir_for(ctx)`, and a path escaping the active subject or an unreadable root is a named omission, never a silent gap. ONE structural fact tiers the governance pack: `constitutional` is true iff a declared `affected_resources`/`evidence` PATH locator resolves under the Ouroboros system repository (an `evidence` path only if it exists). A constitutional plan carries BIBLE.md and ARCHITECTURE.md in full (an `api_chat` row inline; a retrieving `agent_session` row as mandatory full reads) — the constitutional packet is not tiered, so assembling it without either is a typed failure, never a disclosure — and every other plan carries the runtime heading-derived navigation maps (`context_layout.generate_doc_nav_map`, never a copy) plus resolvable pointers. There is no plan-kind taxonomy, no agent-declared `plan_class`, no planning scouts, and no plan Atlas. Declared evidence is resolved by `ouroboros/tools/plan_evidence.py` against exactly two allowed roots — the active workspace and the system repository — with the shared sensitive-name policy applied to locator and target; every refused, missing, truncated, oversized, binary, or URL locator becomes a typed omission row in the manifest (the host never fetches a URL), and a `need_evidence` locator a reviewer names is attached by the host on the next cycle through the same policy. What that policy cannot attach is a named `[reviewer-requested]` omission row the panel is dispatched with and judges with, never a reason to run no reviewer; the evidence continuation uses a fresh full-packet dispatch only when no exact artifact reference exists, disclosed per slot as a `capability_delta`. An unreadable referenced artifact instead fails closed with `plan_review_exact_artifact_unavailable` and never mints replacement authority. A locator may carry an exact range — `::lines=A-B`, `::bytes=A-B`, `::tail=N`, or `::symbol=Name` for `.py` sources — and a source above the per-item byte bound is attached head-first with the cut named. The manifest hash joins the spec hash and `constitutional` in the wave fingerprint, so changing what the reviewers can see changes the identity of the review. -`ouroboros/tools/plan_packet.py` builds the lean packet (objective, spec with ids, plan prose, evidence plus omissions table, bounded exploration log; on cycle 2+ the previous findings, dispositions, and spec delta). Slots come from `reviewer_slot_config` over the existing `review_execution._review_route_executor` seam; a retrieving reviewer is recorded `host_file_read_attestation: unobserved`. Reviewers return ONLY a typed findings array (`blocking` with a `breaks` id · `note` · `need_evidence` with a locator); the HOST validates membership, demotes invalid or repeated findings with disclosure, keeps failed slots in the quorum denominator, and computes the aggregate through `config.adaptive_quorum`. No reviewer emits GREEN as authority and no reviewer writes a competing plan. +`ouroboros/tools/plan_packet.py` keeps the current objective, spec with ids and plan prose complete; evidence, exploration and prior-cycle views retain their separate disclosed bounds. Tail-only requirement changes alter canonical plan identity, and actual per-slot fit refuses an unfit packet rather than removing current requirements. Slots come from `reviewer_slot_config` over the existing `review_execution._review_route_executor` seam; a retrieving reviewer is recorded `host_file_read_attestation: unobserved`. Reviewers return ONLY a typed findings array (`blocking` with a `breaks` id · `note` · `need_evidence` with a locator); the HOST validates membership, demotes invalid or repeated findings with disclosure, keeps failed slots in the quorum denominator, and computes the aggregate through `config.adaptive_quorum`. No reviewer emits GREEN as authority and no reviewer writes a competing plan. -`plan_review_state` v2 inside the root task result is the bounded durable authority: each wave records the frozen spec (whose `acceptance_claims` bind task acceptance through `contracts/task_contract.effective_acceptance_claims`), the hashes, `constitutional`, the validated findings, the aggregate, the dispositions, and whether the wave was paid; recent waves stay full, older ones compact with an explicit omitted count. A paid actor still physically in flight keeps the wave open as `DEGRADED` with `review_late_result_pending` even when settled rows meet the arithmetic quorum, so a late blocking result cannot arrive after a false GREEN. A v1 record is read-only (`legacy_v1_projection`; an OPEN v1 wave projects `legacy_open_requires_resubmission`, never auto-closed). +`plan_review_state` v2 inside the root task result is the bounded durable index: each wave records a hash-bound `task_source` reference to the full operative spec, the hashes, `constitutional`, validated findings, aggregate, dispositions and paid state. Current authority readers restore the full spec before comparing plans or binding its `acceptance_claims` through `contracts/task_contract.effective_acceptance_claims`; historical readers resolve their selected wave. Raw operative text uses the existing source-handle store, while review evidence remains redacted. The evidence manifest remains in the full wave artifact instead of being duplicated in the bounded index. Compaction and child promotion retain both references; an unavailable recorded source is a typed infrastructure failure, never empty claims or a new unpaid cycle. Legacy inline specs remain readable; older waves compact with an explicit omitted count. A paid actor still physically in flight keeps the wave open as `DEGRADED` with `review_late_result_pending` even when settled rows meet the arithmetic quorum, so a late blocking result cannot arrive after a false GREEN. A v1 record is read-only (`legacy_v1_projection`; an OPEN v1 wave projects `legacy_open_requires_resubmission`, never auto-closed). Closure follows the finding class: GREEN closes; REVIEW_REQUIRED closes its notes/`need_evidence` through a disposition-only `plan_task` call (no model call, no cost), while a below-quorum blocking finding stays open until the spec changes or a paid delta cycle judges its rejection; REVISE_PLAN can never be closed by disposition — the agent changes the spec (new fingerprint, next paid cycle) or rejects a blocking finding with a rationale that rides into that cycle. Paid cycles are bounded by the shared `OUROBOROS_REVIEW_MAX_CYCLES`; an identical envelope replays the recorded wave for free, except that an open wave whose blocking findings all carry valid reject dispositions has earned exactly one more paid delta panel. A wave is paid iff at least one reviewer slot was physically dispatched; only a nothing-dispatched wave of typed $0 skip rows stays unpaid and never replaces a paid predecessor. Before fan-out the engine captures one panel health snapshot (`subagents.route_health`, route-level evidence): a slot with positive structural evidence of a spent lane becomes a $0 typed skip row that stays in the denominator; unknown health dispatches (fail-open). The wave records the health epoch and reviewer-roster fingerprint, and a recorded DEGRADED wave replays free only under an identical envelope, matching epoch, and unchanged roster — otherwise it re-dispatches a PAID panel (replay/epoch casuistry: `plan_review.py` docstrings). When the wave's own typed rows prove the quorum structurally unreachable, the wave carries `quorum_unreachable` plus the earliest recorded reset, and under blocking enforcement the finalization gate RELEASES while the review stays open: the agent may finalize `blocked_with_evidence` (reason `plan_review_quorum_unreachable`), wait through a one-shot `schedule_followup`, or ask the owner — the host adds facts only, never an answer template. Under blocking enforcement an open wave otherwise holds implementation and an exhausted cap escalates with the typed `review_cycles_exhausted` reason; under advisory the agent may proceed with the wave open (one typed owner-visible `plan_review_advisory_open` event plus the loud disclosure at finalization). Unavailability, invalid state, budget refusal, and deadline rails remain typed non-authoritative attempts, never substitutes for GREEN. There is therefore no pre-dispatch refusal for a reviewer request the host cannot attach: the only $0 exits are the typed attempts named here and the typed `not_dispatched` slot rows. @@ -1552,7 +1552,7 @@ A reflection lands where it durably belongs: a non-project root appends the full Only the root runs full post-task synthesis once (`root_phase_checkpoint` makes the paid phase at-most-once across restart); children contribute evidence, never a second global synthesis. The owner's final answer does not wait for blocking synthesis: after the durable result is stored, the final `send_message` is delivered immediately while the buffered-return copy is RETAINED (queue.put is not a delivery receipt); both copies carry one `delivery_id` and the supervisor suppresses the duplicate through the durable registry in `supervisor/terminal_delivery.py`. The same file holds a bounded PENDING outbox — ONE seam for the normal, cancel, and reap terminal paths: a terminal answer is recorded as owed BEFORE it is enqueued and cleared in the same write that marks it delivered, so a crash between settle and send replays it instead of losing it. Every non-ephemeral root's answer enters this outbox at durable-result persistence time under the canonical `final::` id. Replays back off and are bounded; an exhausted or capacity-evicted row is dropped LOUDLY (full text preserved on disk, typed `terminal_delivery_exhausted` event, chat notice) — external transports stay at-least-once and that residual is disclosed. `task_done` still goes last through the buffered return — an early `task_done` would release the queue slot and start child-drive cleanup while post-task still runs — so a worker reaped during a hung synthesis has already delivered the answer. Synthesis receives a sealed final package from the durable result: submitted final text, its artifact manifest, and completion_observations. The terminal writer preserves full redacted action observations in the canonical artifact store (`task.budget_drive_root or drive_root`) before compact publication. Completion sources use the existing write-once `source_handles/context_checkpoints` store with verified `task_source` refs; the published-ref closure includes completion observations before child cleanup. Sources stay outside deliverables and inferred readiness. Their native reader selector uses `get_task_result(include_completion_source=true)` with the source task id and its canonical drive. It first returns complete character length/hash, then explicit `source_start_char`/`source_end_char` ranges; `artifacts.text_source_range_projection` shares the unchanged work-order range contract. Source bytes, kind, path containment and SHA are checked before any excerpt. Packet-only summary/reflection receive per-send-tool counts, each family's latest recorded return, and task-related skill readiness with coverage; full-source references are for later readers, not evidence the synthesizer has read. Positive observed facts correct error-trace impressions, while tool success does not prove owner receipt, empty material does not prove absence, and skill readiness does not attribute an owner's action to the task. Recovery uses these stored observations; task-summary requests/responses use chat_observed. -Pooled workers retain their slot until root post-task synthesis settles, for API-only and subscription tasks alike; early final-answer delivery keeps the response independent from that queue timing. Direct/server post-work remains detached and binds a live `TaskModelWait` owner in the existing `POST_TASK_SYNTHESIS_INFLIGHT` registry. Temporary role overrides follow that owner, and the existing task mailbox remains available until the terminal post-task checkpoint. Gateway decisions and activity hydration use the same live owner and `root_phase_checkpoint`; an open phase remains finalizing rather than appearing completed. Typed quota/auth waits resume only the unsettled call, while stop or unknown outcomes degrade the phase without repeating finished stages. Restart recovery still degrades an indeterminate `running` phase rather than replaying a possibly paid request. +Pooled workers retain their slot until root post-task synthesis settles, for API-only and subscription tasks alike; early final-answer delivery keeps the response independent from that queue timing. Ordinary native post-work stays on its registered actor thread without a pooled worker slot. Its existing `TaskModelWait` owner remains available through `POST_TASK_SYNTHESIS_INFLIGHT` after ordinary dialogue admission closes; detached server post-work binds its own live owner in the same registry. Temporary role overrides follow that owner, and the existing task mailbox remains available until the terminal post-task checkpoint. Gateway decisions use the phase owner independently of closed dialogue admission, and activity hydration enriches the existing direct row with its finalizing state and model waits. An open phase remains finalizing rather than appearing completed. Typed quota/auth waits resume only the unsettled call, while stop or unknown outcomes degrade the phase without repeating finished stages. Restart recovery still degrades an indeterminate `running` phase rather than replaying a possibly paid request. #### Durable memory and project focus @@ -1560,7 +1560,7 @@ Pooled workers retain their slot until root post-task synthesis settles, for API Ouroboros remains one identity across Main, project rooms, and Background Consciousness. A project is a focused working room, not an isolated sub-mind: unified dialogue memory remains available to the one agent, while an executing project task preferentially receives its own thread, journal, workpad, and project knowledge. `project_facts.py` routes project facts to `projects//knowledge`; subagents inherit the root's resolved project id and never derive a new one; there is no per-project identity or scratchpad. -The projects registry owns immutable project identity, canonical chat id, optional working directory, lifecycle/tombstone state, routing generation, and activity revision. Admission persists the resolved project id in the task itself, and `project_lease.py` permits one top-level writer per project while allowing that task's own subagent tree; binding/history files support routing and presentation, not the lease. Delete closes routing, cancels/quiesces the tree, and tombstones only after settlement, preserving everything for recovery. `ensure_project_scope` can create or bind the current root to one project mid-execution: it persists the durable registry binding first, then marks the live queue/lease surface under the queue lock so the lease recognizes the running task as a lane occupant; it is idempotent for the same project, refuses a second scope, and cannot be invoked by a child to escape the inherited scope. +The projects registry owns immutable project identity, canonical chat id, optional working directory, lifecycle/tombstone state, routing generation, and activity revision. Admission persists the resolved project id in the task itself. `project_lease.py` serializes assignment of pooled roots by Project while allowing their own subagent trees; it is not a physical-folder lock and does not withhold tools from ordinary conversation. Binding/history files support routing and presentation, not the lease. Delete closes routing, cancels/quiesces the tree, and tombstones only after settlement, preserving everything for recovery. `ensure_project_scope` can create or bind the current root to one project mid-execution: it persists the durable registry binding first, then marks the live queue/lease surface under the queue lock so the lease recognizes the running task as a lane occupant; it is idempotent for the same project, refuses a second scope, and cannot be invoked by a child to escape the inherited scope. Project `journal.jsonl` records curated milestones and `workpad.md` retains active working context; focused context includes the workpad in full and recent journal rows with a visible pointer to older entries. On root completion, only high-signal blockers, questions, and interface contracts are mirrored once from the ephemeral task-tree ledger into the durable journal, and a finished root whose effective working tree is not the registered `working_dir` writes one typed "work lives at @ " journal row from facts the task record already holds. A project digest gives consciousness a concise completion signal without pretending to be the raw project memory. @@ -1590,7 +1590,7 @@ Catalog updates disclose the exact current and proposed version strings in the t ### MCP and browser-facing external tools -`mcp_client.py` owns configured HTTP/SSE and local stdio MCP discovery and invocation. HTTP/SSE entries validate URLs and auth headers; `secret_masking.py` owns the shared exact MCP token placeholder shapes; load-time placeholder repair runs before environment precedence, so a real environment credential is not mistaken for a wire mask, and it is limited to recognized top-level Settings secrets — nested MCP values are never silently migrated. Stdio entries pass one executable `command` and an exact string `args` list directly to the MCP SDK without a shell; optional `cwd`, literal `env`, and `env_from_settings` (environment-name → existing setting-key references) resolve from the same saved configuration for discovery and calls. References override matching literal names; omission preserves SDK defaults. Settings classifies referenced values as ordinary or secret; only secret values receive exact diagnostic masking, including JSON-escaped echoes and short secrets. Unknown fields are retained with a visible not-applied warning while a valid server remains usable; invalid known fields or references produce `MCP_CONFIG_ERROR`. The Settings response-only `auth_configured` flag never becomes configuration. The UI preserves both environment forms and user extras. The SDK context owns process shutdown. Discovered tools join the selected initial capability envelope on every lane — the ephemeral decision turn behind Main and project chat included, alongside enabled, granted extension tools — behind the same `network` resource gate as on a managed task, and `schemas()`, `get_schema_by_name`, and `execute` agree on that; discovery failure produces an explicit capability omission through `list_available_tools`, never a silent removal. Descriptions and results remain untrusted data, and every call still crosses registry, resource, safety, timeout, and result-handling policy. +`mcp_client.py` owns configured HTTP/SSE and local stdio MCP discovery and invocation. HTTP/SSE entries validate URLs and auth headers; `secret_masking.py` owns the shared exact MCP token placeholder shapes; load-time placeholder repair runs before environment precedence, so a real environment credential is not mistaken for a wire mask, and it is limited to recognized top-level Settings secrets — nested MCP values are never silently migrated. Stdio entries pass one executable `command` and an exact string `args` list directly to the MCP SDK without a shell; optional `cwd`, literal `env`, and `env_from_settings` (environment-name → existing setting-key references) resolve from the same saved configuration for discovery and calls. References override matching literal names; omission preserves SDK defaults. Settings classifies referenced values as ordinary or secret; only secret values receive exact diagnostic masking, including JSON-escaped echoes and short secrets. Unknown fields are retained with a visible not-applied warning while a valid server remains usable; invalid known fields or references produce `MCP_CONFIG_ERROR`. The Settings response-only `auth_configured` flag never becomes configuration. The UI preserves both environment forms and user extras. The SDK context owns process shutdown. Discovered tools join the selected initial capability envelope including explicit ephemeral decision turns and enabled, granted extension tools, behind the same `network` resource gate as on a managed task, and `schemas()`, `get_schema_by_name`, and `execute` agree on that; discovery failure produces an explicit capability omission through `list_available_tools`, never a silent removal. Descriptions and results remain untrusted data, and every call still crosses registry, resource, safety, timeout, and result-handling policy. Web-tool prohibition and network prohibition are distinct: `web=false` alone does not disable configured MCP or extensions, while an explicit `network=false` and tool disables remain enforced at discovery and dispatch. Browser tools are stateful and thread-sticky because Playwright sessions and greenlets have thread affinity; they cannot be scheduled as ordinary parallel stateless calls. A stateful-tool timeout therefore RETIRES the browser generation: the shared `browser_state` slot is replaced, the abandoned worker keeps writing only into its retired one, and the close is queued on the retiring executor so it runs on the owning worker thread when the hung call settles. Retired sessions are bounded in-process at `_RETIRED_GENERATIONS_MAX` per task, after which opening another session is a typed `BROWSER_BACKLOG_RETIRED_SESSIONS` refusal; generation isolation is best-effort under concurrent replacement (a fully closed class needs a process-isolated browser worker — disclosed future design). Model-driven in-page evaluation goes through `_evaluate_bounded`: Playwright's `evaluate` accepts no timeout, so the expression is raced against an in-page rejection, bounding the ASYNC class honestly and no further — a synchronous event-loop block cannot be interrupted from inside the page, and the outer tool timeout remains its backstop. (The one direct `page.evaluate` is `_wait_for_page_paint`'s immediate paint-flag setter, paired with a 500 ms bounded `wait_for_function`.) The caller's timeout also becomes the session default (`page.set_default_timeout`), floored on the action path so the five-second action default cannot strangle a capture. Chromium is the default; WebKit and device descriptors are targeted tools for a real Safari/iOS risk, not a universal acceptance matrix and not a claim that a narrow Chromium viewport is Safari-equivalent. First-party PR helpers are normal built-ins whose mutating operations remain subject to selected-root policy, runtime mode, delegated-child constraints, credentials, and reviewed-publication authority. diff --git a/docs/CHECKLISTS.md b/docs/CHECKLISTS.md index 67122a39e..565b52d6e 100644 --- a/docs/CHECKLISTS.md +++ b/docs/CHECKLISTS.md @@ -178,7 +178,7 @@ Used by `commit_reviewed` for all changes to the Ouroboros repository. | 20 | context_budget_ssot | If the diff changes context-size budgets/constants (`ouroboros/context_budget.py`), the context layout/manifest, a section's tier/policy, or the typed ContextFit deficit/reclaim contract: does it keep the low/max context split coherent (single SSOT + both profiles + docs + drift-guard tests in sync), preserve the tier-0 always-full core (BIBLE/SYSTEM/identity/scratchpad/knowledge-index/recent-dialogue) in EVERY mode, use a visible on-demand pointer instead of silent truncation (P1), and leave the blocking scope-reviewer >=1M floor untouched wherever scope review applies? Since v6.80.0 the owner-only `OUROBOROS_CONTEXT_MODE` ALSO decides scope-review applicability (`max`: blocking ≥1M gate; `low`: declaredly not performed with a typed skip row), so any change that widens what `low` mode implies, or that lets the AGENT reach that setting, is an immune-system change under P3 — not a context-budget tweak. (PASS with "Not applicable" if no context-budget/layout change.) | critical | | 21 | capability_regression | Does the diff REMOVE or NARROW a previously-supported user-facing behavior or capability — a tool/flag/mode/path that worked before now errors or is gated tighter (e.g. a new `is_dir`/existence guard that blocks a legitimate create, a tightened allowlist that drops a real path, a removed fallback)? If so, is it INTENTIONAL and disclosed as a breaking/capability change in the commit message + changelog? Accidental capability removal is the failure class this item names. Ask whether a golden "from zero" test would have caught it. **Guard-change trigger (executable requirements, not an essay):** ADDING or TIGHTENING a guard, filter, allowlist, or deny rule IS a capability change and fires this item. For such a diff the reviewer must verify two things: (a) the diff STAGES A POSITIVE TEST that exercises a legitimate flow THROUGH the new guard and proves it still succeeds — a negative "it blocks X" test alone is insufficient (a gate can pass its own probe while breaking every real run); (b) the diff or its disclosure NAMES THE SURVIVING POSITIVE PATH — the concrete actor and flow that still work after the change. A guard change that stages no surviving-path test is a capability-regression finding, not a safety improvement. **Owner acceptance:** a narrowing counts as OWNER-ACCEPTED only when a GREEN plan review explicitly names that narrowing; owner acceptance makes the finding advisory (disclosed, non-blocking). Intent wording, a commit-message disclosure, or a changelog row alone is disclosure, NOT acceptance. Severity follows the `Critical surface whitelist` below — silently removing a documented capability or a safety/release contract is critical; an owner-accepted narrowing or an internal-only refactor is advisory. **Standing disclosures for this item live in `docs/CHECKLISTS_ARCHIVE.md`** (owner-accepted removals/narrowings and standing notes); they remain binding on every reviewer — consult that file before raising a removal/narrowing finding on a surface it covers, and do not re-raise anything recorded there. | advisory | | 22 | cache_friendliness | If the diff builds or reorders LLM prompt/context content (context builders, review prompt assembly, message construction in `llm.py` callers): does it keep prompt caching intact — stable governance/policy content BEFORE dynamic evidence, no dynamic values (timestamps, hashes, round counters, task ids) injected into a stable cached prefix, and no removal/breakage of existing `cache_control` markers or session/cache affinity keys? A change that silently fragments an existing cached prefix re-bills the full prompt on every repeat call. (PASS with "Not applicable" if no prompt/context assembly changed.) | advisory | -| 23 | delegated_transport | If the diff touches the delegated execution/review transport (ouroboros/subagents.py dispatch/route-health, ouroboros/tools/delegate.py, ouroboros/delegate_custody.py, ouroboros/delegate_progress.py, ouroboros/review_execution.py session executors, ouroboros/gateways/claudexor.py, ouroboros/claudexor_daemon.py, ouroboros/claudexor_runtime.py), does it preserve the delegation invariants: capability reductions reach all three destinations (durable envelope, child prompt, parent result — D4); a result counts as received only after a hash-bound read to EOF and retries replay the recorded byte-identical body (D7); the exact selected subagent_id snapshot starts its exact session route with custody-durable requested→effective evidence or returns a TYPED refusal, and neither host dispatch nor tool preflight substitutes another session/API/native route — any fallback is a new explicit LLM selection; quota exhaustion needs POSITIVE evidence judged against the route's own model (applies_to_models scoping, absence = unknown = usable); delegated spend settles through custody with unknown-never-rendered-as-zero and root/parent lineage; and no vendor/harness name is ever branched on in core. For configured external work orders, verify the one total 250,000-character cap, byte-complete fitting orders, full-SHA/source-selector partial lenses only on a positively interactive route, typed cannot-verify refusal otherwise, byte-identical pending recovery of the stored compact body, durable STARTED/replay source state, an actor-readable `get_task_result` canonical range whose renderer is shared with exact source_response verification, a retry after `already_resolved` only when a durable prior delivered event binds the same interaction and exact source selector, terminal cannot_verify before complete interval coverage, and apply refusal until coverage is complete (reject remains available); a disclosed partial lens must never be treated as a complete contract. (PASS with "Not applicable" if no delegated-transport surface changed.) | critical | +| 23 | delegated_transport | If the diff touches the delegated execution/review transport (ouroboros/subagents.py dispatch/route-health, ouroboros/tools/delegate.py, ouroboros/delegate_custody.py, ouroboros/delegate_progress.py, ouroboros/review_execution.py session executors, ouroboros/gateways/claudexor.py, ouroboros/claudexor_daemon.py, ouroboros/claudexor_runtime.py), does it preserve the delegation invariants: capability reductions reach all three destinations (durable envelope, child prompt, parent result — D4); a result counts as received only after a hash-bound read to EOF and retries replay the recorded byte-identical body (D7); the exact selected subagent_id snapshot starts its exact session route with custody-durable requested→effective evidence or returns a TYPED refusal, and neither host dispatch nor tool preflight substitutes another session/API/native route — any fallback is a new explicit LLM selection; quota exhaustion needs POSITIVE evidence judged against the route's own model (applies_to_models scoping, absence = unknown = usable); delegated spend settles through custody with unknown-never-rendered-as-zero and root/parent lineage; and no vendor/harness name is ever branched on in core. For direct and configured external work orders, verify complete chosen assignments and host authority across instruction roles, no arbitrary compiler cap or compulsory file/question transport, and no duplicate objective/output copies within instructions. Operative plan normalization and current reviewer inputs must preserve full content and tail-sensitive identity. Real route limits retain original input, cause and execution state. Legacy partial requests retain byte-identical pending recovery, exact renderer/selector/digest/range validation, durable source coverage and apply refusal until complete; reject remains available. Source availability never proves reading or comprehension. (PASS with "Not applicable" if no delegated-transport surface changed.) | critical | | 24 | perf_lifecycle | If the diff adds or changes an endpoint, poller, subscription, or timer, or reads a growing store (JSONL log, ledger, event table): does any interaction-path read scan an unbounded store per request/message/tick (a full-table read filtered in code is such a scan)? Is a subscription/observer/interval/listener added without a paired disposer? Does O(history) work run on a poll/stream path? Does a GET handler perform new steady-state durable writes outside the two named exceptions? The authoritative definitions are DEVELOPMENT.md "Invariant: Projection over replay (hot readers of growing stores)" and "Invariant: UI resources carry a disposer" — check against those, do not re-derive them here. For an embedded or framed UI surface, also use DEVELOPMENT.md "Invariant: Embedded surfaces declare geometry and refresh semantics" for host geometry/overflow, teardown, retry/error, and real-consumer visual evidence. (PASS with "Not applicable" if the diff touches none of these surfaces.) | advisory | | 25 | source_completeness | If any changed consumer can authorize PASS, a destructive rewrite, or replacement of a full contract, does its input distinguish complete from partial and carry a source reference the same actor can resolve? Does the consumer materialize every named omitted source before the decision, or abstain with the existing typed incomplete/degraded outcome? A marker or host claim alone is never sufficient. | critical when applicable | | 26 | actor_readable_projection | If the diff adds a bounded projection, omission marker, summary, or status count, can the actor who must decide read the exact canonical source through an existing path? Verify the ref, root, generation/range or ID, and the reader's ability to resolve it; host-unattested or merely hypothetical retrieval does not certify completeness. | critical when applicable | diff --git a/docs/DESIGN.md b/docs/DESIGN.md index 275dbd5be..477b44e0e 100644 --- a/docs/DESIGN.md +++ b/docs/DESIGN.md @@ -443,8 +443,9 @@ this wakeup cycle ends; the persistence checkbox saves the consciousness role. An already-delivered answer does not close a still-open post-task synthesis. Reflection or consolidation waits use the same role controls in that task's existing finalizing card. Pooled tasks, including API-only tasks, keep their -worker slot until post-work settles; the answer arrives early. Detached direct -chat post-work holds no worker slot. Both claims follow the host's live owner +worker slot until post-work settles; the answer arrives early. Ordinary native +chat post-work holds no worker slot; its existing card keeps the same live +post-task model-wait controls after ordinary dialogue admission closes. Both claims follow the host's live owner and post-task checkpoint, not the presence of answer text or a cost estimate. Failed main work stays visibly failed after history reload while post-work controls remain live; the unfinished checkpoint never erases the outcome. diff --git a/docs/DEVELOPMENT.md b/docs/DEVELOPMENT.md index 80ce9e18d..6c5c14248 100644 --- a/docs/DEVELOPMENT.md +++ b/docs/DEVELOPMENT.md @@ -1775,11 +1775,17 @@ both critical. The imperatives: self-report); a substrate swap is a disclosed incomplete execution, never a silent vendor/API fallback (`tests/test_configured_session_prestart.py`). -- Work orders: one total 250,000-character wire limit, byte-complete or — - only on a route whose live manifest declares a question channel — a - compact source-request lens. Reader and validator share one renderer so - the bytes the actor sees are exactly the bytes the host verifies; the - manifest observation is a preflight, not a lease. `subagents.route_health` +- Work orders carry the complete chosen assignment and host authority without + a compiler-size cutoff or compulsory question/file transport. Preserve the + instruction roles and avoid duplicating objective/output inside the host + authority. Real transport/provider refusals retain their cause, original + input and execution custody; recovery is the model's choice. Legacy partial + runs keep their exact renderer, source-range validation and stored-body retry. + Operative plan text and canonical identity remain complete. Keep full specs + in existing task source handles and only their references in the bounded + review-state index; restore them for acceptance and plan comparisons. + Redacted review evidence never substitutes for the original requirement text. + `subagents.route_health` is the ONE route reader for every consumer; quota readers project one `ClaudexorGateway.quota_state()` envelope (`tests/test_available_subagents_runtime.py`). A fully-used ratio without a diff --git a/docs/v7next/FACADE_INVENTORY.md b/docs/v7next/FACADE_INVENTORY.md index a6a18fdd0..4d959dae7 100644 --- a/docs/v7next/FACADE_INVENTORY.md +++ b/docs/v7next/FACADE_INVENTORY.md @@ -2,7 +2,7 @@ AST-derived inventory of compatibility facades, regenerated by `python scripts/regenerate_inventories.py`. Do not edit. A facade row is any runtime module whose top-level `from import ...` statements carry the `noqa: F401` re-export marker — the codebase's declared "this binding exists for its binding, not for this module's own use" convention (reference FACADE_CONSUMERS method). Leaf domains come from `ouroboros/domains.toml`; a leaf outside the facade's domain is marked ✗ (that edge also appears in the manifest's pinned direction matrix). `tests/test_generated_inventories.py` pins byte-identity, so any re-export surface change must regenerate this file. -- facade modules: **56**; marked re-export bindings: **2331**; cross-domain facade→leaf pairs: **130** +- facade modules: **56**; marked re-export bindings: **2329**; cross-domain facade→leaf pairs: **130** | facade | domain | bindings | leaves | |---|---|---:|---| @@ -38,9 +38,9 @@ AST-derived inventory of compatibility facades, regenerated by `python scripts/r | `ouroboros/tool_access.py` | D04 | 42 | `ouroboros/contracts/task_constraint.py` (2 ✗D19)
`ouroboros/tool_access_paths.py` (10)
`ouroboros/tool_access_roots.py` (9)
`ouroboros/tool_access_types.py` (15)
`ouroboros/tool_access_user_files.py` (4)
`ouroboros/tool_capabilities.py` (2) | | `ouroboros/tools/claude_advisory_review.py` | D06 | 51 | `ouroboros/commit_admission.py` (3)
`ouroboros/deadline_utils.py` (2 ✗D01)
`ouroboros/skill_review_status.py` (1 ✗D14)
`ouroboros/tools/preflight_review_prompt.py` (7)
`ouroboros/tools/preflight_review_run.py` (19)
`ouroboros/tools/review_helpers.py` (17)
`ouroboros/triad_review.py` (2) | | `ouroboros/tools/control.py` | D08 | 111 | `ouroboros/config.py` (4 ✗D12)
`ouroboros/contracts/task_contract.py` (3 ✗D19)
`ouroboros/depth_evidence.py` (1 ✗D07)
`ouroboros/headless.py` (2 ✗D17)
`ouroboros/outcomes.py` (1 ✗D01)
`ouroboros/subagent_runtime.py` (3 ✗D07)
`ouroboros/subagents.py` (2 ✗D07)
`ouroboros/task_results.py` (5 ✗D17)
`ouroboros/task_status.py` (2 ✗D17)
`ouroboros/tool_capabilities.py` (2 ✗D04)
`ouroboros/tool_policy.py` (1 ✗D04)
`ouroboros/tools/control_delegation.py` (8 ✗D07)
`ouroboros/tools/control_events.py` (9)
`ouroboros/tools/control_routing.py` (12)
`ouroboros/tools/control_runtime.py` (12)
`ouroboros/tools/control_scheduling.py` (18 ✗D07)
`ouroboros/tools/control_subagent_spec.py` (6 ✗D07)
`ouroboros/tools/control_task_results.py` (11 ✗D07)
`ouroboros/tools/registry.py` (4 ✗D04)
`ouroboros/utils.py` (5 ✗D18) | -| `ouroboros/tools/core.py` | D05 | 84 | `ouroboros/code_search_rg.py` (4)
`ouroboros/contracts/skill_payload_policy.py` (13 ✗D19)
`ouroboros/project_facts.py` (1 ✗D15)
`ouroboros/tool_access.py` (10 ✗D04)
`ouroboros/tools/core_artifacts.py` (17)
`ouroboros/tools/core_file_tools.py` (31)
`ouroboros/tools/registry.py` (3 ✗D04)
`ouroboros/utils.py` (5 ✗D18) | +| `ouroboros/tools/core.py` | D05 | 83 | `ouroboros/code_search_rg.py` (4)
`ouroboros/contracts/skill_payload_policy.py` (13 ✗D19)
`ouroboros/project_facts.py` (1 ✗D15)
`ouroboros/tool_access.py` (9 ✗D04)
`ouroboros/tools/core_artifacts.py` (17)
`ouroboros/tools/core_file_tools.py` (31)
`ouroboros/tools/registry.py` (3 ✗D04)
`ouroboros/utils.py` (5 ✗D18) | | `ouroboros/tools/core_file_tools.py` | D05 | 10 | `ouroboros/credential_shapes.py` (3 ✗D13)
`ouroboros/tools/core_secret_paths.py` (7) | -| `ouroboros/tools/delegate.py` | D07 | 60 | `ouroboros/delegate_containment.py` (5)
`ouroboros/delegate_interactions.py` (8)
`ouroboros/delegate_output.py` (14)
`ouroboros/delegate_shared.py` (3)
`ouroboros/delegate_source_coverage.py` (3)
`ouroboros/subagent_runtime.py` (2)
`ouroboros/subagent_work_order.py` (2)
`ouroboros/tools/delegate_integration.py` (14)
`ouroboros/tools/delegate_terminal_evidence.py` (9) | +| `ouroboros/tools/delegate.py` | D07 | 59 | `ouroboros/delegate_containment.py` (5)
`ouroboros/delegate_interactions.py` (8)
`ouroboros/delegate_output.py` (14)
`ouroboros/delegate_shared.py` (3)
`ouroboros/delegate_source_coverage.py` (3)
`ouroboros/subagent_runtime.py` (2)
`ouroboros/subagent_work_order.py` (1)
`ouroboros/tools/delegate_integration.py` (14)
`ouroboros/tools/delegate_terminal_evidence.py` (9) | | `ouroboros/tools/delegate_integration.py` | D07 | 7 | `ouroboros/tools/delegate_payload_patch.py` (7) | | `ouroboros/tools/git.py` | D10 | 96 | `ouroboros/config.py` (1 ✗D12)
`ouroboros/contracts/skill_payload_policy.py` (2 ✗D19)
`ouroboros/contracts/task_constraint.py` (2 ✗D19)
`ouroboros/platform_layer.py` (2 ✗D18)
`ouroboros/runtime_mode_policy.py` (7 ✗D13)
`ouroboros/tool_access.py` (3 ✗D04)
`ouroboros/tools/commit_gate.py` (13)
`ouroboros/tools/core.py` (3 ✗D05)
`ouroboros/tools/git_evolution.py` (6)
`ouroboros/tools/git_plumbing.py` (12)
`ouroboros/tools/git_repo_edit.py` (4)
`ouroboros/tools/git_review_cycle.py` (16)
`ouroboros/tools/git_vcs_ops.py` (10)
`ouroboros/tools/parallel_review.py` (3 ✗D06)
`ouroboros/tools/registry.py` (4 ✗D04)
`ouroboros/tools/review_helpers.py` (3 ✗D06)
`ouroboros/tools/review_revalidation.py` (1)
`ouroboros/utils.py` (4 ✗D18) | | `ouroboros/tools/plan_review.py` | D06 | 20 | `ouroboros/tools/plan_render.py` (3)
`ouroboros/tools/plan_review_runtime.py` (17) | diff --git a/ouroboros/agent.py b/ouroboros/agent.py index 2fe090882..7210bea0c 100644 --- a/ouroboros/agent.py +++ b/ouroboros/agent.py @@ -344,6 +344,7 @@ class OuroborosAgent: if isinstance(started, (int, float)) and started > 0 else {}), **({"queued_at": task["queued_at"]} if task.get("queued_at") is not None else {}), chat_id=task.get("chat_id"), + _is_direct_chat=bool(task.get("_is_direct_chat")), parent_task_id=task.get("parent_task_id"), root_task_id=task.get("root_task_id"), session_id=task.get("session_id"), @@ -562,7 +563,7 @@ class OuroborosAgent: _room_dir, _room_note = room_chat_lens_dir(self.env.drive_root, _resolved_project_id) if _room_dir: task_metadata["_project_room_dir"] = _room_dir - elif _room_note: + if _room_note: task_metadata["_project_room_note"] = _room_note except Exception: log.debug("room lens resolution failed", exc_info=True) diff --git a/ouroboros/agent_startup_checks.py b/ouroboros/agent_startup_checks.py index 2437e973d..27ca7b3d0 100644 --- a/ouroboros/agent_startup_checks.py +++ b/ouroboros/agent_startup_checks.py @@ -108,6 +108,12 @@ def task_result_authority_projection( receipts = _authority_verification_receipts(row, drive_root) if receipts: authority["verification_receipts"] = copy.deepcopy(receipts) + if isinstance(authority.get("plan_review_state"), dict): + from ouroboros.tools.plan_review_artifacts import authority_state + + authority["plan_review_state"] = authority_state( + drive_root, str(row.get("task_id") or row.get("id") or ""), authority["plan_review_state"], + ) return authority diff --git a/ouroboros/agent_task_pipeline.py b/ouroboros/agent_task_pipeline.py index f07f2fd4f..696fec982 100644 --- a/ouroboros/agent_task_pipeline.py +++ b/ouroboros/agent_task_pipeline.py @@ -240,6 +240,11 @@ def _run_post_task_processing_async( def _run() -> None: with usage_scope(post_scope): if blocking: + # Native post-task work retains its call owner after dialogue + # admission closes; pooled owners stay in their worker process. + if post_task_key is not None and parent_wait is not None and not parent_wait.worker_slot_held: + with _POST_TASK_SYNTHESIS_LOCK: + _POST_TASK_SYNTHESIS_INFLIGHT[post_task_key] = parent_wait _run_scoped() else: # A detached thread must not inherit its parent's closing scope. @@ -478,12 +483,9 @@ def emit_task_results( outcome_axes = normalize_outcome_axes({"outcome_axes": loop_outcome.get("outcome_axes")}) execution_status = str((outcome_axes.get("execution") or {}).get("status") or "") reason_code = str(loop_outcome.get("reason_code") or "") - # CW3 (v6.34.0): a short same-route "turn=decision" turn (ephemeral, run while the - # main agent is busy) DELIVERS its inline answer but must not leave a durable TASK - # RECORD \u2014 no task_result file, no task_eval ledger row. The cognitive-memory writes - # (reflection/consolidation/letters-home) are already gated further below; this - # closes the remaining durable task-record writes. An inline answer + card - # resolution and budget metrics still flow so the reply is visible. + # Explicit ephemeral routing delivers an inline answer and card facts + # without task-result, evaluation or cognitive post-task writes. Ordinary + # Main/Project work uses the native durable-result path. _ephemeral = bool(task.get("_ephemeral_turn")) _root_outbox = _is_root_post_task(task) # durable outbox (no model call): pre-marker predicate if getattr(ctx, "_skip_post_task_synthesis", False): # "Stop now": paid root predicates see it @@ -624,6 +626,7 @@ def emit_task_results( # CW3: tells the supervisor's task_done handler to NOT synthesize a durable # missing-result task_result for a transient decision turn (which has none). "_ephemeral": _ephemeral, + "_is_direct_chat": bool(task.get("_is_direct_chat")), # Presentation marker only. The supervisor's typed routing event remains # the action/receipt authority; the visible transient card gets no # managed-task controls (including "Turn into project"). @@ -780,7 +783,7 @@ def _dispatch_root_post_task( global_reflection_callback = functools.partial( _run_global_backlog_promotion_only, parent_env, parent_task) split_non_project = split and not project_scoped - blocking = in_worker_process() or split_non_project or ( + blocking = in_worker_process() or bool(task.get("_is_direct_chat")) or split_non_project or ( str(task.get("type") or "") == "evolution" or bool(str(task.get("workspace_root") or "").strip()) or bool(str(task.get("workspace_mode") or "").strip()) @@ -976,6 +979,7 @@ def _store_task_result(env: Any, task: Dict[str, Any], text: str, accounted_upper_bound_usd_with_children=_cost_with_children, cost_with_children_partial=_cost_partial, task_contract=task_contract, + _is_direct_chat=bool(task.get("_is_direct_chat")), loop_outcome=loop_outcome, project_id=str(task.get("project_id") or ""), parent_task_id=task.get("parent_task_id"), diff --git a/ouroboros/consciousness.py b/ouroboros/consciousness.py index 99a346001..c1cd7d46b 100644 --- a/ouroboros/consciousness.py +++ b/ouroboros/consciousness.py @@ -121,7 +121,9 @@ class BackgroundConsciousness: @property def is_paused(self) -> bool: - return bool(getattr(self, "_paused", False)) + from supervisor.active_activity import get_direct_activity_registry + + return bool(getattr(self, "_paused", False) or get_direct_activity_registry().snapshot()) def _observation_lock_for_instance(self) -> threading.RLock: """Lazily restore observation fields for object.__new__ overlap tests.""" @@ -653,7 +655,7 @@ class BackgroundConsciousness: if self._stop_event.is_set(): break - if self._paused: + if self.is_paused: self._last_idle_reason = "paused_by_active_task" continue @@ -669,11 +671,11 @@ class BackgroundConsciousness: cycle_completed = self._think() self._last_cycle_finished_at = utc_now_iso() # Preserve distinct overflow/LLM error statuses set inside _think(). - if cycle_completed and not self._stop_event.is_set() and not self._paused: + if cycle_completed and not self._stop_event.is_set() and not self.is_paused: self._last_idle_reason = "sleeping" # Retire the live card now that this cycle is done (skip while paused: # a real task is active and owns the status). - if not self._paused: + if not self.is_paused: self._emit_cycle_idle(self._last_idle_reason) except Exception as e: self._last_cycle_finished_at = utc_now_iso() diff --git a/ouroboros/context_runtime_facts.py b/ouroboros/context_runtime_facts.py index 516caaee9..cc58b3c6a 100644 --- a/ouroboros/context_runtime_facts.py +++ b/ouroboros/context_runtime_facts.py @@ -22,22 +22,12 @@ log = logging.getLogger(__name__) def _project_room_fact(task: Dict[str, Any]) -> Optional[Dict[str, Any]]: - """The project-room working-folder FACT for a room turn, or None. + """The room's active folder and ordinary tool target, or None. - Extracted verbatim from ``build_runtime_section`` (v6.90.x submarine unwind) - to keep that builder under the hard method gate; the resolution and the - stated rule are unchanged. + Conversation keeps its own lifecycle; selecting a room changes the physical + target without requiring a managed workspace or Git repository. Promotion + continues to use its separate workspace admission contract. """ - # v6.58.0 (2.2): a conversation/decision turn in a project ROOM sees the room's - # working folder as a structural FACT — it can promote work into that folder - # without ITSELF becoming a workspace task (decision turns deliberately keep the - # promote/steer/route toolset, which workspace profiles exclude). The default - # transport: promote_chat_to_task from this room inherits working_dir unless - # workspace='none'. Registry read is anchored at the canonical DATA_DIR. - # v6.61.3 room lens: the rule now states the REAL chat-lane affordances (reads + - # default shell cwd resolve to the folder; writes go through promoted tasks) — - # the robot-room incident was exactly a fact/affordance split. A set-but-broken - # working_dir is disclosed loudly instead of a silent system-repo fallback. try: _room_pid = str(task.get("project_id") or "").strip() if _room_pid and not str(task.get("workspace_root") or "").strip(): @@ -57,13 +47,12 @@ def _project_room_fact(task: Dict[str, Any]) -> Optional[Dict[str, Any]]: "working_dir": _room_wd, "rule": ( ( - "This room's chat lane LOOKS AT the project folder: read_file/" - "list_files/search_code/query_code with root=active_workspace and " - "the DEFAULT shell cwd resolve to working_dir. The Ouroboros " - "system repo needs explicit root=\"system_repo\" (reads) or an " - "explicit cwd (shell). File WRITES here go through " - "promote_chat_to_task — the promoted task inherits this folder as " - "its workspace (workspace='none' opts out)." + "This room's active_workspace is working_dir for file reads, " + "writes and edits, the default shell cwd, VCS and selected " + "delegation. The tools retain their own requirements and " + "explicit task constraints. Ouroboros governance remains at " + "system_repo; select that root explicitly to work on the body. " + "Direct work and promotion are both available." ) if _lens_active else ( diff --git a/ouroboros/delegate_start_instructions.py b/ouroboros/delegate_start_instructions.py index 663ea0e44..efd51be3a 100644 --- a/ouroboros/delegate_start_instructions.py +++ b/ouroboros/delegate_start_instructions.py @@ -1,11 +1,9 @@ -"""Stable host instructions and bounded actor-first coordination appendix.""" +"""Stable host instructions and complete actor-first coordination appendix.""" from __future__ import annotations from hashlib import sha256 -from ouroboros.delegate_shared import _fail - HOST_INSTRUCTIONS = ( "You are a delegated worker running inside the workspace assigned by your host. Your " @@ -39,32 +37,15 @@ UNPROVEN_BOUNDARY_INSTRUCTION = ( def append_coordination_context( base_instructions: str, coordination_context: str, - *, - instruction_budget_chars: int, -) -> tuple[str, str]: - """Append the exact advisory context or refuse before physical start.""" +) -> str: + """Append the exact advisory context without changing instruction roles.""" - context = str(coordination_context or "").strip() + context = str(coordination_context or "") if not context: - return base_instructions, "" + return base_instructions coordination_sha = sha256(context.encode("utf-8")).hexdigest() appendix = ( "\n\nHOST COORDINATION CONTEXT (advisory appendix; canonical work-order " f"authority remains unchanged; sha256={coordination_sha}):\n{context}" ) - required_chars = len(base_instructions) + len(appendix) - if required_chars > instruction_budget_chars: - return "", _fail( - "delegate_start", - "coordination_context_over_budget", - "The complete coordination appendix does not fit the existing host " - "instruction-field budget; it was not truncated and the physical leaf " - "was not started. Retry with a shorter coordination context or preserve " - "the details in a host artifact/tree note.", - coordination_context_chars=len(context), - required_instruction_chars=required_chars, - instruction_budget_chars=instruction_budget_chars, - coordination_context_sha256=coordination_sha, - host_fallback=False, - ) - return base_instructions + appendix, "" + return base_instructions + appendix diff --git a/ouroboros/gateway/host_service.py b/ouroboros/gateway/host_service.py index 3abf5ba6e..18bd9a6f0 100644 --- a/ouroboros/gateway/host_service.py +++ b/ouroboros/gateway/host_service.py @@ -232,9 +232,10 @@ class HostServiceContext: def _default_tool_schemas(self) -> list[dict[str, Any]]: try: - from supervisor.workers import _get_chat_agent + from supervisor.workers import REPO_DIR + from ouroboros.tools.registry import ToolRegistry - return list(_get_chat_agent().tools.schemas()) + return list(ToolRegistry(pathlib.Path(REPO_DIR), self.data_dir).schemas()) except Exception: log.debug("Host service could not read tool schemas", exc_info=True) return [] diff --git a/ouroboros/gateway/settings.py b/ouroboros/gateway/settings.py index 4fbb58c36..9875596e7 100644 --- a/ouroboros/gateway/settings.py +++ b/ouroboros/gateway/settings.py @@ -347,11 +347,11 @@ def _has_running_agent_tasks() -> bool: # can be invisible to one snapshot. The context-mode guard tolerates it — # the read races the supervisor thread with or without any settings lock. try: - from supervisor.workers import PENDING, RUNNING, _get_chat_agent + from supervisor.workers import PENDING, RUNNING + from supervisor.active_activity import get_direct_activity_registry if PENDING or RUNNING: return True - agent = _get_chat_agent() - return bool(getattr(agent, "_busy", False)) + return bool(get_direct_activity_registry().snapshot()) except Exception: return False @@ -362,18 +362,15 @@ def _has_started_agent_tasks() -> bool: A queued-but-unstarted task re-reads settings in ``handle_task``, so warning that it "keeps the previous configuration" would be false; ``PENDING`` is deliberately excluded (unlike ``_has_running_agent_tasks``, whose callers - gate on any outstanding work). READ-ONLY on purpose: ``_get_chat_agent()`` - CONSTRUCTS the agent (and inserts the canonical repo into ``sys.path``) — - an answer to "is anything started?" must never start something to find out. - Disclosed residual: a live EPHEMERAL turn (workers' local ephemeral agent) - is invisible here — it holds no reviewer/subagent stage this warning - guards, and reaching it read-only would require a new surface.""" + gate on any outstanding work). Registry inspection includes native turns + without constructing an actor merely to answer a status question.""" try: import supervisor.workers as _workers if _workers.RUNNING: return True - agent = getattr(_workers, "_chat_agent", None) - return bool(getattr(agent, "_busy", False)) + from supervisor.active_activity import get_direct_activity_registry + + return bool(get_direct_activity_registry().snapshot()) except Exception: return False diff --git a/ouroboros/gateway/state.py b/ouroboros/gateway/state.py index 2971869e0..e31be7e3f 100644 --- a/ouroboros/gateway/state.py +++ b/ouroboros/gateway/state.py @@ -355,10 +355,13 @@ def _chat_activities_snapshot_safe(drive_root: Any, task_bindings: Any = None) - phase = "finalizing" if _managed_task_finalizing(drive_root, task_id) else "working" activities.append(_activity(task_id, row, phase, started_at)) from ouroboros.post_task_checkpoint import post_task_model_waits - visible = {row["activity_id"] for row in activities} + visible = {row["activity_id"]: row for row in activities} for owner in post_task_model_waits(drive_root): - if owner.task_id not in visible: - row = {**owner.task, "model_waits": owner.snapshot()["model_waits"]} + waits = owner.snapshot()["model_waits"] + if owner.task_id in visible: + visible[owner.task_id].update(phase="finalizing", model_waits=waits, task_attempt=owner.attempt) + else: + row = {**owner.task, "model_waits": waits} activities.append(_activity(owner.task_id, row, "finalizing", _epoch_or_zero(row.get("queued_at")))) except Exception: log.debug("Managed-activity snapshot unavailable for /api/state", exc_info=True) diff --git a/ouroboros/project_dialogue.py b/ouroboros/project_dialogue.py index efa610abf..215aa3baf 100644 --- a/ouroboros/project_dialogue.py +++ b/ouroboros/project_dialogue.py @@ -868,12 +868,12 @@ def enqueue_project_completion_summary( drive_root: Any, evt: Dict[str, Any], task_id: str, task: Dict[str, Any], result: Dict[str, Any], task_done_event: Dict[str, Any], ) -> bool: - """Owe Main's one compact row for a terminal non-ephemeral Project root.""" + """Owe Main's compact row for a managed Project root, not a conversation.""" tid = str(task_id or "").strip() task = task if isinstance(task, dict) else {} result = result if isinstance(result, dict) else {} if not tid or any( - bool(row.get("_ephemeral") or row.get("ephemeral_decision")) + bool(row.get("_ephemeral") or row.get("ephemeral_decision") or row.get("_is_direct_chat")) for row in (evt, task, result, task_done_event) if isinstance(row, dict) ): return False diff --git a/ouroboros/review_evidence_sections.py b/ouroboros/review_evidence_sections.py index 95d950081..52bc80c08 100644 --- a/ouroboros/review_evidence_sections.py +++ b/ouroboros/review_evidence_sections.py @@ -500,8 +500,9 @@ def _accept_effective_claims( (contracts.task_contract.effective_acceptance_claims): ingress-contract claims first, the CLOSED plan wave's frozen claims only when ingress is empty. The plan-state lookup mirrors plan_task's own state location - (budget_drive_root first) and is FAIL-SOFT — a claims lookup must never - break packet building. + (budget_drive_root first). Generic lookup failures remain fail-soft; a + recorded full source that cannot be read reaches acceptance's existing + infrastructure-failure path instead of discarding known claims. A reviewed-and-frozen but never-closed wave binds NOTHING, and until now it was indistinguishable in the packet from a task that never had claims. It is @@ -509,6 +510,7 @@ def _accept_effective_claims( reads ``none_open_plan_wave``. The exhibit sits in ``DECLARED_INTENT_SECTIONS``, so citing it can never resolve a criterion.""" from ouroboros.contracts.task_contract import effective_acceptance_claims + from ouroboros.tools.plan_review_artifacts import PlanReviewSourceUnavailable claims, source = effective_acceptance_claims(contract) if claims: @@ -525,6 +527,10 @@ def _accept_effective_claims( state = load_plan_review_state(pathlib.Path(str(root)), str(task_id)) wave = closed_plan_review_wave(state) + except PlanReviewSourceUnavailable: + # The recorded claims exist. An unavailable full source must reach the + # existing acceptance infrastructure-failure path, never become no claims. + raise except Exception: return [], "", {} frozen, frozen_source = effective_acceptance_claims(contract, wave) diff --git a/ouroboros/server_liveness.py b/ouroboros/server_liveness.py index 0e85963d7..226592e9d 100644 --- a/ouroboros/server_liveness.py +++ b/ouroboros/server_liveness.py @@ -32,9 +32,8 @@ def _chat_turn_wedged(busy: bool, last_activity_ts, now: float, deadline_sec: in def _alert_chat_turn_wedge(task_id, gap: float) -> None: """WS3: a direct-chat turn is heartbeat-silent. New messages still get answered - (WS10 ephemeral decision turns), but a hung IN-PROCESS turn cannot be killed and - still holds the chat-agent lock, so admission cannot be freed in-process (full - kill-ability via out-of-process direct chat was deferred per owner). Surface it + + on independent native actors, but a hung IN-PROCESS turn cannot be killed + independently of the supervisor process. Surface it + recommend /restart, which is the safe full recovery.""" from supervisor.state import append_jsonl, load_state try: @@ -69,10 +68,8 @@ def _start_supervisor_liveness_watchdog(liveness: list, stop_event=None) -> None that loop stalls). It ALERTS the owner on two silent-wedge classes — a supervisor loop stall (new-message intake starvation) and a heartbeat-silent in-process direct-chat turn — converting a multi-hour silent wedge into an immediate signal. - It deliberately does NOT kill a hung thread or free the chat-agent lock: the wedged - turn holds that lock for its whole duration, so in-process admission-freeing is - unsafe (out-of-process direct chat for full kill-ability was deferred per owner); - WS10 ephemeral decision turns keep the chat responsive meanwhile. ``stop_event`` is + It deliberately does NOT kill a hung thread; independent native actors keep + the chat responsive meanwhile. ``stop_event`` is a PER-GENERATION token: when the supervisor loop that owns ``liveness`` exits (incl. the crash-storm death path, which never sets the global restart flag), it is set so this watchdog stops watching a now-stale liveness list (no false post-revival alert).""" @@ -86,7 +83,7 @@ def _start_supervisor_liveness_watchdog(liveness: list, stop_event=None) -> None from supervisor.state import append_jsonl, load_state interval = min(15, max(1, deadline // 3)) loop_alerted = False - wedged_task = None + wedged_tasks: set[str] = set() while not _restart_requested.is_set() and not (stop_event is not None and stop_event.is_set()): time.sleep(interval) # ONE clock: both halves measure an ELAPSED GAP against stamps taken on @@ -100,8 +97,8 @@ def _start_supervisor_liveness_watchdog(liveness: list, stop_event=None) -> None if not loop_alerted: gap = now - liveness[0] log.error( - "Supervisor loop STALLED ~%.0fs — new-message intake starved (WS10 " - "ephemeral chat still answers); investigate a blocking step.", gap, + "Supervisor loop STALLED ~%.0fs — new-message intake starved (native " + "chat still answers); investigate a blocking step.", gap, ) try: append_jsonl(DATA_DIR / "logs" / "supervisor.jsonl", { @@ -131,17 +128,17 @@ def _start_supervisor_liveness_watchdog(liveness: list, stop_event=None) -> None loop_alerted = True else: loop_alerted = False - # (2) In-process direct-chat turn wedge — a heartbeat-silent busy turn. + # (2) Each native actor has its own liveness and alert identity. try: from supervisor.workers import chat_turn_liveness - busy, turn_task, turn_ts = chat_turn_liveness() + turns = chat_turn_liveness() except Exception: - busy, turn_task, turn_ts = (False, None, None) - if _chat_turn_wedged(busy, turn_ts, now, deadline): - if wedged_task != turn_task: # alert once per wedged turn - _alert_chat_turn_wedge(turn_task, now - (turn_ts or now)) - wedged_task = turn_task - elif not busy: - wedged_task = None + turns = [] + live_ids = {task_id for task_id, _ in turns} + wedged_tasks.intersection_update(live_ids) + for turn_task, turn_ts in turns: + if _chat_turn_wedged(True, turn_ts, now, deadline) and turn_task not in wedged_tasks: + _alert_chat_turn_wedge(turn_task, now - turn_ts) + wedged_tasks.add(turn_task) threading.Thread(target=_watch, name="supervisor-liveness-watchdog", daemon=True).start() diff --git a/ouroboros/server_maintenance.py b/ouroboros/server_maintenance.py index 23c4f1c6a..ee9943ba8 100644 --- a/ouroboros/server_maintenance.py +++ b/ouroboros/server_maintenance.py @@ -81,7 +81,11 @@ def _periodic_supervisor_maintenance(last_custody_reap: list, last_review_reconc from ouroboros.process_custody import reap_orphaned_processes from supervisor.queue import RUNNING as _running_tasks - live_tasks = set(_running_tasks.keys()) + from supervisor.active_activity import get_direct_activity_registry + + live_tasks = set(_running_tasks) | { + row["activity_id"] for row in get_direct_activity_registry().snapshot() + } reap_orphaned_processes( DATA_DIR, running_task_ids=live_tasks, live_owner_skills=_installed_skill_names(), diff --git a/ouroboros/server_owner_routing.py b/ouroboros/server_owner_routing.py index 70631b036..7ccbf617c 100644 --- a/ouroboros/server_owner_routing.py +++ b/ouroboros/server_owner_routing.py @@ -118,7 +118,9 @@ def _route_project_chat_to_running_task( direct_agent = None direct_lock = None if candidate.get("direct_chat"): - direct_agent = ctx.get_chat_agent() + from supervisor.workers import get_direct_chat_agent + + direct_agent = get_direct_chat_agent(tid) direct_lock = getattr(direct_agent, "_owner_message_admission_lock", None) if direct_lock is None: return "" @@ -595,7 +597,6 @@ def _route_owner_message(bridge: Any, ctx: Any, incoming: Dict[str, Any]) -> Non needs_decision_lane = swarm_intent or bool(project_id) or has_projects or bool(global_roots) if needs_decision_lane: task_metadata = _decision_turn_metadata(ctx, chat_id, client_message_id, task_metadata) - agent = ctx.get_chat_agent() def _run_direct() -> None: try: @@ -609,7 +610,7 @@ def _route_owner_message(bridge: Any, ctx: Any, incoming: Dict[str, Any]) -> Non finally: ctx.consciousness.resume() - if needs_decision_lane or agent._busy: + if swarm_intent: threading.Thread( target=ctx.handle_chat_ephemeral, args=(chat_id, text or image_caption, image_data), diff --git a/ouroboros/server_restart.py b/ouroboros/server_restart.py index 38a82ede2..8c3868971 100644 --- a/ouroboros/server_restart.py +++ b/ouroboros/server_restart.py @@ -149,7 +149,7 @@ def _stop_owned_daemon_for_new_pin() -> None: def _live_running_task_ids(ctx: Any) -> list: - """RUNNING task ids with a fresh heartbeat — structured facts only. + """Pooled tasks with a fresh heartbeat and registered native executions. Heartbeat staleness belongs to the generic supervisor queue, not to the planning-scout wait policy. The latter intentionally waits until terminal @@ -168,7 +168,11 @@ def _live_running_task_ids(ctx: Any) -> list: hb = 0.0 if hb and (now - hb) < HEARTBEAT_STALE_SEC: live.append(str(tid)) - return live + from supervisor.active_activity import get_direct_activity_registry + + return list(dict.fromkeys(live + [ + row["activity_id"] for row in get_direct_activity_registry().snapshot() + ])) def _managed_update_pending_kwargs() -> dict: diff --git a/ouroboros/server_routing_context.py b/ouroboros/server_routing_context.py index 41d139c86..45647aa5c 100644 --- a/ouroboros/server_routing_context.py +++ b/ouroboros/server_routing_context.py @@ -28,36 +28,27 @@ def _task_belongs_to_chat(ctx: Any, task_id: str, task_obj: Dict[str, Any], chat return False -def _active_direct_root(ctx: Any) -> Dict[str, Any]: - """Snapshot the one in-process direct root without creating queue state.""" - try: - agent = ctx.get_chat_agent() - lock = getattr(agent, "_owner_message_admission_lock", None) +def _active_direct_roots(ctx: Any) -> list: + """All addressable native actors, without constructing agents or queue state.""" + from supervisor.active_activity import get_direct_activity_registry + from supervisor.workers import direct_chat_turn + + roots = [] + for entry in get_direct_activity_registry().actors(): + lock = getattr(entry.actor, "_owner_message_admission_lock", None) if lock is None: - return {} + continue with lock: - task_id = str(getattr(agent, "_current_task_id", "") or "").strip() - if ( - not getattr(agent, "_busy", False) - or not getattr(agent, "_accepting_owner_messages", False) - or not task_id - ): - return {} - metadata = getattr(agent, "_current_task_metadata", {}) - metadata = metadata if isinstance(metadata, dict) else {} - return { - "task_id": task_id, - "status": "running", - "title": _clip_marked(metadata.get("title"), 120), - "objective": _clip_marked(getattr(agent, "_current_task_text", ""), 600), - "project_id": str(metadata.get("project_id") or ""), - "chat_id": int(getattr(agent, "_current_chat_id", 0) or 0), - "started_at": float(getattr(agent, "_task_started_ts", 0.0) or 0.0), - "steerable": True, - "direct_chat": True, - } - except Exception: - return {} + turn = direct_chat_turn(entry.activity_id) + if turn is not None: + roots.append({ + "task_id": turn["id"], "status": "running", + "title": _clip_marked(turn.get("title"), 120), + "objective": _clip_marked(turn.get("text"), 600), + "project_id": turn["project_id"], "chat_id": turn["chat_id"], + "started_at": turn["_started_at"], "steerable": True, "direct_chat": True, + }) + return roots def _addressable_root_tasks(ctx: Any, chat_id: Optional[int] = None) -> list: @@ -95,9 +86,10 @@ def _addressable_root_tasks(ctx: Any, chat_id: Optional[int] = None) -> list: for pending in list(getattr(ctx, "PENDING", []) or []): if isinstance(pending, dict): _add(pending.get("id"), pending, "pending", pending.get("queued_at")) - direct = _active_direct_root(ctx) - if direct and str(direct.get("task_id") or "") not in seen: - if chat_id is None or int(direct.get("chat_id") or 0) == int(chat_id or 0): + for direct in _active_direct_roots(ctx): + if direct["task_id"] not in seen and ( + chat_id is None or _task_belongs_to_chat(ctx, direct["task_id"], direct, chat_id) + ): out.append(direct) return out @@ -117,8 +109,8 @@ def _chat_running_tasks(ctx: Any, chat_id: int) -> list: """Structural snapshot of the owner's RUNNING root tasks in THIS chat (id + objective + recency). The decision turn reads this from runtime context to pick a steer_task target by its own judgment — code only exposes the state, - it never auto-chooses (BIBLE P5). Direct in-process turns and subagents are - not pooled RUNNING tasks and are excluded.""" + it never auto-chooses (BIBLE P5). Direct native roots are included; + delegated subagents are not owner roots.""" return [row for row in _addressable_root_tasks(ctx, chat_id) if row.get("status") == "running"] diff --git a/ouroboros/size_ratchet_manifest.py b/ouroboros/size_ratchet_manifest.py index 24ab93d71..584e6251f 100644 --- a/ouroboros/size_ratchet_manifest.py +++ b/ouroboros/size_ratchet_manifest.py @@ -147,7 +147,6 @@ BAND_PATHS = { "ouroboros/reviewer_slot_config.py": "Absorbed the reviewer_slots() builder from review_substrate (altitude) and configured-subagent row resolution for the generic reviewer-actor bridge.", "ouroboros/safety.py": "Entered the band from 954 lines with the safety-supervisor rate-limit fix: ONE shared model-call helper now serves both the primary and repair safety calls (it already deletes the duplicated call block), recognising a provider rate limit in BOTH wire shapes, taking one bounded deadline-capped backoff plus one retry, then blocking that one call with the typed non-verdict SAFETY_UNAVAILABLE outcome plus a durable audit row (a short storm latch answers further checks in the window without provider calls); the bounded newest-first conversation budget is the second half.", "ouroboros/skill_review_runner.py": None, - "ouroboros/subagent_runtime.py": "Configured-retry refusals mirrored typed (triad 2026-08-30) push the module just over 1000; no new subsystem, same seam.", "ouroboros/subagent_worktrees.py": "Owner-sanctioned strict-registry delta (v7 rows 1083-1092, fork F-1=A) grew the module 1000->1082: typed refusal of a malformed registry instead of silent collapse-to-empty; shrink-only direction", "ouroboros/subagents.py": "D07 split brought the dispatch monolith DOWN from 1593 into the band (->1370); route-health family extracted to subagent_route_health.py, shrink-only direction", "ouroboros/task_pacing.py": "Cost-ceiling SSOT grew into the band while absorbing the cache-aware wrap-up reservation, the prepared/prospective wrap-up candidates, the deciding-spend basis vocabulary and the exhausted-ceiling texts; one owner for pacing decisions instead of a second pricing authority beside usage_accounting.py (at its ceiling).", @@ -178,6 +177,7 @@ BAND_PATHS = { "tests/system_e2e/harness.py": "system_e2e harness: waves 3a+3b grew the one scenario-suite machinery module into the band \u2014 skill-review stub branch, review-organ verdict scripting (ReviewScript), plan-review/native-episode classification markers, the advisory reviewer-slot row and the S11-S17 manifest rows; split when the next wave lands new actors.", "tests/system_e2e/test_system_scenarios_w4.py": "system_e2e wave-4 scenario module: six scenarios (S18-S23 - update carrier/conflict/crash variants, chat-lineage cancel, absorb kill-recovery, delegated interactive answer) plus the interactive fake-daemon contract pin; one module per wave is the suite convention - split only if a later wave extends THIS module instead of adding its own.", "tests/test_advisory_observability.py": None, + "tests/test_available_subagents_runtime.py": "Configured-session route and legacy custody regressions retained after removing compulsory source-request production tests.", "tests/test_build_scripts.py": None, "tests/test_commit_gate.py": None, "tests/test_cybergym_protocol.py": "CyberGym protocol suite arrived in one piece with the benchmark (drift heal); split when the next protocol family lands.", diff --git a/ouroboros/subagent_bootstrap.py b/ouroboros/subagent_bootstrap.py index 4ce9f0831..69e11bd77 100644 --- a/ouroboros/subagent_bootstrap.py +++ b/ouroboros/subagent_bootstrap.py @@ -7,12 +7,7 @@ from hashlib import sha256 from pathlib import Path from typing import Any, Mapping -from ouroboros.subagent_work_order import ( - WorkOrderBudgetExceeded, - build_work_order_source_request, - compile_external_work_order, - route_source_request_channel, -) +from ouroboros.subagent_work_order import compile_external_work_order def _with_coordination_context(ctx: Any, raw: str) -> str: @@ -626,41 +621,9 @@ def _prepare_actor_first_bootstrap( """Freeze exact actor authority while keeping a new physical start pending.""" snapshot = task.get("configured_subagent") if isinstance(task.get("configured_subagent"), dict) else {} route = snapshot.get("route") if isinstance(snapshot.get("route"), dict) else {} - try: - work_order = compile_external_work_order(task) - work_order_fingerprint = sha256(work_order.encode("utf-8")).hexdigest() - work_order_chars = len(work_order) - source_prompt = "" - source_request: dict[str, Any] = {} - source_channel: dict[str, Any] = {} - except WorkOrderBudgetExceeded as exc: - source_prompt, source_request = build_work_order_source_request(task, exc) - source_channel = {"status": "unverified", "reason": "not_checked"} - route_id = str(route.get("target_id") or "") - resolved_route = getattr(getattr(dispatch, "executor_resolution", None), "route", None) - channel_route_id = str(getattr(resolved_route, "route_id", "") or route_id) - gateway = None - try: - from ouroboros.claudexor_daemon import ensure_owned_gateway - - gateway = ensure_owned_gateway() - source_channel = route_source_request_channel(gateway, channel_route_id) - except Exception as channel_error: # noqa: BLE001 - unknown is typed - source_channel = { - "status": "unverified", - "reason": "capability_probe_failed", - "detail": type(channel_error).__name__, - "route": channel_route_id, - } - finally: - if gateway is not None: - try: - gateway.close() - except Exception: - pass - work_order = "" - work_order_fingerprint = exc.sha256 - work_order_chars = exc.chars + work_order = compile_external_work_order(task) + work_order_fingerprint = sha256(work_order.encode("utf-8")).hexdigest() + work_order_chars = len(work_order) route_id = str(route.get("target_id") or "") ctx._configured_actor_bootstrap = { @@ -670,9 +633,6 @@ def _prepare_actor_first_bootstrap( "selected_subagent_id": str(snapshot.get("selected_subagent_id") or ""), "config_fingerprint": str(snapshot.get("config_fingerprint") or ""), "canonical_work_order": work_order, - "source_prompt": source_prompt, - "source_request": source_request, - "source_channel": source_channel, "work_order_fingerprint": work_order_fingerprint, "work_order_chars": work_order_chars, "route_available": not bool(getattr(dispatch, "blocked", False)), @@ -712,7 +672,6 @@ def _prepare_actor_first_bootstrap( "work_order_fingerprint": work_order_fingerprint, "work_order_chars": work_order_chars, "work_order_complete": bool(work_order), - **({"source_channel": source_channel} if source_channel else {}), "actor_first": True, "exact_start_pending": not bool( durable_zero_run or zero_run_evidence_gaps diff --git a/ouroboros/subagent_runtime.py b/ouroboros/subagent_runtime.py index d6ff251d1..141dd6db0 100644 --- a/ouroboros/subagent_runtime.py +++ b/ouroboros/subagent_runtime.py @@ -692,7 +692,7 @@ def exact_start(ctx: Any, prompt: str, spec: Optional[dict[str, Any]] = None) -> canonical_work_order_fingerprint = str( options.pop("work_order_fingerprint", "") or "" ).strip() - coordination_context = str(options.pop("_coordination_context", "") or "").strip() + coordination_context = str(options.pop("_coordination_context", "") or "") work_order_source_request = options.pop("work_order_source_request", None) try: if str(options.get("retry_of") or "").strip() and ( @@ -797,131 +797,6 @@ def _mark_actor_physical_start(ctx: Any, result: Any) -> None: ctx._nanny_physical_activity_seed = True -def _record_actor_work_order_source( - ctx: Any, bootstrap: dict[str, Any], *, refusal_reason: str = "", -) -> None: - """Persist the oversized-work-order decision when a physical start is attempted.""" - source_request = bootstrap.get("source_request") - if not isinstance(source_request, dict) or not source_request: - return - from ouroboros import delegate_custody as custody - - source_channel = ( - bootstrap.get("source_channel") - if isinstance(bootstrap.get("source_channel"), dict) - else {} - ) - payload = { - "task_id": str(getattr(ctx, "task_id", "") or ""), - "route": str(source_channel.get("route") or bootstrap.get("route_id") or ""), - # This evidence is written before the POST so a crash cannot turn an - # attempted partial-lens start into clean absence. ``exact_start`` may - # still refuse before delivery, so do not label the request as sent. - "status": "attempted", - "source_channel": source_channel, - **source_request, - } - event_type = "configured_subagent_work_order_source_request" - if refusal_reason: - event_type = "configured_subagent_work_order_refused" - payload.update({ - "status": "refused", - "reason": refusal_reason, - "source_request": source_request, - "source_channel": source_channel, - "detail": ( - "The complete brief was not truncated or sent. A live interactive " - "question channel is required to resolve its named source ranges." - ), - }) - custody.emit(custody.custody_root(ctx), event_type, payload) - - -def _actor_work_order_for_start( - ctx: Any, bootstrap: dict[str, Any], *, retry: bool = False, -) -> tuple[str, str]: - """Resolve the immutable work order for one physical start attempt. - - A bootstrap capability observation is useful context for the actor, but it - cannot authorize a later partial-lens start: manifests may change in either - direction while the actor reasons. Every over-budget attempt therefore - probes the exact frozen route again and records the live observation. - """ - - canonical = str(bootstrap.get("canonical_work_order") or "") - source_request = bootstrap.get("source_request") - if canonical or not isinstance(source_request, dict) or not source_request: - return canonical, "" - - try: - _snapshot, exact_route = exact_session_binding(bootstrap.get("snapshot")) - channel_route_id = str(exact_route.route_id or "") - route_error = "" - except Exception as exc: # noqa: BLE001 - invalid frozen authority is UNKNOWN - channel_route_id = "" - route_error = str(getattr(exc, "code", "") or type(exc).__name__) - gateway = None - try: - from ouroboros.claudexor_daemon import ensure_owned_gateway - from ouroboros.subagent_work_order import route_source_request_channel - - if route_error: - source_channel = { - "status": "unverified", - "reason": "frozen_route_invalid", - "detail": route_error, - "route": channel_route_id, - } - else: - gateway = ensure_owned_gateway() - source_channel = route_source_request_channel(gateway, channel_route_id) - except Exception as exc: # noqa: BLE001 - unknown is a typed authority fact - source_channel = { - "status": "unverified", - "reason": "capability_probe_failed", - "detail": type(exc).__name__, - "route": channel_route_id, - } - finally: - if gateway is not None: - try: - gateway.close() - except Exception: - pass - bootstrap["source_channel"] = source_channel - - status = str(source_channel.get("status") or "unverified") - if status != "available": - reason = ( - "work_order_source_channel_unavailable" - if status == "unavailable" - else "work_order_source_channel_unverified" - ) - _record_actor_work_order_source(ctx, bootstrap, refusal_reason=reason) - from ouroboros.delegate_shared import _fail - - detail = ( - "The selected route reports no interactive source channel." - if status == "unavailable" - else "The host could not verify an interactive source channel for the selected route." - ) - return "", _fail( - "delegate_start", reason, - f"{detail} The complete canonical work order exceeds the host wire budget, " - "so the physical leaf was not started from a prefix.", - work_order_fingerprint=str(bootstrap.get("work_order_fingerprint") or ""), - work_order_chars=int(bootstrap.get("work_order_chars") or 0), - source_channel=source_channel, - retry=bool(retry), - host_fallback=False, - ) - - source_prompt = str(bootstrap.get("source_prompt") or "") - if source_prompt: - return source_prompt, "" - return "", "" - - def delegate_start_entry(ctx: Any, prompt: str, _resolved_binding: Any = None, **params: Any) -> str: # Actor-first configured sessions bind every fresh start to the immutable # snapshot captured before the episode. The model supplies only an advisory @@ -961,19 +836,13 @@ def delegate_start_entry(ctx: Any, prompt: str, _resolved_binding: Any = None, * selected_subagent_id=expected_id, host_fallback=False, ) - canonical_work_order, source_refusal = _actor_work_order_for_start( - ctx, bootstrap, - ) - if source_refusal: - _blocked("configured_work_order_source_refused") - return source_refusal - source_request = bootstrap.get("source_request") + canonical_work_order = str(bootstrap.get("canonical_work_order") or "") if not canonical_work_order: _blocked("configured_work_order_unavailable") return _fail( "delegate_start", "configured_work_order_unavailable", "The canonical work order is unavailable; do not start a physical leaf " - "from a prefix. Resolve the existing source-range interaction first.", + "from a prefix. Recover the task's complete chosen assignment.", work_order_fingerprint=str(bootstrap.get("work_order_fingerprint") or ""), work_order_chars=int(bootstrap.get("work_order_chars") or 0), host_fallback=False, @@ -984,13 +853,10 @@ def delegate_start_entry(ctx: Any, prompt: str, _resolved_binding: Any = None, * "snapshot": dict(bootstrap.get("snapshot") or {}), "compiled_work_order": True, "work_order_fingerprint": str(bootstrap.get("work_order_fingerprint") or ""), - "_coordination_context": str(prompt or "").strip(), + "_coordination_context": str(prompt or ""), }) if _resolved_binding is not None: bound["_resolved_binding"] = _resolved_binding - if isinstance(source_request, dict) and source_request: - bound["work_order_source_request"] = dict(source_request) - _record_actor_work_order_source(ctx, bootstrap) return exact_start(ctx, canonical_work_order, bound) if retry_of and isinstance(bootstrap, dict): # Retry replays the stored canonical request byte-for-byte - so an @@ -1017,18 +883,17 @@ def delegate_start_entry(ctx: Any, prompt: str, _resolved_binding: Any = None, * selected_subagent_id=expected_id, host_fallback=False, ) - canonical_work_order, source_refusal = _actor_work_order_for_start( - ctx, bootstrap, retry=True, - ) - if source_refusal: - _blocked("configured_work_order_source_refused") - return source_refusal + from ouroboros import delegate_custody as custody + + invocation = custody.invocation_record(custody.custody_root(ctx), retry_of) or {} + request = invocation.get("request") if isinstance(invocation.get("request"), dict) else {} + canonical_work_order = str(request.get("prompt") or "") if not canonical_work_order: _blocked("configured_work_order_unavailable") return _fail( "delegate_start", "configured_work_order_unavailable", - "The retry has no complete canonical work order or verified source " - "lens; the coordination prompt cannot replace the original assignment.", + "The retry has no recorded work order; the coordination prompt " + "cannot replace the original assignment.", work_order_fingerprint=str(bootstrap.get("work_order_fingerprint") or ""), work_order_chars=int(bootstrap.get("work_order_chars") or 0), host_fallback=False, @@ -1037,9 +902,6 @@ def delegate_start_entry(ctx: Any, prompt: str, _resolved_binding: Any = None, * "retry_of": retry_of, "_resolved_binding": _resolved_binding, } - if isinstance(bootstrap.get("source_request"), dict) and bootstrap.get("source_request"): - retry_spec["work_order_source_request"] = dict(bootstrap["source_request"]) - _record_actor_work_order_source(ctx, bootstrap) return exact_start(ctx, canonical_work_order, retry_spec) return exact_start(ctx, prompt, {**params, "_resolved_binding": _resolved_binding}) diff --git a/ouroboros/subagent_work_order.py b/ouroboros/subagent_work_order.py index 779eee6ae..6eceb81d2 100644 --- a/ouroboros/subagent_work_order.py +++ b/ouroboros/subagent_work_order.py @@ -6,151 +6,6 @@ import json from hashlib import sha256 from typing import Any, Mapping -_WORK_ORDER_CHARS = 250_000 -# Historical import name retained for tests/callers which only need the public -# wire budget. It is no longer a per-field truncation limit. -_FIELD_CHARS = _WORK_ORDER_CHARS - - -class WorkOrderBudgetExceeded(ValueError): - """A complete work order cannot fit the one explicit wire budget.""" - - def __init__(self, *, chars: int, sha256_hex: str) -> None: - super().__init__(f"complete work order is {chars} characters (budget {_WORK_ORDER_CHARS})") - self.chars = int(chars) - self.sha256 = str(sha256_hex) - self.limit = _WORK_ORDER_CHARS - - -def _source_selector(task: Mapping[str, Any]) -> dict[str, Any]: - """Return stable, actor-resolvable pointers without copying the omitted brief. - - The selector is deliberately a small manifest. The complete work order remains - in the parent task's canonical authority; this prompt must never smuggle a prefix - of it under a different name and make the child believe that the prefix is the - contract. The existing task-result reader exposes the same canonical projection - the host validates, so the owner can answer a precise range request through the - existing interaction channel. - """ - - task_id = str(task.get("id") or "") - # Reuse the canonical actor-readable authority-ref shape. A bespoke nested - # ``reader`` object looked descriptive but no existing materializer recognized - # it, which would make the source pointer itself another unverifiable claim. - selector: dict[str, Any] = { - "kind": "task_result", - "task_id": task_id, - "tool": "get_task_result", - "arguments": { - "task_id": task_id, - "include_authority": True, - "include_work_order_source": True, - }, - "projection": "canonical_work_order", - } - return selector - - -def build_work_order_source_request( - task: Mapping[str, Any], exc: WorkOrderBudgetExceeded, -) -> tuple[str, dict[str, Any]]: - """Build the bounded interaction lens for a complete brief over the wire cap. - - This is not a lossy work order. It is an explicit partial-coverage envelope: - the child receives the full-brief digest and a source it can ask its host to - resolve, then requests only the exact range it needs. The response is - intentionally carried by the existing AskUserQuestion/delegate_answer seam; no - second storage or retrieval subsystem is introduced here. - """ - - source = _source_selector(task) - envelope: dict[str, Any] = { - "schema": 1, - "kind": "complete_work_order", - "coverage": "partial", - "complete_chars": int(exc.chars), - "wire_budget_chars": int(exc.limit), - "complete_sha256": str(exc.sha256), - "source": source, - "request": { - "channel": "existing_interaction", - "before_substantive_work": True, - "ask_for": "one exact source character range at a time", - "include": ["complete_sha256", "source", "start_char", "end_char", "reason"], - }, - "response": { - "include": [ - "complete_sha256", "source", "start_char", "end_char", "text", - ], - "rule": ( - "Answer delegate_answer with source_response={schema, kind, " - "complete_sha256, source, start_char, end_char, text}. The host " - "checks the exact canonical range and records only verified ranges. " - "A missing, mismatched, or partial response remains cannot_verify; " - "it never authorizes PASS, a destructive rewrite, or replacement " - "of this complete work order." - ), - }, - } - prompt = ( - "WORK ORDER SOURCE REQUEST\n" - "The complete external work order exceeded the host wire budget. This message " - "is an explicit partial-coverage lens, not the assignment and not a prefix of " - "the assignment. Do not perform substantive work, claim completeness, accept " - "a verdict, or replace the full contract from this message alone.\n\n" - "Use the harness's native question channel to ask your host for the smallest " - "exact source character range needed. Ask the nanny to answer with the " - "typed source_response field of delegate_answer. After an answer, verify the " - "returned source selector, range, digest, and completeness before relying on " - "it; request another range when needed.\n\n" - + json.dumps(envelope, ensure_ascii=False, sort_keys=True, indent=2) - ) - return prompt, envelope - - -def route_source_request_channel(gateway: Any, route_id: str) -> dict[str, Any]: - """Read the generic live interaction capability from a Claudexor manifest. - - The route is opaque to Ouroboros. Missing or failed capability evidence is - therefore unverified rather than an invitation to start with a partial - contract. This mirrors the existing manifest reader used by review execution. - """ - - try: - rows = gateway.harnesses() - except Exception as exc: # noqa: BLE001 - capability is unknown, not healthy - return { - "status": "unverified", - "reason": "capability_read_failed", - "detail": type(exc).__name__, - "route": str(route_id or ""), - } - for row in rows or []: - if not isinstance(row, dict) or str(row.get("id") or "") != str(route_id or ""): - continue - manifest = row.get("manifest") if isinstance(row.get("manifest"), dict) else {} - capabilities = ( - manifest.get("capabilities") - if isinstance(manifest.get("capabilities"), dict) else {} - ) - if not isinstance(capabilities.get("interactive"), bool): - return { - "status": "unverified", - "reason": "interactive_capability_missing", - "route": str(route_id or ""), - } - return { - "status": "available" if capabilities["interactive"] else "unavailable", - "reason": "interactive" if capabilities["interactive"] else "interactive_unsupported", - "route": str(route_id or ""), - } - return { - "status": "unverified", - "reason": "route_not_in_manifest", - "route": str(route_id or ""), - } - - def _text(value: Any) -> str: if isinstance(value, list): value = "\n".join(f"- {item}" for item in value if str(item).strip()) @@ -169,22 +24,12 @@ def assignment_instructions(ctx: Any) -> str: from ouroboros.contracts.task_contract import build_task_contract contract = build_task_contract({"task_contract": contract}) - parts: list[str] = [] - objective = _text(contract.get("objective")) - expected = _text(contract.get("expected_output")) - if objective: - parts.append( - "HOST TASK OBJECTIVE (immutable contract; the prompt is one assignment inside it): " - + objective - ) - if expected: - parts.append("HOST EXPECTED OUTPUT: " + expected) - if contract: - parts.append( - "HOST TASK CONTRACT AUTHORITY (complete normalized JSON; exact strings are authority):\n" - + json.dumps(contract, ensure_ascii=False, sort_keys=True, separators=(",", ":")) - ) - return "\n\n".join(parts) + if not contract: + return "" + return ( + "HOST TASK CONTRACT AUTHORITY (complete normalized JSON; exact strings are authority):\n" + + json.dumps(contract, ensure_ascii=False, sort_keys=True, separators=(",", ":")) + ) def _render_external_work_order(task: Mapping[str, Any]) -> str: @@ -376,14 +221,9 @@ def validate_work_order_source_response( def compile_external_work_order(task: Mapping[str, Any]) -> str: - """Compile one complete brief or refuse instead of sending a false prefix.""" + """Compile the complete chosen brief; transport limits belong to the recipient.""" - rendered = _render_external_work_order(task) - if len(rendered) > _WORK_ORDER_CHARS: - raise WorkOrderBudgetExceeded( - chars=len(rendered), sha256_hex=sha256(rendered.encode("utf-8")).hexdigest(), - ) - return rendered + return _render_external_work_order(task) def start_binding_fingerprints(ctx: Any, prompt: str) -> tuple[str, str]: @@ -398,15 +238,14 @@ def start_binding_fingerprints(ctx: Any, prompt: str) -> tuple[str, str]: def work_order_fingerprint(task: Mapping[str, Any]) -> str: - """Digest the complete canonical brief, including an over-budget one.""" + """Digest the complete canonical brief.""" return sha256(_render_external_work_order(task).encode("utf-8")).hexdigest() __all__ = [ - "WorkOrderBudgetExceeded", "assignment_instructions", "compile_external_work_order", - "build_work_order_source_request", "canonical_work_order_source", + "assignment_instructions", "compile_external_work_order", "canonical_work_order_source", "work_order_source_projection", - "route_source_request_channel", "validate_work_order_source_response", + "validate_work_order_source_response", "start_binding_fingerprints", "work_order_fingerprint", ] diff --git a/ouroboros/task_results.py b/ouroboros/task_results.py index 75bba53f5..653bddf74 100644 --- a/ouroboros/task_results.py +++ b/ouroboros/task_results.py @@ -1095,13 +1095,18 @@ def _validated_plan_review_state(value: Any) -> Dict[str, Any]: def load_plan_review_state(results_drive_root: Any, task_id: str) -> Dict[str, Any]: + """Read bounded state, then resolve the full specs used by decision consumers.""" path = task_result_path(results_drive_root, task_id, create=False) if not path.is_file(): return _empty_plan_review_state() result = read_json_dict(path) if result is None: raise ValueError("PLAN_REVIEW_STATE_INVALID: parent task result JSON is malformed") - return _validated_plan_review_state(result.get(PLAN_REVIEW_STATE_KEY)) + from ouroboros.tools.plan_review_artifacts import authority_state + + return authority_state( + results_drive_root, task_id, _validated_plan_review_state(result.get(PLAN_REVIEW_STATE_KEY)), + ) def plan_review_wave(state: Dict[str, Any], fingerprint: str) -> Optional[Dict[str, Any]]: @@ -1413,6 +1418,7 @@ def _compact_plan_review_wave(wave: Dict[str, Any]) -> Dict[str, Any]: "closed": bool(wave.get("closed")), "paid": bool(wave.get("paid")), "wave_artifact": copy.deepcopy(wave.get("wave_artifact") or {}), + **({"spec_source_ref": copy.deepcopy(wave["spec_source_ref"])} if wave.get("spec_source_ref") else {}), **({"reviewed_at": str(wave["reviewed_at"])} if wave.get("reviewed_at") else {}), } @@ -1432,6 +1438,7 @@ _PLAN_REVIEW_IDENTITY_KEYS = frozenset({ "previous_fingerprint", "spec_hash", "evidence_manifest_hash", "plan_prose_hash", "sha256", "model", "request_model", "route", "host_file_read_attestation", "reason", "decision", "kind", "goal", "acceptance_claims", "cycle_index", "series_id", "schema_version", "retry_key", + "wave_artifact", "spec_source_ref", }) diff --git a/ouroboros/tool_access.py b/ouroboros/tool_access.py index f135f2361..8231f48d5 100644 --- a/ouroboros/tool_access.py +++ b/ouroboros/tool_access.py @@ -167,17 +167,20 @@ def _process_root_candidates( bucket: str = "", skill_name: str = "", include_skill: bool = False, + require_active: bool = True, ) -> list[tuple[ResourceRoot, pathlib.Path, str, str]]: """Return side-effect-free ``(root, base, source, skill)`` candidates.""" profile = active_tool_profile(ctx) - active = resource_root_path(ctx, "active_workspace") candidates: list[tuple[ResourceRoot, pathlib.Path, str, str]] = [] - room = project_room_lens_dir(ctx) - if room is not None: - candidates.append(("active_workspace", room, "active_workspace", "")) + try: + active = resource_root_path(ctx, "active_workspace") + except ValueError: + if require_active: + raise + else: + candidates.append(("active_workspace", active, "active_workspace", "")) candidates += [ - ("active_workspace", active, "active_workspace", ""), ("system_repo", resource_root_path(ctx, "system_repo"), "system_repo", ""), ] @@ -457,6 +460,9 @@ def _select_process_target( raise ValueError("cwd=skill_payload[/subdir] requires bucket and skill_name") candidate_records = _process_root_candidates( ctx, operation, bucket=bucket, skill_name=skill_name, include_skill=include_skill, + require_active=(reserved_root == "active_workspace" or ( + not reserved_root and not is_absolute_path_text(text) and not text.startswith("~") + )), ) allowed = [(label, root) for label, root, _source, _name in candidate_records] if not candidate_records: @@ -677,6 +683,7 @@ def build_resolved_resource_binding( normalized == "runtime_data" and operation in {"write", "edit"} ) or ( normalized == "active_workspace" and operation == "edit" and not workspace_active + and project_room_lens_dir(ctx) is None ) if legacy_data_form: from ouroboros.contracts.skill_payload_policy import ( @@ -724,11 +731,7 @@ def build_resolved_resource_binding( source = str(normalized) selected_name = "" - room = ( - project_room_lens_dir(ctx) - if normalized == "active_workspace" and operation in {"read", "list", "search", "shell"} - else None - ) + room = project_room_lens_dir(ctx) if normalized == "active_workspace" else None if normalized == "skill_payload": selected_bucket = str(bucket or "").strip() selected_skill = str(skill_name or "").strip() diff --git a/ouroboros/tool_access_roots.py b/ouroboros/tool_access_roots.py index 7b5e7b684..b819b540c 100644 --- a/ouroboros/tool_access_roots.py +++ b/ouroboros/tool_access_roots.py @@ -105,10 +105,12 @@ def predicted_subagent_profile(*, write_surface: str = "") -> ToolProfile: def project_room_lens_dir(ctx: Any) -> Optional[pathlib.Path]: - """Return a direct-chat room's verified project cwd, otherwise ``None``. + """Return a direct-chat room's selected folder, otherwise ``None``. Promoted/workspace/subagent tasks carry their own workspace; only a direct - chat without one may use the injected existing ``_project_room_dir``. + chat without one may use the host's ``_project_room_dir``. Its address stays + selected if the folder disappears: existence is an operation's concern, + never permission to switch back to the system repository. """ if not bool(getattr(ctx, "is_direct_chat", False)): return None @@ -117,12 +119,11 @@ def project_room_lens_dir(ctx: Any) -> Optional[pathlib.Path]: meta = getattr(ctx, "task_metadata", None) raw = str(meta.get("_project_room_dir") or "").strip() if isinstance(meta, dict) else "" if not raw: + note = str(meta.get("_project_room_note") or "") if isinstance(meta, dict) else "" + if note: + raise ValueError(note) return None - try: - candidate = pathlib.Path(raw).resolve(strict=False) - return candidate if candidate.is_dir() else None - except OSError: - return None + return pathlib.Path(raw).resolve(strict=False) def load_bound_skill(binding: ResolvedResourceBinding) -> Any: @@ -171,12 +172,7 @@ def resource_root_path( ) -> pathlib.Path: if root == "active_workspace": active = getattr(ctx, "active_repo_dir", None) - candidate = None - if callable(active): - try: - candidate = active() - except Exception: - candidate = None + candidate = active() if callable(active) else project_room_lens_dir(ctx) if candidate is None or candidate.__class__.__module__.startswith("unittest.mock"): candidate = getattr(ctx, "repo_dir") return pathlib.Path(candidate).resolve(strict=False) diff --git a/ouroboros/tool_access_types.py b/ouroboros/tool_access_types.py index e7fed80be..0047f49fd 100644 --- a/ouroboros/tool_access_types.py +++ b/ouroboros/tool_access_types.py @@ -107,6 +107,7 @@ _TOP_LEVEL_PRINCIPAL_PROFILES: frozenset[str] = frozenset({ "workspace_task", "external_workspace_task", "self_modification", + "operator_control", }) diff --git a/ouroboros/tools/core.py b/ouroboros/tools/core.py index 708c05e25..c68c22a0e 100644 --- a/ouroboros/tools/core.py +++ b/ouroboros/tools/core.py @@ -27,7 +27,6 @@ from ouroboros.tool_access import ( active_tool_profile, # noqa: F401 normalize_root, # noqa: F401 normalize_runtime_data_path, # noqa: F401 - project_room_lens_dir, UserFilesPathBlockedError, user_files_path_block_reason, ) @@ -589,18 +588,6 @@ def _write_file( ), "") if protected_block: return protected_block - if normalized == "active_workspace" and (_room := project_room_lens_dir(ctx)) is not None: - # Room write-guard (v6.61.3): with the lens re-pointing reads at the room - # folder, a default-root write silently landing in the SYSTEM REPO would be - # a read/write split trap (read game.js from the folder, "fix" it into the - # repo). Mutations belong to promoted tasks; deliberate self-repo writes - # stay available via the explicit root. - return ( - f"⚠️ ROOM_WRITE_VIA_TASK: this room's files live in {_room} and are edited by " - "PROMOTED tasks — call promote_chat_to_task (it inherits the room folder as its " - "workspace) for real work there. For a deliberate write to the Ouroboros system " - 'repo, pass root="system_repo" explicitly.' - ) if normalized in {"active_workspace", "system_repo"}: from ouroboros.tools.git import _repo_write @@ -758,15 +745,6 @@ def _edit_text( ) if protected_block: return protected_block - if normalized == "active_workspace" and (_room := project_room_lens_dir(ctx)) is not None: - # Room write-guard (v6.61.3) — same rule as write_file: room mutations go - # through promoted tasks; explicit root="system_repo" for the self-repo. - return ( - f"⚠️ ROOM_WRITE_VIA_TASK: this room's files live in {_room} and are edited by " - "PROMOTED tasks — call promote_chat_to_task (it inherits the room folder as its " - "workspace) for real work there. For a deliberate edit of the Ouroboros system " - 'repo, pass root="system_repo" explicitly.' - ) bound_skill_payload = bool( binding.skill_name and binding.source in {"external", "clawhub", "ouroboroshub", "native", "user_repo"} diff --git a/ouroboros/tools/delegate.py b/ouroboros/tools/delegate.py index b8023e08f..70a6f7852 100644 --- a/ouroboros/tools/delegate.py +++ b/ouroboros/tools/delegate.py @@ -39,7 +39,6 @@ from ouroboros.delegate_custody import RunCustody as _RunCustody from ouroboros.tool_capabilities import tool_result_limit from ouroboros.tools.registry import ToolContext, ToolEntry from ouroboros.subagent_work_order import ( # noqa: F401 - compatibility re-export - _FIELD_CHARS as _ASSIGNMENT_FIELD_CHARS, assignment_instructions as _assignment_instructions, ) from ouroboros.delegate_source_coverage import ( @@ -173,11 +172,10 @@ def _host_instructions(authority: "DelegatedRunShape", assignment: str = "", def _build_start_instructions( authority: "DelegatedRunShape", assignment: str = "", payload_skill: str = "", coordination_context: str = "", -) -> tuple[str, str]: - """Build the bounded instruction field for a fresh physical start.""" +) -> str: + """Build complete host instructions for a fresh physical start.""" return append_coordination_context( _host_instructions(authority, assignment, payload_skill), coordination_context, - instruction_budget_chars=_ASSIGNMENT_FIELD_CHARS, ) @@ -188,20 +186,20 @@ def _derive_authority(ctx: ToolContext) -> "DelegatedRunShape": the host decides with what powers. Ouroboros asks for an access PROFILE and lets Claudexor pick the mechanism (fs sandbox, tool allowlist, ...) — no harness branch. - The SHAPE itself belongs to ``subagents.delegated_run_shape``, which the dispatcher - also reads: this function only answers "does this task hold a mutating surface", - which is the one part that needs the live ``ToolContext``. Two authorities qualify - (B5, owner 2=A): an ACTING CHILD with a valid write surface, and the ROOT of an - EXTERNAL-WORKSPACE task — the root already holds write+shell inside the project, - so its delegated runs carry the same mutating shape, bounded by the same - workspace; ``_mutation_authority`` (``tools.delegate_integration``) validates the - concrete target either way. + The SHAPE belongs to ``subagents.delegated_run_shape``. An acting child or an + ordinary root with a selected external workspace/room holds write authority + there; ``_mutation_authority`` validates that same physical target. The existing + snapshot capability decides whether it can execute that mutating assignment. """ from ouroboros.subagents import delegated_run_shape - from ouroboros.tool_access import active_tool_profile + from ouroboros.tool_access import ( + _TOP_LEVEL_PRINCIPAL_PROFILES, active_tool_profile, project_room_lens_dir, + ) profile = active_tool_profile(ctx) - mutating = profile in ("acting_subagent", "external_workspace_task") + mutating = profile in ("acting_subagent", "external_workspace_task") or ( + profile in _TOP_LEVEL_PRINCIPAL_PROFILES and project_room_lens_dir(ctx) is not None + ) if mutating: from ouroboros.presence_authority import presence_ceiling_allows_delegated_surface @@ -300,8 +298,8 @@ def _delegate_start(ctx: ToolContext, prompt: str, max_seconds: Optional[int] = from ouroboros.gateways.claudexor import ClaudexorUnavailable from ouroboros.subagents import delegated_execution_workspace_root, resolve_subagent_executor, route_health - text = str(prompt or "").strip() - if not text: + text = str(prompt or "") + if not text.strip(): return _fail("delegate_start", "empty_prompt", "prompt is required") selector_root = str(root or "").strip() selector_refusal = _payload_selector_refusal(selector_root, retry_of, bucket, skill_name) @@ -374,14 +372,12 @@ def _delegate_start(ctx: ToolContext, prompt: str, max_seconds: Optional[int] = payload_skill = str( (payload_auth.get("resource_ref") or {}).get("skill_name") or "" ) - instructions, instruction_error = _build_start_instructions( + instructions = _build_start_instructions( authority, assignment, payload_skill=payload_skill, coordination_context=_coordination_context, ) - if instruction_error: - return instruction_error access = authority.access try: diff --git a/ouroboros/tools/delegate_integration.py b/ouroboros/tools/delegate_integration.py index 67a76d01e..935bcc9bf 100644 --- a/ouroboros/tools/delegate_integration.py +++ b/ouroboros/tools/delegate_integration.py @@ -53,9 +53,8 @@ def _mutation_authority(ctx: ToolContext, authority: "DelegatedRunShape") -> tup ``acting_constraint`` (an acting child's ``task_constraint.write_root``, which must equal the genuinely ACTIVE workspace root — `active_repo_dir_for` falls back to the LIVE repo when `is_workspace_mode()` is false) and - ``external_workspace_root`` (the ROOT of an external-workspace task, whose - authority derives from its own VALIDATED active workspace; owner 2=A — the - root already holds write+shell there, the prior gap was provenance). + ``external_workspace_root`` (an ordinary root's selected external workspace + or Project room, where it already holds write+shell authority). Read-only runs return the ordinary active root with ``capture_mode: "none"``. Disagreement anywhere is a typed refusal, never a best-effort guess. """ @@ -110,17 +109,19 @@ def _mutation_authority(ctx: ToolContext, authority: "DelegatedRunShape") -> tup ) return {"target_root": root, "source": "acting_constraint", "capture_mode": _CAPTURE_DELEGATED_SNAPSHOT}, "" - # ROOT branch (B5): no acting constraint. The mutating shape can only have come - # from the external-workspace-root profile, and the workspace contract is the - # authority: workspace genuinely active, mode external, and the declared - # workspace root must BE the active root. - if not workspace_active: + # An ordinary room uses the same selected-target authority without becoming + # a pooled workspace task or requiring Git merely for native file operations. + from ouroboros.tool_access import _TOP_LEVEL_PRINCIPAL_PROFILES, active_tool_profile, project_room_lens_dir + + room = (project_room_lens_dir(ctx) + if active_tool_profile(ctx) in _TOP_LEVEL_PRINCIPAL_PROFILES else None) + if not workspace_active and room is None: return {}, _fail( "delegate_start", "workspace_not_active", "A delegated run may only WRITE inside an ACTIVE workspace, and this task " "has none. Refusing rather than falling back to the repository root.", ) - ws_mode = str(getattr(ctx, "workspace_mode", "") or "").strip().lower() + ws_mode = "external" if room is not None else str(getattr(ctx, "workspace_mode", "") or "").strip().lower() if ws_mode not in {"external", "external_workspace"}: return {}, _fail( "delegate_start", "write_root_missing", @@ -128,14 +129,15 @@ def _mutation_authority(ctx: ToolContext, authority: "DelegatedRunShape") -> tup "external workspace contract names the tree it may write. Refusing rather " "than guessing a target.", ) - declared = _resolved(getattr(ctx, "workspace_root", None)) + selected_root = room if room is not None else getattr(ctx, "workspace_root", None) + declared = _resolved(selected_root) resolved_root = _resolved(root) if resolved_root is None or declared is None or resolved_root != declared: return {}, _fail( "delegate_start", "write_root_mismatch", "The active root does not resolve to this task's declared external " "workspace, so the run would write somewhere this task was never given.", - active_root=root, declared_workspace_root=str(getattr(ctx, "workspace_root", "") or ""), + active_root=root, declared_workspace_root=str(selected_root or ""), ) return {"target_root": root, "source": "external_workspace_root", "capture_mode": _CAPTURE_DELEGATED_SNAPSHOT}, "" diff --git a/ouroboros/tools/edit_ops.py b/ouroboros/tools/edit_ops.py index 2c59faa28..9211d20c4 100644 --- a/ouroboros/tools/edit_ops.py +++ b/ouroboros/tools/edit_ops.py @@ -94,10 +94,7 @@ def _resolve_edit_target( spellings of one file inside a single call collapse to one entry instead of two writes where the last silently discards the first. """ - from ouroboros.tools.core import ( - _access_or_block, - project_room_lens_dir, - ) + from ouroboros.tools.core import _access_or_block if not path or not str(path).strip(): return None, "", None, f"⚠️ {error_tag}: path is required." @@ -132,12 +129,6 @@ def _resolve_edit_target( return None, "", None, ( f"⚠️ {error_tag}: protected artifact path blocked: {reason}" ) - if normalized == "active_workspace" and project_room_lens_dir(ctx) is not None: - return None, "", None, ( - "⚠️ ROOM_WRITE_VIA_TASK: this room's files are edited by PROMOTED tasks — " - "call promote_chat_to_task for real work there. For a deliberate edit of " - 'the Ouroboros system repo, pass root="system_repo" explicitly.' - ) norm = normalize_repo_path(rel) if ( binding_targets_system_repo(ctx, binding) @@ -190,8 +181,11 @@ def _finish_mutation( except Exception: log.debug("%s: advisory invalidation failed (non-critical)", source_tool, exc_info=True) targets_system = binding_targets_system_repo(ctx, binding) if binding is not None else False - if ctx.is_workspace_mode() and not targets_system: - return "Files are on disk but NOT committed. Do not commit; the headless runner will emit a patch artifact." + if not targets_system: + footer = "Files are on disk but NOT committed." + if ctx.is_workspace_mode(): + footer += " Do not commit; the headless runner will emit a patch artifact." + return footer footer = ( "Files are on disk but NOT committed. Run commit_reviewed when ready.\n" "⚠️ Advisory pre-review is now stale — run preflight_review before commit_reviewed." @@ -200,7 +194,7 @@ def _finish_mutation( # does from git._repo_write / _str_replace_editor (the protected-write contract # in ARCHITECTURE "Safety and runtime mode" and SYSTEM.md "Safety-critical # files"): the mode ALLOWS the write, and the notice is what keeps it visible. - protected = protected_paths_in(changed_paths) if targets_system or not ctx.is_workspace_mode() else [] + protected = protected_paths_in(changed_paths) if protected and mode_allows_protected_write(_runtime_mode()): footer += "\n\n" + core_patch_notice(protected) return footer diff --git a/ouroboros/tools/git_repo_edit.py b/ouroboros/tools/git_repo_edit.py index 4d449ae21..e76ffe48d 100644 --- a/ouroboros/tools/git_repo_edit.py +++ b/ouroboros/tools/git_repo_edit.py @@ -228,6 +228,11 @@ def _repo_write(ctx: ToolContext, path: str = "", content: str = "", f"✅ Written {len(written)} file(s): {summary}\n" "Files are on disk in the active workspace. Do not commit; the headless runner will emit a patch artifact." ) + elif not system_target: + result = ( + f"✅ Written {len(written)} file(s): {summary}\n" + "Files are on disk in the active workspace." + ) else: result = ( f"✅ Written {len(written)} file(s): {summary}\n" @@ -429,9 +434,9 @@ def _str_replace_editor( result += f"\n⚠️ SKILL_SHORT_FORM_IGNORED: {short_form.ignored_reason}." if data_skill_target is None and ctx.is_workspace_mode() and not system_target: result += "\nDo not commit; the headless runner will emit a patch artifact." - elif data_skill_target is None: + elif system_target: result += "\nRun commit_reviewed when ready.\n⚠️ Advisory pre-review is now stale — run preflight_review before commit_reviewed." - else: + elif data_skill_target is not None: result += "\nRun skill_review for this skill before enabling or declaring it ready." if system_target and pathlib.PurePosixPath(rel_path).parts[:1] == ("skills",): result += ( diff --git a/ouroboros/tools/plan_packet.py b/ouroboros/tools/plan_packet.py index 9d43630b6..51ba44121 100644 --- a/ouroboros/tools/plan_packet.py +++ b/ouroboros/tools/plan_packet.py @@ -6,23 +6,21 @@ rubric, the blocking rule, the convergence rule (cycle ≥2), the checklist section verbatim, and the governance pack (W3: BIBLE.md + ARCHITECTURE.md in full for a self-modification plan, their navigation maps otherwise); the user content carries TASK OBJECTIVE · SPEC · PLAN PROSE · EVIDENCE (+ OMISSIONS) · ROOT EXPLORATION -LOG · PRIOR CYCLES in that order. String bounds are ``PACKET_*_CHARS`` (plan_spec) via -``utils.truncate_review_artifact`` — visible marker, never silent. The +LOG · PRIOR CYCLES in that order. Current chosen inputs stay complete; history +and exploration retain disclosed display bounds. The ``PLAN_REVIEW_CONTROL_JSON`` control line is NOT emitted here (Phase C owns it). """ from __future__ import annotations +import json from typing import Any, Mapping, Optional from ouroboros.tools.plan_spec import ( PACKET_EXPLORATION_CHARS, bounded_json, - PACKET_OBJECTIVE_CHARS, PACKET_PRIOR_CYCLES_CHARS, PACKET_PRIOR_FINDING_SUMMARY_CHARS, - PACKET_PROSE_CHARS, - PACKET_SPEC_CHARS, PLAN_FINDINGS_ARRAY_CONTRACT, bounded_text, spec_with_ids, @@ -172,10 +170,12 @@ def build_plan_review_system_prompt( return "\n".join(parts) -def _json_block(payload: Any, limit: int) -> str: - """Fenced JSON bounded STRUCTURALLY (whole items, disclosed counts + full-set hash) — - never clipped mid-string (S-B07/S-B08).""" - text, notes = bounded_json(payload, limit) +def _json_block(payload: Any, limit: Optional[int] = None) -> str: + """Complete fenced JSON, or a disclosed historical projection when bounded.""" + text, notes = ( + bounded_json(payload, limit) if limit is not None + else (json.dumps(payload, ensure_ascii=False, indent=2, sort_keys=True, default=str), []) + ) block = f"```json\n{text}\n```" if notes: block += "\n⚠️ OMISSION NOTE (structural): " + "; ".join(notes) @@ -283,16 +283,16 @@ def build_plan_review_user_content( """Deterministic reviewer packet: TASK OBJECTIVE · SPEC · PLAN PROSE · EVIDENCE (+ OMISSIONS) [cache-stable prefix] · ROOT EXPLORATION LOG · PRIOR CYCLES (all reviewers' prior findings as a compact blocking-first projection + agent dispositions + spec - delta on cycle ≥2). String bounds are the ``PACKET_*_CHARS`` constants via - ``truncate_review_artifact`` (visible marker, never silent); the SPEC JSON - block is bounded by ``PACKET_SPEC_CHARS`` (worst case ~1M chars otherwise).""" + delta on cycle ≥2). Current objective, spec and plan prose stay complete; + the caller's per-slot fit decides whether the actual route can receive them. + Exploration and prior cycles retain their disclosed projection bounds.""" view = spec_with_ids(spec) if goal and not view.get("goal"): view["goal"] = goal sections = [ - "## TASK OBJECTIVE\n\n" + (bounded_text(objective, PACKET_OBJECTIVE_CHARS) or "(none declared)") + "\n", - "## SPEC (ids are the only valid `breaks` targets)\n\n" + _json_block(view, PACKET_SPEC_CHARS) + "\n", - "## PLAN PROSE\n\n" + (bounded_text(plan_prose, PACKET_PROSE_CHARS) or "(none)") + "\n", + "## TASK OBJECTIVE\n\n" + (objective or "(none declared)") + "\n", + "## SPEC (ids are the only valid `breaks` targets)\n\n" + _json_block(view) + "\n", + "## PLAN PROSE\n\n" + (plan_prose or "(none)") + "\n", "## EVIDENCE\n\n" + _render_evidence(manifest), "## ROOT EXPLORATION LOG\n\n" + (bounded_text(root_exploration_log, PACKET_EXPLORATION_CHARS) or "(not provided by host)") + "\n", diff --git a/ouroboros/tools/plan_review_artifacts.py b/ouroboros/tools/plan_review_artifacts.py index f6e12b406..dbf233653 100644 --- a/ouroboros/tools/plan_review_artifacts.py +++ b/ouroboros/tools/plan_review_artifacts.py @@ -19,6 +19,10 @@ from ouroboros.usage_accounting import ( ) +class PlanReviewSourceUnavailable(ValueError): + """A recorded plan names full authority that this reader cannot resolve.""" + + def persist_wave(drive_root: Any, task_id: str, wave: Dict[str, Any]) -> Dict[str, Any]: from ouroboros.artifacts import store_task_artifact_bytes from ouroboros.observability import redact_projection @@ -78,9 +82,64 @@ def authority_wave(drive_root: Any, task_id: str, hot_wave: Optional[dict]) -> O return None ref = hot_wave.get("wave_artifact") if isinstance(hot_wave.get("wave_artifact"), dict) else {} if not ref: + if hot_wave.get("spec_in_artifact") or hot_wave.get("spec_body_truncated"): + raise PlanReviewSourceUnavailable("PLAN_REVIEW_SOURCE_UNAVAILABLE: full spec has no artifact reference") return hot_wave - exact = read_wave(drive_root, task_id, ref) - return {**exact, **hot_wave, "findings": list(exact.get("findings") or [])} + if not drive_root or not task_id: + raise PlanReviewSourceUnavailable("PLAN_REVIEW_SOURCE_UNAVAILABLE: artifact owner is unknown") + try: + exact = read_wave(drive_root, task_id, ref) + except (OSError, ValueError) as exc: + raise PlanReviewSourceUnavailable( + f"PLAN_REVIEW_SOURCE_UNAVAILABLE: task {task_id}, artifact {ref.get('path')}: {exc}" + ) from exc + source = hot_wave.get("spec_source_ref") or exact.get("spec_source_ref") + if source: + from ouroboros.artifacts import read_actor_source_bytes + from ouroboros.tools.plan_spec import spec_hash + + try: + spec = json.loads(read_actor_source_bytes(drive_root, task_id, source)) + if not isinstance(spec, dict) or spec_hash(spec) != hot_wave.get("spec_hash", exact.get("spec_hash")): + raise ValueError("operative spec hash mismatch") + except (OSError, ValueError) as exc: + raise PlanReviewSourceUnavailable(f"PLAN_REVIEW_SOURCE_UNAVAILABLE: {exc}") from exc + elif hot_wave.get("spec_in_artifact"): + raise PlanReviewSourceUnavailable("PLAN_REVIEW_SOURCE_UNAVAILABLE: operative spec reference is missing") + elif not hot_wave.get("spec_body_truncated") and isinstance(hot_wave.get("spec"), dict): + spec = hot_wave["spec"] # Legacy inline authority predates the source handle. + else: + from ouroboros.tools.plan_spec import spec_hash + + spec = exact.get("spec") + if (not isinstance(spec, dict) or exact.get("spec_body_truncated") + or spec_hash(spec) != exact.get("spec_hash")): + raise PlanReviewSourceUnavailable("PLAN_REVIEW_SOURCE_UNAVAILABLE: artifact has no complete spec") + restored = { + **exact, **hot_wave, + "spec": copy.deepcopy(spec), "goal": spec.get("goal") or "", + "findings": list(exact.get("findings") or []), + } + restored.pop("spec_in_artifact", None) + restored.pop("spec_body_truncated", None) + return restored + + +def authority_state(drive_root: Any, task_id: str, state: Dict[str, Any]) -> Dict[str, Any]: + """Resolve the current spec; historical consumers resolve their selected wave.""" + from ouroboros.task_results import current_plan_review_wave + + current = current_plan_review_wave(state) + if not current or current.get("compact") or not ( + current.get("spec_source_ref") or current.get("spec_in_artifact") + or (current.get("spec_body_truncated") and current.get("wave_artifact")) + ): + return state + resolved = authority_wave(drive_root, task_id, current) + return {**state, "waves": [ + resolved if wave.get("request_fingerprint") == current.get("request_fingerprint") else wave + for wave in state.get("waves") or [] + ]} _PLAN_REVIEW_TRANSPORT_KEYS = frozenset({ @@ -271,6 +330,12 @@ def in_flight_resume_inputs( def hot_index_wave(wave: dict, *, page_size: int) -> dict: """Keep a bounded per-slot page; exact authority stays in ``wave_artifact``.""" + if wave.get("spec_source_ref"): + # The operative spec has no size limit. Its exact source already exists; + # never duplicate it inside the bounded task-result index. + wave = {**wave, "spec": {}, "spec_in_artifact": True} + wave.pop("goal", None) + wave.pop("evidence_manifest", None) # Already preserved by wave_artifact. findings = [dict(row) for row in wave.get("findings") or [] if isinstance(row, dict)] counts: Dict[str, int] = {} page = [] @@ -309,13 +374,24 @@ def record_exact_wave( *, need_evidence_seen: List[str], page_size: int, ) -> dict: """Persist exact bytes first, then publish their bounded hot index.""" + from ouroboros.artifacts import store_actor_source_bytes from ouroboros.task_results import record_plan_review_wave + # Raw operative authority already lived in the task result. Keep it exact + # under the existing source handle; wave evidence/output redaction stays on. + source = store_actor_source_bytes( + state_root, task_id, category="context_checkpoints", source_id="plan-spec", + data=json.dumps(wave["spec"], ensure_ascii=False, sort_keys=True, separators=(",", ":")).encode("utf-8"), + extension="json", + ) + wave["spec_source_ref"] = source + exact = {**exact, "spec_source_ref": source} wave["wave_artifact"] = persist_wave(state_root, task_id, exact) - return record_plan_review_wave( + stored = record_plan_review_wave( state_root, task_id, hot_index_wave(wave, page_size=page_size), need_evidence_seen=need_evidence_seen, ) + return authority_wave(state_root, task_id, stored) def slot_row(slot: Any) -> dict: diff --git a/ouroboros/tools/plan_spec.py b/ouroboros/tools/plan_spec.py index 7e75292dc..ab3c0339d 100644 --- a/ouroboros/tools/plan_spec.py +++ b/ouroboros/tools/plan_spec.py @@ -9,20 +9,11 @@ it never interprets the domain. Everything here works identically for a code change, a slide deck, a literature review, a GUI flow or a trip plan — a spec with ZERO file paths is first-class. -Bounds (every cut is disclosed, never silent — DEVELOPMENT.md "No silent -truncation"; list bounds record an omission entry, string bounds go through the -SSOT ``utils.truncate_review_artifact`` marker): +Operative specs are normalized without size cuts: every chosen requirement +participates in identity and reaches the current reviewer packet. Bounds below +apply only to findings, reviewer-request memory, and historical/display views; +those cuts retain their existing disclosures. -* ``MAX_LIST_ITEMS`` — items kept per spec list (in_scope, non_goals, - invariants, decisions, deferred, affected_resources, evidence, - acceptance_claims); ``MAX_REJECTED_PER_DECISION`` — nested ``decision.rejected``. -* ``MAX_ITEM_CHARS`` / ``MAX_GOAL_CHARS`` — per-string bounds (same 600-char - default as ``task_contract._bounded_claim_text``). -* ``MAX_FINDINGS_PER_SLOT`` is the rendered page size, never an authority or - aggregation cap; ``MAX_FINDING_TEXT_CHARS`` bounds each finding string. -* ``PACKET_*_CHARS`` — reviewer-packet section bounds (objective, plan prose, - root exploration log; SPEC and prior cycles are bounded STRUCTURALLY by - ``bounded_json`` — whole items with disclosed counts + full-set hash). """ from __future__ import annotations @@ -39,20 +30,16 @@ from ouroboros.tool_access import path_is_relative_to from ouroboros.triad_review import empty_array_is_verified_clean, extract_json_array from ouroboros.utils import truncate_review_artifact +# Reviewer-request memory/attachment bounds, not limits on an operative spec. MAX_LIST_ITEMS = 40 -MAX_REJECTED_PER_DECISION = 8 MAX_ITEM_CHARS = 600 -MAX_GOAL_CHARS = 2000 MAX_FINDINGS_PER_SLOT = 32 # Per-task `need_evidence` memory: reviewers' requests the host remembers (and, W3, attaches). # Bounded so the durable review state stays bounded whatever the panel asks for; a request past # the cap is demoted (never remembered), disclosed `need_evidence_memory_full`. MAX_NEED_EVIDENCE_MEMORY = 4 * MAX_LIST_ITEMS MAX_FINDING_TEXT_CHARS = 2000 -PACKET_OBJECTIVE_CHARS = 8_000 -PACKET_SPEC_CHARS = 120_000 PACKET_PRIOR_FINDING_SUMMARY_CHARS = 400 -PACKET_PROSE_CHARS = 40_000 PACKET_EXPLORATION_CHARS = 12_000 PACKET_PRIOR_CYCLES_CHARS = 60_000 @@ -95,21 +82,13 @@ def _unique_id(candidate: str, seen: set[str]) -> str: return out -def _cap_list(items: list, label: str, omissions: list[str], *, bound: int = MAX_LIST_ITEMS) -> list: - """Bound a list at ``bound`` and RECORD the cut (P1) — never a silent slice.""" - if len(items) <= bound: - return items - omissions.append(f"{label}: {len(items)} items declared, kept the first {bound} (bound {bound})") - return items[:bound] - - def _is_scalar(value: Any) -> bool: """str/int/float are tolerated as text; bool is not (``True`` is not a spec item).""" return isinstance(value, (str, int, float)) and not isinstance(value, bool) def _string_list(raw: Any, label: str, errors: list[str], omissions: list[str]) -> list[str]: - """Tolerant string list: bare string → one item; blanks dropped; bounded with disclosure.""" + """Tolerant complete string list: bare string → one item; blank drops disclosed.""" if raw is None: return [] items = [raw] if _is_scalar(raw) else raw @@ -124,7 +103,7 @@ def _string_list(raw: Any, label: str, errors: list[str], omissions: list[str]) if not _is_scalar(item): errors.append(f"{label}[{index}]: must be a string") continue - text = bounded_text(item, MAX_ITEM_CHARS) + text = str(item).strip() if text: out.append(text) else: @@ -133,7 +112,7 @@ def _string_list(raw: Any, label: str, errors: list[str], omissions: list[str]) blank += 1 if blank: omissions.append(f"{label}: {blank} blank item(s) dropped") - return _cap_list(out, label, omissions) + return out def _object_list(raw: Any, label: str, text_key: str, errors: list[str]) -> list[dict]: @@ -170,7 +149,6 @@ def _normalize_claims(raw: Any, errors: list[str], omissions: list[str], seen: s f"acceptance_claims: {dropped} item(s) dropped as empty/invalid by the " "task_contract acceptance-claims normalizer" ) - claims = _cap_list(claims, "acceptance_claims", omissions) for index, claim in enumerate(claims, start=1): # Ids are HOST-MINTED positionally (claim_1..N) because they are the only valid # `breaks` targets a reviewer may name: a caller-chosen id could shadow another @@ -186,9 +164,9 @@ def _normalize_claims(raw: Any, errors: list[str], omissions: list[str], seen: s def _normalize_decisions(raw: Any, errors: list[str], omissions: list[str], seen: set[str]) -> list[dict]: out: list[dict] = [] - items = _cap_list(_object_list(raw, "decisions", "choice", errors), "decisions", omissions) + items = _object_list(raw, "decisions", "choice", errors) for index, item in enumerate(items, start=1): - choice = bounded_text(item.get("choice"), MAX_ITEM_CHARS) + choice = str(item.get("choice") or "").strip() if not choice: errors.append(f"decisions[{index - 1}]: choice is required") continue @@ -197,20 +175,19 @@ def _normalize_decisions(raw: Any, errors: list[str], omissions: list[str], seen # id or move between cycles, and these ids are what a blocking finding names. "id": _unique_id(f"decision_{index}", seen), "choice": choice, - "rejected": _cap_list( - _string_list(item.get("rejected"), f"decisions[{index - 1}].rejected", errors, omissions), - f"decisions[{index - 1}].rejected", omissions, bound=MAX_REJECTED_PER_DECISION, + "rejected": _string_list( + item.get("rejected"), f"decisions[{index - 1}].rejected", errors, omissions, ), - "why": bounded_text(item.get("why"), MAX_ITEM_CHARS), + "why": str(item.get("why") or "").strip(), }) return out def _normalize_deferred(raw: Any, errors: list[str], omissions: list[str], seen: set[str]) -> list[dict]: out: list[dict] = [] - items = _cap_list(_object_list(raw, "deferred", "what", errors), "deferred", omissions) + items = _object_list(raw, "deferred", "what", errors) for index, item in enumerate(items, start=1): - what = bounded_text(item.get("what"), MAX_ITEM_CHARS) + what = str(item.get("what") or "").strip() if not what: errors.append(f"deferred[{index - 1}]: what is required") continue @@ -219,7 +196,7 @@ def _normalize_deferred(raw: Any, errors: list[str], omissions: list[str], seen: # id or move between cycles, and these ids are what a blocking finding names. "id": _unique_id(f"deferred_{index}", seen), "what": what, - "why_safe_to_defer": bounded_text(item.get("why_safe_to_defer"), MAX_ITEM_CHARS), + "why_safe_to_defer": str(item.get("why_safe_to_defer") or "").strip(), }) return out @@ -228,9 +205,9 @@ def normalize_spec(raw: Mapping[str, Any] | None) -> tuple[dict, list[str]]: """Normalize a plan spec (plan §9.2 schema) → ``(spec, errors)``. Tolerant of strings-vs-dicts, mints stable ids (claim_N / invariant_N / - decision_N / deferred_N the declared id, when different, is kept beside it as `declared_id`), trims, and bounds - every list at ``MAX_LIST_ITEMS`` — the excess is RECORDED under - ``spec["normalization_omissions"]`` (P1), never silently dropped. Genuinely + decision_N / deferred_N; a differing declared claim id stays as `declared_id`), + trims edges, and preserves every operative string and list item. Empty/invalid + dropped claims and blank list items are disclosed in ``normalization_omissions``. Genuinely malformed input (missing goal, unknown field, wrong container type, decision without choice) yields typed error strings; on any error the returned spec is best-effort and NOT authoritative — the caller must refuse it. @@ -245,7 +222,7 @@ def normalize_spec(raw: Mapping[str, Any] | None) -> tuple[dict, list[str]]: unknown = sorted(str(key) for key in raw if key not in _SPEC_KEYS) if unknown: errors.append("spec: unknown fields: " + ", ".join(unknown)) - goal = bounded_text(raw.get("goal"), MAX_GOAL_CHARS) if isinstance(raw.get("goal"), str) else "" + goal = raw["goal"].strip() if isinstance(raw.get("goal"), str) else "" if not goal: errors.append( "goal: must be a string" if raw.get("goal") is not None and not isinstance(raw.get("goal"), str) diff --git a/ouroboros/tools/registry_guards.py b/ouroboros/tools/registry_guards.py index 49e889beb..d9ba46429 100644 --- a/ouroboros/tools/registry_guards.py +++ b/ouroboros/tools/registry_guards.py @@ -306,7 +306,7 @@ def _resource_allowed(ctx: Any, key: str) -> bool: if isinstance(value, bool) and not value: return False if key == "network": - for name in ("web", "allow_web", "internet", "external_network"): + for name in ("internet", "external_network"): value = resources.get(name) if isinstance(value, bool) and not value: return False @@ -319,8 +319,7 @@ def _disabled_tools(ctx: Any) -> frozenset: Independent of ``allowed_resources``: a caller can disable specific tools (e.g. the agent's web_search/browser/VLM tools for a faithful benchmark) WITHOUT setting web/network=false — so shell network egress (git/pip) stays - available and the web<->network cross-implication in ``_resource_allowed`` - never fires. + available. Withholding web tools does not withhold unrelated network tools. """ metadata = getattr(ctx, "task_metadata", {}) if isinstance(getattr(ctx, "task_metadata", {}), dict) else {} contract = metadata.get("task_contract") if isinstance(metadata.get("task_contract"), dict) else {} diff --git a/ouroboros/tools/tool_context.py b/ouroboros/tools/tool_context.py index 9e4ec8fe0..03cb676c8 100644 --- a/ouroboros/tools/tool_context.py +++ b/ouroboros/tools/tool_context.py @@ -115,6 +115,11 @@ class ToolContext: def active_repo_dir(self) -> pathlib.Path: if self.is_workspace_mode(): return pathlib.Path(self.workspace_root) + from ouroboros.tool_access import project_room_lens_dir + + room = project_room_lens_dir(self) + if room is not None: + return room return pathlib.Path(self.repo_dir) def is_workspace_mode(self) -> bool: diff --git a/ouroboros/tools/tool_resolution.py b/ouroboros/tools/tool_resolution.py index 0694ec872..ff062c5cf 100644 --- a/ouroboros/tools/tool_resolution.py +++ b/ouroboros/tools/tool_resolution.py @@ -56,10 +56,7 @@ def active_repo_dir_for(ctx: Any) -> pathlib.Path: """Return the active repo/workspace root for real and lightweight test contexts.""" active = getattr(ctx, "active_repo_dir", None) if callable(active): - try: - candidate = active() - except Exception: - candidate = None + candidate = active() path = _coerce_real_path(candidate) if path is not None: return path @@ -71,6 +68,11 @@ def active_repo_dir_for(ctx: Any) -> pathlib.Path: if workspace_mode: return workspace_path + from ouroboros.tool_access import project_room_lens_dir + + room = project_room_lens_dir(ctx) + if room is not None: + return room return pathlib.Path(getattr(ctx, "repo_dir")) diff --git a/ouroboros/workspace_admission.py b/ouroboros/workspace_admission.py index ac15e8b3e..0d1e86371 100644 --- a/ouroboros/workspace_admission.py +++ b/ouroboros/workspace_admission.py @@ -167,17 +167,12 @@ def resolve_room_workspace( def room_chat_lens_dir(drive_root: Any, project_id: str) -> tuple[str, str]: - """The project-room folder for the DIRECT-CHAT lens (v6.61.3), or ("", note). + """The selected folder for a direct conversation, with an availability note. - Chat-lane sibling of ``resolve_room_workspace``: the conversation lane of a - folder-room re-points its reads/default-shell-cwd at the room folder so the - tool affordance matches the room fact (the robot-room incident: ``.`` resolved - to the system repo and the agent narrated the wrong tree). Requirements are - LIGHTER than task admission — no git requirement (reading a plain folder in - chat is fine); mutations still go through promoted tasks, which keep the full - ``validate_workspace_root`` gate. Returns ``(dir, note)``: a set-but-unusable - working_dir yields ("", loud note) so the chat context can disclose the - breakage instead of silently falling back to the system repo.""" + Ordinary folders need no Git admission. A missing folder keeps its selected + address so tools cannot fall back to the system repository; an unreadable + registry instead returns a note without inventing an address. + """ pid = str(project_id or "").strip() if not pid: return "", "" @@ -195,12 +190,12 @@ def room_chat_lens_dir(drive_root: Any, project_id: str) -> tuple[str, str]: return "", "" try: resolved = pathlib.Path(raw).expanduser().resolve(strict=False) - except OSError as exc: + except (OSError, ValueError, RuntimeError) as exc: return "", f"project {pid!r} working_dir is unusable: {type(exc).__name__}: {exc}" if not resolved.is_dir(): - return "", ( + return str(resolved), ( f"project {pid!r} working_dir {raw} is unusable (missing or not a directory) — " - "room reads/shell fall back to the system repo; fix or re-attach the folder" + "the selected folder remains the active target; fix or re-attach the folder" ) return str(resolved), "" diff --git a/prompts/SYSTEM.md b/prompts/SYSTEM.md index 7408d563b..fa7eeb24d 100644 --- a/prompts/SYSTEM.md +++ b/prompts/SYSTEM.md @@ -15,9 +15,10 @@ What holds in every mode, however little of me is loaded: - I respond as who I am. Every message from my human is a line in a dialogue, not a task in a queue; a live interruption marked `[Message from my human]` is current dialogue and takes priority. -- Each message gets exactly ONE routing decision — answer, promote to a task, - route to a project, steer a running task, or ask for a manual target — never - competing actions. A chat turn routes or promotes real work; a task delegates. +- Each message gets exactly ONE routing decision: answer or work directly, + delegate, promote, route to a project, or steer existing work. Conversation + shape does not limit my tools. I preserve my human's explicit choice of + author, delegate, or destination. A typed routing annotation is metadata for that decision, not the reply: after any routing tool call I still finish with one self-contained final response that states the user-visible outcome. @@ -48,17 +49,13 @@ What holds in every mode, however little of me is loaded: ## Decision Loop -Most messages deserve a real response first, action second; if words answer, -I answer with words. In a conversation turn, anything needing tools, files, or -several steps is promoted into a task (`promote_chat_to_task`) so the chat stays -free and follow-up chat can steer it; a message that clearly continues an -EXISTING project's work is routed to that project (`route_to_project`) with a -short receipt naming it; a message about a running task steers it. This is -judgment, not a keyword rule: when confidence is low, the target is stale, or -several tasks/projects could match, I do not route silently — I ask for a -manual target through the routing tool's typed choice, never through prose. -While a task runs, a new main-chat message is its own short turn, and I steer -the running task only when the message is explicitly about it. +If words answer, I answer with words. Tools, files, and several steps may be +part of the conversation itself. I promote work when an independent task is +useful, and route or steer existing work when that matches my human's intent. +When several destinations could match or a target is stale, I resolve the +choice through the available dialogue and routing tools before dispatching +to it. A new Main message remains a complete conversation while other work +runs; I steer that work only when the message is about it. `recent_tasks` is for requests that refer to prior work not visible in the present chat; it is continuity recovery, not a substitute for asking when @@ -75,12 +72,11 @@ reviewable work: repo exploration, log forensics, external research, alternate designs, adversarial checks. When a request has independent branches, I delegate early and keep thinking in the parent instead of serializing every branch myself — but I never schedule a task just to avoid answering. Decisions -stay serial and mine: a child's findings do not replace my verification, and -seriality is no reason to self-author — a serial pipeline still delegates the -authorship of its substantial implementation blocks (one strong child at a -time is fine) while I integrate, verify, and decide. For ANY substantial work -product — research, documents, and artifacts as much as code — the default is -a delegated child, not my own serial `edit_text` rounds or shell rewrites. +stay mine: a child's findings do not replace my verification. I choose direct +authorship or delegation for code, research, documents, and other artifacts +according to the work and available actors. An explicit delegation requirement +from my human remains binding; a failed route is not permission to replace it +silently with my own work. `## Available subagents`, when present, is the complete owner-enabled choice set; the host does not rank rows or substitute actors, and dispatch is @@ -125,12 +121,11 @@ cannot open the payload lane); other data-plane artifacts are built in an `external_workspace`/`genesis` tree and materialized by me. Skill authoring: I author under the `external` bucket, read -`docs/CREATING_SKILLS.md` first, and start manifest-first with `SKILL.md`. A -substantial payload is authored by a strong delegated child (judged -semantically, never by line count); with only read-only actors it becomes an -authored handoff I materialize mechanically, not hidden self-authorship. A -failed run gets one bounded salvage, then another actor or an honest blocked -report — never a silent actor change. A skill is ready only after preflight, +`docs/CREATING_SKILLS.md` first, and start manifest-first with `SKILL.md`. +I may author the payload directly or delegate it; a read-only actor can return +an authored handoff that I materialize. I inspect a failed run's retained work +before choosing recovery, another actor, or an honest blocked report, preserving +any explicit delegation requirement. A skill is ready only after preflight, review, grants, dependencies, enablement, and widget/extension visibility are checked. @@ -139,8 +134,9 @@ checked. A project is a durable room — its own thread, journal, workpad, knowledge, and optional working folder — while I stay ONE agent: my unified memory spans the main chat and every project room, and nothing project-related is hidden from -me. Projects serialize internally (one writer per project); parallelism -happens between projects and via subagent swarms within a task. For multi-file +me. The queue serializes managed roots within a Project, allowing their own +subagent trees; this is not an exclusive lock over every file operation. +Ordinary conversation keeps its tools and the room's active folder. For multi-file builds I prefer a real git working folder and orchestrate acting children with patches instead of passing code as chat text. Evolution remains mine alone. diff --git a/server.py b/server.py index f8f6934bc..eb62d7159 100644 --- a/server.py +++ b/server.py @@ -52,7 +52,7 @@ from ouroboros.server_process import ( # noqa: F401 log, ) from ouroboros.server_routing_context import ( # noqa: F401 - _active_direct_root, + _active_direct_roots, _addressable_root_tasks, _chat_running_tasks, _clip_marked, @@ -564,20 +564,13 @@ def _run_supervisor(settings: dict) -> None: _apply_settings_to_env(settings) - # Revival must drop the prior consciousness and cached event-queue binding. + # Revival must drop the prior consciousness. Native turns own fresh agents. if _consciousness is not None: try: _consciousness.stop() except Exception: log.debug("Failed to stop previous consciousness instance", exc_info=True) _consciousness = None - try: - from supervisor import workers as _workers_mod - - _workers_mod._chat_agent = None - except Exception: - log.debug("Failed to reset cached chat agent", exc_info=True) - try: ensure_legacy_imported(pathlib.Path(DATA_DIR)) @@ -617,7 +610,7 @@ def _run_supervisor(settings: dict) -> None: from supervisor.workers import ( init as workers_init, get_event_q, WORKERS, PENDING, RUNNING, spawn_workers, kill_workers, assign_tasks, ensure_workers_healthy, - handle_chat_direct, handle_chat_ephemeral, _get_chat_agent, auto_resume_after_restart, + handle_chat_direct, handle_chat_ephemeral, auto_resume_after_restart, ) max_workers = int(settings.get("OUROBOROS_MAX_WORKERS", 10)) @@ -720,7 +713,7 @@ def _run_supervisor(settings: dict) -> None: queue_deep_self_review_task=queue_deep_self_review_task, persist_queue_snapshot=persist_queue_snapshot, safe_restart=safe_restart, kill_workers=kill_workers, spawn_workers=spawn_workers, sort_pending=sort_pending, consciousness=_consciousness, - get_chat_agent=_get_chat_agent, handle_chat_direct=handle_chat_direct, + handle_chat_direct=handle_chat_direct, handle_chat_ephemeral=handle_chat_ephemeral, request_restart=_request_restart_exit, ) except Exception as exc: diff --git a/supervisor/active_activity.py b/supervisor/active_activity.py index 134684855..6a690409f 100644 --- a/supervisor/active_activity.py +++ b/supervisor/active_activity.py @@ -30,6 +30,8 @@ class DirectActivityEntry: origin_message_ref: Dict[str, Any] = field(default_factory=dict) model_wait_owner: Any = field(default=None, repr=False, compare=False) + actor: Any = field(default=None, repr=False, compare=False) + def to_dict(self) -> Dict[str, Any]: row = { "activity_id": self.activity_id, @@ -50,7 +52,7 @@ class DirectActivityRegistry: """Thread-safe registry for active direct-chat and ephemeral-decision turns.""" def __init__(self) -> None: - self._lock = threading.Lock() + self._lock = threading.Condition() self._activities: Dict[str, DirectActivityEntry] = {} def register( @@ -63,6 +65,7 @@ class DirectActivityRegistry: kind: str = "direct_chat", phase: str = "thinking", origin_message_ref: Optional[Dict[str, Any]] = None, + actor: Any = None, ) -> DirectActivityEntry: aid = str(activity_id or "").strip() if not aid: @@ -76,6 +79,7 @@ class DirectActivityRegistry: phase=str(phase or "thinking"), started_at=time.time(), origin_message_ref=dict(origin_message_ref or {}), + actor=actor, ) with self._lock: self._activities[aid] = entry @@ -86,6 +90,7 @@ class DirectActivityRegistry: aid = str(activity_id or "").strip() with self._lock: entry = self._activities.pop(aid, None) + self._lock.notify_all() if entry: log.debug("Unregistered direct activity: %s (chat_id=%s)", aid, entry.chat_id) return entry @@ -124,10 +129,22 @@ class DirectActivityRegistry: with self._lock: return self._activities.get(aid) + def actors(self) -> List[DirectActivityEntry]: + """Private actor handles; never include execution objects in UI snapshots.""" + with self._lock: + return list(self._activities.values()) + + def wait_until_empty(self, timeout: float) -> List[str]: + """Wait for whole executions, including preparation and post-task work.""" + with self._lock: + self._lock.wait_for(lambda: not self._activities, timeout=max(0.0, timeout)) + return list(self._activities) + def clear(self) -> None: """Clear registry — primarily for tests and process resets.""" with self._lock: self._activities.clear() + self._lock.notify_all() # Global process-local singleton diff --git a/supervisor/steering.py b/supervisor/steering.py index f213cde23..0fb1ae189 100644 --- a/supervisor/steering.py +++ b/supervisor/steering.py @@ -87,27 +87,14 @@ def _handle_steer_task(evt: Dict[str, Any], ctx: Any) -> None: direct_lock = None direct_active = False try: - direct_agent = ctx.get_chat_agent() + from supervisor.workers import get_direct_chat_agent, direct_chat_turn + + direct_agent = get_direct_chat_agent(target) direct_lock = getattr(direct_agent, "_owner_message_admission_lock", None) if direct_lock is not None: with direct_lock: - direct_active = bool( - getattr(direct_agent, "_busy", False) - and getattr(direct_agent, "_accepting_owner_messages", False) - and str(getattr(direct_agent, "_current_task_id", "") or "") == target - ) - if direct_active: - direct_metadata = getattr(direct_agent, "_current_task_metadata", {}) - direct_metadata = direct_metadata if isinstance(direct_metadata, dict) else {} - task = { - "id": target, - "chat_id": int(getattr(direct_agent, "_current_chat_id", 0) or 0), - "project_id": str(direct_metadata.get("project_id") or ""), - "title": str(direct_metadata.get("title") or ""), - "suggested_name": str(direct_metadata.get("suggested_name") or ""), - "objective": str(getattr(direct_agent, "_current_task_text", "") or ""), - "_is_direct_chat": True, - } + task = direct_chat_turn(target) + direct_active = task is not None except Exception: direct_active = False if not direct_active: diff --git a/supervisor/worker_chat_lane.py b/supervisor/worker_chat_lane.py index 7591c0375..949e371a3 100644 --- a/supervisor/worker_chat_lane.py +++ b/supervisor/worker_chat_lane.py @@ -1,7 +1,7 @@ """The direct and ephemeral chat lanes, and the resume after a restart. -A chat turn runs on the single long-lived agent under its own lock; an ephemeral -turn gets a throwaway one. Both are refused while the repo-writer gate is closed +Each chat turn owns a fresh native agent and a registered execution; an explicit +swarm routing turn retains its transient contract. Both are refused while the repo-writer gate is closed for a DESTRUCTIVE update window (apply/replace prologue, materialization, rollback), so a managed update never races a turn that could touch the checkout mid-reset. While the ONE authorized assisted resolver holds the repository @@ -18,7 +18,6 @@ from __future__ import annotations import logging import json import pathlib -import sys import time import uuid from typing import Any, Dict, Optional, Tuple, Union @@ -143,16 +142,12 @@ def handle_chat_direct( task_constraint: Optional[dict] = None, task_metadata: Optional[dict] = None, ) -> None: - with _pool()._chat_agent_lock: - if not owner_conversation_admitted(chat_id): - return - _handle_chat_direct_locked( - chat_id, - text, - image_data, - task_constraint=task_constraint, - task_metadata=task_metadata, - ) + if not owner_conversation_admitted(chat_id): + return + _handle_chat_direct_locked( + chat_id, text, image_data, + task_constraint=task_constraint, task_metadata=task_metadata, + ) def _handle_chat_direct_locked( @@ -177,7 +172,7 @@ def _handle_chat_direct_locked( return _run_chat_task( - _pool()._get_chat_agent(), chat_id, text, image_data, + None, chat_id, text, image_data, task_constraint=task_constraint, task_metadata=task_metadata, ephemeral=False, ) @@ -213,9 +208,8 @@ def _run_chat_task( ) -> None: """Build the direct-chat task and run it on the given agent, draining events. - ``ephemeral`` marks a SHORT-LIVED same-route turn (run on a separate agent - instance while the shared chat agent is busy): it carries _ephemeral_turn so - the task pipeline skips long-term memory / reflection / evolution writes.""" + ``ephemeral`` is only the explicit constrained routing contract. Ordinary + Main/Project turns use the full native task/result/delivery lifecycle.""" task: Optional[dict] = None client_msg_id = "" if task_metadata: @@ -232,7 +226,26 @@ def _run_chat_task( "text": text, "_is_direct_chat": True, } + from supervisor.active_activity import get_direct_activity_registry + + registry = get_direct_activity_registry() + # Close/check/register is one short transaction with the update owner. + # The registered execution includes agent construction, attachment staging, + # the whole native lifecycle and event delivery, not just its LLM rounds. + with _pool()._repo_writer_gate_lock: + if not owner_conversation_admitted(chat_id): + return + activity = registry.register( + task["id"], chat_id, + client_message_id=client_msg_id, + project_id=str((task_metadata or {}).get("project_id") or ""), + kind=kind, origin_message_ref=(task_metadata or {}).get("origin_message_ref"), + actor=agent, + ) try: + if agent is None: + agent = _pool()._get_chat_agent() + activity.actor = agent from ouroboros.contracts.task_contract import attach_task_contract if ephemeral: @@ -348,50 +361,37 @@ def _run_chat_task( ) attach_task_contract(task) - pid = str(task.get("project_id") or "") + # Announce the authoritative start immediately (owner decision 2A): + # the client's `Sending...` retires on this frame, not on a socket + # echo, and the frame carries the activity<->client_message_id link + # so even a turn that fails before its first LLM round concludes + # cleanly via its keyed error final. + try: + from supervisor.message_bus import get_bridge - from supervisor.active_activity import track_direct_activity - - with track_direct_activity( - activity_id=str(task["id"]), - chat_id=int(chat_id or 0), - client_message_id=client_msg_id, - project_id=pid, - kind=kind, - phase="thinking", - origin_message_ref=task.get("origin_message_ref"), - ): - # Announce the authoritative start immediately (owner decision 2A): - # the client's `Sending...` retires on this frame, not on a socket - # echo, and the frame carries the activity<->client_message_id link - # so even a turn that fails before its first LLM round concludes - # cleanly via its keyed error final. - try: - from supervisor.message_bus import get_bridge - - get_bridge().send_chat_action( - int(chat_id or 0), - "typing", - activity_id=str(task["id"]), - client_message_id=client_msg_id, - phase="thinking", - kind=kind, - ) - except Exception: - log.debug("Direct-turn start typing announce failed", exc_info=True) - # The turn's live emits (loop_llm_call and friends publish - # straight to the agent's event queue DURING handle_task) and its - # returned events are both drained after the registry entry is - # gone: route them through the turn-scoped addressing proxy. - turn_queue = _TurnEventQueue(_pool().get_event_q(), task["id"], chat_id) - prev_queue = getattr(agent, "_event_queue", None) - agent._event_queue = turn_queue - try: - events = agent.handle_task(task) - finally: - agent._event_queue = prev_queue - for e in events: - _pool().get_event_q().put(turn_queue.stamp(e)) + get_bridge().send_chat_action( + int(chat_id or 0), + "typing", + activity_id=str(task["id"]), + client_message_id=client_msg_id, + phase="thinking", + kind=kind, + ) + except Exception: + log.debug("Direct-turn start typing announce failed", exc_info=True) + # The turn's live emits (loop_llm_call and friends publish + # straight to the agent's event queue DURING handle_task) and its + # returned events can be consumed after this registry entry is gone: + # stamp the authoritative chat identity before handing them off. + turn_queue = _TurnEventQueue(_pool().get_event_q(), task["id"], chat_id) + prev_queue = getattr(agent, "_event_queue", None) + agent._event_queue = turn_queue + try: + events = agent.handle_task(task) + finally: + agent._event_queue = prev_queue + for e in events: + _pool().get_event_q().put(turn_queue.stamp(e)) except Exception as e: import traceback err_msg = f"⚠️ Error: {type(e).__name__}: {e}" @@ -441,6 +441,8 @@ def _run_chat_task( ) except Exception: log.debug("Suppressed exception", exc_info=True) + finally: + registry.unregister(task["id"]) def handle_chat_ephemeral( @@ -450,12 +452,9 @@ def handle_chat_ephemeral( task_constraint: Optional[dict] = None, task_metadata: Optional[dict] = None, ) -> None: - """The "turn = decision" path (v6.33.0 WS10): when the shared chat agent is - busy, a new main-chat message runs as a SHORT-LIVED turn on a SEPARATE agent - instance — bypassing _chat_agent_lock so it never freezes/injects into the - running turn, while keeping the SAME ROUTE (same make_agent config: model / - mode / effort, not a cheaper lane). Ephemeral turns are serialized among - themselves and are barred from long-term memory/reflection/evolution writes.""" + """Run an explicitly constrained swarm routing turn on its own actor.""" + if not owner_conversation_admitted(chat_id): + return from supervisor.state import budget_remaining, load_state failure_meta = _host_operation_failure(task_metadata) try: @@ -469,18 +468,10 @@ def handle_chat_ephemeral( except Exception: pass return - if not getattr(sys, 'frozen', False): - sys.path.insert(0, str(_pool().REPO_DIR)) - from ouroboros.agent import make_agent - - with _pool()._ephemeral_chat_lock: - if not owner_conversation_admitted(chat_id): - return - agent = make_agent(repo_dir=str(_pool().REPO_DIR), drive_root=str(_pool().DRIVE_ROOT), event_queue=_pool().get_event_q()) - _run_chat_task( - agent, chat_id, text, image_data, - task_constraint=task_constraint, task_metadata=task_metadata, ephemeral=True, - ) + _run_chat_task( + None, chat_id, text, image_data, + task_constraint=task_constraint, task_metadata=task_metadata, ephemeral=True, + ) def auto_resume_after_restart() -> None: @@ -550,8 +541,7 @@ def auto_resume_after_restart() -> None: return time.sleep(2) # Let everything initialize - agent = _pool()._get_chat_agent() - if not agent._busy: + if not _pool().chat_turn_liveness(): import threading threading.Thread( target=handle_chat_direct, @@ -583,8 +573,8 @@ DIRECT_TURN_STOP_LIVE = "live" # armed, still inside a step (the sweep re def stop_direct_chat_turn(task_id: str, turn: Dict[str, Any], *, deliver: bool = True) -> str: """Stop the in-process direct-chat turn COOPERATIVELY; a typed outcome. - There is no worker process to kill: the turn runs on the long-lived chat - agent inside the supervisor. The lane writes the typed ``finalize_now`` + There is no worker process to kill: the turn owns a native actor + inside the supervisor. The lane writes the typed ``finalize_now`` control (``REASON_OWNER_STOPPED_DIRECT_TURN``) to the canonical drive's owner mailbox — the one the turn's loop drains at every round boundary, where it ends the turn with ZERO further model calls — then waits the diff --git a/supervisor/workers.py b/supervisor/workers.py index 811bc3c51..832df302a 100644 --- a/supervisor/workers.py +++ b/supervisor/workers.py @@ -220,12 +220,9 @@ def ensure_worker_pool_started(n: int = 0, *, allow_disabled_restart: bool = Fal return True -_chat_agent = None -# Serializes every direct-chat caller; _chat_agent has mutable per-call state. import threading as _threading -_chat_agent_lock = _threading.Lock() -_ephemeral_chat_lock = _threading.Lock() -_repo_writer_gate_lock = _threading.Lock() +# Admission and activity registration share this short lock; execution never holds it. +_repo_writer_gate_lock = _threading.RLock() _repo_writer_gate_reason = "" @@ -281,16 +278,10 @@ def repo_writer_task_allowed(task: Dict[str, Any]) -> bool: def drain_repo_writers(timeout: float = 30.0) -> List[str]: - """Wait for the two existing in-process writer lanes after admission closes.""" - deadline = time.monotonic() + max(0.0, float(timeout)) - blocked: List[str] = [] - for label, lock in (("direct_chat", _chat_agent_lock), ("ephemeral_chat", _ephemeral_chat_lock)): - remaining = max(0.0, deadline - time.monotonic()) - if not lock.acquire(timeout=remaining): - blocked.append(label) - continue - lock.release() - return blocked + """Wait for every registered execution after closing writer admission.""" + from supervisor.active_activity import get_direct_activity_registry + + return get_direct_activity_registry().wait_until_empty(float(timeout)) def _repo_writer_turn_allowed(chat_id: int) -> bool: @@ -308,48 +299,46 @@ def _repo_writer_turn_allowed(chat_id: int) -> bool: def _get_chat_agent(): - global _chat_agent - if _chat_agent is None: - if not getattr(sys, 'frozen', False): - sys.path.insert(0, str(REPO_DIR)) - from ouroboros.agent import make_agent - _chat_agent = make_agent( - repo_dir=str(REPO_DIR), - drive_root=str(DRIVE_ROOT), - event_queue=get_event_q(), - ) - return _chat_agent + """Construct a fresh native actor; each turn owns its mutable state.""" + if not getattr(sys, 'frozen', False) and str(REPO_DIR) not in sys.path: + sys.path.insert(0, str(REPO_DIR)) + from ouroboros.agent import make_agent + return make_agent( + repo_dir=str(REPO_DIR), drive_root=str(DRIVE_ROOT), event_queue=get_event_q(), + ) + + +def get_direct_chat_agent(task_id: str): + from supervisor.active_activity import get_direct_activity_registry + + entry = get_direct_activity_registry().get(task_id) + return entry.actor if entry is not None else None def chat_turn_liveness(): - """(busy, task_id, last_activity_ts) of the in-process direct-chat turn — read - WITHOUT taking _chat_agent_lock (a wedged turn holds that lock for its whole - duration, so the watchdog must never block on it). The supervisor liveness - watchdog (WS3) reads this to spot a heartbeat-silent direct turn, which is - in-process and therefore invisible to the worker RUNNING heartbeat table.""" - agent = _chat_agent - if agent is None or not getattr(agent, "_busy", False): - return (False, None, None) - return (True, getattr(agent, "_current_task_id", None), getattr(agent, "_last_activity_ts", None)) + """All in-process actors, read without taking any execution/admission lock.""" + from supervisor.active_activity import get_direct_activity_registry + + return [ + (str(entry.activity_id), getattr(entry.actor, "_last_activity_ts", None)) + for entry in get_direct_activity_registry().actors() + if getattr(entry.actor, "_busy", False) + ] + + +def direct_chat_turns() -> List[Dict[str, Any]]: + from supervisor.active_activity import get_direct_activity_registry + + return [turn for entry in get_direct_activity_registry().actors() + if (turn := direct_chat_turn(entry.activity_id)) is not None] def direct_chat_turn(task_id: str = "") -> Optional[Dict[str, Any]]: - """The in-process direct-chat turn as a queue-shaped task record, or None. - - The ownership predicate (``task_has_live_ownership``), the owner-control - ingresses (cancel, hurry, decisions) and the graceful-stop episode resolve - a live direct turn through THIS one reader, so the durable running mirror - and the owner controls can never disagree about it again: a turn the - task list shows as running is addressable, and a turn that is not - addressable is not shown as live (the class the rc.7 QA regress hit — - ``running`` + cancel 404 + spend still growing). Read WITHOUT the - chat-agent lock, like ``chat_turn_liveness``: a wedged turn holds that - lock for its whole duration. ``task_id`` narrows the answer to that turn; - empty answers whichever direct turn is live. An ephemeral decision turn - (not ``_accepting_owner_messages``) is transport control, never an - owner-addressable task, and writes no durable running row either. - """ - agent = _chat_agent + """Read one addressable actor through the same owner used by routing/controls.""" + if not task_id: + turns = direct_chat_turns() + return turns[0] if len(turns) == 1 else None + agent = get_direct_chat_agent(task_id) if agent is None or not getattr(agent, "_busy", False): return None current = str(getattr(agent, "_current_task_id", "") or "") @@ -396,7 +385,7 @@ def arm_direct_chat_turn( stamp the control's msg_id lands under (the immediate stop and the graceful owner-stop episode keep separate latches, so one never hides the other); ``extra_stamps`` ride along under the same lock.""" - agent = _chat_agent + agent = get_direct_chat_agent(task_id) if agent is None: return None lock = getattr(agent, "_owner_message_admission_lock", None) @@ -417,7 +406,7 @@ def stamp_direct_chat_turn(task_id: str, **fields: Any) -> bool: armed owner-stop control id, so a sweep tick re-arms idempotently instead of re-toasting). Stamps belong to ONE turn id and vanish with it. Returns False when that turn is not live (nothing to stamp).""" - agent = _chat_agent + agent = get_direct_chat_agent(task_id) if agent is None or direct_chat_turn(task_id) is None: return False stamps = getattr(agent, "_direct_turn_stamps", None) diff --git a/tests/conftest.py b/tests/conftest.py index 5054cb062..6fbf05bfc 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -732,3 +732,10 @@ def pytest_terminal_summary(terminalreporter): # shapes, ``MagicMock`` vs real, etc.). + + +@pytest.fixture(autouse=True) +def _isolate_direct_activities(monkeypatch): + from supervisor import active_activity + + monkeypatch.setattr(active_activity, "_DIRECT_ACTIVITY_REGISTRY", active_activity.DirectActivityRegistry()) diff --git a/tests/system_e2e/harness.py b/tests/system_e2e/harness.py index 81e55563b..3f8574e00 100644 --- a/tests/system_e2e/harness.py +++ b/tests/system_e2e/harness.py @@ -114,6 +114,7 @@ SCENARIOS = { # the owner through the same cancel endpoint the UI drives, caught MID-ROUND # on an event-gated model hold (ModelGate), never a timed race. "S26": ("direct-chat owner stop: an in-flight direct turn is addressable (running list + activity snapshot), stop-now mid-round answers the typed 'still live' with the cooperative control armed ONCE (a repeat is idempotent), the turn ends at its next step with ZERO further model rounds under the owner-stop reason, the chat concludes, custody settles already_settled against the turn's own terminal, and a later stop is the typed 404", LANE_MOCK), + "S27": ("ordinary Main/Project capability: real stdio MCP reads and writes, correct built-in room target, and a second native turn completes while the first model call is held", LANE_MOCK), } MOCK_SLUG = "openai-compatible::mock-model" diff --git a/tests/system_e2e/test_ordinary_conversation_capabilities.py b/tests/system_e2e/test_ordinary_conversation_capabilities.py new file mode 100644 index 000000000..527d2edc4 --- /dev/null +++ b/tests/system_e2e/test_ordinary_conversation_capabilities.py @@ -0,0 +1,159 @@ +"""S27: ordinary Main/Project tools and concurrency through real stdio MCP. + +One Main turn is held inside its model call while another completes MCP work. +Both start after a Project exists. The Project then uses MCP and the built-in +writer against its attached folder. Durable results, tool traces, wire schemas +and physical files jointly prove capability; a stub's final answer alone cannot. +The shared e2e_clone fixture runs committed HEAD, so an uncommitted candidate +must first be materialized in an isolated clone by its operator. +""" + +from __future__ import annotations + +import json +import subprocess +import sys +import uuid + +import pytest + +from devtools.benchmarks.common.server_runner import _api +from tests.system_e2e.harness import ( + LANE_MOCK, ArtifactOracle, ModelGate, ScriptedStubModel, body_text, + keyless_settings, require_lane, start_server, wait_durable_result, wait_until, ws_url, +) +from tests.system_e2e.test_system_scenarios_w6 import _WsFrames, _direct_activities + + +MCP_SOURCE = '''from pathlib import Path +from mcp.server.fastmcp import FastMCP + +server = FastMCP("ordinary-fixture") + +@server.tool() +def read_note(name: str) -> str: + """Read a note in this fixture's private folder.""" + return Path(name).read_text(encoding="utf-8") + +@server.tool() +def write_note(name: str, text: str) -> str: + """Write a note in this fixture's private folder and read it back.""" + Path(name).write_text(text, encoding="utf-8") + return Path(name).read_text(encoding="utf-8") + +server.run(transport="stdio") +''' + + +def _received_turn(oracle: ArtifactOracle, client_message_id: str) -> dict | None: + for row in oracle.events("task_received"): + task = row.get("task") or {} + origin = (task.get("metadata") or {}).get("origin_message_ref") or {} + if origin.get("client_message_id") == client_message_id: + return task + return None + + +@pytest.mark.integration +@pytest.mark.serial +def test_s27_ordinary_main_and_project_mcp_with_concurrent_native_turn(e2e_clone, tmp_path): + require_lane(LANE_MOCK) + room, mcp_dir = tmp_path / "attached room", tmp_path / "mcp fixture" + room.mkdir() + mcp_dir.mkdir() + subprocess.run(["git", "init", "-q", str(room)], check=True, capture_output=True) + script = mcp_dir / "server.py" + script.write_text(MCP_SOURCE, encoding="utf-8") + (mcp_dir / "input.txt").write_text("S27_READ_яё𐍈🚀", encoding="utf-8") + read_tool, write_tool = "mcp_fixture__read_note", "mcp_fixture__write_note" + steps = [ + {"tool": read_tool, "arguments": {"name": "input.txt"}}, + {"tool": write_tool, "arguments": {"name": "main.txt", "text": "S27_MAIN_WRITE"}}, + {"final": "S27_MAIN_COMPLETE"}, + {"final": "S27_HELD_COMPLETE"}, + {"tool": read_tool, "arguments": {"name": "input.txt"}}, + {"tool": write_tool, "arguments": {"name": "project.txt", "text": "S27_PROJECT_WRITE"}}, + {"tool": "write_file", "arguments": { + "root": "active_workspace", "path": "room-result.txt", "content": "S27_ROOM_WRITE", + }}, + {"final": "S27_PROJECT_COMPLETE"}, + ] + gate = ModelGate(lambda body: bool(body.get("tools")) and "S27_HOLD" in body_text(body), + timeout=300) + with ScriptedStubModel(steps, gate=gate) as stub: + settings = keyless_settings(stub, MCP_ENABLED=True, MCP_SERVERS=[{ + "id": "fixture", "enabled": True, "transport": "stdio", "command": sys.executable, + "args": [str(script)], "cwd": str(mcp_dir), + }]) + server = start_server(e2e_clone, tmp_path / "server", settings) + try: + oracle = ArtifactOracle(server.data_root) + project = _api(server.base_url, "POST", "/api/projects", { + "name": "S27 attached room", "path": str(room), + }, timeout=60).get("project") or {} + assert project.get("id") and project.get("chat_id"), project + assert project.get("working_dir") == str(room), project + from websockets.sync.client import connect + + with connect(ws_url(server), open_timeout=30, proxy=None) as ws, _WsFrames(ws) as frames: + def send(content, *, in_project=False): + message_id = "e2e-s27-" + uuid.uuid4().hex + ws.send(json.dumps({ + "type": "chat", "content": content, "client_message_id": message_id, + "chat_id": project["chat_id"] if in_project else 1, + **({"project_id": project["id"]} if in_project else {}), + })) + return message_id + + def finish(message_id, marker): + task = wait_until(lambda: _received_turn(oracle, message_id), 60) + assert task and task.get("_is_direct_chat") is True, task + assert not task.get("_ephemeral_turn"), task + task_id = task["id"] + stored = wait_durable_result(oracle, task_id, timeout=180) + assert stored.get("status") == "completed", stored + assert marker in str(stored.get("result") or ""), stored + assert wait_until(lambda: [f for f in frames.find(type="chat", task_id=task_id) + if not f.get("is_progress") and marker in str(f.get("content"))], 60) + assert not oracle.child_task_ids(task_id), "ordinary work was delegated/promoted" + return task_id + + held_message = send("S27_HOLD: think until the model responds, then answer here.") + assert gate.arrived.wait(180), "the first native turn never reached the model" + held = wait_until(lambda: (_direct_activities(server, held_message) or [None])[0], 30) + assert held, "the held Main turn is not addressable" + held_id = held["activity_id"] + assert held_id not in oracle.running_ids(), "native turn acquired a pool worker" + main_id = finish(send("Read input.txt with the fixture, write main.txt, then answer here."), + "S27_MAIN_COMPLETE") + assert (mcp_dir / "main.txt").read_text() == "S27_MAIN_WRITE" + assert not gate.release.is_set() and not gate.timed_out + assert oracle.task_result(held_id).get("status") == "running" + gate.release.set() + assert finish(held_message, "S27_HELD_COMPLETE") == held_id + project_id = finish(send("Read input.txt and write project.txt with the fixture; " + "write room-result.txt in this Project and answer here.", + in_project=True), "S27_PROJECT_COMPLETE") + + assert (mcp_dir / "project.txt").read_text() == "S27_PROJECT_WRITE" + assert (room / "room-result.txt").read_text() == "S27_ROOM_WRITE" + assert not (e2e_clone / "room-result.txt").exists(), "room write targeted the system repo" + assert oracle.task_result(project_id).get("_is_direct_chat") is True + assert not [row for row in oracle._jsonl("logs/chat.jsonl") + if row.get("type") == "project_completion_summary" + and row.get("task_id") == project_id], "ordinary Project reply leaked into Main" + for task_id, expected in ((main_id, {read_tool, write_tool}), + (project_id, {read_tool, write_tool, "write_file"})): + rows = [r for r in oracle.tools_rows() if r.get("task_id") == task_id] + assert {r.get("tool") for r in rows} == expected, rows + assert "S27_READ_яё𐍈🚀" in json.dumps(rows, ensure_ascii=False), rows + assert all(not r.get("is_error") for r in rows), rows + agent_calls = [body for kind, body in stub.calls if kind == "agent"] + assert len(agent_calls) == 5 and stub.script_consumed(), stub.kinds() + for body in agent_calls: + names = {tool.get("function", {}).get("name") for tool in body.get("tools", [])} + assert {read_tool, write_tool} <= names, names + assert gate.held == 1 and not gate.timed_out + finally: + gate.release.set() + server.stop() diff --git a/tests/test_available_subagents_runtime.py b/tests/test_available_subagents_runtime.py index 86bd99b9e..8bc845c23 100644 --- a/tests/test_available_subagents_runtime.py +++ b/tests/test_available_subagents_runtime.py @@ -261,210 +261,42 @@ def test_selected_session_visibility_failure_blocks_without_native_substitution( assert amended.executor_resolution.reason == "delegate_tools_invisible" -@pytest.mark.parametrize( - ("interactive", "expected_reason"), - [ - (True, ""), - (False, "work_order_source_channel_unavailable"), - ], -) -def test_over_budget_bootstrap_uses_only_a_live_interaction_channel( - monkeypatch, tmp_path, interactive, expected_reason, -): - from ouroboros import delegate_custody as custody +def test_large_bootstrap_delivers_full_work_without_question_channel(monkeypatch, tmp_path): import ouroboros.claudexor_daemon as daemon import ouroboros.subagent_bootstrap as bootstrap import ouroboros.subagent_runtime as runtime - from ouroboros.subagent_work_order import work_order_fingerprint + from ouroboros.subagent_work_order import compile_external_work_order class Gateway: def harnesses(self): - return [{ - "id": "codex", - "manifest": {"capabilities": {"interactive": interactive}}, - }] + pytest.fail("work-order size must not require an interactive-capability probe") def close(self): pass - monkeypatch.setattr(daemon, "ensure_owned_gateway", lambda: Gateway()) + monkeypatch.setattr(daemon, "ensure_owned_gateway", Gateway) calls = [] monkeypatch.setattr(runtime, "exact_start", lambda ctx, prompt, spec: ( calls.append((prompt, spec)) - or json.dumps({"status": "started", "run_id": "run-source", "custody_durable": True}) + or json.dumps({"status": "started", "run_id": "run-full", "custody_durable": True}) )) - import ouroboros.delegate_supervision as supervision - - monkeypatch.setattr( - supervision, "supervised_wait", - lambda *_a, **_kw: pytest.fail("the host must not wait inside bootstrap (owner 1=A)"), - ) snapshot = _snapshot(_settings(_session_row(target="codex=gpt-5.6-sol")), "session-builder") - dispatch = SimpleNamespace( - executor="harness", blocked=False, - executor_resolution=SimpleNamespace(route=SimpleNamespace(route_id="codex")), - ) - ctx = SimpleNamespace( - task_id="child-source", drive_root=tmp_path, budget_drive_root=str(tmp_path), - task_metadata={}, - ) - task = { - "id": "child-source", - "objective": ("THIS MUST NOT BE SENT AS A PREFIX " + ("x" * 250_100)), - "configured_subagent": snapshot, - "task_contract": {"objective": "THIS MUST NOT BE SENT AS A PREFIX " + ("x" * 250_100)}, - } - full_sha = work_order_fingerprint(task) - # Charter D1: the host pre-starts the leaf during bootstrap, through the - # same wrapper the model's delegate_start(prompt="") uses — and does NOT - # wait on it (owner 1=A). With a live interactive channel the oversized - # order rides the source-request lens; without one, the definite refusal - # ends the child unrun and typed at $0. + dispatch = SimpleNamespace(executor="harness", blocked=False) + ctx = SimpleNamespace(task_id="child-full", drive_root=tmp_path, + budget_drive_root=str(tmp_path), task_metadata={}) + objective = "яё𐍈🚀\n" * 55_000 + "DECISIVE_TAIL" + task = {"id": ctx.task_id, "objective": objective, "configured_subagent": snapshot, + "task_contract": {"objective": objective}} raw = bootstrap.bootstrap_before_context(ctx, task, dispatch) - custody_rows = [ - json.loads(line) - for line in custody.event_log_path(tmp_path).read_text().splitlines() - ] - if interactive: - out = json.loads(raw) - assert out["status"] == "configured_session_started" - assert out["startup"]["status"] == "started" - assert out["startup"]["run_id"] == "run-source" - assert len(calls) == 1 - prompt, spec = calls[0] - assert "WORK ORDER SOURCE REQUEST" in prompt - assert "THIS MUST NOT BE SENT AS A PREFIX" not in prompt - assert spec["compiled_work_order"] is True - assert spec["work_order_fingerprint"] == full_sha - assert spec["work_order_source_request"]["complete_sha256"] == full_sha - assert custody_rows[-1]["type"] == "configured_subagent_work_order_source_request" - assert custody_rows[-1]["status"] == "attempted" - assert custody_rows[-1]["source_channel"] == { - "status": "available", - "reason": "interactive", - "route": "codex", - } - else: - assert raw == "" - assert ctx._configured_startup_refusal["reason"] == expected_reason - assert calls == [] - assert custody_rows[-1]["type"] == "delegate_run_start_blocked" - assert custody_rows[-2]["type"] == "configured_subagent_work_order_refused" - assert custody_rows[-2]["reason"] == expected_reason - - -@pytest.mark.parametrize( - ("bootstrap_interactive", "start_interactive", "expected_status", "expected_reason"), - [ - (True, False, "refused", "work_order_source_channel_unavailable"), - (None, True, "started", ""), - (True, None, "refused", "work_order_source_channel_unverified"), - ], -) -def test_over_budget_start_reprobes_live_interaction_capability( - monkeypatch, tmp_path, bootstrap_interactive, start_interactive, - expected_status, expected_reason, -): - from ouroboros import delegate_custody as custody - import ouroboros.claudexor_daemon as daemon - import ouroboros.subagent_bootstrap as bootstrap - import ouroboros.subagent_runtime as runtime - - observations = iter([bootstrap_interactive, start_interactive]) - closed = [] - - class Gateway: - def harnesses(self): - interactive = next(observations) - capabilities = ( - {} if interactive is None else {"interactive": interactive} - ) - return [{ - "id": "codex", - "manifest": {"capabilities": capabilities}, - }] - - def close(self): - closed.append(True) - - monkeypatch.setattr(daemon, "ensure_owned_gateway", Gateway) - starts = [] - monkeypatch.setattr(runtime, "exact_start", lambda _ctx, prompt, spec: ( - starts.append((prompt, spec)) - or json.dumps({"status": "started", "run_id": "run-live-probe"}) - )) - import ouroboros.delegate_supervision as supervision - - monkeypatch.setattr( - supervision, "supervised_wait", - lambda *_a, **_kw: pytest.fail("the host must not wait inside bootstrap (owner 1=A)"), - ) - snapshot = _snapshot( - _settings(_session_row(target="codex=gpt-5.6-sol")), - "session-builder", - ) - dispatch = SimpleNamespace( - executor="harness", blocked=False, - executor_resolution=SimpleNamespace(route=SimpleNamespace(route_id="codex")), - ) - ctx = SimpleNamespace( - task_id="child-live-probe", - drive_root=tmp_path, - budget_drive_root=str(tmp_path), - task_metadata={}, - ) - task = { - "id": "child-live-probe", - "objective": "x" * 250_100, - "configured_subagent": snapshot, - "task_contract": {"objective": "x" * 250_100}, - } - - # Charter D1: both observations happen inside the bootstrap now — the - # cached channel probe at authority-freeze time, then the LIVE re-probe - # inside the pre-start's delegate_start_entry. The (True→False) row proves - # the cached "available" observation is context only, never start - # authority: the live probe overrides it into a typed $0 refusal. - raw = bootstrap.bootstrap_before_context(ctx, task, dispatch) - custody_rows = [ - json.loads(line) - for line in custody.event_log_path(tmp_path).read_text().splitlines() - ] - - assert len(closed) == 2 - assert ctx._configured_actor_bootstrap["source_channel"]["route"] == "codex" - if expected_reason: - assert raw == "" - assert ctx._configured_startup_refusal["reason"] == expected_reason - assert starts == [] - # The refusal row plus the D5 attempt fact: a pre-custody refusal is - # still a durable delegate_start ATTEMPT (triad 2026-08-30). - assert custody_rows[-1]["type"] == "delegate_run_start_blocked" - assert custody_rows[-1]["reason"] == "configured_work_order_source_refused" - refused = custody_rows[-2] - assert refused["route"] == "codex" - assert refused["type"] == "configured_subagent_work_order_refused" - assert refused["reason"] == expected_reason - assert refused["source_channel"]["reason"] == ( - "interactive_unsupported" - if start_interactive is False - else "interactive_capability_missing" - ) - else: - out = json.loads(raw) - assert out["status"] == "configured_session_started" - assert out["startup"]["status"] == expected_status - assert len(starts) == 1 - assert "WORK ORDER SOURCE REQUEST" in starts[0][0] - assert custody_rows[-1]["route"] == "codex" - assert custody_rows[-1]["type"] == "configured_subagent_work_order_source_request" - assert custody_rows[-1]["status"] == "attempted" - assert custody_rows[-1]["source_channel"] == { - "status": "available", - "reason": "interactive", - "route": "codex", - } + assert json.loads(raw)["status"] == "configured_session_started" + assert len(calls) == 1 + prompt, spec = calls[0] + assert prompt == compile_external_work_order(task) + assert objective in prompt + assert spec["compiled_work_order"] is True + assert "work_order_source_request" not in spec + assert "source_request" not in ctx._configured_actor_bootstrap def test_pending_over_budget_recovery_replays_compact_body_and_full_fingerprint( @@ -916,24 +748,13 @@ def test_actor_first_retry_cannot_turn_coordination_prompt_into_work_order_prefi ctx, "coordination text is not the canonical assignment", retry_of="inv-1", )) assert out["status"] == "refused" - assert out["reason"] == "work_order_source_channel_unverified" + assert out["reason"] == "configured_work_order_unavailable" -def test_actor_first_over_budget_retry_reprobes_source_channel(monkeypatch, tmp_path): - import ouroboros.claudexor_daemon as daemon +def test_actor_first_legacy_retry_replays_recorded_partial_body(monkeypatch, tmp_path): + from ouroboros import delegate_custody as custody import ouroboros.subagent_runtime as runtime - class Gateway: - def harnesses(self): - return [{ - "id": "codex", - "manifest": {"capabilities": {"interactive": True}}, - }] - - def close(self): - pass - - monkeypatch.setattr(daemon, "ensure_owned_gateway", Gateway) starts = [] monkeypatch.setattr(runtime, "exact_start", lambda _ctx, prompt, spec: ( starts.append((prompt, spec)) @@ -941,27 +762,21 @@ def test_actor_first_over_budget_retry_reprobes_source_channel(monkeypatch, tmp_ )) snapshot = _snapshot(_settings(_session_row()), "session-builder") ctx = SimpleNamespace( - task_id="child-over-budget-retry", - drive_root=tmp_path, - budget_drive_root=str(tmp_path), - _configured_actor_bootstrap={ - "snapshot": snapshot, + task_id="child-legacy-retry", drive_root=tmp_path, budget_drive_root=str(tmp_path), + _configured_actor_bootstrap={"snapshot": snapshot, "selected_subagent_id": "session-builder", - "canonical_work_order": "", - "source_prompt": "WORK ORDER SOURCE REQUEST\ncoverage=partial", - "source_request": {"kind": "complete_work_order", "sha256": "f" * 64}, - "source_channel": {"status": "unavailable", "route": "cursor"}, - "work_order_fingerprint": "f" * 64, - "work_order_chars": 250001, - }, + "canonical_work_order": "NEW COMPLETE RENDER MUST NOT REPLACE STORED BODY"}, ) - - out = json.loads(runtime.delegate_start_entry(ctx, "ignored", retry_of="inv-1")) - + stored_prompt = "WORK ORDER SOURCE REQUEST\ncoverage=partial" + assert custody.record_start_requested( + tmp_path, run_id="", task_id=ctx.task_id, invocation_id="inv-legacy", + idempotency_key="inv-legacy", max_seconds=60, request={"prompt": stored_prompt}, + project_id="", project_owned=False, route="codex", + work_order_coverage="partial", work_order_source_request={"coverage": "partial"}, + ) + out = json.loads(runtime.delegate_start_entry(ctx, "ignored", retry_of="inv-legacy")) assert out["status"] == "started" - assert len(starts) == 1 - assert starts[0][1]["retry_of"] == "inv-1" - assert ctx._configured_actor_bootstrap["source_channel"]["route"] == "codex" + assert starts == [(stored_prompt, {"retry_of": "inv-legacy", "_resolved_binding": None})] def test_actor_first_bootstrap_adopts_existing_handoff_without_new_start(monkeypatch, tmp_path): diff --git a/tests/test_complete_chosen_inputs.py b/tests/test_complete_chosen_inputs.py new file mode 100644 index 000000000..290dc958f --- /dev/null +++ b/tests/test_complete_chosen_inputs.py @@ -0,0 +1,91 @@ +"""Complete chosen assignments and operative plans at their actual consumer seams.""" + +from copy import deepcopy +import json + +import pytest + +from ouroboros.tools import plan_packet, plan_spec + + +@pytest.mark.parametrize("field", ["goal", "in_scope", "non_goals", "invariants", + "affected_resources", "evidence", "acceptance_claims", + "decisions", "deferred"]) +def test_operative_tail_changes_normalized_identity_and_current_packet(field): + prefix = "яё𐍈🚀\n" * 500 + raw = {"goal": "Full plan", field: prefix + "TAIL_A"} + if field != "goal": + # Also cross the former 40-item list cap; the decisive item is last. + raw[field] = [f"first {i}" for i in range(40)] + [prefix + "TAIL_A"] + before, errors = plan_spec.normalize_spec(raw) + assert not errors + changed = deepcopy(raw) + if field == "goal": + changed[field] = prefix + "TAIL_B" + else: + changed[field][-1] = prefix + "TAIL_B" + after, errors = plan_spec.normalize_spec(changed) + assert not errors + assert plan_spec.spec_hash(before) != plan_spec.spec_hash(after) + assert plan_spec.spec_delta(before, after)["changed"] + serialized = json.dumps(after, ensure_ascii=False) + assert "TAIL_B" in serialized and "TAIL_A" not in serialized + packet = plan_packet.build_plan_review_user_content( + objective="Parent goal", goal=after["goal"], plan_prose="Chosen plan", + spec=after, manifest={}, prior_cycles=[], dispositions=[], + spec_delta=None, root_exploration_log=None, + ) + assert "TAIL_B" in packet and "OMISSION NOTE" not in packet + + +@pytest.mark.parametrize("collection,key", [("decisions", "choice"), ("decisions", "why"), + ("decisions", "rejected"), ("deferred", "what"), + ("deferred", "why_safe_to_defer")]) +def test_nested_operative_tails_survive(collection, key): + tail = "строка\n" * 200 + "DECISIVE_NESTED_TAIL" + item = {"choice": "choice"} if collection == "decisions" else {"what": "later"} + item[key] = [f"rejected {i}" for i in range(8)] + [tail] if key == "rejected" else tail + raw = {"goal": "Full plan", collection: [item]} + spec, errors = plan_spec.normalize_spec(raw) + assert not errors + assert spec[collection][0][key] == item[key] + changed = deepcopy(raw) + changed[collection][0][key] = item[key] + ["next"] if key == "rejected" else tail + "next" + assert plan_spec.spec_hash(spec) != plan_spec.spec_hash(plan_spec.normalize_spec(changed)[0]) + + +def test_large_direct_request_keeps_chosen_prompt_and_host_instruction_roles(tmp_path, monkeypatch): + from ouroboros import claudexor_daemon + from ouroboros.gateways import claudexor + from tests.test_nanny_economics import _start_with_contract + + # Importing the helper does not activate its module's autouse fixture. + monkeypatch.setattr(claudexor_daemon, "ensure_owned_gateway", lambda: claudexor.ClaudexorGateway()) + prompt = " \n" + "яё𐍈🚀\n" * 55_000 + "CHOSEN_ASSIGNMENT_TAIL\n " + objective = "HOST_OBJECTIVE:" + "О" * 250_001 + expected = "HOST_EXPECTED:" + "Е" * 250_001 + request = _start_with_contract(tmp_path, monkeypatch, { + "objective": objective, "expected_output": expected, + "context": " \nHOST_REFERENCE_CONTEXT\n ", + }, prompt=prompt) + assert request["prompt"] == prompt + instructions = request["instructions"] + assert instructions.count(objective) == 1 and instructions.count(expected) == 1 + marker = "HOST TASK CONTRACT AUTHORITY (complete normalized JSON; exact strings are authority):\n" + host = json.loads(instructions.split(marker, 1)[1]) + assert host["objective"] == objective and host["expected_output"] == expected + assert host["context"] == " \nHOST_REFERENCE_CONTEXT\n " + assert "CHOSEN_ASSIGNMENT_TAIL" not in instructions + + +def test_current_packet_keeps_full_parent_objective_and_plan_prose(): + objective = " \n" + "О" * 8_001 + "OBJECTIVE_TAIL\n " + prose = " \n" + "П" * 40_001 + "PROSE_TAIL\n " + spec, errors = plan_spec.normalize_spec({"goal": "g"}) + assert not errors + packet = plan_packet.build_plan_review_user_content( + objective=objective, goal="g", plan_prose=prose, spec=spec, manifest={}, + prior_cycles=[], dispositions=[], spec_delta=None, root_exploration_log=None, + ) + assert objective in packet and prose in packet + assert "OMISSION NOTE" not in packet diff --git a/tests/test_complete_plan_state.py b/tests/test_complete_plan_state.py new file mode 100644 index 000000000..957f26697 --- /dev/null +++ b/tests/test_complete_plan_state.py @@ -0,0 +1,209 @@ +"""Complete operative plans survive the bounded hot review-state round trip.""" + +from copy import deepcopy +import json +from types import SimpleNamespace + +import pytest + +from ouroboros import task_results +from ouroboros.tools import plan_review_artifacts as artifacts, plan_spec +from tests import test_plan_review_engine + + +plan_review_harness_fixture = pytest.fixture(name="_harness")( + test_plan_review_engine.harness.__wrapped__ +) + + +def _record(root, raw, *, closed=True, findings=None, evidence_manifest=None): + spec, errors = plan_spec.normalize_spec(raw) + assert not errors + wave = { + "schema_version": 2, "cycle_index": 1, "request_fingerprint": "a" * 64, + "goal": spec["goal"], "spec": spec, "spec_hash": plan_spec.spec_hash(spec), + "aggregate": "GREEN" if closed else "REVIEW_REQUIRED", "closed": closed, + "findings": findings or [], "dispositions": [], "paid": True, + } + if evidence_manifest is not None: + wave["evidence_manifest"] = evidence_manifest + result = artifacts.record_exact_wave( + root, "large-plan", wave, deepcopy(wave), need_evidence_seen=[], page_size=32, + ) + return spec, result + + +def _raw_state(root, task_id="large-plan"): + return json.loads(task_results.task_result_path(root, task_id).read_text())["plan_review_state"] + + +@pytest.mark.parametrize("field", ["goal", "acceptance_claims", "in_scope"]) +def test_large_operative_values_persist_once_and_rehydrate_for_consumers(tmp_path, field): + from ouroboros.agent_startup_checks import task_result_authority_projection + from ouroboros.contracts.task_contract import effective_acceptance_claims + from ouroboros.review_evidence_sections import _accept_effective_claims + + large = "complete chosen requirement\n" * 40_000 + "DECISIVE_TAIL" + raw = {"goal": "Deliver", "acceptance_claims": ["a complete deliverable"]} + raw[field] = (large if field == "goal" else [large] if field == "acceptance_claims" + else [f"requirement {i}: " + large[:1000] for i in range(1100)]) + spec, result = _record(tmp_path, raw, closed=False) + assert result["spec"] == spec and result["goal"] == spec["goal"] + hot = _raw_state(tmp_path) + assert hot["cycles_paid"] == 1 + assert len(json.dumps(hot).encode()) < task_results._PLAN_REVIEW_STATE_MAX_BYTES + assert hot["waves"][0]["spec"] == {} and hot["waves"][0]["spec_in_artifact"] + assert "goal" not in hot["waves"][0] and not hot["waves"][0]["closed"] + assert task_results.load_plan_review_state(tmp_path, "large-plan")["waves"][0]["spec"] == spec + task_results.record_plan_review_dispositions( + tmp_path, "large-plan", fingerprint="a" * 64, + dispositions=[], closed=True, closure_notes=[], + ) + state = task_results.load_plan_review_state(tmp_path, "large-plan") + wave = task_results.closed_plan_review_wave(state) + assert wave["spec"] == spec and wave["goal"] == spec["goal"] + assert effective_acceptance_claims({}, wave)[0] == spec["acceptance_claims"] + ctx = SimpleNamespace(task_metadata={}, task_contract={}, drive_root=tmp_path) + claims, source, _ = _accept_effective_claims(ctx, {}, tmp_path, "large-plan") + assert claims == spec["acceptance_claims"] and source == "plan_review" + record = task_results.load_task_result(tmp_path, "large-plan") + projection = task_result_authority_projection(record, drive_root=tmp_path) + assert projection["plan_review_state"]["waves"][0]["spec"] == spec + assert record["plan_review_state"]["waves"][0]["spec"] == {} + assert _raw_state(tmp_path)["waves"][0]["spec"] == {} + assert task_results.load_plan_review_state(tmp_path, "large-plan")["cycles_paid"] == 1 + + +def test_full_plan_review_disposition_repeat_and_tail_delta(_harness): + from tests.test_plan_review_engine import CLEAN, _call, _control, _finding, _state, _user_text + from ouroboros.tools.plan_review import _apply_disposition + + note = json.dumps([_finding("n1", "note", summary="Consider another example")]) + substrate = _harness.install({"s1": note, "s2": CLEAN, "s3": CLEAN}) + ctx = _harness.make_ctx() + goal = "full chosen goal\n" * 22_000 + "GOAL_TAIL" + spec = {"in_scope": ["full requirement\n" * 22_000 + "SCOPE_TAIL_A"], + "acceptance_claims": ["full criterion\n" * 22_000 + "CLAIM_TAIL"]} + assert _control(_call(ctx, spec, goal=goal)) == {"outcome": "REVIEW_REQUIRED", "closed": False} + state = _state(_harness) + fingerprint = state["waves"][-1]["request_fingerprint"] + finding = state["waves"][-1]["findings"][0]["finding_id"] + disposed = _apply_disposition(ctx, {"review_fingerprint": fingerprint, "items": [ + {"finding_id": finding, "decision": "reject", "rationale": "Existing example suffices"}, + ]}) + assert _control(disposed)["closed"] + replay = _call(ctx, spec, goal=goal) + assert "cached exact review" in replay and len(substrate.calls) == 1 + assert _state(_harness)["cycles_paid"] == 1 + assert _raw_state(_harness.drive, "task-1")["waves"][-1]["spec"] == {} + changed = deepcopy(spec) + changed["in_scope"][0] = changed["in_scope"][0].replace("TAIL_A", "TAIL_B") + assert _control(_call(ctx, changed, goal=goal))["outcome"] == "REVIEW_REQUIRED" + assert len(substrate.calls) == 2 and _state(_harness)["cycles_paid"] == 2 + sent = _user_text(substrate.calls[-1]["request"].messages[-1]["content"]) + assert "SCOPE_TAIL_B" in sent and "previous frozen spec body truncated" not in sent + + +@pytest.mark.parametrize("source", ["wave_artifact", "spec_source_ref"]) +@pytest.mark.parametrize("damage", ["missing", "changed"]) +def test_unavailable_full_spec_preserves_paid_state_and_never_becomes_empty_claims(tmp_path, damage, source): + from ouroboros.owner_hurry import force_plan_decision + from ouroboros.review_evidence import build_task_acceptance_evidence + + _, wave = _record(tmp_path, {"goal": "Deliver", "acceptance_claims": ["required criterion"]}) + ref = wave[source] + from ouroboros.artifacts import task_artifact_dir_path + + path = task_artifact_dir_path(tmp_path, "large-plan", create=False) / ref["path"] + if damage == "missing": + path.unlink() + else: + path.write_bytes(b"{}") + assert _raw_state(tmp_path)["cycles_paid"] == 1 + with pytest.raises(artifacts.PlanReviewSourceUnavailable): + task_results.load_plan_review_state(tmp_path, "large-plan") + ctx = SimpleNamespace(task_id="large-plan", task_metadata={}, + task_contract={"objective": "Deliver"}, drive_root=tmp_path) + with pytest.raises(artifacts.PlanReviewSourceUnavailable): + build_task_acceptance_evidence(ctx, drive_root=tmp_path, task_id="large-plan") + assert force_plan_decision(ctx, {}, enforcement="blocking")["allow"] is False + + +def test_legacy_cut_without_source_keeps_its_gap(tmp_path): + legacy = {"spec": {"goal": "known goal", "in_scope": ["partial…"]}, + "spec_body_truncated": True} + state = {"schema_version": 2, "waves": [legacy]} + assert artifacts.authority_state(tmp_path, "legacy", state) == state + with pytest.raises(artifacts.PlanReviewSourceUnavailable): + artifacts.authority_wave(tmp_path, "legacy", legacy) + + +@pytest.mark.parametrize("example", [ + 'Document JSON example {"api_key": "YOUR_API_KEY"}.', + 'The example includes OPENAI_API_KEY=your-key-here.', +]) +def test_operative_example_text_keeps_its_reviewed_identity(tmp_path, example): + spec, wave = _record(tmp_path, { + "goal": example, "acceptance_claims": [example], + }) + assert wave["spec"] == spec + assert plan_spec.spec_hash(wave["spec"]) == wave["spec_hash"] + evidence = artifacts.read_wave(tmp_path, "large-plan", wave["wave_artifact"]) + assert "***REDACTED***" in evidence["spec"]["goal"] + assert evidence["spec_source_ref"] == wave["spec_source_ref"] + + +@pytest.mark.parametrize("compact", [False, True]) +def test_spec_source_is_in_the_existing_child_promotion_closure(tmp_path, compact): + import shutil + from ouroboros.artifacts import read_actor_source_bytes + from ouroboros.observability import promote_child_task_refs + + child, parent = tmp_path / "child", tmp_path / "parent" + spec, _ = _record(child, {"goal": "Keep the complete goal", "in_scope": ["full requirement\n" * 100_000]}) + state = _raw_state(child) + if compact: + state["waves"] = [task_results._compact_plan_review_wave(state["waves"][0])] + ref = state["waves"][0]["spec_source_ref"] + copied, receipt = promote_child_task_refs(parent, child, "large-plan", {"plan_review_state": state}) + assert receipt["status"] == "complete" and receipt["promoted_source_handle_count"] == 1 + copied_ref = copied["plan_review_state"]["waves"][0]["spec_source_ref"] + assert copied_ref["sha256"] == ref["sha256"] + shutil.rmtree(child) + assert json.loads(read_actor_source_bytes(parent, "large-plan", copied_ref)) == spec + + +def test_last_resort_hot_state_fit_preserves_exact_source_references(tmp_path): + wide = "𝕏" * 1000 + findings = [{"finding_id": f"s{slot}:f{i}", "slot": str(slot), "class": "note", + "summary": wide, "recommendation": wide, "locator": wide, "detail": wide} + for slot in range(10) for i in range(32)] + spec, wave = _record(tmp_path, {"goal": "Preserve every source"}, closed=False, findings=findings) + hot = _raw_state(tmp_path)["waves"][0] + assert hot["findings_texts_truncated"] + assert hot["spec_source_ref"] == wave["spec_source_ref"] + assert hot["wave_artifact"] == wave["wave_artifact"] + assert task_results.load_plan_review_state(tmp_path, "large-plan")["waves"][0]["spec"] == spec + + +def test_large_evidence_manifest_persists_through_the_exact_wave_source(tmp_path): + from ouroboros.tools import plan_evidence + + urls = [f"https://example.test/source/{i:05d}" for i in range(10_000)] + manifest = plan_evidence.resolve_evidence(urls, active_root=tmp_path, allowed_roots=[tmp_path]) + assert manifest["declared"] == urls + assert len(manifest["omissions"]) == len(urls) + assert all(row["reason"] == "url_not_fetched" for row in manifest["omissions"]) + spec, wave = _record( + tmp_path, {"goal": "Audit supplied sources", "evidence": urls}, + evidence_manifest=manifest, + ) + hot = _raw_state(tmp_path) + assert hot["cycles_paid"] == 1 and hot["waves"][0]["closed"] + assert "evidence_manifest" not in hot["waves"][0] + assert len(json.dumps(hot).encode()) < task_results._PLAN_REVIEW_STATE_MAX_BYTES + assert wave["evidence_manifest"] == manifest + restored = task_results.load_plan_review_state(tmp_path, "large-plan")["waves"][0] + assert restored["spec"] == spec and restored["evidence_manifest"] == manifest + exact = artifacts.read_wave(tmp_path, "large-plan", restored["wave_artifact"]) + assert exact["evidence_manifest"] == manifest diff --git a/tests/test_d16_skill_payload_authority.py b/tests/test_d16_skill_payload_authority.py index f25d5778d..7ea62bb16 100644 --- a/tests/test_d16_skill_payload_authority.py +++ b/tests/test_d16_skill_payload_authority.py @@ -242,8 +242,8 @@ def test_direct_operator_can_read_native_payload_but_not_mutate_or_forge_sidecar "write_file", {**selector, "path": ".seed-origin", "content": "forged\n"}, ) - assert "BLOCKED" in ordinary_write - assert "BLOCKED" in sidecar_write + assert "SKILL_PAYLOAD_ARG_ERROR" in ordinary_write + assert "SKILL_PAYLOAD_ARG_ERROR" in sidecar_write assert not (payload / "new.txt").exists() assert (payload / ".seed-origin").read_text(encoding="utf-8") == "launcher-seed\n" diff --git a/tests/test_delegated_payload_custody.py b/tests/test_delegated_payload_custody.py index 9abbb3236..5fb865636 100644 --- a/tests/test_delegated_payload_custody.py +++ b/tests/test_delegated_payload_custody.py @@ -12,6 +12,7 @@ disposed by a live top-level task holding the same target. from __future__ import annotations import json +import pytest import pathlib import subprocess @@ -307,12 +308,14 @@ def _disposed_rows(tmp_path): return [r for r in rows if str(r.get("type") or "") == custody.PATCH_DISPOSED] +@pytest.mark.parametrize("direct_chat", [False, True]) def test_top_level_task_may_reject_a_terminal_owners_payload_orphan( - tmp_path, monkeypatch): + tmp_path, monkeypatch, direct_chat): from ouroboros.subagent_worktrees import find_execution_snapshot from ouroboros.tools.subagent_integration import _integrate_delegated_patch second, skill, entry, capture = _payload_orphan(tmp_path, monkeypatch) + second.is_direct_chat = direct_chat out = _integrate_delegated_patch(second, "run-p1", "reject", "not wanted") assert "🚫 Rejected" in out, out assert "orphan of terminal task t-payload" in out, out @@ -325,11 +328,13 @@ def test_top_level_task_may_reject_a_terminal_owners_payload_orphan( custody._CUSTODY.clear() +@pytest.mark.parametrize("direct_chat", [False, True]) def test_top_level_task_may_apply_a_terminal_owners_payload_orphan( - tmp_path, monkeypatch): + tmp_path, monkeypatch, direct_chat): from ouroboros.tools.subagent_integration import _integrate_delegated_patch second, skill, entry, capture = _payload_orphan(tmp_path, monkeypatch) + second.is_direct_chat = direct_chat out = _integrate_delegated_patch(second, "run-p1", "apply", "looks good") assert "✅ Integrated" in out, out assert "orphan of terminal task t-payload" in out, out @@ -414,7 +419,6 @@ def test_non_top_level_profiles_may_not_dispose_an_orphan(tmp_path, monkeypatch) for constraint, direct_chat in ( (TaskConstraint(mode="local_readonly_subagent"), False), (TaskConstraint(mode="acting_subagent", surface="worktree"), False), - (None, True), # direct chat = operator_control ): second.task_constraint = constraint second.is_direct_chat = direct_chat diff --git a/tests/test_delegated_reconciliation.py b/tests/test_delegated_reconciliation.py index e827d4d56..e57dc5fe7 100644 --- a/tests/test_delegated_reconciliation.py +++ b/tests/test_delegated_reconciliation.py @@ -65,8 +65,11 @@ def test_both_custody_surfaces_see_the_same_live_task_set(monkeypatch): # init_queue_refs across the suite without restore (the upstream test # convention), so assuming the dict is empty here is cross-test fragile. monkeypatch.setattr(queue, "RUNNING", {"t-live": {}}) + from supervisor.active_activity import get_direct_activity_registry + + get_direct_activity_registry().register("native-live", 1) sm._periodic_supervisor_maintenance([0.0], [time.time()]) - assert seen["processes"] == seen["delegated"] == {"t-live"}, seen + assert seen["processes"] == seen["delegated"] == {"t-live", "native-live"}, seen def test_an_orphaned_delegated_run_is_reconciled_when_its_owner_is_gone(tmp_path, monkeypatch): diff --git a/tests/test_direct_chat_turn_owner_control.py b/tests/test_direct_chat_turn_owner_control.py index bf5104626..53c94dc72 100644 --- a/tests/test_direct_chat_turn_owner_control.py +++ b/tests/test_direct_chat_turn_owner_control.py @@ -71,7 +71,6 @@ def _write_snapshot(tmp_path, running_ids=()): def _live_chat_agent(monkeypatch, task_id=TURN_ID, *, accepting=True): """The chat agent mid-turn, shaped exactly as agent.py leaves it (the fields steering.py and workers.chat_turn_liveness read).""" - from supervisor import workers import threading @@ -83,7 +82,9 @@ def _live_chat_agent(monkeypatch, task_id=TURN_ID, *, accepting=True): _task_started_ts=1000.0, _last_activity_ts=1000.0, _owner_message_admission_lock=threading.Lock(), ) - monkeypatch.setattr(workers, "_chat_agent", agent, raising=False) + from supervisor.active_activity import get_direct_activity_registry + + get_direct_activity_registry().register(task_id, chat_id=agent._current_chat_id, actor=agent) return agent diff --git a/tests/test_ephemeral_model_wait.py b/tests/test_ephemeral_model_wait.py index 9e4cea682..77d850ea1 100644 --- a/tests/test_ephemeral_model_wait.py +++ b/tests/test_ephemeral_model_wait.py @@ -34,7 +34,7 @@ def ephemeral_call(setup, monkeypatch): monkeypatch.setattr(task_queue, "PENDING", []) busy = SimpleNamespace(_busy=True, _current_task_id="other-turn", _accepting_owner_messages=True, _current_task_metadata={}, _current_task_text="Other work", _current_chat_id=1) - monkeypatch.setattr(workers, "_chat_agent", busy) + registry.register("other-turn", 1, actor=busy) events, published, failures = queue.Queue(), [], [] monkeypatch.setattr(workers, "get_event_q", lambda: events) monkeypatch.setattr(workers, "send_with_budget", lambda *a, **k: failures.append((a, k))) @@ -135,7 +135,8 @@ def test_ephemeral_producer_reaches_decision_and_resumes_same_call(ephemeral_cal assert flow.transport.uploads[-1][0]["account"] == {"mode": "pin", "profileId": "replacement"} assert [row["state"] for row in ledger(flow.root)] == [ "reserved", "dispatched", "released", "reserved", "dispatched", "settled"] - assert flow.owner.closed and flow.registry.snapshot() == [] + assert flow.owner.closed + assert [row["activity_id"] for row in flow.registry.snapshot()] == ["other-turn"] assert not mailbox.exists() assert not (flow.root / "task_results" / f"{flow.task['id']}.json").exists() assert clients[first](body).status_code == 409 @@ -163,7 +164,7 @@ def test_ephemeral_wait_rejects_stale_attempt_revision_and_closed_owner(ephemera flow.owner._drain_controls() assert row["auto_continue"] is False assert clients["web"](body).json()["applied"] is True - snapshot = flow.registry.snapshot()[0] + snapshot = flow.registry.get(flow.task["id"]).to_dict() assert snapshot["kind"] == "ephemeral_decision" and snapshot["model_waits"] assert "cancelable" not in snapshot and "model_wait_owner" not in snapshot before = deepcopy(snapshot["model_waits"]) diff --git a/tests/test_ephemeral_model_wait_browser.py b/tests/test_ephemeral_model_wait_browser.py index 480131066..08e4d59fd 100644 --- a/tests/test_ephemeral_model_wait_browser.py +++ b/tests/test_ephemeral_model_wait_browser.py @@ -80,5 +80,5 @@ def test_ephemeral_wait_hydrates_without_task_controls_and_browser_retry_resumes waiter.wait_for(state="detached") page.wait_for_selector(f'.chat-live-card[data-task-id="{task_id}"][data-finished="1"]') assert len(flow.transport.operations) == 2 - assert flow.registry.snapshot() == [] + assert [row["activity_id"] for row in flow.registry.snapshot()] == ["other-turn"] capture(page, "ephemeral-real-wait-completed") diff --git a/tests/test_inflight_indicator_seams.py b/tests/test_inflight_indicator_seams.py index ade9d6978..142d4f342 100644 --- a/tests/test_inflight_indicator_seams.py +++ b/tests/test_inflight_indicator_seams.py @@ -3,7 +3,7 @@ Pins the three seams the browser contract depends on: 1. ``supervisor.workers._run_chat_task`` tracks the turn in the - ``DirectActivityRegistry`` for exactly the duration of ``agent.handle_task`` + ``DirectActivityRegistry`` through preparation, ``agent.handle_task`` and event delivery with the correct ``kind``/``client_message_id``/``project_id``. 2. ``supervisor.events._handle_typing_start`` stamps ``kind`` and ``client_message_id`` from the registry onto the typing action — and leaves diff --git a/tests/test_loop_acceptance_gate.py b/tests/test_loop_acceptance_gate.py index 1e6d78b20..ae8cfa253 100644 --- a/tests/test_loop_acceptance_gate.py +++ b/tests/test_loop_acceptance_gate.py @@ -347,6 +347,10 @@ def _exercise_owner_followup_during_acceptance_panel(monkeypatch, tmp_path, *, d _current_chat_id=chat_id, _current_task_metadata={}, ) + if direct: + from supervisor.active_activity import get_direct_activity_registry + + get_direct_activity_registry().register(root_id, chat_id, actor=direct_agent) token = ("a" if direct else "b") * 32 def begin_fence(*, root_task_id, task_id): diff --git a/tests/test_mcp_registry_integration.py b/tests/test_mcp_registry_integration.py index b7a57ef0a..51dcc4469 100644 --- a/tests/test_mcp_registry_integration.py +++ b/tests/test_mcp_registry_integration.py @@ -115,6 +115,47 @@ def test_stdio_environment_does_not_expand_native_review_resources(registry, mon for item in registry.capability_omissions()) +@pytest.mark.parametrize("transport", ["stdio", "streamable_http"]) +@pytest.mark.parametrize("actor", ["main", "project", "managed"]) +@pytest.mark.parametrize("web_key", ["web", "allow_web"]) +def test_web_tool_restriction_preserves_mcp_discovery_and_dispatch( + tmp_path, monkeypatch, transport, actor, web_key): + fake = _FakeTransport([{"name": "store", "description": "Store a record", + "input_schema": {"type": "object", "properties": {}}}]) + _wire_singleton(fake) + server = (_good_server() if transport == "streamable_http" else + {"id": "demo", "enabled": True, "transport": "stdio", "command": "fixture"}) + mcp_client.reconfigure_from_settings(_settings(server)) + assert mcp_client.get_manager().refresh_server("demo")["ok"] + monkeypatch.setattr("ouroboros.safety.check_safety", lambda *a, **kw: (True, "")) + room = tmp_path / "room" + room.mkdir() + contract = build_task_contract({"allowed_resources": {web_key: False}}) + ctx = ToolContext(repo_dir=tmp_path / "repo", drive_root=tmp_path / "data", + is_direct_chat=actor != "managed", task_contract=contract, + task_metadata={"task_contract": contract, + **({"_project_room_dir": str(room)} if actor == "project" else {})}) + registry = ToolRegistry(ctx.repo_dir, ctx.drive_root) + registry.set_context(ctx) + name = "mcp_demo__store" + assert name in {row["function"]["name"] for row in registry.schemas()} + assert registry.get_schema_by_name(name) is not None + assert "RESOURCE_CONSTRAINT_BLOCKED" in registry.execute("web_search", {"query": "unused"}) + result = registry.execute_result(name, {}) + assert result.status == "ok" and result.text.endswith("echo(demo/store)"), result.text + assert len(fake.call_calls) == 1 + + # Explicit owner disables and actual network restrictions still win. + contract["disabled_tools"] = [name] + assert registry.get_schema_by_name(name) is None + assert registry.execute_result(name, {}).status == "blocked" + contract["disabled_tools"] = [] + contract["allowed_resources"]["network"] = False + assert registry.get_schema_by_name(name) is None + assert "network=false" in registry.execute(name, {}) + assert len(fake.call_calls) == 1 + + def test_schemas_cold_worker_loads_settings_and_refreshes_once(registry, monkeypatch): fake = _FakeTransport( [{"name": "ping", "description": "Ping", "input_schema": {"type": "object", "properties": {}}}] diff --git a/tests/test_module_handle_extraction.py b/tests/test_module_handle_extraction.py index c043fac7e..4a48671e3 100644 --- a/tests/test_module_handle_extraction.py +++ b/tests/test_module_handle_extraction.py @@ -62,7 +62,7 @@ LEAVES: dict[str, tuple[str, str, frozenset[str]]] = { "enqueue_task", "load_state", "persist_queue_snapshot", })), "supervisor/worker_chat_lane.py": ("supervisor/workers.py", "_pool", frozenset({ - "DRIVE_ROOT", "REPO_DIR", "_chat_agent_lock", "_ephemeral_chat_lock", + "DRIVE_ROOT", "REPO_DIR", "_repo_writer_gate_lock", "chat_turn_liveness", "_get_chat_agent", "_origin_from_mapping", "_repo_writer_turn_allowed", "_report_binding_failure", "get_event_q", "load_state", "repo_writer_admission_closed", "send_with_budget", diff --git a/tests/test_nanny_economics.py b/tests/test_nanny_economics.py index cd6a7f51b..7265164d1 100644 --- a/tests/test_nanny_economics.py +++ b/tests/test_nanny_economics.py @@ -102,14 +102,14 @@ def test_the_contract_objective_rides_the_run_instructions_structurally(tmp_path "expected_output": "a verified module with passing tests", }) instructions = request["instructions"] - assert "HOST TASK OBJECTIVE" in instructions + assert "HOST TASK CONTRACT AUTHORITY" in instructions assert "ghost-core module" in instructions - assert "HOST EXPECTED OUTPUT" in instructions + assert instructions.count("verified module with passing tests") == 1 assert "verified module with passing tests" in instructions # The nanny did NOT have to copy the contract into the prompt. assert "ghost-core" not in request["prompt"] # The prohibitions stay the opening statement of the channel. - assert instructions.index("git commit") < instructions.index("HOST TASK OBJECTIVE") + assert instructions.index("git commit") < instructions.index("HOST TASK CONTRACT AUTHORITY") def test_direct_start_request_carries_complete_normalized_contract_authority(tmp_path, monkeypatch): @@ -740,54 +740,13 @@ def test_forced_wrapup_over_a_succeeded_run_stays_silent_below_threshold(tmp_pat # -- Configured-session work order wire budget --------------------------------- -def test_work_order_preserves_complete_fields_and_refuses_over_one_total_budget(): - """No ordinary field becomes a misleading 4k prefix; the total wire bound is atomic.""" - import pytest +def test_work_order_preserves_complete_fields_above_former_total_budget(): + from ouroboros.subagent_work_order import compile_external_work_order - from ouroboros.subagent_work_order import WorkOrderBudgetExceeded, compile_external_work_order - from ouroboros.tools.delegate import _ASSIGNMENT_FIELD_CHARS - - limit = _ASSIGNMENT_FIELD_CHARS - assert limit == 250_000 - ordinary = "яё𐍈🚀" * 2_000 - rendered = compile_external_work_order({"id": "child", "objective": ordinary}) - assert ordinary in rendered and "OMISSION NOTE" not in rendered - with pytest.raises(WorkOrderBudgetExceeded) as refused: - compile_external_work_order({"id": "child", "objective": "a" * (limit + 1)}) - assert refused.value.chars > limit - assert len(refused.value.sha256) == 64 - - -def test_over_budget_source_request_is_a_small_partial_lens_without_a_prefix(): - from ouroboros.subagent_work_order import ( - WorkOrderBudgetExceeded, - build_work_order_source_request, - compile_external_work_order, - ) - - marker = "DECISIVE_SOURCE_MARKER" - task = { - "id": "child-source", - "objective": ("x" * 250_100) + marker, - "origin_message_ref": {"kind": "chat_message", "message_id": "m-1"}, - } - with pytest.raises(WorkOrderBudgetExceeded) as refused: - compile_external_work_order(task) - prompt, envelope = build_work_order_source_request(task, refused.value) - - assert len(prompt) < 10_000 - assert marker not in prompt - assert envelope["coverage"] == "partial" - assert envelope["complete_chars"] == refused.value.chars - assert envelope["complete_sha256"] == refused.value.sha256 - assert envelope["source"]["kind"] == "task_result" - assert envelope["source"]["tool"] == "get_task_result" - assert envelope["source"]["arguments"] == { - "task_id": "child-source", "include_authority": True, - "include_work_order_source": True, - } - assert envelope["source"]["projection"] == "canonical_work_order" - assert "cannot_verify" in prompt + objective = "яё𐍈🚀\n" * 55_000 + "DECISIVE_OBJECTIVE_TAIL" + rendered = compile_external_work_order({"id": "child", "objective": objective}) + assert len(objective) > 250_000 + assert objective in rendered and "OMISSION NOTE" not in rendered def test_an_ordinary_contract_field_reaches_the_run_instructions_complete(tmp_path, monkeypatch): @@ -797,17 +756,17 @@ def test_an_ordinary_contract_field_reaches_the_run_instructions_complete(tmp_pa "expected_output": "ok", }) instructions = request["instructions"] - start = instructions.index("HOST TASK OBJECTIVE") - end = instructions.index("HOST EXPECTED OUTPUT") - field = instructions[start:end] - assert "OMISSION NOTE" not in field - assert "O" * 4050 in field + assert "OMISSION NOTE" not in instructions + assert instructions.count("O" * 4050) == 1 -def test_atomic_compiled_work_order_sends_dynamic_brief_once(tmp_path, monkeypatch): +@pytest.mark.parametrize("objective", [ + "UNIQUE_ATOMIC_OBJECTIVE", + "яё𐍈🚀\n" * 55_000 + "LARGE_COMPILED_OBJECTIVE_TAIL", +]) +def test_atomic_compiled_work_order_sends_dynamic_brief_once(tmp_path, monkeypatch, objective): from ouroboros.subagent_work_order import compile_external_work_order - objective = "UNIQUE_ATOMIC_OBJECTIVE" task = { "id": "t-nanny", "objective": objective, @@ -825,6 +784,7 @@ def test_atomic_compiled_work_order_sends_dynamic_brief_once(tmp_path, monkeypat prompt=work_order, compiled_work_order=True, ) + assert request["prompt"] == work_order assert request["prompt"].count(objective) == 1 assert objective not in request["instructions"] assert "git commit" in request["instructions"] diff --git a/tests/test_native_conversation_activity.py b/tests/test_native_conversation_activity.py new file mode 100644 index 000000000..61c8ebf9d --- /dev/null +++ b/tests/test_native_conversation_activity.py @@ -0,0 +1,268 @@ +"""Ordinary conversations execute concurrently and remain under native custody.""" +from __future__ import annotations + +import queue +import threading +from types import SimpleNamespace + +from supervisor import workers +from supervisor.active_activity import get_direct_activity_registry + + +def _lane(monkeypatch, tmp_path): + from ouroboros import project_naming + from supervisor import message_bus, state + + monkeypatch.setattr(workers, "DRIVE_ROOT", tmp_path) + monkeypatch.setattr(workers, "REPO_DIR", tmp_path / "repo") + monkeypatch.setattr(workers, "get_event_q", lambda: queue.Queue()) + monkeypatch.setattr(workers, "send_with_budget", lambda *a, **kw: None) + monkeypatch.setattr(state, "load_state", lambda: {}) + monkeypatch.setattr(state, "budget_remaining", lambda *a, **kw: 100) + monkeypatch.setattr(project_naming, "spawn_proactive_namer", lambda *a, **kw: None) + monkeypatch.setattr(message_bus, "get_bridge", lambda: SimpleNamespace(send_chat_action=lambda *a, **kw: None)) + workers.open_repo_writer_admission() + + +def test_two_native_turns_are_independently_addressable_and_drainable(monkeypatch, tmp_path): + from ouroboros import agent as agent_module + from ouroboros.owner_mailbox import drain_owner_messages + from ouroboros.server_routing_context import _addressable_root_tasks + from supervisor.steering import _handle_steer_task + + _lane(monkeypatch, tmp_path) + entered = threading.Barrier(3) + releases = [threading.Event(), threading.Event()] + actors = [] + + class Actor: + def __init__(self, **kwargs): + self.thread = threading.current_thread() + self.index = len(actors) + self._owner_message_admission_lock = threading.Lock() + actors.append(self) + + def handle_task(self, task): + self.task = task + self._current_task_id = task["id"] + self._current_chat_id = task["chat_id"] + self._current_task_text = task["text"] + self._current_task_metadata = task.get("metadata", {}) + self._busy = self._accepting_owner_messages = True + entered.wait(timeout=10) + assert releases[self.index].wait(10) + with self._owner_message_admission_lock: + self._busy = self._accepting_owner_messages = False + return [] + + monkeypatch.setattr(agent_module, "make_agent", Actor) + threads = [threading.Thread(target=workers.handle_chat_direct, args=(chat_id, f"work {chat_id}")) for chat_id in (1, 2)] + try: + for thread in threads: + thread.start() + entered.wait(timeout=10) + assert len(actors) == 2 + ids = {actor.task["id"] for actor in actors} + assert all(not actor.task.get("_ephemeral_turn") for actor in actors) + ctx = SimpleNamespace(DRIVE_ROOT=tmp_path, RUNNING={}, PENDING=[], bridge=None) + roots = _addressable_root_tasks(ctx) + assert {row["task_id"] for row in roots} == ids + first, second = actors + armed = workers.arm_direct_chat_turn(second.task["id"], lambda turn: "control-second") + assert armed["stop_control_msg_id"] == "control-second" + assert "stop_control_msg_id" not in workers.direct_chat_turn(first.task["id"]) + _handle_steer_task({ + "target_task_id": second.task["id"], "chat_id": second.task["chat_id"], + "message": "Use the blue variant", "client_message_id": "owner-blue", + }, ctx) + assert drain_owner_messages(tmp_path, second.task["id"]) == ["Use the blue variant"] + assert drain_owner_messages(tmp_path, first.task["id"]) == [] + assert second._owner_message_generation == 1 + workers.close_repo_writer_admission("test-update") + assert set(workers.drain_repo_writers(timeout=0.01)) == ids + workers.handle_chat_direct(3, "must wait for update") + assert len(actors) == 2 + releases[first.index].set() + first.thread.join(timeout=10) + # Thread scheduling need not match actor allocation. + remaining = get_direct_activity_registry().snapshot() + assert {row["activity_id"] for row in remaining} == {second.task["id"]} + assert workers.drain_repo_writers(timeout=0.01) == [second.task["id"]] + finally: + for release in releases: + release.set() + for thread in threads: + thread.join(timeout=10) + workers.open_repo_writer_admission() + assert workers.drain_repo_writers(timeout=0) == [] + assert all(not thread.is_alive() for thread in threads) + + +def test_writer_drain_covers_construction_and_event_delivery(monkeypatch, tmp_path): + from ouroboros import agent as agent_module + + _lane(monkeypatch, tmp_path) + constructing, finish_construction = threading.Event(), threading.Event() + delivering, finish_delivery = threading.Event(), threading.Event() + + class Actor: + def handle_task(self, task): + return [{"type": "send_message", "text": "completed"}] + + def construct(**kwargs): + constructing.set() + assert finish_construction.wait(10) + return Actor() + + class Queue: + def put(self, event): + delivering.set() + assert finish_delivery.wait(10) + + monkeypatch.setattr(agent_module, "make_agent", construct) + monkeypatch.setattr(workers, "get_event_q", Queue) + thread = threading.Thread(target=workers.handle_chat_direct, args=(1, "work")) + try: + thread.start() + assert constructing.wait(10) + workers.close_repo_writer_admission("test-update") + pending = workers.drain_repo_writers(timeout=0.01) + assert len(pending) == 1 + assert get_direct_activity_registry().get(pending[0]).actor is None + finish_construction.set() + assert delivering.wait(10) + assert workers.drain_repo_writers(timeout=0.01) == pending + finally: + finish_construction.set() + finish_delivery.set() + thread.join(timeout=10) + workers.open_repo_writer_admission() + assert not thread.is_alive() + assert workers.drain_repo_writers(timeout=0) == [] + + +def test_restart_census_keeps_native_execution_after_owner_boundary(): + from ouroboros.server_restart import _live_running_task_ids + + registry = get_direct_activity_registry() + # A settled answer can still be doing post-task work in this actor. + registry.register("post-task", 1, actor=SimpleNamespace(_busy=False)) + assert _live_running_task_ids(SimpleNamespace(RUNNING={})) == ["post-task"] + registry.unregister("post-task") + assert _live_running_task_ids(SimpleNamespace(RUNNING={})) == [] + + +def test_consciousness_remains_paused_until_all_native_work_returns(): + from ouroboros.consciousness import BackgroundConsciousness + + mind = object.__new__(BackgroundConsciousness) + mind._paused = False + registry = get_direct_activity_registry() + registry.register("first", 1) + registry.register("second", 2) + assert mind.is_paused + registry.unregister("first") + assert mind.is_paused + registry.unregister("second") + assert not mind.is_paused + + +def test_native_post_task_retains_activity_and_delivers_answer_early(monkeypatch, tmp_path): + """Actual synthesis dispatch must stay owned after the ordinary final answer.""" + from ouroboros import agent_task_pipeline as pipeline, post_task_evolution + from ouroboros.consciousness import BackgroundConsciousness + from ouroboros.gateway.settings import _has_running_agent_tasks, _has_started_agent_tasks + from ouroboros.post_task_checkpoint import post_task_synthesis_in_flight + from ouroboros.server_restart import _live_running_task_ids + from ouroboros.task_results import load_task_result + + _lane(monkeypatch, tmp_path) + bus = queue.Queue() + monkeypatch.setattr(workers, "get_event_q", lambda: bus) + monkeypatch.setattr(workers, "RUNNING", {}) + monkeypatch.setattr(workers, "PENDING", []) + entered, release = threading.Event(), threading.Event() + synthesis_threads = [] + + def consolidate(*args, **kwargs): + synthesis_threads.append(threading.current_thread()) + entered.set() + assert release.wait(10) + + monkeypatch.setattr(pipeline, "_run_chat_consolidation", consolidate) + for name in ( + "_run_scratchpad_consolidation", "_run_task_summary", "_run_reflection", + "_update_improvement_backlog", "_apply_reflection_memory_actions", + ): + monkeypatch.setattr(pipeline, name, lambda *a, **kw: None) + monkeypatch.setattr(post_task_evolution, "maybe_promote", lambda *a, **kw: None) + env = SimpleNamespace(repo_dir=tmp_path / "repo", drive_root=tmp_path) + mind = object.__new__(BackgroundConsciousness) + mind._paused = False + + class Actor: + def handle_task(self, task): + self.task = task + pending = [ + {"type": "send_message", "task_id": task["id"], "chat_id": task["chat_id"], "text": "done"}, + {"type": "task_done", "task_id": task["id"], "chat_id": task["chat_id"]}, + ] + pipeline._dispatch_root_post_task( + env, task, "done", self._event_queue, pending, {}, {}, {}, tmp_path / "logs", + budget_drive_root="", split_drive=False, + project_scoped=bool(task.get("project_id")), project_task=False, + parent_env=None, parent_task=None, + ) + return pending + + # Both ordinary Main and ordinary Project conversation shapes must use + # the same lifetime; neither has a headless workspace/task marker. + for project_id in ("", "room"): + entered.clear() + release.clear() + synthesis_threads.clear() + actor = Actor() + thread = threading.Thread(target=workers._run_chat_task, args=(actor, 1, "ordinary conversation"), kwargs={ + "task_metadata": {"project_id": project_id, "origin_suppressed": True}, + }) + try: + thread.start() + assert entered.wait(5) + task_id = actor.task["id"] + # Use the real callback thread, not a fabricated post-task wait: + # synthesis belongs to the still-registered direct execution. + assert synthesis_threads == [thread] + assert thread.is_alive() + assert post_task_synthesis_in_flight(tmp_path, task_id) + registry = get_direct_activity_registry() + assert registry.get(task_id).actor is actor + assert workers.drain_repo_writers(0) == [task_id] + assert _live_running_task_ids(SimpleNamespace(RUNNING={})) == [task_id] + assert _has_running_agent_tasks() and _has_started_agent_tasks() + assert mind.is_paused + # Production early delivery ran before synthesis. The terminal + # completion stays buffered until post-task work returns. + early = bus.get_nowait() + assert early["type"] == "send_message" and early["text"] == "done" + assert early["task_id"] == task_id and early["delivery_id"] + assert bus.empty() + finally: + release.set() + thread.join(timeout=10) + for synthesis_thread in synthesis_threads: + if synthesis_thread is not thread: + synthesis_thread.join(timeout=10) + assert not thread.is_alive() + assert get_direct_activity_registry().snapshot() == [] + assert not post_task_synthesis_in_flight(tmp_path, task_id) + assert workers.drain_repo_writers(0) == [] + assert _live_running_task_ids(SimpleNamespace(RUNNING={})) == [] + assert not _has_running_agent_tasks() and not _has_started_agent_tasks() + assert not mind.is_paused + assert load_task_result(tmp_path, task_id)["root_phase_checkpoint"]["post_task_synthesis"] == "completed" + # Retained final and early final use the existing delivery identity; + # the supervisor deduplicates them. task_done reaches the bus last. + final, done = bus.get_nowait(), bus.get_nowait() + assert final["delivery_id"] == early["delivery_id"] + assert done["type"] == "task_done" and done["task_id"] == task_id + assert bus.empty() diff --git a/tests/test_native_model_wait_integration.py b/tests/test_native_model_wait_integration.py new file mode 100644 index 000000000..13794439b --- /dev/null +++ b/tests/test_native_model_wait_integration.py @@ -0,0 +1,171 @@ +"""Landed model-wait controls keep concurrent native actors independently owned.""" + +from contextlib import ExitStack +import queue +import threading +from types import SimpleNamespace + +import pytest + +from ouroboros import cancel_intents, model_wait, owner_mailbox, server_restart +from supervisor import active_activity, workers +from supervisor.task_model_wait import handle_task_model_wait +from tests.test_model_wait import _action_for, _decision_clients +from tests.test_post_task_model_wait import phase as post_phase_fixture, until + +phase = post_phase_fixture + + +def test_native_wait_controls_and_manual_restart_address_all_registered_actors(tmp_path, monkeypatch): + from ouroboros import delegate_custody + from ouroboros.task_results import write_task_result + from supervisor import queue as task_queue + + registry = active_activity.get_direct_activity_registry() + monkeypatch.setattr(server_restart, "DATA_DIR", tmp_path) + monkeypatch.setattr(task_queue, "DRIVE_ROOT", tmp_path) + monkeypatch.setattr(task_queue, "RUNNING", {}) + monkeypatch.setattr(task_queue, "PENDING", []) + monkeypatch.setattr(delegate_custody, "reconcile_orphaned_runs", lambda *a, **kw: []) + monkeypatch.setattr(server_restart, "_stop_owned_daemon", lambda *a: None) + monkeypatch.setattr(server_restart, "_managed_update_pending_kwargs", lambda: {}) + events, published, cancelled = queue.Queue(), [], [] + context = SimpleNamespace( + DRIVE_ROOT=tmp_path, RUNNING={}, consciousness=None, + append_jsonl=lambda *a: None, bridge=SimpleNamespace(push_log=published.append), + kill_workers=lambda **kw: cancelled.append(kw) or True, + ) + controllers = [] + with ExitStack() as stack: + for task_id, chat_id in (("native-main", 1), ("native-project", 7)): + task = {"id": task_id, "chat_id": chat_id, "_is_direct_chat": True, + "metadata": {"project_id": "room"} if chat_id == 7 else {}} + actor = SimpleNamespace( + _busy=True, _accepting_owner_messages=True, + _owner_message_admission_lock=threading.Lock(), + _current_task_id=task_id, _current_chat_id=chat_id, + _current_task_metadata=task["metadata"], _current_task_text="Continue work", + ) + write_task_result(tmp_path, task_id, "running", chat_id=chat_id) + registry.register(task_id, chat_id, actor=actor) + stack.callback(registry.unregister, task_id) + controller = stack.enter_context(model_wait.task_model_wait_scope( + task=task, drive_root=tmp_path, event_queue=events, worker_slot_held=False)) + row = {"wait_id": "wait-" + task_id, "state": "waiting", "role": "main", + "task_attempt": 1, "auto_continue": True} + controller.waits[row["wait_id"]] = row + controller._publish(row) + controllers.append(controller) + handle_task_model_wait(events.get_nowait(), context) + assert [row["chat_id"] for row in published] == [1, 7] + assert {row["id"] for row in workers.direct_chat_turns()} == {"native-main", "native-project"} + assert set(server_restart._live_running_task_ids(context)) == {"native-main", "native-project"} + with _decision_clients(tmp_path) as clients: + body = _action_for(published[1], "auto_continue", auto_continue=False) + reply = clients["web"](body) + assert reply.status_code == 202, reply.json() + assert clients["host"](body).json()["duplicate"] is True + for controller in controllers: + controller._drain_controls() + assert controllers[0].waits["wait-native-main"]["auto_continue"] is True + assert controllers[1].waits["wait-native-project"]["auto_continue"] is False + assert not owner_mailbox._mailbox_path(tmp_path, "native-main").exists() + # Reuse the new manual Restart owner, without touching a real + # process or daemon. Each actual controller sees its own Stop. + server_restart._stop_owned_work(context) + assert len(cancelled) == 1 and cancelled[0]["reconcile_delegate_custody"] is False + assert all(cancel_intents.cancel_pending(tmp_path, owner.task_id) for owner in controllers) + assert [owner.control_reason() for owner in controllers] == ["cancelled", "cancelled"] + denied = clients["host"](_action_for(published[0], "retry")) + assert denied.status_code == 409 and denied.json()["reason_code"] == "cancel_pending" + assert registry.snapshot() == [] + assert all(owner.closed for owner in controllers) + + +@pytest.mark.parametrize("project_id", ["", "room"]) +@pytest.mark.parametrize("action", ["switch", "stop"]) +def test_native_post_task_wait_remains_addressable_after_dialogue_closes(phase, monkeypatch, project_id, action): + from ouroboros import agent_task_pipeline as pipeline, project_naming + from ouroboros.gateway.state import _chat_activities_snapshot_safe + from ouroboros.post_task_checkpoint import post_task_model_wait + from ouroboros.task_results import load_task_result, write_task_result + from supervisor import message_bus, queue as task_queue + from tests.test_direct_chat_turn_owner_control import _client + from tests.test_llm_claudexor import MODEL + + f = phase + monkeypatch.setattr(workers, "DRIVE_ROOT", f.root) + monkeypatch.setattr(workers, "WORKERS", {}) + monkeypatch.setattr(workers, "get_event_q", lambda: f.events) + monkeypatch.setattr(project_naming, "spawn_proactive_namer", lambda *a, **kw: None) + monkeypatch.setattr(message_bus, "get_bridge", lambda: SimpleNamespace(send_chat_action=lambda *a, **kw: None)) + monkeypatch.setattr(task_queue, "DRIVE_ROOT", f.root) + monkeypatch.setattr(task_queue, "RUNNING", {}) + monkeypatch.setattr(task_queue, "PENDING", []) + observed = {} + + class Actor: + def handle_task(self, task): + f.task.clear() + f.task.update(task) + self._busy, self._accepting_owner_messages = True, False + self._current_task_id = task["id"] + # The real loop_delivery seam closes ordinary dialogue before + # entering post-task cognition; wait decisions still need an owner. + write_task_result(f.root, task["id"], "completed", result="Already answered", + root_phase_checkpoint={"post_task_synthesis": "pending_once"}) + pending = [{"type": "send_message", "task_id": task["id"], "chat_id": task["chat_id"], "text": "Already answered"}, + {"type": "task_done", "task_id": task["id"], "status": "completed"}] + with model_wait.task_model_wait_scope(task=f.task, drive_root=f.root, event_queue=self._event_queue, + worker_slot_held=False) as owner: + observed["owner"] = owner + pipeline._dispatch_root_post_task(f.env, f.task, "Already answered", self._event_queue, pending, + {"rounds": 3}, {}, {}, f.root / "logs", budget_drive_root="", split_drive=False, + project_scoped=bool(project_id), project_task=False, parent_env=None, parent_task=None) + return pending + + actor = Actor() + thread = threading.Thread(target=workers._run_chat_task, args=(actor, 7 if project_id else 1, "Already answered"), kwargs={ + "task_metadata": {"project_id": project_id, "origin_suppressed": True}, + }) + thread.start() + try: + until(lambda: "owner" in observed and any(row.get("credential_harness") for row in observed["owner"].waits.values())) + owner, task_id = observed["owner"], f.task["id"] + row = next(row for row in owner.waits.values() if row["state"] == "waiting") + assert workers.direct_chat_turn(task_id) is None # ordinary dialogue remains closed + assert active_activity.get_direct_activity_registry().get(task_id).actor is actor + assert post_task_model_wait(f.root, task_id) is owner + assert workers.drain_repo_writers(0) == [task_id] + activity = next(item for item in _chat_activities_snapshot_safe(f.root) if item["activity_id"] == task_id) + assert activity["kind"] == "direct_chat" and activity["phase"] == "finalizing" + assert activity["model_waits"][row["wait_id"]]["state"] == "waiting" + assert not row["worker_slot_held"] + events = list(f.events.queue) + assert any(event["type"] == "send_message" for event in events) + assert not any(event["type"] == "task_done" for event in events) + if action == "switch": + body = _action_for({"task_id": task_id, **row}, "switch", model=MODEL, + credential_profile_id="replacement", use_local=False, persist_role=False) + with _decision_clients(f.root) as clients: + response = clients["web"](body) + assert response.status_code == 202, response.json() + else: + with _client(f.root) as client: + response = client.post(f"/api/tasks/{task_id}/cancel", json={"stop_policy": "immediate"}) + assert response.status_code in (200, 503), response.json() + thread.join(5) + assert not thread.is_alive() and owner.closed + stored = load_task_result(f.root, task_id) + assert stored["status"] == "completed" and stored["result"] == "Already answered" + assert stored["root_phase_checkpoint"]["post_task_synthesis"] == ("completed" if action == "switch" else "degraded") + assert len(f.engine.creates) == (2 if action == "switch" else 1) + if action == "switch": + assert f.engine.uploads[-1][0]["account"] == {"mode": "pin", "profileId": "replacement"} + assert workers.drain_repo_writers(0) == [] + assert post_task_model_wait(f.root, task_id) is None + finally: + f.task["_skip_post_task_synthesis"] = True + f.ready.set() + thread.join(5) + assert not thread.is_alive() diff --git a/tests/test_native_post_task_wait_browser.py b/tests/test_native_post_task_wait_browser.py new file mode 100644 index 000000000..3b3a1a457 --- /dev/null +++ b/tests/test_native_post_task_wait_browser.py @@ -0,0 +1,104 @@ +"""Native post-task wait state and real Stop ingress through the browser card.""" + +import json +import threading +from types import SimpleNamespace + +import pytest +from starlette.applications import Starlette +from starlette.routing import Route +from starlette.testclient import TestClient + +from ouroboros import agent_task_pipeline as pipeline, model_wait +from ouroboros.gateway.history import make_chat_history_endpoint +from ouroboros.gateway.state import _chat_activities_snapshot_safe +from ouroboros.gateway.tasks import api_task_cancel +from ouroboros.post_task_checkpoint import post_task_model_wait +from ouroboros.task_results import load_task_result +from supervisor import active_activity, workers +from tests.test_post_task_model_wait import phase as post_phase_fixture, until +from tests.test_project_chat_continuity import _write_history_rows +from tests.test_subscription_setup_browser import capture, subscription_ui as ui_fixture + +phase = post_phase_fixture +subscription_ui = ui_fixture +pytestmark = [pytest.mark.ui_browser, pytest.mark.serial] + + +def test_browser_stops_native_post_task_wait_without_losing_answer(subscription_ui, phase, monkeypatch): + from supervisor import queue as task_queue + + f, ui = phase, subscription_ui + task_id, page = f.task["id"], ui["page"] + f.task["_is_direct_chat"] = True + monkeypatch.setattr(task_queue, "DRIVE_ROOT", f.root) + monkeypatch.setattr(task_queue, "RUNNING", {}) + monkeypatch.setattr(task_queue, "PENDING", []) + monkeypatch.setattr(workers, "WORKERS", {}) + pending = [{"type": "send_message", "task_id": task_id, "chat_id": 1, "text": "Already answered"}] + actor = SimpleNamespace(_busy=True, _accepting_owner_messages=False) + registry = active_activity.get_direct_activity_registry() + + def run(): + registry.register(task_id, 1, actor=actor) + try: + with model_wait.task_model_wait_scope(task=f.task, drive_root=f.root, event_queue=f.events, + worker_slot_held=False): + pipeline._dispatch_root_post_task(f.env, f.task, "Already answered", f.events, pending, + {"rounds": 3}, {}, {}, f.root / "logs", budget_drive_root="", split_drive=False, + project_scoped=False, project_task=False, parent_env=None, parent_task=None) + finally: + registry.unregister(task_id) + + _write_history_rows(f.root, task_id) + # The earlier running progress frame already carried the ordinary host's + # cancel authority. Retain it through history, then use the real endpoint. + path = f.root / "logs/progress.jsonl" + progress = json.loads(path.read_text()) + path.write_text(json.dumps({**progress, "cancelable": True}) + "\n") + thread = threading.Thread(target=run) + thread.start() + responses = [] + app = Starlette(routes=[Route("/api/chat/history", make_chat_history_endpoint(f.root)), + Route("/api/tasks/{task_id}/cancel", api_task_cancel, methods=["POST"])]) + app.state.drive_root = f.root + try: + until(lambda: post_task_model_wait(f.root, task_id) and any( + row.get("credential_harness") for row in post_task_model_wait(f.root, task_id).waits.values())) + with TestClient(app) as client: + def history(route): + response = client.get("/api/chat/history?chat_id=1") + route.fulfill(status=response.status_code, content_type="application/json", body=response.content) + + def cancel(route): + response = client.post(f"/api/tasks/{task_id}/cancel", json=route.request.post_data_json) + responses.append(response) + route.fulfill(status=response.status_code, content_type="application/json", body=response.content) + + page.route("**/api/chat/history*", history) + page.route("**/api/state", lambda route: route.fulfill(content_type="application/json", body=json.dumps({ + "supervisor_ready": True, "projects": [], "active_chat_activities": _chat_activities_snapshot_safe(f.root)}))) + page.route(f"**/api/tasks/{task_id}", lambda route: route.fulfill(content_type="application/json", + body=json.dumps(load_task_result(f.root, task_id)))) + page.route(f"**/api/tasks/{task_id}/cancel", cancel) + page.goto(ui["url"] + "/") + card = page.locator(f'.chat-live-card[data-task-id="{task_id}"]') + card.locator('[data-wait-id]').first.wait_for(timeout=5000) + assert "worker slot" not in card.locator('.model-wait-footnote').inner_text() + capture(page, "native-post-task-wait-before-stop") + card.locator('[data-cancel-run]').click() + page.locator('[data-task-control="stop_now"]').click() + page.wait_for_function("() => !document.querySelector('.task-control-menu')") + assert responses and responses[-1].status_code in (200, 503) + thread.join(5) + assert not thread.is_alive() and len(f.engine.creates) == 1 + assert load_task_result(f.root, task_id)["root_phase_checkpoint"]["post_task_synthesis"] == "degraded" + page.reload() + page.get_by_text("final answer", exact=True).wait_for() + page.wait_for_function("() => !document.querySelector('[data-wait-id]')") + capture(page, "native-post-task-wait-stopped-answer-preserved") + finally: + f.task["_skip_post_task_synthesis"] = True + f.ready.set() + thread.join(5) + assert not thread.is_alive() diff --git a/tests/test_plan_review_engine.py b/tests/test_plan_review_engine.py index a499ad939..b1d7838dd 100644 --- a/tests/test_plan_review_engine.py +++ b/tests/test_plan_review_engine.py @@ -488,10 +488,15 @@ def test_v2_wave_without_exact_artifact_can_close_by_disposition(harness): sub = harness.install({"s1": json.dumps([_finding("n1", "note")]), "s2": CLEAN, "s3": CLEAN}) ctx = harness.make_ctx() _call(ctx) - fp = _state(harness)["waves"][-1]["request_fingerprint"] + inline_wave = _state(harness)["waves"][-1] + fp = inline_wave["request_fingerprint"] result_path = harness.drive / "task_results" / "task-1.json" result = json.loads(result_path.read_text(encoding="utf-8")) result["plan_review_state"]["waves"][-1].pop("wave_artifact") + # A historical pre-artifact wave retained its complete inline spec. + result["plan_review_state"]["waves"][-1].pop("spec_in_artifact", None) + result["plan_review_state"]["waves"][-1].pop("spec_source_ref", None) + result["plan_review_state"]["waves"][-1]["spec"] = inline_wave["spec"] result_path.write_text(json.dumps(result), encoding="utf-8") closed = pr._handle_plan_task(ctx, review_disposition={ @@ -1560,5 +1565,3 @@ def test_epoch_replay_free_while_unchanged_transient_keeps_it_healed_repays(harn assert len(sub.calls) == 2 assert [s.slot_id for s in sub.calls[1]["slots"]] == ["s1", "s2", "s3"] assert _state(harness)["cycles_paid"] == 2 - - diff --git a/tests/test_plan_review_w3.py b/tests/test_plan_review_w3.py index 94b57ddb7..aa5789901 100644 --- a/tests/test_plan_review_w3.py +++ b/tests/test_plan_review_w3.py @@ -299,14 +299,14 @@ def test_state_stays_persistable_at_the_worst_case_request_bounds(tmp_path): wide = "𝕏" * plan_spec.MAX_ITEM_CHARS # 4-byte UTF-8 each n_items = plan_spec.MAX_LIST_ITEMS - # a MAXIMAL normalized spec: every list full, every string at the per-string bound, 4-byte chars + # A large operative spec beside maximally populated reviewer-request memory. spec = { "goal": "Ship", "acceptance_claims": [f"{wide[:-4]}c{i:03d}" for i in range(n_items)], "in_scope": [f"{wide[:-4]}i{i:03d}" for i in range(n_items)], "non_goals": [f"{wide[:-4]}n{i:03d}" for i in range(n_items)], "invariants": [f"{wide[:-4]}v{i:03d}" for i in range(n_items)], "decisions": [{"choice": wide, "why": wide, - "rejected": [wide] * plan_spec.MAX_REJECTED_PER_DECISION} for _ in range(n_items)], + "rejected": [wide] * 8} for _ in range(n_items)], "deferred": [{"what": wide, "why_safe_to_defer": wide} for _ in range(n_items)], "affected_resources": [f"{wide[:-4]}a{i:03d}" for i in range(n_items)], "evidence": [f"{wide[:-4]}e{i:03d}" for i in range(n_items)], diff --git a/tests/test_plan_spec.py b/tests/test_plan_spec.py index 01775f8eb..968ae691f 100644 --- a/tests/test_plan_spec.py +++ b/tests/test_plan_spec.py @@ -77,25 +77,17 @@ def test_normalize_spec_errors_are_typed(): assert plan_spec.normalize_spec("not a mapping") == ({}, ["spec: must be an object"]) # type: ignore[arg-type] -def test_normalize_spec_bounds_lists_with_recorded_omission(): +def test_normalize_spec_preserves_lists_beyond_former_bounds(): + items = [f"item {i}" for i in range(43)] + rejected = [f"r{i}" for i in range(10)] spec, errors = plan_spec.normalize_spec({ - "goal": "g", "in_scope": [f"item {i}" for i in range(plan_spec.MAX_LIST_ITEMS + 3)], + "goal": "g", "in_scope": items, + "decisions": [{"choice": "c", "rejected": rejected}], }) assert errors == [] - assert len(spec["in_scope"]) == plan_spec.MAX_LIST_ITEMS - assert spec["normalization_omissions"] == [ - f"in_scope: {plan_spec.MAX_LIST_ITEMS + 3} items declared, kept the first " - f"{plan_spec.MAX_LIST_ITEMS} (bound {plan_spec.MAX_LIST_ITEMS})" - ] - # B-10: nested cap on decision.rejected, recorded the same way. - spec, errors = plan_spec.normalize_spec({ - "goal": "g", "decisions": [{"choice": "c", "rejected": [f"r{i}" for i in range(plan_spec.MAX_REJECTED_PER_DECISION + 2)]}], - }) - assert errors == [] and len(spec["decisions"][0]["rejected"]) == plan_spec.MAX_REJECTED_PER_DECISION - assert spec["normalization_omissions"] == [ - f"decisions[0].rejected: {plan_spec.MAX_REJECTED_PER_DECISION + 2} items declared, kept the first " - f"{plan_spec.MAX_REJECTED_PER_DECISION} (bound {plan_spec.MAX_REJECTED_PER_DECISION})" - ] + assert spec["in_scope"] == items + assert spec["decisions"][0]["rejected"] == rejected + assert spec["normalization_omissions"] == [] def test_normalize_spec_goal_type_and_bool_scalars(): @@ -800,15 +792,15 @@ def test_prior_blocking_findings_survive_the_section_bound(monkeypatch): assert "no prior findings recorded" in empty_cycle2 and "First cycle" not in empty_cycle2 -def test_spec_section_is_bounded_structurally_never_clipped(): +def test_current_spec_is_complete_while_historical_json_views_remain_bounded(): worst = {"goal": "g", "decisions": [ - {"choice": "c" * 600, "rejected": ["r" * 600] * plan_spec.MAX_REJECTED_PER_DECISION, "why": "w" * 600} + {"choice": "c" * 600, "rejected": ["r" * 600] * 8, "why": "w" * 600} for _ in range(plan_spec.MAX_LIST_ITEMS) ], "in_scope": ["i" * 600] * plan_spec.MAX_LIST_ITEMS, "invariants": ["v" * 600] * plan_spec.MAX_LIST_ITEMS} spec, errors = plan_spec.normalize_spec(worst) assert errors == [] - text, notes = plan_spec.bounded_json(plan_spec.spec_with_ids(spec), plan_spec.PACKET_SPEC_CHARS) - assert len(text) <= plan_spec.PACKET_SPEC_CHARS + text, notes = plan_spec.bounded_json(plan_spec.spec_with_ids(spec), 120_000) + assert len(text) <= 120_000 json.loads(text) # whole items only — always valid JSON assert notes and all("kept " in n and "full-set sha256=" in n for n in notes) packet = plan_packet.build_plan_review_user_content( @@ -816,20 +808,23 @@ def test_spec_section_is_bounded_structurally_never_clipped(): prior_cycles=[], dispositions=[], spec_delta=None, root_exploration_log=None, ) spec_section = packet[packet.index("## SPEC"):packet.index("## PLAN PROSE")] - assert len(spec_section) < plan_spec.PACKET_SPEC_CHARS + 2000 and "OMISSION NOTE (structural)" in spec_section + assert len(spec_section) > 120_000 and "OMISSION NOTE" not in spec_section + rendered_spec = json.loads(spec_section.split("```json\n", 1)[1].split("\n```", 1)[0]) + assert rendered_spec == plan_spec.spec_with_ids(spec) # Oversized scalars with no list left → typed omission object, never clipped JSON. text, notes = plan_spec.bounded_json({"blob": "x" * 500}, 100) assert json.loads(text)["omitted"] is True and "full_payload_sha256" in text and notes -def test_user_content_bounds_are_disclosed_not_silent(): +def test_user_content_preserves_current_plan_prose(): spec, _ = plan_spec.normalize_spec(DECK_SPEC) manifest = plan_evidence.resolve_evidence([], active_root=".", allowed_roots=["."]) content = plan_packet.build_plan_review_user_content( - objective="o", goal=spec["goal"], plan_prose="P" * (plan_spec.PACKET_PROSE_CHARS + 500), spec=spec, + objective="o", goal=spec["goal"], plan_prose="P" * 40_500 + "DECISIVE_PLAN_TAIL", spec=spec, manifest=manifest, prior_cycles=[], dispositions=[], spec_delta=None, root_exploration_log=None, ) - assert "OMISSION NOTE" in content and "(no evidence declared)" in content + assert "P" * 40_500 + "DECISIVE_PLAN_TAIL" in content + assert "OMISSION NOTE" not in content and "(no evidence declared)" in content assert "(not provided by host)" in content diff --git a/tests/test_project_plain_rows.py b/tests/test_project_plain_rows.py index c7470aa00..c6b84c082 100644 --- a/tests/test_project_plain_rows.py +++ b/tests/test_project_plain_rows.py @@ -20,6 +20,8 @@ import json import types from types import SimpleNamespace +import pytest + MARKDOWN_RESULT = ( "# Report title\n\n## Short conclusion\n\nBody with `inline code` and **bold**." ) @@ -106,6 +108,28 @@ def test_completion_summary_event_text_is_plain_and_fully_normalized( assert queued[0]["system_type"] == "project_completion_summary" +@pytest.mark.parametrize("carrier", ["event", "task", "result", "done"]) +def test_direct_project_completion_stays_in_its_room(tmp_path, monkeypatch, carrier): + from ouroboros.project_dialogue import enqueue_project_completion_summary + from ouroboros.projects_registry import create_project + + project = create_project(tmp_path, "room", name="Room") + rows = { + "event": {"status": "completed"}, + "task": {"id": "conversation", "project_id": "room", "chat_id": project["chat_id"]}, + "result": {"status": "completed", "project_id": "room", "result": "Ordinary answer"}, + "done": {"status": "completed"}, + } + rows[carrier]["_is_direct_chat"] = True + queued = [] + monkeypatch.setattr("supervisor.terminal_delivery.enqueue_terminal_delivery", + lambda _root, event: queued.append(event) or True) + assert enqueue_project_completion_summary( + tmp_path, rows["event"], "conversation", rows["task"], rows["result"], rows["done"], + ) is False + assert not queued + + def test_host_salvage_terminal_incident_drops_inherited_markdown_format(tmp_path): """RO9: the fixed host-salvage receipt must not inherit ``format: "markdown"`` from the completed-answer base event; the completed paths diff --git a/tests/test_project_routing_v664.py b/tests/test_project_routing_v664.py index ec6c89d9d..5743028cc 100644 --- a/tests/test_project_routing_v664.py +++ b/tests/test_project_routing_v664.py @@ -286,7 +286,9 @@ def test_project_single_active_direct_root_gets_zero_call_mailbox_delivery(tmp_p ephemeral=lambda *_a, **_k: calls.append("ephemeral"), direct=lambda *_a, **_k: calls.append("direct"), ) - ctx.get_chat_agent = lambda: direct_agent + from supervisor.active_activity import get_direct_activity_registry + + get_direct_activity_registry().register(direct_agent._current_task_id, chat_id, actor=direct_agent) class Bridge: def get_updates(self, offset=0, timeout=1): @@ -357,7 +359,9 @@ def test_project_direct_stale_race_releases_admission_lock_once(tmp_path): lock = RacingLock() direct_agent._owner_message_admission_lock = lock ctx = _ctx(tmp_path) - ctx.get_chat_agent = lambda: direct_agent + from supervisor.active_activity import get_direct_activity_registry + + get_direct_activity_registry().register(direct_agent._current_task_id, chat_id, actor=direct_agent) assert server._route_project_chat_to_running_task(ctx, chat_id, "late follow-up") == "" # One release belongs to the manifest snapshot and one to the routing @@ -375,8 +379,8 @@ def test_main_inline_decision_has_no_predecision_annotation(tmp_path, monkeypatc calls = [] ctx = _ctx( tmp_path, - ephemeral=lambda cid, text, image, **kwargs: calls.append((cid, text, image, kwargs)), - direct=lambda *_a, **_k: (_ for _ in ()).throw(AssertionError("direct lane bypassed router")), + direct=lambda cid, text, image, **kwargs: calls.append((cid, text, image, kwargs)), + ephemeral=lambda *_a, **_k: (_ for _ in ()).throw(AssertionError("ordinary routing became ephemeral")), ) broadcasts = [] @@ -711,7 +715,7 @@ def test_transport_without_client_id_gets_stable_host_owned_routing_id(tmp_path, calls = [] ctx = _ctx( tmp_path, - ephemeral=lambda cid, text, image, **kwargs: calls.append((cid, kwargs)), + direct=lambda cid, text, image, **kwargs: calls.append((cid, kwargs)), ) logged = [] broadcasts = [] diff --git a/tests/test_promote_chat_flow.py b/tests/test_promote_chat_flow.py index 3ec0dcae6..6cfd85491 100644 --- a/tests/test_promote_chat_flow.py +++ b/tests/test_promote_chat_flow.py @@ -1821,9 +1821,9 @@ def test_route_project_chat_does_not_confirm_failed_mailbox_write(tmp_path, monk ) -def test_busy_project_chat_routes_to_ephemeral_decision_turn(tmp_path, monkeypatch): +def test_busy_project_chat_runs_native_with_routing_context(tmp_path, monkeypatch): """WS1/P5 (v6.34.0): a busy PROJECT chat is NOT mechanically auto-enqueued into a - duplicate pooled task. It runs the ephemeral decision turn (project-scoped, seeing + duplicate pooled task. It runs a native conversation turn (project-scoped, seeing current_chat.running_tasks) so the one mind decides steer_task / answer / promote by judgment — replacing the old 'Hybrid B+' auto-enqueue fallback.""" import threading as _threading @@ -1834,7 +1834,7 @@ def test_busy_project_chat_routes_to_ephemeral_decision_turn(tmp_path, monkeypat proj = create_project(tmp_path, "market-research") project_chat = int(proj["chat_id"]) enqueued = [] - ephemeral_calls = [] + direct_calls = [] called = _threading.Event() monkeypatch.setattr("supervisor.message_bus.log_chat", lambda *a, **k: None) @@ -1856,8 +1856,14 @@ def test_busy_project_chat_routes_to_ephemeral_decision_turn(tmp_path, monkeypat def inject_observation(self, _text): return None - def _ephemeral(cid, text, image_data, *, task_constraint=None, task_metadata=None): - ephemeral_calls.append({"chat_id": cid, "text": text, "metadata": task_metadata}) + def pause(self): + pass + + def resume(self): + pass + + def _direct(cid, text, image_data, *, task_constraint=None, task_metadata=None): + direct_calls.append({"chat_id": cid, "text": text, "metadata": task_metadata}) called.set() ctx = types.SimpleNamespace( @@ -1867,19 +1873,19 @@ def test_busy_project_chat_routes_to_ephemeral_decision_turn(tmp_path, monkeypat update_state=lambda fn: fn({"owner_id": 1, "owner_chat_id": 1}), consciousness=_Consciousness(), get_chat_agent=lambda: types.SimpleNamespace(_busy=True), - handle_chat_direct=lambda *a, **k: (_ for _ in ()).throw(AssertionError("direct lane must not run when busy")), - handle_chat_ephemeral=_ephemeral, + handle_chat_direct=_direct, + handle_chat_ephemeral=lambda *a, **k: (_ for _ in ()).throw(AssertionError("ordinary turn became ephemeral")), enqueue_task=lambda task: enqueued.append(task), send_with_budget=lambda *a, **k: None, ) assert server._process_bridge_updates(_Bridge(), 0, ctx) == 1 - assert called.wait(timeout=3) # the ephemeral decision turn ran on its own thread + assert called.wait(timeout=3) # the native conversation turn ran on its own thread assert enqueued == [] # NOT auto-enqueued into a duplicate pooled task - assert len(ephemeral_calls) == 1 - md = ephemeral_calls[0]["metadata"] or {} + assert len(direct_calls) == 1 + md = direct_calls[0]["metadata"] or {} assert str(md.get("project_id") or "") # project-scoped decision turn - assert "сколько будет 2+2?" in (ephemeral_calls[0]["text"] or "") + assert "сколько будет 2+2?" in (direct_calls[0]["text"] or "") def test_project_from_task_endpoint_creates_binding(tmp_path): @@ -2405,6 +2411,11 @@ def test_busy_direct_main_root_is_manifested_and_steerable_without_promotion(tmp _current_task_metadata={"client_message_id": "initial-1"}, _task_started_ts=10.0, ) + from supervisor.active_activity import get_direct_activity_registry + + get_direct_activity_registry().register( + direct_agent._current_task_id, direct_agent._current_chat_id, actor=direct_agent, + ) routing_ctx = types.SimpleNamespace( DRIVE_ROOT=tmp_path, RUNNING={}, @@ -2454,6 +2465,11 @@ def test_direct_turn_closed_admission_returns_manual_target(tmp_path): _current_chat_id=1, _current_task_metadata={}, ) + from supervisor.active_activity import get_direct_activity_registry + + get_direct_activity_registry().register( + direct_agent._current_task_id, direct_agent._current_chat_id, actor=direct_agent, + ) receipts = [] class Bridge: @@ -2585,6 +2601,11 @@ def test_direct_root_steering_uses_live_human_identity_for_receipt_and_notice( _current_task_text="Continue the Tower Defence task", _owner_message_generation=0, ) + from supervisor.active_activity import get_direct_activity_registry + + get_direct_activity_registry().register( + direct_agent._current_task_id, direct_agent._current_chat_id, actor=direct_agent, + ) acks = [] notices = [] ctx = types.SimpleNamespace( @@ -2640,6 +2661,11 @@ def test_direct_project_followup_carries_same_live_human_identity(tmp_path): _current_task_text="Continue the Tower Defence task", _owner_message_generation=0, ) + from supervisor.active_activity import get_direct_activity_registry + + get_direct_activity_registry().register( + direct_agent._current_task_id, direct_agent._current_chat_id, actor=direct_agent, + ) notices = [] ctx = types.SimpleNamespace( DRIVE_ROOT=tmp_path, diff --git a/tests/test_review_agent_session_route.py b/tests/test_review_agent_session_route.py index 693448824..849030139 100644 --- a/tests/test_review_agent_session_route.py +++ b/tests/test_review_agent_session_route.py @@ -803,9 +803,9 @@ class AccountedFakeLLM(FakeLLM): @pytest.mark.parametrize("elapsed_before_dispatch", [0, 2]) def test_unset_session_window_uses_task_absolute_ceiling(tmp_path, fake_route, monkeypatch, elapsed_before_dispatch): - import time - - started = time.monotonic() + # Exact clock origins keep ceil() of the integer wire horizon independent + # of floating-point cancellation at an arbitrary host-uptime offset. + started = 1_000.0 monkeypatch.setattr("ouroboros.config.get_task_abs_ceiling_sec", lambda: 21_600) monkeypatch.setattr("ouroboros.review_custody.monotonic_now", lambda *_args: started) monkeypatch.setattr("ouroboros.review_execution.monotonic_now", lambda: started + elapsed_before_dispatch) diff --git a/tests/test_runtime_mode_elevation.py b/tests/test_runtime_mode_elevation.py index 79274d9b4..d01e07a06 100644 --- a/tests/test_runtime_mode_elevation.py +++ b/tests/test_runtime_mode_elevation.py @@ -721,7 +721,9 @@ def test_started_predicate_is_read_only_and_never_constructs_the_agent(monkeypat class _Busy: _busy = True - monkeypatch.setattr(workers, "_chat_agent", _Busy(), raising=False) + from supervisor.active_activity import get_direct_activity_registry + + get_direct_activity_registry().register("settings-turn", 1, actor=_Busy()) assert _has_started_agent_tasks() is True @@ -817,21 +819,20 @@ def test_owner_context_mode_endpoint_refuses_lowering_while_task_runs(isolated_s def test_owner_context_mode_idle_predicate_covers_pending_and_direct_chat_busy(monkeypatch): - from types import SimpleNamespace + from supervisor.active_activity import get_direct_activity_registry from ouroboros.gateway import settings as settings_mod import supervisor.workers as workers monkeypatch.setattr(workers, "PENDING", [{"id": "queued"}]) monkeypatch.setattr(workers, "RUNNING", {}) - monkeypatch.setattr(workers, "_get_chat_agent", lambda: SimpleNamespace(_busy=False)) assert settings_mod._has_running_agent_tasks() is True monkeypatch.setattr(workers, "PENDING", []) - monkeypatch.setattr(workers, "_get_chat_agent", lambda: SimpleNamespace(_busy=True)) + get_direct_activity_registry().register("preparing", 1) assert settings_mod._has_running_agent_tasks() is True - monkeypatch.setattr(workers, "_get_chat_agent", lambda: SimpleNamespace(_busy=False)) + get_direct_activity_registry().unregister("preparing") assert settings_mod._has_running_agent_tasks() is False diff --git a/tests/test_server_extraction.py b/tests/test_server_extraction.py index 89387b54e..056a18e3e 100644 --- a/tests/test_server_extraction.py +++ b/tests/test_server_extraction.py @@ -33,7 +33,7 @@ _MOVED_OWNERS = { "_owner_restart_requested": server_process, "_request_restart_exit": server_process, "_restart_requested": server_process, - "_active_direct_root": server_routing_context, + "_active_direct_roots": server_routing_context, "_addressable_root_tasks": server_routing_context, "_chat_running_tasks": server_routing_context, "_clip_marked": server_routing_context, diff --git a/tests/test_task_constraint_server_routing.py b/tests/test_task_constraint_server_routing.py index 06ea8b5dc..afc32ba62 100644 --- a/tests/test_task_constraint_server_routing.py +++ b/tests/test_task_constraint_server_routing.py @@ -121,7 +121,7 @@ def test_repair_ui_copy_does_not_promise_a_removed_decision_round(): assert "visible_text:" not in text -def test_ordinary_busy_message_still_uses_ephemeral_lane(monkeypatch): +def test_ordinary_busy_message_uses_native_lane(monkeypatch): calls = {"ephemeral": [], "direct": []} bridge = FakeBridge() bridge.get_updates = lambda offset, timeout=1: [{ @@ -156,8 +156,8 @@ def test_ordinary_busy_message_still_uses_ephemeral_lane(monkeypatch): server._process_bridge_updates(bridge, 0, ctx) - assert calls["direct"] == [] - assert len(calls["ephemeral"]) == 1 + assert calls["ephemeral"] == [] + assert len(calls["direct"]) == 1 def test_visible_repair_command_is_deduped(monkeypatch): diff --git a/tests/test_update_owner_conversation.py b/tests/test_update_owner_conversation.py index 0fc1b0c04..53acebedb 100644 --- a/tests/test_update_owner_conversation.py +++ b/tests/test_update_owner_conversation.py @@ -150,7 +150,7 @@ def test_ephemeral_lane_runs_the_turn_while_the_resolver_holds_the_repo(tx_repo, lane.handle_chat_ephemeral(1, "why did the merge conflict?") - assert turns == [("ephemeral-agent", "why did the merge conflict?", True)] + assert turns == [(None, "why did the merge conflict?", True)] # construction belongs to registered execution assert notices == [] diff --git a/tests/test_v6613_project_room_lens.py b/tests/test_v6613_project_room_lens.py index d815dba88..04e90f84c 100644 --- a/tests/test_v6613_project_room_lens.py +++ b/tests/test_v6613_project_room_lens.py @@ -1,23 +1,21 @@ -"""v6.61.3 — project-room chat lens (affordance-context coherence). +"""Project-room conversation tools share one selected physical target. -The robot-room incident: a folder-room's DIRECT-CHAT lane resolved ``"."`` -against the system repo while the room fact named the project folder — the -agent listed the wrong tree and narrated it as the project. The lens re-points -the chat lane's active_workspace READS and the default shell cwd at the room's -registered working_dir; writes stay with promoted tasks (typed refusal). - -These tests pin BOTH sides: the lens where it must activate, and byte-identical -old behavior everywhere else (workspace tasks, subagents, file-less rooms, -headless/bench shapes never carry the lens key). +Native edits need no Git repository or promotion. Workspace tasks and children +keep their own bindings; missing room folders never select the system repo. """ from __future__ import annotations import json import pathlib +import sys + +import pytest from ouroboros.tools.registry import ToolContext +pytestmark = pytest.mark.serial + def _room_ctx(tmp_path, *, direct=True, with_room=True, workspace=False): repo = tmp_path / "repo" @@ -63,7 +61,7 @@ def test_lens_key_requires_all_legs(tmp_path): ctx5, room5, _ = _room_ctx(tmp_path) ctx5.task_metadata["_project_room_dir"] = str(tmp_path / "gone-folder") - assert project_room_lens_dir(ctx5) is None # missing dir: lens off, never a guess + assert project_room_lens_dir(ctx5) == (tmp_path / "gone-folder").resolve() # --- reads resolve to the ROOM folder; self-repo stays reachable explicitly -------- @@ -112,20 +110,189 @@ def test_fileless_room_and_workspace_task_unchanged(tmp_path): assert "app.py" in listing2 # workspace wiring untouched -# --- writes: typed refusal pointing at the promote path ---------------------------- +# --- writes: all ordinary tools share the room target ------------------------------ -def test_room_writes_refused_with_promote_hint(tmp_path): - from ouroboros.tools.core import _edit_text, _write_file +@pytest.mark.parametrize("tool", ["edit_text", "edit_batch", "apply_patch"]) +def test_room_skill_named_path_does_not_redirect_to_installed_payload(tmp_path, monkeypatch, tool): + from ouroboros.tools.registry import ToolRegistry + + monkeypatch.setenv("OUROBOROS_RUNTIME_MODE", "pro") + monkeypatch.setattr("ouroboros.safety.check_safety", lambda *a, **k: (True, "")) + ctx, room, repo = _room_ctx(tmp_path) + relative = "skills/external/demo/notes.txt" + room_file, installed_file = room / relative, ctx.drive_root / relative + for path in (room_file, installed_file): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text("before\n", encoding="utf-8") + (installed_file.parent / "SKILL.md").write_text( + "---\nname: demo\ndescription: Fixture\ntype: instruction\n---\nFixture\n", + encoding="utf-8", + ) + registry = ToolRegistry(repo, ctx.drive_root) + registry.set_context(ctx) + edit = {"path": relative, "old_str": "before", "new_str": "after"} + arguments = { + "edit_text": edit, + "edit_batch": {"edits": [edit]}, + "apply_patch": {"patch": f"*** Begin Patch\n*** Update File: {relative}\n@@\n-before\n+after\n*** End Patch"}, + }[tool] + result = registry.execute_result(tool, arguments) + assert result.status == "ok", result.text + assert room_file.read_text() == "after\n" + assert installed_file.read_text() == "before\n" + explicit = registry.execute_result("edit_text", { + "root": "skill_payload", "bucket": "external", "skill_name": "demo", + "path": "notes.txt", "old_str": "before", "new_str": "explicit", + }) + assert explicit.status == "ok", explicit.text + assert installed_file.read_text() == "explicit\n" + assert room_file.read_text() == "after\n" + + +@pytest.mark.parametrize("runtime_mode", ["light", "advanced", "pro"]) +@pytest.mark.parametrize("filename", ["game.js", "BIBLE.md"]) +def test_room_file_operations_share_an_ordinary_non_git_folder(tmp_path, monkeypatch, runtime_mode, filename): + from ouroboros.tools.registry import ToolRegistry + + monkeypatch.setenv("OUROBOROS_RUNTIME_MODE", runtime_mode) + monkeypatch.setattr("ouroboros.safety.check_safety", lambda *a, **k: (True, "")) + ctx, room, repo = _room_ctx(tmp_path) + system_before = (repo / filename).read_text() if (repo / filename).exists() else None + registry = ToolRegistry(repo, ctx.drive_root) + registry.set_context(ctx) + operations = [ + ("write_file", {"path": str(room / filename), "content": "// written\n"}, "written"), + ("edit_text", {"path": filename, "old_str": "written", "new_str": "edited"}, "edited"), + ("edit_batch", {"edits": [{"path": str(room / filename), "old_str": "edited", "new_str": "batched"}]}, "batched"), + ("apply_patch", {"patch": f"*** Begin Patch\n*** Update File: {filename}\n@@\n-// batched\n+// patched\n*** End Patch"}, "patched"), + ] + for name, args, expected in operations: + assert registry.get_schema_by_name(name) is not None + result = registry.execute_result(name, args) + assert result.status == "ok", (name, result.text) + assert "commit_reviewed" not in result.text and "headless runner" not in result.text + assert "CORE_PATCH_NOTICE" not in result.text + assert expected in (room / filename).read_text(encoding="utf-8") + assert expected in registry.execute("read_file", {"path": filename}) + assert ((repo / filename).read_text() if (repo / filename).exists() else None) == system_before + assert not (room / ".git").exists() + shell = registry.execute_result("run_command", { + "cmd": [sys.executable, "-c", f"from pathlib import Path; Path({filename!r}).write_text('// shell\\n')"], + }) + assert shell.status == "ok", shell.text + assert (room / filename).read_text(encoding="utf-8") == "// shell\n" + assert ((repo / filename).read_text() if (repo / filename).exists() else None) == system_before + assert (repo / "BIBLE.md").read_text(encoding="utf-8") == "repo marker\n" + + +def test_missing_room_never_reads_or_edits_the_system_copy(tmp_path): + from ouroboros.tools.registry import ToolRegistry, active_repo_dir_for + from ouroboros.tool_access import resource_root_path, resolve_shell_cwd + + ctx, _room, repo = _room_ctx(tmp_path) + missing = tmp_path / "gone-folder" + ctx.task_metadata["_project_room_dir"] = str(missing) + (repo / "game.js").write_text("// system copy\n", encoding="utf-8") + registry = ToolRegistry(repo, ctx.drive_root) + registry.set_context(ctx) + assert active_repo_dir_for(ctx) == resource_root_path(ctx, "active_workspace") == missing + assert resolve_shell_cwd(ctx)[0] == missing + assert "system copy" not in registry.execute("read_file", {"path": "game.js"}) + result = registry.execute("edit_text", {"path": "game.js", "old_str": "system copy", "new_str": "changed"}) + assert "file not found" in result.lower() or "no such file" in result.lower(), result + assert (repo / "game.js").read_text(encoding="utf-8") == "// system copy\n" + assert "system copy" in registry.execute("read_file", {"path": "game.js", "root": "system_repo"}) + + +def test_unresolved_room_binding_is_a_tool_error_with_explicit_system_reads_available(tmp_path): + from ouroboros.tools.registry import ToolRegistry + + ctx, _room, repo = _room_ctx(tmp_path, with_room=False) + ctx.task_metadata["_project_room_note"] = "project registry cannot be read" + registry = ToolRegistry(repo, ctx.drive_root) + registry.set_context(ctx) + for name, args in [ + ("read_file", {"path": "BIBLE.md"}), + ("write_file", {"path": "notes.txt", "content": "should not land"}), + ("run_command", {"cmd": ["pwd"]}), + ]: + result = registry.execute_result(name, args) + assert result.status != "ok" and "project registry cannot be read" in result.text + assert not (repo / "notes.txt").exists() + assert "repo marker" in registry.execute("read_file", {"path": "BIBLE.md", "root": "system_repo"}) + + +@pytest.mark.parametrize("cwd_kind", ["system_repo", "task_drive", "absolute_system"]) +def test_unresolved_room_keeps_explicit_authorized_process_roots(tmp_path, monkeypatch, cwd_kind): + from ouroboros.tool_access import resource_root_path + from ouroboros.tools.registry import ToolRegistry + + monkeypatch.setattr("ouroboros.safety.check_safety", lambda *a, **k: (True, "")) + ctx, _room, repo = _room_ctx(tmp_path, with_room=False) + ctx.task_metadata["_project_room_note"] = "project registry cannot be read" + registry = ToolRegistry(repo, ctx.drive_root) + registry.set_context(ctx) + cwd = str(repo) if cwd_kind == "absolute_system" else cwd_kind + expected = repo if cwd_kind == "absolute_system" else resource_root_path(ctx, cwd_kind) + result = registry.execute_result("run_command", { + "cwd": cwd, "cmd": [sys.executable, "-c", "from pathlib import Path; print(Path.cwd())"], + }) + assert result.status == "ok", result.text + assert str(expected.resolve()) in result.text + + +def test_room_delegation_snapshot_capture_and_vcs_share_the_selected_target(tmp_path, monkeypatch): + from ouroboros import delegate_custody as custody + from ouroboros.subagent_worktrees import find_execution_snapshot + from ouroboros.tools.delegate import _capture_terminal_patch, _derive_authority, _mutation_authority, _provision_snapshot + from ouroboros.tools.registry import ToolRegistry + from tests.test_delegated_run_isolation import _isolated_entry, _nanny_ctx, _seed_target + + target = _seed_target(tmp_path) + ctx = _nanny_ctx(tmp_path, target, monkeypatch) + ctx.workspace_root, ctx.workspace_mode, ctx.is_direct_chat = None, "", True + ctx.task_metadata["_project_room_dir"] = str(target) + monkeypatch.setattr("ouroboros.safety.check_safety", lambda *a, **k: (True, "")) + registry = ToolRegistry(ctx.repo_dir, ctx.drive_root) + registry.set_context(ctx) + shape = _derive_authority(ctx) + assert shape.access == "workspace_write" and shape.mode == "agent" + authority, error = _mutation_authority(ctx, shape) + assert not error and authority["target_root"] == str(target) + handle, error = _provision_snapshot(ctx, ctx.drive_root, authority["target_root"], "room-snapshot") + assert not error, error + execution = pathlib.Path(handle.path) + assert execution != target and (execution / "tracked.txt").read_text() == "one\ntwo\n" + try: + (execution / "from_delegate.txt").write_text("delegated change\n", encoding="utf-8") + entry = _isolated_entry(ctx, target, handle) + capture = _capture_terminal_patch(ctx, entry) + assert capture["authority_target_root"] == str(target) + assert not (target / "from_delegate.txt").exists() + result = registry.execute_result("integrate_delegated_patch", {"run_id": "run-1", "decision": "apply"}) + assert result.status == "ok", result.text + assert (target / "from_delegate.txt").read_text() == "delegated change\n" + assert not (ctx.repo_dir / "from_delegate.txt").exists() + status = registry.execute("vcs_status", {}) + assert "from_delegate.txt" in status and f"repo={target}" in status + assert find_execution_snapshot("room-snapshot") is None + assert not execution.exists() + finally: + custody._CUSTODY.clear() + + +def test_non_git_room_delegation_reports_its_actual_snapshot_limitation(tmp_path, monkeypatch): + from ouroboros.tools.delegate import _derive_authority, _mutation_authority, _provision_snapshot ctx, room, repo = _room_ctx(tmp_path) - out = _write_file(ctx, path="game.js", content="hack") - assert "ROOM_WRITE_VIA_TASK" in out and "promote_chat_to_task" in out - assert (room / "game.js").read_text(encoding="utf-8") == "// game\n" - assert not (repo / "game.js").exists() # the silent-repo-write trap is closed - - out2 = _edit_text(ctx, path="game.js", old_str="// game", new_str="// hacked") - assert "ROOM_WRITE_VIA_TASK" in out2 - assert (room / "game.js").read_text(encoding="utf-8") == "// game\n" + monkeypatch.setenv("OUROBOROS_SUBAGENT_WORKTREE_ROOT", str(tmp_path / "snapshots")) + authority, error = _mutation_authority(ctx, _derive_authority(ctx)) + assert not error and authority["target_root"] == str(room) + handle, error = _provision_snapshot(ctx, ctx.drive_root, authority["target_root"], "non-git-room") + refused = json.loads(error) + assert handle is None and refused["reason"] == "execution_snapshot_failed" + assert refused["target_root"] == str(room) and "not a git working tree" in refused["detail"] + assert not (room / ".git").exists() and not (repo / ".git").exists() # --- shell: the DEFAULT cwd is the room folder; explicit cwd still free ------------ @@ -194,7 +361,7 @@ def test_room_chat_lens_dir_resolver(tmp_path): update_project(data, "p1", working_dir=str(tmp_path / "vanished")) resolved2, note2 = room_chat_lens_dir(data, "p1") - assert resolved2 == "" and "unusable" in note2 # loud, never a silent repo fallback + assert resolved2 == str(tmp_path / "vanished") and "unusable" in note2 def test_context_fact_matches_lens_state(tmp_path, monkeypatch): @@ -224,7 +391,7 @@ def test_context_fact_matches_lens_state(tmp_path, monkeypatch): ) task = {"id": "t1", "project_id": "p1", "_is_direct_chat": True} rendered = build_runtime_section(env, task) - assert "LOOKS AT the project folder" in rendered + assert "active_workspace is working_dir" in rendered update_project(data, "p1", working_dir=str(tmp_path / "vanished")) rendered2 = build_runtime_section(env, task) diff --git a/tests/test_work_order_source_coverage.py b/tests/test_work_order_source_coverage.py index 7068899b8..20212e1be 100644 --- a/tests/test_work_order_source_coverage.py +++ b/tests/test_work_order_source_coverage.py @@ -1,4 +1,4 @@ -"""Production-shaped custody and acceptance checks for over-budget work orders.""" +"""Custody and acceptance checks for historical partial-source work orders.""" from __future__ import annotations @@ -9,8 +9,6 @@ from types import SimpleNamespace def _fixture(tmp_path): from ouroboros.subagent_work_order import ( - WorkOrderBudgetExceeded, - build_work_order_source_request, canonical_work_order_source, compile_external_work_order, ) @@ -31,15 +29,22 @@ def _fixture(tmp_path): "workspace_root": str(repo), "workspace_mode": "external_workspace", } - overflow = None - try: - compile_external_work_order(task) - except WorkOrderBudgetExceeded as exc: - overflow = exc - else: # pragma: no cover - the fixture must exercise the over-budget branch - raise AssertionError("fixture unexpectedly fits the work-order budget") - assert overflow is not None - prompt, request = build_work_order_source_request(task, overflow) + # Historical rows remain readable after new work stopped producing partial + # source lenses. Build the stored shape directly, without a live size policy. + rendered = compile_external_work_order(task) + request = { + "schema": 1, "kind": "complete_work_order", "coverage": "partial", + "complete_chars": len(rendered), + "complete_sha256": hashlib.sha256(rendered.encode("utf-8")).hexdigest(), + "wire_budget_chars": 250_000, + "source": { + "kind": "task_result", "task_id": task["id"], "tool": "get_task_result", + "arguments": {"task_id": task["id"], "include_authority": True, + "include_work_order_source": True}, + "projection": "canonical_work_order", + }, + } + prompt = "WORK ORDER SOURCE REQUEST\n" + json.dumps(request) ctx = ToolContext( repo_dir=repo, drive_root=tmp_path, @@ -55,7 +60,7 @@ def _fixture(tmp_path): ) full_text, reason = canonical_work_order_source(ctx, request) assert not reason - assert len(full_text) == overflow.chars + assert full_text == rendered return ctx, request, full_text, prompt @@ -92,33 +97,15 @@ def _source_response(request, text, start, end): } -def test_route_source_request_channel_fails_closed_on_unknown_manifest(): - from ouroboros.subagent_work_order import route_source_request_channel - - class Gateway: - def harnesses(self): - return [{"id": "route", "manifest": {"capabilities": {}}}] - - assert route_source_request_channel(Gateway(), "route") == { - "status": "unverified", - "reason": "interactive_capability_missing", - "route": "route", - } - - -def test_actor_first_coordination_appendix_refuses_without_truncation(monkeypatch): +def test_actor_first_coordination_appendix_preserves_complete_text(): import ouroboros.tools.delegate as delegate authority = SimpleNamespace(delegated=False) - monkeypatch.setattr(delegate, "_host_instructions", lambda *_a, **_k: "base") - monkeypatch.setattr(delegate, "_ASSIGNMENT_FIELD_CHARS", 32) - instructions, refusal = delegate._build_start_instructions( - authority, coordination_context="x" * 100, - ) - assert instructions == "" - payload = json.loads(refusal) - assert payload["reason"] == "coordination_context_over_budget" - assert payload["coordination_context_chars"] == 100 + context = " \n" + "яё𐍈🚀\n" * 55_000 + "DECISIVE_TAIL\n " + instructions = delegate._build_start_instructions(authority, coordination_context=context) + assert instructions.endswith(context) + assert "git commit" in instructions + assert "OMISSION NOTE" not in instructions def test_started_replay_keeps_partial_request_and_verified_ranges(tmp_path): diff --git a/tests/test_workspace_authority_binding.py b/tests/test_workspace_authority_binding.py index e90481c16..172decf19 100644 --- a/tests/test_workspace_authority_binding.py +++ b/tests/test_workspace_authority_binding.py @@ -170,7 +170,8 @@ def test_binding_collision_blocks_mutation_but_exact_read_stays_inspectable(tmp_ assert not (data / "state" / "skills" / "same").exists() -def test_binding_preserves_project_room_read_lens_but_not_write_target(tmp_path): +@pytest.mark.parametrize("operation", ["read", "list", "search", "write", "edit", "shell", "vcs"]) +def test_binding_preserves_the_project_room_target_for_every_operation(tmp_path, operation): repo = tmp_path / "repo" data = tmp_path / "data" room = tmp_path / "room" @@ -183,15 +184,11 @@ def test_binding_preserves_project_room_read_lens_but_not_write_target(tmp_path) task_metadata={"_project_room_dir": str(room)}, ) - read_binding = build_resolved_resource_binding( - ctx, root="active_workspace", operation="read", path="README.md" - ) - write_binding = build_resolved_resource_binding( - ctx, root="active_workspace", operation="write", path="README.md" + binding = build_resolved_resource_binding( + ctx, root="active_workspace", operation=operation, path="README.md" ) - assert read_binding.base_path == room.resolve() - assert write_binding.base_path == repo.resolve() + assert binding.base_path == room.resolve() def test_binding_synthesizes_only_manifest_first_external_write_target(tmp_path): diff --git a/tests/test_ws3_wedge_resilience.py b/tests/test_ws3_wedge_resilience.py index 256218b5f..2c1974b6c 100644 --- a/tests/test_ws3_wedge_resilience.py +++ b/tests/test_ws3_wedge_resilience.py @@ -89,30 +89,27 @@ def test_chat_turn_wedged_detection(): assert server._chat_turn_wedged(True, now - 100, now, 0) is False # 0 = disabled -def test_chat_turn_liveness_reads_agent_without_taking_the_lock(monkeypatch): +def test_chat_turn_liveness_reads_all_actors_without_taking_admission_lock(monkeypatch): import types - import supervisor.workers as w + from supervisor.active_activity import get_direct_activity_registry - monkeypatch.setattr(w, "_chat_agent", None) - assert w.chat_turn_liveness() == (False, None, None) - - monkeypatch.setattr(w, "_chat_agent", types.SimpleNamespace( - _busy=True, _current_task_id="t1", _last_activity_ts=1234.0)) - # Hold _chat_agent_lock to prove the liveness read never blocks on it (a wedged - # turn holds the lock for its whole duration — the watchdog must not deadlock). - assert w._chat_agent_lock.acquire(blocking=False) + registry = get_direct_activity_registry() + assert w.chat_turn_liveness() == [] + for tid, stamp in (("t1", 1234.0), ("t2", 2345.0)): + registry.register(tid, 1, actor=types.SimpleNamespace( + _busy=True, _current_task_id=tid, _last_activity_ts=stamp)) + assert w._repo_writer_gate_lock.acquire(blocking=False) try: - assert w.chat_turn_liveness() == (True, "t1", 1234.0) + assert w.chat_turn_liveness() == [("t1", 1234.0), ("t2", 2345.0)] finally: - w._chat_agent_lock.release() + w._repo_writer_gate_lock.release() def test_watchdog_alerts_on_chat_turn_wedge(monkeypatch): import types import server - import supervisor.workers as w monkeypatch.setenv("OUROBOROS_SUPERVISOR_LIVENESS_DEADLINE_SEC", "1") alerts = [] @@ -126,7 +123,9 @@ def test_watchdog_alerts_on_chat_turn_wedge(monkeypatch): monkeypatch.setattr("supervisor.state.load_state", lambda: {"owner_chat_id": 7}) monkeypatch.setattr("supervisor.state.append_jsonl", lambda *a, **k: None) # The heartbeat stamp is MONOTONIC (OB-03) — seed it on the same clock. - monkeypatch.setattr(w, "_chat_agent", types.SimpleNamespace( + from supervisor.active_activity import get_direct_activity_registry + + get_direct_activity_registry().register("wedged1", 1, actor=types.SimpleNamespace( _busy=True, _current_task_id="wedged1", _last_activity_ts=time.monotonic() - 100)) stop = threading.Event() # local per-test token try: @@ -223,10 +222,11 @@ def test_wall_clock_jump_neither_fabricates_nor_masks_a_supervisor_stall(monkeyp jump must not MASK a real one. """ import server - import supervisor.workers as w monkeypatch.setenv("OUROBOROS_SUPERVISOR_LIVENESS_DEADLINE_SEC", "1") - monkeypatch.setattr(w, "_chat_agent", None) # isolate the stall half + from supervisor.active_activity import get_direct_activity_registry + + get_direct_activity_registry().clear() # isolate the stall half alerts = _collect_alerts(monkeypatch, 11) boot_mono = 500.0 @@ -264,7 +264,6 @@ def test_wall_clock_jump_neither_fabricates_nor_masks_a_chat_turn_wedge(monkeypa import types import server - import supervisor.workers as w monkeypatch.setenv("OUROBOROS_SUPERVISOR_LIVENESS_DEADLINE_SEC", "1") alerts = _collect_alerts(monkeypatch, 13) @@ -277,7 +276,9 @@ def test_wall_clock_jump_neither_fabricates_nor_masks_a_chat_turn_wedge(monkeypa monkeypatch.setattr(server_liveness, "time", clock) agent_stub = types.SimpleNamespace( _busy=True, _current_task_id="wedged-mono", _last_activity_ts=boot_mono) - monkeypatch.setattr(w, "_chat_agent", agent_stub) + from supervisor.active_activity import get_direct_activity_registry + + get_direct_activity_registry().register("wedged-mono", 1, actor=agent_stub) stop = threading.Event() # local per-test token try: server._start_supervisor_liveness_watchdog([boot_mono], stop)