diff --git a/devtools/benchmarks/continual_learning/RUNBOOK.md b/devtools/benchmarks/continual_learning/RUNBOOK.md index d2047948f..2d29bf487 100644 --- a/devtools/benchmarks/continual_learning/RUNBOOK.md +++ b/devtools/benchmarks/continual_learning/RUNBOOK.md @@ -26,16 +26,22 @@ Field-tested configuration and operational hazards from the 2026-07-20 full 1-se Disclosure (net-resilience sprint): `OUROBOROS_TRANSIENT_RETRY_MAX` no longer bounds a REMOTE pre-dispatch transport outage. That class (`transport_unavailable`, $0 released attempts) now waits and redials at the round level. CLB solve tasks carry no - `deadline_at` and the waiting itself spends $0, so the binding rail here is the - supervisor's absolute per-attempt ceiling (`OUROBOROS_TASK_ABS_CEILING_SEC` when set; the - runtime ships `unlimited`, and then the wait's own 6h operation window from episode entry - binds), not a deadline or budget rail: a dead egress holds the task up to that bound - instead of failing it after the burst. The wait is visible as durable `network_wait` events in the - isolated server's `events.jsonl`. Note: idle-rail survival via waiting progress notes - requires a real chat thread; headless tasks without a `chat_id` keep the idle rail - (reaper) as an additional bound on the wait. A transport death AFTER dispatch (the - socket dies mid-request) is repeated by the primary dispatch at most twice per round as - new physical attempts. One call has one outer attempt budget + `deadline_at` and the waiting itself spends $0, so the rails that can bind here are the + OPTIONAL ones you configure, not a window the wait invents. A managed (queued) task's + outage episode carries no wait bound of its own: `loop_transport` measures its window from + the owner deadline, and with no `deadline_at` there is none — the 6h + `OPERATION_WINDOW_FALLBACK_SEC` belongs to other operations (deep self-review, plan + review, vision) and is NOT applied to this wait. Set `OUROBOROS_TASK_ABS_CEILING_SEC` if + you want a finite per-attempt lifetime (the runtime ships `unlimited`); otherwise the + binding rail for a headless CLB solve task is the supervisor's idle reaper + (`OUROBOROS_TASK_IDLE_TIMEOUT_SEC`, measured from last real progress), plus Stop/cancel. + A dead egress holds the task up to whichever of those is actually set instead of failing + it after the burst — and with none of them set it holds indefinitely. The wait is visible + as durable `network_wait` events in the isolated server's `events.jsonl`. Note: idle-rail + survival via waiting progress notes requires a real chat thread; headless tasks without a + `chat_id` keep the idle rail (reaper) as an additional bound on the wait. A transport + death AFTER dispatch (the socket dies mid-request) is repeated by the primary dispatch at + most twice per round as new physical attempts. One call has one outer attempt budget (`OUROBOROS_TRANSIENT_RETRY_MAX` bounds every attempt of the call, repeats included), within which up to three `llm_api_error` rows can be typed transport-death failures (the first death plus at most two repeats), reserving up to three upper bounds against diff --git a/docs/DOMAIN_MAP.md b/docs/DOMAIN_MAP.md index 4e4cfa1e7..ae954533b 100644 --- a/docs/DOMAIN_MAP.md +++ b/docs/DOMAIN_MAP.md @@ -8,14 +8,14 @@ The manifest is the SSOT of the module→domain assignment (1:1, complete over t | domain | name | modules | proposed | |---|---|---:|---:| -| D01 | Agent core & main loop | 34 | 0 | +| D01 | Agent core & main loop | 35 | 0 | | D02 | LLM client, routing & providers | 38 | 0 | | D03 | Context assembly, fit & compaction | 11 | 0 | | D04 | Tool execution: registry, access & typed results | 21 | 0 | | D05 | Tool surfaces: files, code, shell, media, external | 28 | 0 | | D06 | Review stack | 67 | 0 | -| D07 | Delegation, subagents & Claudexor | 53 | 0 | -| D08 | Supervisor: queue, workers, events & runtime control | 47 | 0 | +| D07 | Delegation, subagents & Claudexor | 54 | 0 | +| D08 | Supervisor: queue, workers, events & runtime control | 48 | 0 | | D09 | Cancellation, owner control & process custody | 13 | 0 | | D10 | Git, update & release machinery | 28 | 0 | | D11 | Gateway, server & Web UI | 56 | 0 | @@ -28,7 +28,7 @@ The manifest is the SSOT of the module→domain assignment (1:1, complete over t | D18 | Launcher, packaging, platform & shared substrate | 15 | 0 | | D19 | Frozen contracts (ABI) | 10 | 0 | | D20 | Presence | 10 | 0 | -| **total** | | **566** | **0** | +| **total** | | **569** | **0** | ## Dependency direction matrix (strict, pinned) @@ -187,6 +187,7 @@ No function body (≥ 10 normalized lines) is shared verbatim across domains. Ne - `ouroboros/agent_dispatch.py` - `ouroboros/agent_startup_checks.py` - `ouroboros/agent_task_pipeline.py` +- `ouroboros/budget_pause.py` - `ouroboros/deadline_utils.py` - `ouroboros/focus.py` - `ouroboros/loop.py` @@ -402,6 +403,7 @@ No function body (≥ 10 normalized lines) is shared verbatim across domains. Ne - `ouroboros/claudexor_startup_failure.py` - `ouroboros/configured_subagents.py` - `ouroboros/delegate_containment.py` +- `ouroboros/delegate_continuation.py` - `ouroboros/delegate_custody.py` - `ouroboros/delegate_custody_memo.py` - `ouroboros/delegate_custody_reconcile.py` @@ -462,6 +464,7 @@ No function body (≥ 10 normalized lines) is shared verbatim across domains. Ne - `ouroboros/tools/followup.py` - `supervisor/__init__.py` - `supervisor/active_activity.py` +- `supervisor/budget_resume.py` - `supervisor/cognitive_operations.py` - `supervisor/direct_roots.py` - `supervisor/event_taxonomy.py` diff --git a/docs/architecture/01-high-level-architecture.md b/docs/architecture/01-high-level-architecture.md index 4aa417482..d62125757 100644 --- a/docs/architecture/01-high-level-architecture.md +++ b/docs/architecture/01-high-level-architecture.md @@ -46,6 +46,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de │ ├── task_admission.py ← Token-owned admission reservations fence duplicate user-ingress ids before Project/workspace/attachment side effects; queue.py stays the state authority (§5) │ ├── task_lifecycle.py ← Cancellation custody — the ONE settle owner of durable cancel intents — plus the `sweep_cancel_intents` watchdog and the queue-owned root-budget admission fence (flow: §5; rules: §10 invariant 14) │ ├── cancel_publication.py ← Cancellation settlement publication for `task_lifecycle.py`: typed CANCEL_* outcomes, artifact-honest cancelled result fields, ledger cost reconstruction, salvage, owed-before-settle registration, capture-miss terminalization (§5) + │ ├── budget_resume.py ← Exact-continuation budget Resume (#1196): the single-use, pause-id and generation-bound grant, its revocation, hold release/re-binding; re-exported by queue_transitions.py │ ├── queue_transitions.py ← Queue-owned transitions outside cancellation custody: acceptance-fence open/inspect/seal, explicit budget resume, typed `stop_evolution_tasks` (an incomplete stop leaves the campaign OPEN under the durable `evolution_owner_stopped` flag, cleared only by an owner start ingress) and fenced Project deletion (lineage ROOTS only; tombstone after provable quiescence); imports nothing from task_lifecycle (§5) │ ├── terminal_delivery.py ← Durable terminal-answer delivery seam for final answers, cancel salvage, cascade digests and non-retry reaps: restart-surviving `delivery_id` dedupe (the id digests only the stable part of the answer, so a rebuilt replay dedups) + the bounded PENDING outbox `state/terminal_deliveries.json`; typed `terminal_delivery_exhausted` and `terminal_delivery_handoff`; per-origin projection `host_salvage` / `host_notice` / `custody_notice` / `model_final` (§5; §6 Task lifecycle; §10 invariant 15) │ ├── task_reaper.py ← Single-owner off-loop reaper for timeout teardown and health-prepared terminal-file/crash jobs; an unconfirmed death keeps the slot reaping with `task_reaper_wedged`; mints no cancel intents (§5) @@ -228,6 +229,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de ├── delegate_start_instructions.py ← Stable host start instructions + a complete separately-hashed coordination appendix; host pre-start sends no appendix ├── delegate_target_drift.py ← Read-only authority-tree drift evidence for delegated capture; records changed paths without attributing them to the child or blocking a normal no-change disposition (§6 Delegated subagents) ├── delegate_recovery.py ← Narrow exact-leaf recovery for proven crash + planned self-restart; vetoes every no-resume cause + ├── delegate_continuation.py ← Explicit continuation of a settled run the engine cancelled at its wall-clock cap (`continue_from`): typed gate over durable custody (own run, confirmed `wall_clock_exceeded`, result read, patch disposed, same executor/authority) + the host block; not recovery ├── delegate_registration_policy.py ← `persistent_registration` + the STARTED-row field tables ├── delegate_pending.py ← Durable pending-invocation replay preserving the original idempotency key + canonical start body ├── delegate_custody_memo.py ← Process-local memo of the custody rows (`custody_rows`): an ordered `(st_dev, st_ino, consumed, st_mtime_ns)` fingerprint of the rotated events chain prefix plus a hash of the live file's consumed bytes, advanced by folding only appended bytes, refolded on any doubt, bypassed (never cached) while the chain is unreadable; inline legacy request bodies replaced by a re-readable locator; a warm cache with an exact fallback, not a durable projection diff --git a/docs/architecture/03-web-ui-pages-and-buttons.md b/docs/architecture/03-web-ui-pages-and-buttons.md index 5a4f2ba84..1b5b1e340 100644 --- a/docs/architecture/03-web-ui-pages-and-buttons.md +++ b/docs/architecture/03-web-ui-pages-and-buttons.md @@ -155,7 +155,7 @@ An addressing call (`promote_chat_to_task`, `route_to_project`, `steer_task`, `e #### Liveness census and the chat header -`GET /api/state` unites direct turns and ROOT managed queue tasks in `active_chat_activities`. Managed phases are `queued`, `budget_paused` (PENDING fenced until explicit resume), `working`, and `finalizing` (RUNNING with an open post-task checkpoint). Late-mounted chats hydrate from the queue. The census alone inserts live-set entries; finals and census delete them. Only `active_chat_activities_complete === true` with `supervisor_ready === true` clears every absent id, regardless of `kind`; incompleteness retains positive rows and clears nothing. Completeness requires successful reads of every required live-identity source. The page-wide snapshot sequencer projects reads every 3 seconds on Chat, otherwise every 20 seconds, without per-entry generations. +`GET /api/state` unites direct turns and ROOT managed queue tasks in `active_chat_activities`. Managed phases are `queued`, `budget_paused` (PENDING fenced until explicit resume), `working`, and `finalizing` (RUNNING with an open post-task checkpoint); a direct turn paused on its budget rail (#1196) is parked under its own id and reports those phases as `kind="direct_chat"`. Late-mounted chats hydrate from the queue. The census alone inserts live-set entries; finals and census delete them. Only `active_chat_activities_complete === true` with `supervisor_ready === true` clears every absent id, regardless of `kind`; incompleteness retains positive rows and clears nothing. Completeness requires successful reads of every required live-identity source. The page-wide snapshot sequencer projects reads every 3 seconds on Chat, otherwise every 20 seconds, without per-entry generations. Typing is a submission receipt: match `client_message_id`, retire local `Sending...`, request census; never insert, revive or extend liveness. `kind` selects `Thinking` for direct turns or `Working`/`Queued`/`Paused` for managed roots, never deletion immunity. Child typing still lacks kind; Telegram native typing ignores it. The header reducer reads connection, census, unconfirmed owner sends and live managed cards. A direct block is not a managed card: the header retains the census verdict, while a mounted unfinished block hosts its running indicator instead of the typing bubble. Terminal failure remains a task result, never reasonless header `Attention`. Only typing receipt, snapshot turn, durable routing receipt, replayed user row, turn conclusion or offline-queue eviction retires `Sending...`; live user echoes and socket writes cannot. Descendants enter the header through their cards, so typing alone cannot expose a child before progress creates its card. Unenumerated roots (including Presence, absent from the direct registry) likewise appear through cards alone. Only the reducer writes the badge, except the panel-boot `Online` seed. diff --git a/docs/architecture/05-supervisor-loop.md b/docs/architecture/05-supervisor-loop.md index 9fb200b3f..2eac976e7 100644 --- a/docs/architecture/05-supervisor-loop.md +++ b/docs/architecture/05-supervisor-loop.md @@ -12,7 +12,7 @@ A headless task is ADDRESSED when it is admitted, not when it is displayed (`log The run is also NAMED at admission and chat promotion, without a new model call: a caller-supplied `title` (`ouroboros run --title` or the top-level contract field; `metadata.title` is refused with a 400 like `metadata.project_id`) is authorship and fills both `title` and `suggested_name`; otherwise the request's first line fills `suggested_name` ALONE, so a truncated prompt never outranks a real name coined later. A `task_named` frame is broadcast on admission so the live card is never born showing its status phrase as a title. -`queue_snapshot.json` is an atomic recovery and diagnostic projection, not a second scheduler: pending and running rows, acceptance and root-budget fences, worker counts, assignable capacity, any pool-disabled reason, and the latest bounded root focus. Startup restores a recent snapshot into an empty pending queue and never resurrects ordinary RUNNING work: it FENCES every surviving RUNNING row with a durable cancel intent (`reason='server_shutdown'`, ledgered as `terminalized_running`), which cancellation custody terminalizes a watchdog window later, expiring its open quiz and closing the paired owner wait; a PENDING child below it is marked `pending_parent_interrupted` and settled by the boot's `kill_workers`, so a closed window leaves neither ghost nor orphan. Only an owner-wait handoff with an acknowledged planned-restart transaction outlives snapshot age — and an EXACT budget pause (#1196): a `_budget_pause` marker carrying `exact_continuation` and its `checkpoint` locator is never assignable, is restored across any snapshot age without waking while its task-result `budget_pause` row and source stay readable, and its grant rides `_budget_pause_resume`; a grant that never reached a worker before a restart, or meets an exhausted wallet before dispatch, returns to the pause (`revoke_exact_budget_resume`), never to a replay or a terminal. Assignment carries the grant's `paused_duration_sec` into the RUNNING row as `budget_paused_sec`, a carrier the timeout rail subtracts from execution time beside the quota clock; `started_at` is the original. The granted row also passes a still-standing root fence (its own root's Resume lifted eligibility). Terminal tasks stay terminal, a task with an active cancel intent is left to custody, descendants of an accepted or sealed root finalize as cancelled, and malformed fence evidence fails closed. Assignment mirrors RUNNING into the durable task result for EVERY assigned task, not only a subagent: both orphan healers read the STORED status, and an unmirrored root is a ghost no snapshot-less boot can settle, so orphan reconciliation is a terminal writer for roots too, closing the same quiz and wait the task-done seam closes. Focus updates merge through the existing worker event path under the queue lock; no awareness timer, ledger, or wake is created. `direct_roots.json` carries the symmetric direct-root projection and one aggregate gap/freshness fact. +`queue_snapshot.json` is an atomic recovery and diagnostic projection, not a second scheduler: pending and running rows, acceptance and root-budget fences, worker counts, assignable capacity, any pool-disabled reason, and the latest bounded root focus. Startup restores a recent snapshot into an empty pending queue and never resurrects ordinary RUNNING work: it FENCES every surviving RUNNING row with a durable cancel intent (`reason='server_shutdown'`, ledgered as `terminalized_running`), which cancellation custody terminalizes a watchdog window later, expiring its open quiz and closing the paired owner wait; a PENDING child below it is marked `pending_parent_interrupted` and settled by the boot's `kill_workers`, so a closed window leaves neither ghost nor orphan. Only an owner-wait handoff with an acknowledged planned-restart transaction outlives snapshot age — and an EXACT budget pause (#1196): a `_budget_pause` marker carrying `exact_continuation` and its `checkpoint` locator is never assignable and is restored across any snapshot age without waking; a marker whose task-result `budget_pause` row or source is unreadable, mismatched or missing, and one whose root holds an acceptance fence or whose snapshot has malformed acceptance/budget fences, is retained under a typed `_budget_pause_hold` (never dropped or cancelled); a RUNNING row whose durable pause was complete at shutdown is parked, not fenced; a direct turn's parked record keeps `_is_direct_chat`; its grant rides `_budget_pause_resume`, and a grant that never reached a worker before a restart, or meets an exhausted wallet before dispatch, returns to the pause (`revoke_exact_budget_resume`), never to a replay or a terminal. Assignment carries the grant's `paused_duration_sec` into the RUNNING row as `budget_paused_sec`, a carrier the timeout rail subtracts from execution time beside the quota clock; `started_at` is the original. Assignment rechecks an exact child's root grant and fence identity: a new root pause revokes pending child grants, and another root Resume only restores eligibility for explicit child selection. A legacy zero-dispatch root Resume selects only that root through the existing hold; the root fence remains over its unselected children. Terminal tasks stay terminal, a task with an active cancel intent is left to custody, descendants of an accepted or sealed root finalize as cancelled, and malformed fence evidence fails closed. Assignment mirrors RUNNING into the durable task result for EVERY assigned task, not only a subagent: both orphan healers read the STORED status, and an unmirrored root is a ghost no snapshot-less boot can settle, so orphan reconciliation is a terminal writer for roots too, closing the same quiz and wait the task-done seam closes. Focus updates merge through the existing worker event path under the queue lock; no awareness timer, ledger, or wake is created. `direct_roots.json` carries the symmetric direct-root projection and one aggregate gap/freshness fact. `supervisor/queue_schedules.py` owns the existing `state/scheduled_tasks.json` table and is its only writer. Every read-modify-write — the tick, the skill resync, the gateway upsert, `schedule_followup`'s cap-and-write, `manage_schedules` — enters through `schedule_transaction`, which takes BOTH locks itself, queue lock then the table's sidecar file lock. The order lives in the transaction, not in its callers: a caller-composed order is what inverted, when the follow-up tool's one-transaction cap-read-and-write reached for the queue lock inside that hold against the tick's queue-then-table. It is reentrant PER TABLE (keyed by the resolved lock path), so a nested call on another drive root still takes that root's lock rather than riding a depth counter. `load_schedule_store` is the strict read every writer and owner surface uses: only an ABSENT path is an empty table it may create, while a present non-regular file, bytes that do not parse and a row that is not an object each raise `ScheduleStoreUnreadable` naming which it was. A write is refused on it — the next atomic write would replace real rows with an empty document or drop the rows it could not parse — and a READ answers unavailable, because "no schedules" is a claim while an unparseable table means the state is unknown. @@ -51,7 +51,7 @@ Heartbeat and progress are different evidence: a heartbeat proves a process or l A spawned or respawned slot is not assignable until its child's PID-bound `worker_ready` row arrives (`supervisor/worker_pool_lifecycle.py`). A live child's own `worker_starting` row, emitted before extension loading and agent construction, permits one extension of `WORKER_READY_WINDOW_SEC` to `WORKER_READY_CEILING_SEC` (300 seconds from birth, both in `runtime_limits.py`); foreign or pre-spawn rows cannot extend another slot. `worker_ready_window_extended` records that decision. A silent child keeps the original window, and logging failure cannot block startup. After `WORKER_READY_MAX_ATTEMPTS` failed attempts, `Worker.readiness_exhausted` is final for that exact slot — late events cannot reopen it. Total exhaustion, distinguished from busy/booting/reaping capacity and from a live owner-wait stack, closes pooled ingress (owner `/review` included) without blocking direct chat/control or boot/update recovery; once RUNNING completion custody has settled, `disable_exhausted_worker_pool` fails unstarted PENDING work honestly with a Restart hint, and a new task cannot clear the latch. Readiness stays separate from liveness and task idle time; a watcher error releases only still-booting, non-exhausted slots to the crash detector (`worker_ready_released`). Linux workers use forkserver; macOS and Windows use spawn. -Unexpected worker death reserves exact custody under the queue lock and enqueues `confirmed_dead_worker` on the reaper (`worker_health.recover_confirmed_dead_worker`). A saved terminal source wins even after signal death; unknown or incomplete file publication keeps the same job (`TerminalFileRecoveryPending`); only confirmed absence of one reaches the crash policy: a signal is an infrastructure failure, an otherwise eligible non-signal crash retries within `QUEUE_MAX_RETRIES`, preserving owner-wait replay restrictions and cost. A crash storm suppresses respawn while terminal sources settle, then its fence stops pooled admission; direct chat stays available. Startup runs the same terminal-file recovery in `_run_supervisor` after process custody and before `_startup_prune_sweeps` (the no-provider lifespan runs it too, spawning nothing); unknown or still-live ownership defers it rather than racing a writer, and any unresolved or protected source, or an ownership/read error, sets `preserve_task_sources`, skipping task-drive deletion for that pass. For older canonical scheduled rows, `_recover_terminal_task_files` restores that start binding only from a known non-direct child's positive running/started-at record when the existing fresh-queue and later-worker-boot checks prove it orphaned, with no pending queue owner or active cancel; the normal orphan reconciler and terminal guards retain authority, without resuming work. The recovery report includes `rebound`. +Unexpected worker death reserves exact custody under the queue lock and enqueues `confirmed_dead_worker` on the reaper (`worker_health.recover_confirmed_dead_worker`). A saved terminal source wins even after signal death; unknown or incomplete file publication keeps the same job (`TerminalFileRecoveryPending`); only confirmed absence of one reaches the crash policy: a signal is an infrastructure failure, an otherwise eligible non-signal crash retries within `QUEUE_MAX_RETRIES`, preserving owner-wait replay restrictions and cost. Budget-continuation evidence, including a consumed grant, forbids ordinary retry; a failed revocation of an unconsumed grant retains the exact source and grant in a nonterminal hold, with snapshot persistence failure disclosed rather than claimed durable. A crash storm suppresses respawn while terminal sources settle, then its fence stops pooled admission; direct chat stays available. Startup runs the same terminal-file recovery in `_run_supervisor` after process custody and before `_startup_prune_sweeps` (the no-provider lifespan runs it too, spawning nothing); unknown or still-live ownership defers it rather than racing a writer, and any unresolved or protected source, or an ownership/read error, sets `preserve_task_sources`, skipping task-drive deletion for that pass. For older canonical scheduled rows, `_recover_terminal_task_files` restores that start binding only from a known non-direct child's positive running/started-at record when the existing fresh-queue and later-worker-boot checks prove it orphaned, with no pending queue owner or active cancel; the normal orphan reconciler and terminal guards retain authority, without resuming work. The recovery report includes `rebound`. Startup and throttled maintenance reconcile three residue classes, the ~600 s pass off the loop thread (§10). Process custody checks strict PID, start-time, command, owner-task, session and generation evidence before it reaps. Delegated-run reconciliation applies the same owner-gone reasoning to harness rows (§6 Delegated subagents). Task, review and project reconciliation repair records whose producer no longer exists. None of these are command-line-class kill sweeps, and one instance never reaps another. The dedicated watchdog separately observes phase-stamped loop liveness and every native actor; a wedged chat turn alerts with a `/restart` hint, a loop stall only journals, and neither kills a thread. Other owner conversations run on independent native actors, without a second scheduler. diff --git a/docs/architecture/06-agent-core.md b/docs/architecture/06-agent-core.md index 4cafeb7ed..f52af738f 100644 --- a/docs/architecture/06-agent-core.md +++ b/docs/architecture/06-agent-core.md @@ -285,7 +285,7 @@ An undisclosed spend contributes `0.0` to `accounted_usd` — inventing a conser **No terminal or cancel claim without a verified receipt.** `delegate_cancel` returns `confirmed` (read back terminal), `requested`, `failed` or `containment_fault_run_may_still_be_live`; the last two hold a durable CRITICAL containment fault until a receipt or settlement clears it — an overpowered run that may still be alive is an incident, not a reassuring string — and the state read decides, so a refused control is never a verdict about the RUN. One `daemon_says_absent` predicate decides everywhere that a 404 is the daemon ANSWERING that the resource is gone (scoped to the daemon that answered), never a failure to find out; custody closes such a run `delegate_run_closed_absent` (unreachable, not settled), inventing no terminal detail, usage or spend. Results are delivered, not severed: `delegate_wait` stages the whole terminal detail atomically under `task_drive/delegated_runs/.json` with a typed `output_delivery` block, and cut fields are renamed `*_preview` so a partial read of head-truncated JSON fails loudly instead of looking like an answer. -**Four nanny verbs** (`tools/delegate.py`): `delegate_start`, `delegate_wait`, `delegate_cancel`, `delegate_answer`; supervision, recovery, custody and transport live in the leaf modules this section names. There is deliberately no fake `hurry`: Claudexor truthfully exposes cancel and answers, not in-place steering. One refusal author, `delegate_shared._fail`, writes `ok: false` plus `host_code` beside the domain payload, never instead of it; `delegate_shared._owned_run` (OWNED/FOREIGN/UNKNOWN from durable rows) governs wait, cancel and answer and is deliberately not widened. `delegate_start` takes an exact `agent_session` `subagent_id` (or recovery-only `retry_of`); API actor ids are refused. A scheduled configured nanny's bootstrap (`subagent_bootstrap`) starts the exact snapshotted leaf before the first model round through the same `delegate_start(prompt="")` wrapper the model itself uses; recovery adoption precedes the durable zero-run/unknown-evidence fences because a fence may hide a live prior run, a fence-wake outranks every terminal, and the host never waits (`configured_session_started` receipt; waiting is the model's own `delegate_wait` decision). The orthogonal exact-resource selector (`root="skill_payload"`, `bucket`, `skill_name`) NAMES a resource, authorized through a fresh `ResolvedResourceBinding` for `skill_payload.write` — it never grants one, and the id still chooses transport. A DEFINITE start refusal (typed, no custody handle, and either the producer's own `definitely_unrun` marker or the closed `_DEFINITE_UNRUN_REASONS` set) ends the child UNRUN at $0; everything ambiguous wakes the model, because a false "spent nothing" terminal over a possibly-live run is the one direction classification must never fail toward. A geometry a read-only or payload-selector shape can never serve (`directory_execution_unavailable`) is refused with that marker before the daemon call and records a durable start-blocked row; an unregistered project root is another such refusal — the nanny registers first, and a registration WE created is retired at settlement (`delegate_registration_policy`). The replacement and zero-run fences count the ACTOR's own runs: an unsettled run whose durable custody source is the review substrate is that substrate's obligation and never occupies this task's delegation slot — every delegation-domain reader of custody rows consumes the same `review_owned` predicate (`tests/test_custody_owner_kinds.py` is the consumer matrix a new reader joins), while physical custody (settlement, the ledger, containment, registration retirement) keeps seeing every run. +**Four nanny verbs** (`tools/delegate.py`): `delegate_start`, `delegate_wait`, `delegate_cancel`, `delegate_answer`; supervision, recovery, custody and transport live in the leaf modules this section names. A run the engine cancelled at its `maxSeconds` cap settles `cancelled` with the typed reason `wall_clock_exceeded` (recorded on the SETTLED row as `outcome_reason`, replayed as `terminal_reason`); `delegate_start(continue_from=)` continues it explicitly (`delegate_continuation.py`, #1196): a NEW run under a NEW cap and a NEW key, admitted only over this task's OWN settled run with that confirmed cause — never an owner deadline, Stop/Panic, user cancel, failure, unknown or unrecorded ending — after its full output was read to EOF and its captured patch was explicitly applied or rejected (an apply-ambiguous or undisposed patch refuses), on the same actor/route and the same access/mode/isolation/authority target; the started row carries `continuation_of`, the host block names the predecessor, its cause and its disposition and states that NO session state is transferred, and the model writes the remaining work in `prompt`. `delegate_recovery.NO_RESUME_CAUSES` is untouched: this is not crash recovery. There is deliberately no fake `hurry`: Claudexor truthfully exposes cancel and answers, not in-place steering. One refusal author, `delegate_shared._fail`, writes `ok: false` plus `host_code` beside the domain payload, never instead of it; `delegate_shared._owned_run` (OWNED/FOREIGN/UNKNOWN from durable rows) governs wait, cancel and answer and is deliberately not widened. `delegate_start` takes an exact `agent_session` `subagent_id` (or recovery-only `retry_of`); API actor ids are refused. A scheduled configured nanny's bootstrap (`subagent_bootstrap`) starts the exact snapshotted leaf before the first model round through the same `delegate_start(prompt="")` wrapper the model itself uses; recovery adoption precedes the durable zero-run/unknown-evidence fences because a fence may hide a live prior run, a fence-wake outranks every terminal, and the host never waits (`configured_session_started` receipt; waiting is the model's own `delegate_wait` decision). The orthogonal exact-resource selector (`root="skill_payload"`, `bucket`, `skill_name`) NAMES a resource, authorized through a fresh `ResolvedResourceBinding` for `skill_payload.write` — it never grants one, and the id still chooses transport. A DEFINITE start refusal (typed, no custody handle, and either the producer's own `definitely_unrun` marker or the closed `_DEFINITE_UNRUN_REASONS` set) ends the child UNRUN at $0; everything ambiguous wakes the model, because a false "spent nothing" terminal over a possibly-live run is the one direction classification must never fail toward. A geometry a read-only or payload-selector shape can never serve (`directory_execution_unavailable`) is refused with that marker before the daemon call and records a durable start-blocked row; an unregistered project root is another such refusal — the nanny registers first, and a registration WE created is retired at settlement (`delegate_registration_policy`). The replacement and zero-run fences count the ACTOR's own runs: an unsettled run whose durable custody source is the review substrate is that substrate's obligation and never occupies this task's delegation slot — every delegation-domain reader of custody rows consumes the same `review_owned` predicate (`tests/test_custody_owner_kinds.py` is the consumer matrix a new reader joins), while physical custody (settlement, the ledger, containment, registration retirement) keeps seeing every run. **Execution evidence.** `delegate_evidence.task_execution_evidence` projects the custody rows read-side; `delegate_start_attempted` counts blocked and uncustodied attempts too, so a refused-but-obedient nanny is never disclosed as nudge-ignoring (`nanny_nudge_recorded`). `applied_access_profiles` is read off SETTLED rows only (empty = no receipt disclosed it, never "no access"); `acceptance_patch_dispositions` is the bounded section over `delegate_run_patch_verdict` rows (cap 20 with the exact omitted count, `unreviewed_delegated_apply` headline) whose absence means NO disposition was recorded, never "reviewed clean"; an unreadable custody log is the typed `evidence_read_failed` marker, never an empty-therefore-clean section. @@ -335,7 +335,7 @@ Disclosed delegated-isolation residuals (deliberate): a live top-level task with **A mutating run requests its captured native profile, reads back what it got, and DISCLOSES the gap instead of refusing the work.** Full requests no filesystem sandbox; the private snapshot owns delivery, not OS confinement. The owned gateway creates a scoped full-access grant only when no trust record exists; an existing denial is preserved and can be addressed by lowering the invocation or capping the row. Grants persist per scope just like stable project registrations; no automatic trust cleanup is implied. Workspace-write keeps its existing boundary disclosure. In place because Claudexor otherwise hands the harness the operator's real `$HOME` — which holds the daemon token, so a compromised child could start its own runs at any access level. Four facts, one mechanism. (1) The `execution.delegated: true` marker rides in the same record as `isolation: live`, built from `delegated_run_shape` in one place, so one cannot be sent without the other. (2) The version floor is about the SCHEMA: `config.CLAUDEXOR_DELEGATED_MARKER_MIN_VERSION` (3.3.0) is the oldest engine whose strict `RunExecution` accepts `delegated`; below it the start is a 400 and no run exists (`route_health` blocks dispatcher and nanny identically before a token is spent). Read-only delegation sends no marker and keeps the 3.2.0 transport floor; an engine between the two floors serves read-only and refuses mutating. The floor is a schema fact, never a containment fact — the OS boundary is platform-dependent while a build declares one version everywhere; threat model and measured bands: `docs/DELEGATED_ADMISSION.md`. (3) What was APPLIED is asked of the attempt, never of the OS: `delegate_wait` reads `harness_home_isolated`, `confinement_mechanism` and the proven denied path from the attempt record (`gateways.claudexor.attempt_containment`, `delegate_containment.py`); `sys.platform` appears nowhere in the decision. (4) A missing boundary is disclosed (durable `delegate_run_unconfined` event, the child's instructions, the parent's terminal payload), not refused — the child already holds a shell in this worktree, and cutting the lane on every boundary-less host costs more than the marginal step it prevents; a home nested under the operator's own is disclosed-unconfined (`home_nested_under_operator_home`), never relabelled as isolation. A recorded FALSE is still a fault: `harness_home_isolated: false`, or a scoped home that IS the operator's own, cancels as a typed containment fault exactly like a widened access profile — those two exact facts are the WHOLE breach rule; a MISSING home fact is neither breach nor proof, so unproven is REPORTED. -**The model cannot widen its explicitly bounded delegated authority.** `delegate_start` exposes `prompt`, exact session `subagent_id`, `max_seconds`, optional lowering-only `access` (`readonly` or `workspace_write`), recovery-only `retry_of`, and the exact-resource selector; there is no mode, isolation, or scope argument. A retry rejects new access choices and cannot change route, root, access, or permissions, and every run carries host-authored `instructions` the model can neither widen nor forge. The typed sentence selects native process access; explicit task constraints and the assigned edit target still bind. Runtime reviewer sessions remain readonly/ask regardless of a referenced actor row's mutating profile. Because Claudexor DERIVES effective access rather than echoing the request, `delegate_wait` verifies rather than assumes: every fetched run detail goes through `_containment_breach`, the one reader for both halves of containment (access profile and harness HOME) because they fail identically — a verification written for one half leaves the other trusting an echo. A run enforced WIDER than the task is entitled to is cancelled as a typed `access_profile_widened` refusal; narrower is fine. +**The model cannot widen its explicitly bounded delegated authority.** `delegate_start` exposes `prompt`, exact session `subagent_id`, `max_seconds`, optional lowering-only `access` (`readonly` or `workspace_write`), recovery-only `retry_of`, the continuation-only `continue_from` (refused beside `retry_of` or the payload selector), and the exact-resource selector; there is no mode, isolation, or scope argument. A retry rejects new access choices and cannot change route, root, access, or permissions, and every run carries host-authored `instructions` the model can neither widen nor forge. The typed sentence selects native process access; explicit task constraints and the assigned edit target still bind. Runtime reviewer sessions remain readonly/ask regardless of a referenced actor row's mutating profile. Because Claudexor DERIVES effective access rather than echoing the request, `delegate_wait` verifies rather than assumes: every fetched run detail goes through `_containment_breach`, the one reader for both halves of containment (access profile and harness HOME) because they fail identically — a verification written for one half leaves the other trusting an echo. A run enforced WIDER than the task is entitled to is cancelled as a typed `access_profile_widened` refusal; narrower is fine. **Read provenance on the accounts surface.** `GET /api/claudexor/status` carries a `reads` block (`ClaudexorStatusReads`: `catalog`/`accounts`/`quota`, each `ok`|`not_read`|`failed`) because the owned daemon starts lazily: an idle daemon serves empty collections under a 200, and "no account connected" must not be inferred from a collection that was never read — `ok` makes the matching collection authoritative (empty means empty); one parity-tested client reader (`facetReadState`) applies the same rule, and the aggregate `daemon.state` is never the negative answer. Login jobs: the daemon stays the sole process/fence authority and reconcile is an explicit POST, never passive polling (`gateway/claudexor_accounts.py`; the route inventory is its §1 row and §4). @@ -607,7 +607,7 @@ The terminal checkpoint also retains `root_phase_checkpoint.accounting`: one cum The in-task pacing stop is an explicitly unreserved planning threshold in the shared global pool, resolved once per task as a typed `CostCeiling` (`disabled`, `active`, `exhausted_soft_land`, `unknown`; `task_pacing.py`), published in the start-of-task runtime budget block as `in_task_cost_ceiling` and consumed by the loop as the SAME object, so the number the mind is shown and the number that stops the task cannot differ. The root of a tree resolves the minimum of the configured percentage of global remaining budget and the root-tree cap, minus a small planning margin; enabled descendants keep that original number instead of taking a percentage of a later wallet. Explicitly disabled profiles keep the early axis disabled; a root without a task cap resolves from the starting wallet while actual global/root monetary admission stays independent; every host cost surface prints the bound that binds first and names it. The loop decides against subtree-accounted spend including in-flight holds (an own-cost fallback is disclosed as a lower bound), for every task that has a root, with or without a per-task cap — a global-only ceiling decides on the subtree, not the parent alone — and checks the axis only after tool-call rounds; graceful finalization runs before the ledger fence and never weakens it: unknown is not zero, an unknown price fails open, a disabled ceiling never arms it. The margin pulls the stop earlier so a post-round affordability crossing can borrow the fence's OWN per-attempt reservation — cache-aware from the task's last settled split for the same provider, normalized route and review surface (`_usage_cache_splits.py`, process-local; a lost entry only re-prices a full write) — and soft-land while one wrap-up call is still admitted instead of dying answerless at the fence. The admitted candidate takes the send's wire options from their one owner (`task_pacing.main_loop_wire_options`; a lane never adds a payload key after its builder), because a candidate that differs from the real send is refused by the identity predicate before dispatch; such a refusal is recorded (`forced_candidate_drift`) and the answer is asked for once more unpredicated — the fence prices the send it sees, and drift must never cost the owner the final answer. The comparison reserves nothing (a competing task can consume it first; a cache expiry may under-reserve by one write) — the ledger fence at the full cap still binds. The bounded context proxy only pre-screens: a proxy stop, and any prompt the proxy can understate (native image parts), is decided by exact pricing on a copy of the transcript before service finalization. A task that cannot reserve even one wrap-up says so in its own words; the exhausted-ceiling soft landing prices the same prepared candidate and ends as `budget_wrapup_unaffordable` rather than a fence pause when it cannot fit. A reply that still asks for a tool is incomplete on every rail, even beside a `replace` delivery control. -Exact budget pause (#1196, `ouroboros/budget_pause.py`). After real work, every monetary rail of a POOLED task — global exhaustion, the graceful ceiling, both last-fit wrap-up stops, the soft landing, a dispatch the ledger refused — pauses the SAME task id instead of spending a wrap-up call. Order is the invariant: a process-local dispatch fence closes for the task at `usage_accounting.reserve_attempt` (`DispatchFenced`, a `BudgetExceeded`) and explicitly at the two review seams that could otherwise send first — `review_execution` before any POST, a fresh start and a pending-invocation replay alike (`budget_pausing_no_send`), and Light verdict extraction (`budget_pausing_no_extraction`); a durable `budget_pause` row opens as `pausing` BEFORE anything else waits, so a death while the task is still settling meets the crash-retry fence (`has_budget_pause_checkpoint`, fail-closed on an unreadable record) instead of a replay; the task then HOLDS — keeping its native worker and its nonterminal `pausing` state, never buying a wrap-up call — until BOTH local producer families are quiescent: review attempts already sent (drained through their `review_custody` events) and tool futures still running after their logical timeout, each tracked in a task/attempt-scoped registry from its submit until its late settlement callback has FINISHED (`register_tool_future`/`hold_tool_settlement`, fed by `loop_tool_execution`; `done()` is not callback-complete, and a row this registry cannot observe is unknown, not quiescent), with nothing settled, refused, passed or refunded here; an unsettled producer or a failed write publishes a typed `hold_reason` on the existing `task_checkpoint` path (`budget_pause_hold`, owner-visible, on change) and retries, so persistent storage failure is a visible retained hold and never a claimed durable pause; only the task's existing Stop/Panic/deadline/cancel controls (`_hold_control_reason`, the live `model_wait` reader or the same owner-stop flags the restore gate reads) end a hold without a pause — that closes the opened row as `abandoned` and leaves on the control rail, and the dispatch fence never reopens here; every unsettled delegated run the task holds is read from the durable custody rows and, because pre-terminal subscription coverage is unprovable, a stop is REQUESTED through `cancel_and_verify` with its typed requested/confirmed/unknown outcome per run (unreadable custody is held as `custody_read=failed`, never an empty list); the ONE continuation serializer (`owner_wait.continuation_state`) stores the exact cognition plus a program counter (`resume_point`: the interrupted batch's unanswered tool-call ids as execution UNKNOWN — a missing result never proves a call did not run, so the host re-executes nothing and a resume closes them with host tool rows); the `budget_pause` row on the task result is written (`pausing`); only then `BudgetPauseRequested` unwinds the loop with no task_done, no result text and no Main final, and the worker's `budget_pause` event carries the locator. Ineligible actors (direct chat, no queue continuation owner) keep the historical terminal, `resource_limit.exact_pause_unavailable` naming why. The supervisor (`events_budget.install_exact_budget_pause`) validates the event against THAT row, parks the task in PENDING under `_budget_pause` (root scope raises the root admission fence), persists the snapshot and only then marks `paused`; a worker death during pausing completes the same park from the durable rows (`worker_health._complete_exact_budget_pause_after_death`) instead of a crash retry; a paused row is restored across any snapshot age without waking (`restore_budget_pause_allowed`). Resume is the owner's explicit act (`POST /api/tasks/{id}/resume` → `queue_transitions.grant_exact_budget_resume`): money above zero and under the root cap, no cancel intent, the deadline, the finite lifetime computed with a SEPARATE `paused_duration_sec` carrier (`started_at` is never moved, the quota clock is not a pause clock, `None` stays unlimited), a readable source, and for a descendant a root that is not itself paused; it mints ONE single-use grant that returns to the pause if money vanishes before dispatch or a restart intervenes (`revoke_exact_budget_resume`). The loop consumes the grant (`resume_paused_loop`), restores cognition despite workspace drift, discloses drift and external custody before any new effect, and for a graceful rail refreshes the planning threshold within the money still authorized (Q10; the ledger fence is untouched). A root's Resume only makes its own budget-paused descendants ELIGIBLE: the model selects each with `resume_child_task` (`budget_resume_child`, lineage-checked, the same grant seam); cancelled, completed or otherwise-stopped members are never revived. +Exact budget pause (#1196, `ouroboros/budget_pause.py`). After real work, every monetary rail — global exhaustion, the graceful ceiling, both last-fit wrap-up stops, the soft landing, a dispatch the ledger refused — pauses the SAME task id instead of spending a wrap-up call; pooled task and direct owner-chat turn alike (a context with no task id, durable root or continuation owner keeps the terminal, `resource_limit.exact_pause_unavailable` naming why). Order is the invariant: a process-local dispatch fence closes at `usage_accounting.reserve_attempt` (`DispatchFenced`) and at the two review seams that could send first — `review_execution` before any POST (`budget_pausing_no_send`) and Light verdict extraction (`budget_pausing_no_extraction`); a durable `budget_pause` row opens as `pausing` BEFORE anything waits (a direct turn without an admission-written RUNNING row gets one first), so a death while settling meets the crash-retry fence (`has_budget_pause_checkpoint`, fail-closed on an unreadable record), not a replay; the task then HOLDS — worker kept, nonterminal, no wrap-up call — until BOTH local producer families are quiescent: review attempts already sent (drained through their `review_custody` events) and tool futures still running after their logical timeout, tracked per attempt from submit until the late settlement callback has FINISHED (`register_tool_future`/`hold_tool_settlement`; `done()` is not callback-complete, an unobservable row is unknown); an unsettled producer or a failed write publishes a typed `hold_reason` on the `task_checkpoint` path (`budget_pause_hold`, on change) and retries — storage failure is a visible retained hold, never a claimed pause; only the task's existing Stop/Panic/deadline/cancel controls (`_hold_control_reason`) end a hold without a pause, closing the row `abandoned`; every unsettled delegated run is read from custody and, since pre-terminal subscription coverage is unprovable, a stop is REQUESTED through `cancel_and_verify` with its typed requested/confirmed/unknown outcome (unreadable custody is `custody_read=failed`, never an empty list); the ONE serializer (`owner_wait.continuation_state`) stores the cognition plus a program counter (`resume_point`: the interrupted batch's unanswered tool-call ids as execution UNKNOWN — nothing is re-executed; a resume closes them with host rows); the row is written, then `BudgetPauseRequested` unwinds the loop with no task_done, result text or Main final. The worker's `budget_pause` event carries the locator, and for a direct turn the turn's own task record with its `_is_direct_chat` lane fact (`parkable_direct_task`; the live actor ends as on any ending — no second live actor): `events_budget.install_exact_budget_pause` validates the event against THAT row and parks the RUNNING row or the carried record in PENDING under `_budget_pause` (root scope raises the root fence; the snapshot persists the lane fact), persists, then marks `paused` — park, snapshot, confirmation, owner projection and event all under the SAME queue lock a Resume grant holds, and the confirmation is a compare-and-set on the pause id AND its state, so a late park can never publish `paused` over a live grant or rewrite its projection (a row that moved on is published as `budget_pause_park_superseded` instead); a worker death during pausing completes the same park (`worker_health._complete_exact_budget_pause_after_death`), and a worker that died holding an UNCONSUMED grant has that grant revoked under its own identity and the same id parked back on its exact pause rather than terminalized — a CONSUMED grant takes terminal crash custody without ordinary retry and is never reopened; failed revocation retains a nonterminal hold with its exact source/grant, explicitly memory-only when persistence fails; a restart parks a RUNNING row whose durable pause was already complete instead of fencing it (`queue_snapshot._park_pausing_running_rows`; an unconsumed grant is revoked first). A paused row is restored across any snapshot age without waking; a refusal (`budget_pause_restore_refusal`: unreadable/terminal/mismatched record, missing or unreadable source) or an acceptance fence over the root or malformed snapshot fences never drops or cancels it — the row keeps its marker under a typed durable HOLD (`events_budget.budget_hold_fact`, `_budget_pause_hold`), un-dispatchable until the next grant re-validates the authority. Resume is the owner's explicit act (`POST /api/tasks/{id}/resume` → `queue_transitions.resume_budget_paused_task` → `budget_resume.grant_exact_budget_resume`): authoritative ledger money above zero and under the root cap, no cancel intent, the deadline, the finite lifetime on the ONE shared clock (`model_wait.execution_elapsed_seconds`: wall time minus the quota union minus the separate `paused_duration_sec` carrier, which rides EVERY same-ID continuation the one serializer writes — an owner wait parked after a pause resumes on the same clock; `started_at` never moves, `None` stays unlimited), a readable source, custody re-read when the pause could not read it, and for a descendant a root not itself paused; it mints ONE single-use grant bound to the pause id and a per-pause `resume_generation`, releases a hold beside the marker, and returns to the pause when money vanishes or a restart intervenes (`revoke_exact_budget_resume`: a newer pause or grant is never overwritten — `pauseA → Resume → pauseB → restart` re-reads the current pause; an unwritable revocation HOLDS the row and is written by the next Resume before a new grant). An exact root's Resume lifts its fence but makes descendants only ELIGIBLE (`events_budget.hold_root_resume_descendants`): exact-continuation children keep their rows, zero-dispatch siblings take a `root_fence_lifted_pending_selection` hold (re-bound to the live grant on a later Resume) — including a sibling whose only marker was minted from THAT latch, which takes the hold instead of being stranded on `root_budget_fence_missing` — and the model selects each with `resume_child_task` (`budget_resume_child`, lineage-checked, `selected_by` the requester) under its root's live grant (Q9); the same call is the owner's explicit selection. One member of a tree whose latch is still UP is released the same way and only for itself: the recorded selection names the fence generation it was granted against (`events_budget.budget_fence_selected`), so the latch stays up for its siblings and a newer fence is not pre-released. Legacy zero-dispatch root Resume selects only the root, retaining the fence for individual child selection. A new root pause invalidates pending child grants; assignment checks each exact selection against the still-current root grant and fence. The root tool can select any stored descendant of its own tree; intermediate parents retain direct-child selection. After a direct actor parks inline, its process-local dispatch fence is released because PENDING now holds that same id. Hold controls read the canonical budget root, including for a split execution drive. The loop consumes the grant (`resume_paused_loop`: refuses a spent, revoked or foreign grant), restores cognition despite drift, discloses drift and external custody before any new effect, and for a graceful rail refreshes the planning threshold within the money the LEDGER says is still authorized and relaxes the last-fit rail to ONE reservation (`loop_budget._second_reservation_fits`, disclosed as `budget_resume_last_fit_admitted`), so the one affordable call is placed instead of re-pausing on the number that paused it (Q10); the ledger fence at the full cap still binds. A resumed direct turn runs on a pooled worker under its own id, its frames re-stamped with the lane fact (`log_addressing.address_task_event`); cancelled, completed or otherwise-stopped members are never revived. One resolver answers the configured global budget for every agent-side reader (the supervisor's own startup and settings-reload carriers still parse the raw setting and map absence to no limit — a pre-existing, tracked gap), and it answers LIVE: a worker re-projects settings into its environment only at task start, so the resolver reads the saved document first (unlocked, re-parsed only when the file changed; any read failure, a refused benchmark pin included, leaves the environment answering and is never remembered) and no task captures the limit — the fence and every wallet projection follow the owner's current `TOTAL_BUDGET`, raised or lowered, at the next reservation, while the in-task ceiling stays the number resolved at start. An absent setting is silence, resolving to the shipped default, while an explicitly non-positive value means no finite global budget and silences the loop-side global axis. Pre-dispatch pricing (`pricing.py`) is an exact-route, bounded, best-effort lookup from the provider's current catalog — only the normalized exact model id and provider-supplied fields count; no manual price table, prefix inheritance, numeric fallback or admission allowlist disguised as pricing, which would silently invent authority and grow stale. Unknown price is nullable and fail-open for admission while known spend stays below its limits: it reserves `None` and settles from provider-reported cost or a later exact price, else cost stays `None` with `cost_final=false`. A rejection settles at confirmed zero only when structural provider evidence proves it happened before upstream generation with zero usage, releasing the reservation so a provider storm cannot manufacture phantom budget exhaustion (`_usage_response.py` normalizes that block for accounting; adapters still read the raw `usage` dict themselves). `review_wave_admission` applies the same per-attempt math against the tighter of the global and root remainders — the two fences `reserve_attempt` enforces, the binding axis named — before skill, plan, task-acceptance or P3 commit-gate reviewers launch, and the managed-update assisted-apply floor reuses the same estimator before any destructive merge step. An unpayable reviewer row is bypassed, never swapped to another model, so the audit stays honest about which model reviewed; an unpriced slot is disclosed and contributes no invented price while priced siblings still bind — one unknown route cannot disable admission control for a paid wave. diff --git a/docs/architecture/11-frozen-contracts-v1.md b/docs/architecture/11-frozen-contracts-v1.md index 9118a55b3..ab46e0cb2 100644 --- a/docs/architecture/11-frozen-contracts-v1.md +++ b/docs/architecture/11-frozen-contracts-v1.md @@ -21,7 +21,7 @@ This chapter owns the ABI promise: which typed shapes and their parsing, normali | `execution_evidence` — started/settled/succeeded/failed counts, `delegated_run_failure_states`, `evidence_read_failed`, `nanny_nudge_recorded`, `subscription_cost_usd` (None while undisclosed — never 0), `subscription_cost_estimated`, `harness_models`, `applied_access_profiles`; derived from durable custody rows by `delegate_evidence.task_execution_evidence`, attached in `subagents.envelope_from_task` at terminal statuses only (never overwriting `effective_executor`/`executor_route`), and enriched onto the pushed `task_done` frame by `enrich_task_done_event` in `supervisor/subagent_task_truth.py`; `actual_substrate` ∈ harness_used/harness_attempted/native_only from custody evidence only; `substrate_result_fields` = {actual_substrate, delegated_runs_started, delegated_runs_settled, delegated_runs_succeeded, delegated_runs_failed, delegated_runs_source_unresolved, native_contribution}; the `wait_tasks` compact projection carries `dispatch_executor` plus that same set; an unreadable custody log omits the substrate claim and counts everywhere — `evidence_read_failed` means UNKNOWN, never "no run", and absence of evidence on a pre-evidence stored result is never a zero-run receipt; `native_contribution` is the constant "unknown" (no share/ratio is derivable from custody rows); a verifiable `native_only` amends `capability_delta` with `delegated_substrate_unused`; the `log_events.js` executor chip renders layered truth with unverified work-order counts spelled out | `ouroboros/delegate_evidence.py`, `ouroboros/subagents.py`, `supervisor/subagent_task_truth.py`, `web/modules/log_events.js` | `tests/test_execution_evidence.py`, `tests/test_terminal_delegation_receipt.py`, `web/tests/review_truth.test.js`, `tests/test_task_status_flow.py` | | `TaskCostBreakdown` (root-only, read-time, never persisted; `accounted_upper_bound_usd`; `authority="physical_attempt_ledger"`) + `cancel_state: "pending"` with `cancel_reason` beside it; the browser's one consumer is `log_events.js` `taskCancelPending` | `ouroboros/gateway/contracts.py` | `tests/test_gateway_parity.py`, `web/tests/cancel_run.test.js` | | Task hurry ABI — `POST /api/tasks/{task_id}/hurry` with exactly `{request_id}` (extra fields refused), `duplicate` = idempotent success; `OwnerHurryProjection` attempt-keyed states; consumers `log_events.js` `taskSoftStopPending`/`ownerHurryProjection` (task-card only) | `ouroboros/gateway/task_hurry.py`, `ouroboros/gateway/contracts.py` | `tests/test_owner_hurry_s3.py`, `tests/test_owner_stop_s3.py`, `web/tests/task_control_menu.test.js` | -| `StateResponse.active_direct_turns`/`active_chat_activities` (phases queued/working/finalizing/budget_paused, plus the additive `budget_pausing` for a RUNNING root whose durable `budget_pause` row says it is writing its exact pause — one predicate `budget_pause_fact` decides budget pause; a consciousness wake-up is one of those `direct_chat` turns and introduces no third `kind`); `TypingOutbound` activity fields; `ChatOutbound.task_phase`/`task_terminal_status` | `ouroboros/gateway/contracts.py`, `supervisor/active_activity.py`, `web/modules/chat_activity.js` | `tests/test_gateway_parity.py` + the activity test files | +| `StateResponse.active_direct_turns`/`active_chat_activities` (phases queued/working/finalizing/budget_paused, plus the additive `budget_pausing` for a RUNNING root whose durable `budget_pause` row says it is writing its exact pause — one predicate `budget_pause_fact` decides budget pause; a consciousness wake-up is one of those `direct_chat` turns and introduces no third `kind`; a direct turn parked under its exact budget pause and later resumed on a pooled worker reports the managed phases under `kind="direct_chat"`, additive #1196); `TypingOutbound` activity fields; `ChatOutbound.task_phase`/`task_terminal_status` | `ouroboros/gateway/contracts.py`, `supervisor/active_activity.py`, `web/modules/chat_activity.js` | `tests/test_gateway_parity.py` + the activity test files | | `ChatOutbound.card_row`/`card_row_id` — additive host-stamped placement fact on a task-keyed System row (`"timeline"` = a timeline item of the task's card; `"reviews"` = the card's Reviews group carries the fact and the row is still attached to the card; absent = an ordinary row) plus the row's stable identity across live delivery, outbox replay and history; stamped by the producer's `progress_meta`, persisted by `log_chat`, replayed by history | `ouroboros/gateway/contracts.py`, `supervisor/message_bus.py`, `ouroboros/gateway/history.py`, `supervisor/events_chat_delivery.py`, `ouroboros/acceptance_settlement.py`, `web/modules/api_types.js` | `tests/test_gateway_parity.py`, `tests/test_card_row_plumbing.py`, `tests/test_terminal_host_notice.py`, `web/tests/chat_card_row_placement.test.js` | | `ChatOutbound.completion_answer` — additive: a Project root's model-authored final answer, whole, on Main's `project_completion_summary` row (absent for any other ending or row); stamped by the producer's `progress_meta`, persisted by `log_chat`, replayed by history unnormalized | `ouroboros/gateway/contracts.py`, `ouroboros/projects_registry.py`, `ouroboros/project_dialogue.py`, `supervisor/message_bus.py`, `ouroboros/gateway/history.py`, `web/modules/api_types.js` | `tests/test_gateway_parity.py`, `tests/test_project_completion_mirror.py`, `web/tests/chat_plain_system_rows.test.js` | | `ChatOutbound.initiator` — additive origin label of a self-initiated turn (`"consciousness"` on every frame and chat/progress/summary row of a consciousness wake-up and of the roots it starts; absent on an owner's turn); stamped by the turn's own event queue and the agent's frame meta, persisted by `log_chat`/the authored summary row/the task result, replayed by history on each row | `ouroboros/gateway/contracts.py`, `supervisor/log_addressing.py`, `ouroboros/subagent_messages.py`, `supervisor/message_bus.py`, `ouroboros/gateway/history.py`, `web/modules/api_types.js` | `tests/test_consciousness_initiator_label.py`, `tests/test_consciousness_wake_lane.py`, `web/tests/consciousness_label.test.js` | diff --git a/ouroboros/agent.py b/ouroboros/agent.py index d8934a53c..267b6f6f3 100644 --- a/ouroboros/agent.py +++ b/ouroboros/agent.py @@ -669,8 +669,14 @@ class OuroborosAgent: from ouroboros.budget_pause import load_budget_pause saved_wait = load_budget_pause(ctx) # same-ID budget continuation (#1196) if saved_wait and ctx.model_wait_context is not None: + # started_at stays the ORIGINAL start; the granted paused interval is + # the separate carrier the finite lifetime subtracts (#1196). A budget + # grant supplies the CURRENT cumulative value; an owner-wait restart + # of a previously paused task has none to supply, and the serializer's + # own saved carrier is used instead (``restore_continuation``, F5). ctx.model_wait_context.restore_continuation( - saved_wait.get("model_wait") or {}, started_at=ctx.task_started_at) + saved_wait.get("model_wait") or {}, started_at=ctx.task_started_at, + budget_paused_sec=(ctx.budget_pause_resume or {}).get("paused_duration_sec")) if self._event_queue is not None: # Optional runtime seam consumed by loop.py. Unit/direct contexts diff --git a/ouroboros/budget_pause.py b/ouroboros/budget_pause.py index cded3482c..a900a6362 100644 --- a/ouroboros/budget_pause.py +++ b/ouroboros/budget_pause.py @@ -50,11 +50,28 @@ completed and otherwise-stopped members are never revived; a root's Resume only makes its own budget-paused descendants ELIGIBLE — the model selects each one explicitly through the same task control (Q9). -Direct chat actors are NOT silently excluded: they keep the existing terminal -rail, and the recorded ``resource_limit`` names why (``exact_pause_unavailable``). -This module adds no scheduler, ledger or recovery framework: it reuses the -queue's ``_budget_pause`` carrier, the actor source store, the task-result -authority and the delegated custody rows that already exist. +A direct owner-chat turn pauses the SAME way and under the SAME task id: the +live direct actor ends (its registry entry is released as on any other ending, +so no second live actor ever exists for that id), its ``budget_pause`` event +carries the turn's own task record with its ``_is_direct_chat`` lane fact, and +the supervisor parks that record in the existing PENDING carrier under the +same ``_budget_pause`` marker. The existing Resume endpoint grants it and a +pooled worker continues the checkpointed cognition cold — origin, workspace, +attempt, accounting and the opaque acceptance identities travel in the +checkpoint; browser and service handles are never resurrected. Only a turn +with no task id or no durable root is excluded, and loudly +(``resource_limit.exact_pause_unavailable``). + +Every refusal to restore, grant or dispatch is typed and RETAINS the pause: a +corrupt or missing source, an unwritable revocation, a lifted root fence over +a zero-dispatch sibling all become a durable non-dispatch HOLD on the queue +row (``events_budget.budget_hold_fact``) beside the saved pause, never a +cancellation and never a silent dispatch; a grant is bound to ONE pause id +and ONE resume generation, so ``pauseA -> Resume -> pauseB -> restart`` can +never dispatch on the earlier grant. This module adds no scheduler, ledger or +recovery framework: it reuses the queue's ``_budget_pause`` carrier, the actor +source store, the task-result authority and the delegated custody rows that +already exist. """ from __future__ import annotations @@ -146,7 +163,15 @@ def dispatch_fenced(task_id: str) -> bool: # seen on the timeout path is invisible in exactly the window that matters. # ``future.done()`` is not the fact recorded here: a late settlement callback # may still be producing effects after it, so a row settles only when every -# callback that claimed it has FINISHED. +# callback that claimed it has FINISHED. Settlement OWNERSHIP is pinned from +# the registration until the registering caller releases it (its own result +# arrived, or it claimed the late-callback hold): a future that completes +# between the caller's timeout and its ``hold_tool_settlement`` claim would +# otherwise read as settled-and-unheld for one instant, and a concurrent +# registration could prune it — the late callback's effects then escape the +# quiescence gate. Pruning is per SCOPE and only over rows nobody owns or +# holds; rows of another attempt are never touched (``forget_tool_scope`` is +# the only cross-scope drop, at that attempt's own end). _TOOL_LOCK = threading.Lock() _TOOL_FUTURES: Dict[str, Dict[str, Dict[str, Any]]] = {} @@ -156,27 +181,45 @@ def tool_scope_key(ctx: Any) -> str: return f"{str(getattr(ctx, 'task_id', '') or '')}|{int(getattr(ctx, 'task_attempt', 1) or 1)}" -def register_tool_future(ctx: Any, operation_id: str, tool: str, future: Any) -> None: - """Track one tool future until its settlement callbacks have finished.""" +def _row_settled_locked(row: Dict[str, Any]) -> bool: + """Settled = the future finished, no late callback holds it, and the + registering owner has released it. Read with ``_TOOL_LOCK`` held.""" + return bool(row.get("done")) and int(row.get("holds") or 0) <= 0 and not row.get("owner_open") + + +def register_tool_future(ctx: Any, operation_id: str, tool: str, future: Any) -> Callable[[], None]: + """Track one tool future until its settlement callbacks have finished. + + Returns the OWNER release: the registering caller calls it once its own + handling of the call is over — after the result arrived, or after the + timeout path claimed the late-settlement hold — so the row cannot be + pruned in between. Calling it twice is harmless. + """ op = str(operation_id or "") if future is None or not op or not str(getattr(ctx, "task_id", "") or ""): - return + return lambda: None row: Dict[str, Any] = {"operation_id": op, "tool": str(tool or ""), "settled": threading.Event(), "holds": 0, - "done": False, "observable": True} + "done": False, "observable": True, "owner_open": True} scope = tool_scope_key(ctx) with _TOOL_LOCK: - for key, rows in list(_TOOL_FUTURES.items()): - for op_id in [i for i, r in rows.items() if r["settled"].is_set() and not r["holds"]]: - rows.pop(op_id, None) - if not rows and key != scope: - _TOOL_FUTURES.pop(key, None) - _TOOL_FUTURES.setdefault(scope, {})[op] = row + rows = _TOOL_FUTURES.setdefault(scope, {}) + # Prune ONLY this scope's fully settled rows: done, unheld AND released. + for op_id in [i for i, r in rows.items() if _row_settled_locked(r)]: + rows.pop(op_id, None) + rows[op] = row def _done(_future: Any) -> None: with _TOOL_LOCK: row["done"] = True - settled = row["holds"] <= 0 + settled = _row_settled_locked(row) + if settled: + row["settled"].set() + + def _owner_release() -> None: + with _TOOL_LOCK: + row["owner_open"] = False + settled = _row_settled_locked(row) if settled: row["settled"].set() @@ -188,6 +231,7 @@ def register_tool_future(ctx: Any, operation_id: str, tool: str, future: Any) -> log.warning("Tool future %s cannot be observed for budget-pause quiescence", op, exc_info=True) with _TOOL_LOCK: row["observable"] = False + return _owner_release def hold_tool_settlement(ctx: Any, operation_id: str) -> Callable[[], None]: @@ -195,8 +239,11 @@ def hold_tool_settlement(ctx: Any, operation_id: str) -> Callable[[], None]: The caller attaches its callback AFTER claiming and calls the release as the callback's last act, so the registry reports the tool as settled only - once that callback's own effects are over. An unregistered row returns a - no-op: this registry never invents an observation it does not hold. + once that callback's own effects are over. The claim is taken while the + registering owner still pins the row (it releases only after this claim), + so a future that finished a moment earlier is still here to be claimed. + An unregistered row returns a no-op: this registry never invents an + observation it does not hold. """ op = str(operation_id or "") with _TOOL_LOCK: @@ -204,12 +251,12 @@ def hold_tool_settlement(ctx: Any, operation_id: str) -> Callable[[], None]: if row is None: return lambda: None row["holds"] += 1 - row["settled"].clear() + row["settled"].clear() def _release() -> None: with _TOOL_LOCK: row["holds"] = max(0, int(row["holds"]) - 1) - settled = row["holds"] <= 0 and bool(row["done"]) + settled = _row_settled_locked(row) if settled: row["settled"].set() @@ -217,7 +264,7 @@ def hold_tool_settlement(ctx: Any, operation_id: str) -> Callable[[], None]: def forget_tool_scope(ctx: Any) -> None: - """Drop one attempt's rows (task end, and test hygiene for this global).""" + """Drop one attempt's rows (that attempt's end, and test hygiene for this global).""" with _TOOL_LOCK: _TOOL_FUTURES.pop(tool_scope_key(ctx), None) @@ -228,7 +275,8 @@ def drain_local_tool_futures(ctx: Any, *, timeout_sec: float) -> Dict[str, Any]: A call abandoned at its logical timeout keeps running, and its late settlement callback keeps producing effects after the worker returns. Releasing the native process while either is outstanding is exactly the - escape this gate exists to refuse. + escape this gate exists to refuse. A row its owner has not released yet is + unsettled too: the owner is still deciding whether a late callback claims it. """ with _TOOL_LOCK: rows = list(_TOOL_FUTURES.get(tool_scope_key(ctx), {}).values()) @@ -250,13 +298,15 @@ def drain_local_tool_futures(ctx: Any, *, timeout_sec: float) -> Dict[str, Any]: def pause_ineligibility(ctx: Any) -> str: """Empty when an exact pause is possible; otherwise the typed reason. - The reason is recorded on the terminal ``resource_limit`` so a direct actor - or a context without a queue continuation owner is excluded LOUDLY. + The reason is recorded on the terminal ``resource_limit`` so a context + without a continuation owner, a task id or a durable root is excluded + LOUDLY. A direct owner-chat actor is ELIGIBLE: its continuation owner is + the queue carrier the supervisor parks its task record in (the direct lane + binds ``direct_owner_wait`` as its wait callback, the same seam a pooled + worker binds ``worker_owner_wait`` to). """ - if bool(getattr(ctx, "is_direct_chat", False)): - return "direct_actor" if not callable(getattr(ctx, "owner_wait_callback", None)): - return "no_queue_continuation_owner" + return "no_continuation_owner" if not str(getattr(ctx, "task_id", "") or ""): return "no_task_id" if not (getattr(ctx, "budget_drive_root", None) or getattr(ctx, "drive_root", None)): @@ -305,32 +355,35 @@ def resume_point(messages: List[Dict[str, Any]], round_idx: int) -> Dict[str, An # --- external custody (owner Q8) --------------------------------------------------- +def _task_run_rows(root: Any, task_id: str) -> tuple[List[Any], str]: + """``(unsettled runs held by task_id on this custody root, read_error)``: an + unreadable custody store is a typed failure, never an empty (clean-looking) list.""" + try: + from ouroboros import delegate_custody as custody + + mine = str(task_id or "") + return [run for run in custody.replay(pathlib.Path(root)).values() + if str(getattr(run, "task_id", "") or "") == mine and not getattr(run, "settled", True)], "" + except Exception as exc: + log.warning("External custody rows unreadable for %s", task_id, exc_info=True) + return [], f"{type(exc).__name__}: {str(exc)[:200]}" + + def _external_run_rows(ctx: Any) -> tuple[List[Any], str]: - """``(unsettled runs held by this task, read_error)``: an unreadable custody - store is a typed failure, never an empty (clean-looking) list.""" + """The loop-side reader: this context's custody root and task id.""" try: from ouroboros import delegate_custody as custody root = custody.custody_root(ctx) - mine = str(ctx.task_id) - return [run for run in custody.replay(root).values() - if str(getattr(run, "task_id", "") or "") == mine and not getattr(run, "settled", True)], "" except Exception as exc: - log.warning("External custody rows unreadable at budget pause", exc_info=True) + log.warning("External custody root unresolvable at budget pause", exc_info=True) return [], f"{type(exc).__name__}: {str(exc)[:200]}" + return _task_run_rows(root, str(ctx.task_id)) -def observe_external_runs(ctx: Any, *, request_stop: bool = True) -> Dict[str, Any]: - """Observe every unsettled delegated run this task holds; request stops. - - Pre-terminal subscription cost coverage cannot be proved from the ledger - (the reservation of one call never covers a whole session), so the owner's - Q8 branch for UNPROVEN coverage applies to every open run: a stop is - requested through the verified cancel seam and its typed outcome recorded. - ``requested`` and ``unknown`` are NOT death: the run stays under this - task's custody and no second writer may be started over it. - """ - runs, read_error = _external_run_rows(ctx) +def _observe_runs(root: Any, task_id: str, runs: List[Any], read_error: str, *, + request_stop: bool, reason: str) -> Dict[str, Any]: + """The ONE observer body behind the loop-side pause and the supervisor-side grant.""" if read_error: # Held as UNKNOWN on the pause row: the grant re-reads custody and # refuses while it stays unreadable (never "no runs"). @@ -352,11 +405,10 @@ def observe_external_runs(ctx: Any, *, request_stop: bool = True) -> Dict[str, A try: from ouroboros import delegate_custody as custody - root = custody.custody_root(ctx) for run in runs: row = { "run_id": str(getattr(run, "run_id", "") or ""), - "route": str(getattr(run, "route", "") or ""), + "route": str(getattr(run, "route", "") or getattr(run, "route_id", "") or ""), "cost_coverage": "unproven_preterminal", "stop_policy": "request_stop", "state": EXTERNAL_RUNNING, @@ -367,7 +419,7 @@ def observe_external_runs(ctx: Any, *, request_stop: bool = True) -> Dict[str, A row.update(state=EXTERNAL_STOP_UNKNOWN, stop_outcome="not_issued_gateway_unavailable") else: try: - result = custody.cancel_and_verify(root, gateway, run, "budget_pause_uncovered_cost") + result = custody.cancel_and_verify(pathlib.Path(root), gateway, run, reason) outcome = str(result.get("outcome") or "") row.update(stop_outcome=outcome, detail=str(result.get("detail") or "")) if outcome == custody.CANCEL_CONFIRMED: @@ -390,6 +442,48 @@ def observe_external_runs(ctx: Any, *, request_stop: bool = True) -> Dict[str, A "coverage_basis": "preterminal_subscription_coverage_unprovable"} +def observe_external_runs(ctx: Any, *, request_stop: bool = True) -> Dict[str, Any]: + """Observe every unsettled delegated run this task holds; request stops. + + Pre-terminal subscription cost coverage cannot be proved from the ledger + (the reservation of one call never covers a whole session), so the owner's + Q8 branch for UNPROVEN coverage applies to every open run: a stop is + requested through the verified cancel seam and its typed outcome recorded. + ``requested`` and ``unknown`` are NOT death: the run stays under this + task's custody and no second writer may be started over it. + """ + runs, read_error = _external_run_rows(ctx) + root = "" + if not read_error: + from ouroboros import delegate_custody as custody + + root = custody.custody_root(ctx) + return _observe_runs(root, str(ctx.task_id), runs, read_error, request_stop=request_stop, + reason="budget_pause_uncovered_cost") + + +def observe_task_runs(root: Any, task_id: str, *, request_stop: bool = True, + reason: str = "budget_resume_uncovered_cost") -> Dict[str, Any]: + """The supervisor-side twin of ``observe_external_runs``: a FRESH custody read + for ``task_id`` on ``root`` at grant time (never the pause row's saved + summary), requesting a stop for every run still open — its remaining cost + is uncovered/unknown while it runs — and recording the typed outcome.""" + runs, read_error = _task_run_rows(root, task_id) + return _observe_runs(root, str(task_id), runs, read_error, request_stop=request_stop, reason=reason) + + +def unsettled_external_runs(observation: Dict[str, Any]) -> List[Dict[str, Any]]: + """The runs of one observation that are NOT proven terminal. + + ``stop_confirmed`` is the only terminal fact (a verified receipt, a + settlement, or the daemon answering absent); running, requested and + unknown all leave the run under custody, where a second writer may not be + started over it. + """ + return [run for run in (observation.get("runs") or []) + if isinstance(run, dict) and str(run.get("state") or "") != EXTERNAL_STOP_CONFIRMED] + + def drain_local_review_attempts(task_id: str, *, timeout_sec: float) -> Dict[str, Any]: """Bounded quiescence barrier over THIS task's in-flight review attempts. @@ -457,14 +551,29 @@ def local_producer_observation(ctx: Any, *, timeout_sec: float) -> Dict[str, Any # --- durable row -------------------------------------------------------------------- def set_budget_pause(root: Any, task_id: str, row: Dict[str, Any], - expected_pause_id: Optional[str] = None) -> Dict[str, Any]: - """Update only the ``budget_pause`` projection of the task result row.""" + expected_pause_id: Optional[str] = None, *, + expected_state: Any = None, + expected_grant_id: Optional[str] = None) -> Dict[str, Any]: + """Update only the ``budget_pause`` projection of the task result row. + + Compare-and-set: ``expected_pause_id``, ``expected_state`` (one state or a + collection of them) and ``expected_grant_id`` are each checked under the + file lock against the row as it is NOW, so a grant, a revocation or a + consumption written from a stale reading refuses (``ValueError``) instead + of overwriting a newer pause, grant or state — pauseA -> Resume -> pauseB + -> late revoke of A must never land on B (#1196). + """ from ouroboros.task_results import ( _TRULY_TERMINAL_STATUSES, require_writable_task_result_schema, stamp_task_result_schema, task_result_path, ) from ouroboros.utils import update_json_locked + expected_states = ( + None if expected_state is None + else ({expected_state} if isinstance(expected_state, str) else set(expected_state)) + ) + def update(current: dict) -> dict: require_writable_task_result_schema(current) if current.get("status") in _TRULY_TERMINAL_STATUSES: @@ -472,6 +581,10 @@ def set_budget_pause(root: Any, task_id: str, row: Dict[str, Any], old = current.get("budget_pause") or {} if expected_pause_id is not None and old.get("pause_id") != expected_pause_id: raise ValueError("budget pause identity changed") + if expected_states is not None and str(old.get("state") or "") not in expected_states: + raise ValueError(f"budget pause state changed: {old.get('state')!r} is not {sorted(expected_states)}") + if expected_grant_id is not None and str((old.get("grant") or {}).get("grant_id") or "") != str(expected_grant_id): + raise ValueError("budget pause grant changed") return stamp_task_result_schema({**current, "budget_pause": dict(row)}) update_json_locked(task_result_path(root, task_id), update, strict_existing_dict=True) @@ -487,7 +600,7 @@ def budget_pause_row(root: Any, task_id: str) -> Dict[str, Any]: def has_budget_pause_checkpoint(root: Any, task_id: str, task_attempt: int) -> bool: - """A live pause record for THIS attempt: automatic crash retry must not replay it. + """Pause/consumption evidence for THIS attempt: crash retry must not replay it. Fail-closed: an UNREADABLE record cannot authorize an ordinary retry, and a ``pausing`` row without a source yet (death during a hold) still fences the @@ -497,7 +610,8 @@ def has_budget_pause_checkpoint(root: Any, task_id: str, task_attempt: int) -> b pause = budget_pause_row(root, task_id) except Exception: return True - return bool(pause and pause.get("state") in LIVE_PAUSE_STATES + return bool(pause and (pause.get("state") in LIVE_PAUSE_STATES + or pause.get("state") == STATE_RESUMED or (pause.get("grant") or {}).get("consumed_at")) and int(pause.get("task_attempt") or 0) == int(task_attempt)) @@ -555,7 +669,7 @@ def _hold_control_reason(ctx: Any) -> str: except Exception: log.debug("Model-wait control reader unavailable during a budget pause hold", exc_info=True) try: - root = pathlib.Path(ctx.drive_root) + root = pathlib.Path(getattr(ctx, "budget_drive_root", None) or ctx.drive_root) if any((root / "state" / name).exists() for name in ("panic_stop.flag", "owner_restart_no_resume.flag")): return "panic" @@ -630,10 +744,12 @@ def _exact_continuation_row(limit_ctx: Any, ctx: Any, *, pause_id: str, rail: st log.debug("Physical call count unavailable at budget pause", exc_info=True) return { "pause_id": pause_id, "state": STATE_PAUSING, "reason": "budget", + "pause_generation": int(getattr(ctx, "_budget_pause_generation", 0) or 0), "rail": rail, "scope": str(scope or "global"), "root_task_id": str(root_task_id or getattr(ctx, "root_task_id", "") or ""), "reason_text": str(reason_text or ""), "task_attempt": int(ctx.task_attempt or 1), + "is_direct_chat": bool(getattr(ctx, "is_direct_chat", False)), "source_ref": source, "execution_drive_root": str(ctx.drive_root), "started_at": getattr(ctx, "task_started_at", None), @@ -677,9 +793,16 @@ def request_pause(limit_ctx: Any, *, rail: str, scope: str, reason_text: str, setattr(ctx, "_budget_pausing", True) pause_id = uuid.uuid4().hex root = pathlib.Path(ctx.budget_drive_root or ctx.drive_root) + _ensure_pausable_result_row(root, ctx) + # One monotonic generation per pause of this task (pauseA=1, pauseB=2, ...): + # every grant, handoff and revocation names the generation beside the pause + # id, so a carrier from an earlier pause can never read as the current one. + generation = _next_pause_generation(root, task_id) + setattr(ctx, "_budget_pause_generation", generation) # Durable "pausing" FIRST (before any wait), so a death while the task is # still settling meets the crash-retry fence instead of a replay. seed = {"pause_id": pause_id, "state": STATE_PAUSING, "reason": "budget", "rail": rail, + "pause_generation": generation, "scope": str(scope or "global"), "task_attempt": int(ctx.task_attempt or 1), "source_ref": None, "started_at": getattr(ctx, "task_started_at", None), "pausing_since": time.time(), "exact_continuation": True, @@ -747,6 +870,47 @@ def request_pause(limit_ctx: Any, *, rail: str, scope: str, reason_text: str, STATE_ABANDONED = "abandoned" +def _next_pause_generation(root: pathlib.Path, task_id: str) -> int: + """The generation of the pause about to open: one past the row's last one. + + An unreadable row yields generation 1 with no claim about the past — the + seed write that follows meets the same fault and HOLDS, so nothing is + numbered over a record nobody could read. + """ + try: + prior = budget_pause_row(root, task_id) + except Exception: + return 1 + return int(prior.get("pause_generation") or 0) + 1 + + +def _ensure_pausable_result_row(root: pathlib.Path, ctx: Any) -> None: + """A direct owner-chat turn has no admission-written RUNNING row (the pool + mirrors one at dispatch; the direct lane writes a stub only for a turn with + an origin ref). The pause row is a projection ON the task result, so an + absent row is written as a plain RUNNING row first — merge-write, never a + replacement of an existing one. A failed write is not swallowed into a + fake pause: the seed write that follows meets the same fault and HOLDS.""" + from ouroboros.task_results import STATUS_RUNNING, load_task_result, write_task_result + + task_id = str(ctx.task_id) + try: + if load_task_result(root, task_id, strict=False): + return + except Exception: + return # an unreadable row is the seed write's typed hold, not ours to guess + try: + chat_id = getattr(ctx, "current_chat_id", None) + write_task_result( + root, task_id, STATUS_RUNNING, + **({"chat_id": int(chat_id)} if chat_id not in (None, "") else {}), + _is_direct_chat=bool(getattr(ctx, "is_direct_chat", False)), + result="Task is pausing on its budget rail; its exact continuation is being stored.", + ) + except Exception: + log.debug("Pausable result row for %s could not be written ahead of the seed", task_id, exc_info=True) + + def _abandon_pausing_row(root: Any, task_id: str, pause_id: str, reason: str, detail: Dict[str, Any]) -> None: """Close an opened ``pausing`` row that will not become a pause (typed, never silent). @@ -764,11 +928,38 @@ def _abandon_pausing_row(root: Any, task_id: str, pause_id: str, reason: str, log.warning("Abandoned pausing row for %s could not be closed", task_id, exc_info=True) +# Fields of a direct turn's task record that never travel in its pause event: +# the inline image bytes were consumed into the checkpointed transcript, and a +# multi-megabyte base64 blob has no business in the supervisor's event queue or +# the queue snapshot (which whitelists its own fields anyway). +_DIRECT_TASK_EVENT_OMITTED = frozenset({"image_base64"}) + + +def parkable_direct_task(task: Dict[str, Any]) -> Dict[str, Any]: + """The direct turn's own task record, as the queue may park it (#1196). + + A direct turn is never in RUNNING, so the event is the only carrier of the + record the SAME task id continues under: origin, chat, contract, metadata, + workspace, attachments and the lane fact all ride it unchanged. + """ + row = {key: value for key, value in task.items() if key not in _DIRECT_TASK_EVENT_OMITTED} + row["_is_direct_chat"] = True + row.setdefault("_attempt", int(task.get("_attempt") or 1)) + row.setdefault("depth", 0) + return row + + def pause_event(task: Dict[str, Any], pause: Dict[str, Any]) -> Dict[str, Any]: - """The worker->supervisor ``budget_pause`` event for an exact continuation.""" + """The worker->supervisor ``budget_pause`` event for an exact continuation. + + A direct owner-chat turn's event additionally carries the turn's own task + record (``task``) and the lane fact, because that turn was never in the + queue's RUNNING table: the supervisor parks THAT record. + """ from ouroboros.utils import utc_now_iso task_id = str(task.get("id") or "") + direct = bool(task.get("_is_direct_chat")) return { "type": "budget_pause", "task_id": task_id, @@ -777,6 +968,7 @@ def pause_event(task: Dict[str, Any], pause: Dict[str, Any]) -> Dict[str, Any]: "chat_id": task.get("chat_id"), "root_task_id": str(pause.get("root_task_id") or task.get("root_task_id") or task_id), "resource_limit": exact_pause_marker(pause, default_root=str(task.get("root_task_id") or task_id)), + **({"_is_direct_chat": True, "task": parkable_direct_task(task)} if direct else {}), "ts": utc_now_iso(), } @@ -803,7 +995,7 @@ def exact_pause_marker(row: Dict[str, Any], *, default_root: str = "") -> Dict[s "paused_at": row.get("paused_at"), "checkpoint": { key: row.get(key) for key in ( - "pause_id", "task_attempt", "source_ref", "execution_drive_root", + "pause_id", "pause_generation", "task_attempt", "source_ref", "execution_drive_root", "started_at", "paused_at", "paused_duration_sec", "resume_point", "model_wait_quota_clock", "external_runs", "rail", "scope", "reason_text", ) @@ -832,9 +1024,11 @@ def load_budget_pause(ctx: Any, handoff: Optional[Dict[str, Any]] = None) -> Dic if (row.get("status") in _TRULY_TERMINAL_STATUSES or current.get("state") != STATE_RESUME_GRANTED or current.get("pause_id") != handoff.get("pause_id") + or int(current.get("pause_generation") or 0) != int(handoff.get("pause_generation") or 0) or not grant.get("grant_id") or grant.get("grant_id") != handoff.get("grant_id") - or grant.get("consumed_at")): + or int(grant.get("generation") or 0) != int(handoff.get("grant_generation") or grant.get("generation") or 0) + or grant.get("consumed_at") or grant.get("revoked_at")): raise ValueError("budget pause continuation has no live single-use grant") state = json.loads(read_actor_source_bytes(root, ctx.task_id, current["source_ref"])) if (state.get("task_id") != ctx.task_id @@ -844,47 +1038,96 @@ def load_budget_pause(ctx: Any, handoff: Optional[Dict[str, Any]] = None) -> Dic return {**state, "_pause_row": dict(current)} -def _refresh_planning_threshold(ctx: Any, budget_remaining_usd: Optional[float]) -> Dict[str, Any]: +# A root-tree snapshot older than this at refresh time is a STALE fallback the +# accounting refresher returned because the ledger could not answer now. +_FRESH_TREE_MAX_AGE_SEC = 5.0 + + +def _refresh_planning_threshold(ctx: Any, budget_remaining_usd: Optional[float], + usage: Optional[Dict[str, Any]] = None) -> Dict[str, Any]: """Owner Q10: after an explicit Resume of a GRACEFUL stop, the planning threshold moves forward within the money still authorized, so the task is not paused again on the very number that paused it. The hard tree cap and the global ledger fence are untouched; the planning margin is what the - owner's explicit act spends. Returns the disclosure row.""" + owner's explicit act spends. Returns the disclosure row. + + Every number is read from the AUTHORITATIVE ledger NOW: the global wallet + from the usage projection, the tree's cumulative spend and its ACTUAL + root cap from a fresh root-accounting read (an owner may have raised the + cap while the task was paused), the task's own cumulative cost from the + restored usage. A wallet the ledger cannot answer, a degraded or stale + tree read, or unknown spend REFUSES the refresh (the dispatch-time + ``budget_remaining_usd`` is disclosed, never used as room): unknown money + is not room. + """ from ouroboros import task_pacing - from ouroboros.loop_budget import _loop_tree_accounting + from ouroboros.loop_budget import _loop_tree_accounting, _wrapup_global_remaining old = getattr(ctx, "_cost_ceiling", None) if not isinstance(old, task_pacing.CostCeiling): return {"refreshed": False, "reason": "no_ceiling"} + try: + fresh = _wrapup_global_remaining() + except Exception: + log.debug("Ledger wallet read failed at resume refresh", exc_info=True) + fresh = None + if fresh is None: + return {"refreshed": False, "reason": "wallet_unavailable", "wallet_basis": "ledger_unavailable", + "dispatch_time_remaining_usd": budget_remaining_usd} tree = _loop_tree_accounting(refresh=True, max_age_sec=0.0) - usage = getattr(ctx, "_accumulated_usage", None) or {} + tree = tree if isinstance(tree, dict) else None + tree_cap = tree.get("root_limit_usd") if tree else None + tree_capped = old.root_cap_usd is not None or tree_cap is not None + root_cap: Optional[float] = None + root_cap_basis = "none" + if tree_capped: + if tree is None: + return {"refreshed": False, "reason": "tree_spend_unavailable", "wallet_basis": "ledger_projection"} + if tree.get("integrity_degraded"): + return {"refreshed": False, "reason": "tree_accounting_degraded", "wallet_basis": "ledger_projection"} + if float(tree.get("age_sec") or 0.0) > _FRESH_TREE_MAX_AGE_SEC: + return {"refreshed": False, "reason": "tree_accounting_stale", + "tree_age_sec": round(float(tree.get("age_sec") or 0.0), 1), "wallet_basis": "ledger_projection"} + if tree.get("accounted_usd") is None: + return {"refreshed": False, "reason": "tree_spend_unknown", "wallet_basis": "ledger_projection"} + if tree_cap is not None: + root_cap, root_cap_basis = float(tree_cap), "root_accounting" + else: + root_cap, root_cap_basis = float(old.root_cap_usd), "start_of_task" + usage = usage if isinstance(usage, dict) else (getattr(ctx, "_accumulated_usage", None) or {}) task_cost = usage.get("cost") deciding, basis = task_pacing.resolve_deciding_spend( - tree_cost_usd=tree.get("accounted_usd") if isinstance(tree, dict) else None, + tree_cost_usd=tree.get("accounted_usd") if tree else None, task_cost_usd=float(task_cost) if task_cost is not None else None, - root_cap_usd=old.root_cap_usd, + root_cap_usd=root_cap, ) - spent = float(deciding or 0.0) + if deciding is None: + return {"refreshed": False, "reason": "spend_unknown", "basis": basis, "wallet_basis": "ledger_projection"} + spent = float(deciding) components: List[float] = [] - if old.root_cap_usd is not None: - components.append(float(old.root_cap_usd) - spent) - if budget_remaining_usd is not None and float(budget_remaining_usd) > 0: + if root_cap is not None: + components.append(root_cap - spent) + if float(fresh) > 0: profile = task_pacing.resolve_budget_profile(ctx) pct = profile.get("cost_hard_stop_pct") pct = task_pacing._DEFAULT_COST_HARD_STOP_PCT if pct is None else max(0, min(100, int(pct))) if pct > 0: - components.append(float(budget_remaining_usd) * pct / 100.0) + components.append(float(fresh) * pct / 100.0) room = min(components) if components else None if room is None or room <= 0: - return {"refreshed": False, "reason": "no_authorized_room", "spent_usd": spent, "basis": basis} + return {"refreshed": False, "reason": "no_authorized_room", "spent_usd": spent, "basis": basis, + "wallet_basis": "ledger_projection", "global_remaining_usd": float(fresh), + "root_cap_usd": root_cap, "root_cap_basis": root_cap_basis} refreshed = task_pacing.CostCeiling( state=task_pacing.COST_CEILING_ACTIVE, ceiling_usd=spent + room, - root_cap_usd=old.root_cap_usd, planning_margin_usd=old.planning_margin_usd, + root_cap_usd=root_cap, planning_margin_usd=old.planning_margin_usd, basis=f"owner_resume_refresh({old.basis or 'previous'})", ) ctx._cost_ceiling = refreshed return {"refreshed": True, "previous_ceiling_usd": old.ceiling_usd, - "ceiling_usd": refreshed.ceiling_usd, "spent_usd": spent, "basis": basis} + "ceiling_usd": refreshed.ceiling_usd, "spent_usd": spent, "basis": basis, + "wallet_basis": "ledger_projection", "global_remaining_usd": float(fresh), + "root_cap_usd": root_cap, "root_cap_basis": root_cap_basis} def resume_paused_loop(tools: Any, state: Dict[str, Any], messages: list, trace: dict, @@ -904,11 +1147,17 @@ def resume_paused_loop(tools: Any, state: Dict[str, Any], messages: list, trace: root = pathlib.Path(ctx.budget_drive_root or ctx.drive_root) grant = dict(row.get("grant") or {}) grant["consumed_at"] = time.time() + # Compare-and-set on the exact pause, state and grant this worker was + # handed: a grant revoked or superseded between the dispatch and this + # write refuses here instead of consuming a grant that is no longer live. set_budget_pause(root, ctx.task_id, {**row, "state": STATE_RESUMED, "grant": grant, "resumed_at": grant["consumed_at"]}, - expected_pause_id=str(row.get("pause_id") or "")) + expected_pause_id=str(row.get("pause_id") or ""), + expected_state=STATE_RESUME_GRANTED, + expected_grant_id=str(grant.get("grant_id") or "")) ctx._budget_paused_sec = float(grant.get("paused_duration_sec") or row.get("paused_duration_sec") or 0.0) ctx.budget_pause_resume = None + setattr(ctx, "_budget_pause_generation", int(row.get("pause_generation") or 0)) end_dispatch_fence(str(ctx.task_id)) # The pause is over: its quiescence rows are spent observations of a worker # that no longer exists, and the resumed attempt registers its own. @@ -919,12 +1168,33 @@ def resume_paused_loop(tools: Any, state: Dict[str, Any], messages: list, trace: ctx._cost_ceiling = CostCeiling(**state["cost_ceiling"]) refresh: Dict[str, Any] = {"refreshed": False, "reason": "hard_rail"} - if str(row.get("rail") or "") in GRACEFUL_RAILS: - refresh = _refresh_planning_threshold(ctx, budget_remaining_usd) + graceful = str(row.get("rail") or "") in GRACEFUL_RAILS + if graceful: + refresh = _refresh_planning_threshold(ctx, budget_remaining_usd, usage) + # Owner Q10, the other half: the last-fit rail stops one reservation EARLY + # (it wants room for two so a wrap-up call stays affordable). After an + # explicit Resume of a graceful stop that early margin is exactly what the + # owner spent, so the resumed task may place the one call that fits in the + # already-authorized remainder; the ledger fence at the full cap still binds + # every send. Hard rails keep both reservations. + ctx._budget_resume_last_fit_relaxed = bool(graceful) usage["budget_pause_resume"] = {"pause_id": row.get("pause_id"), "grant_id": grant.get("grant_id"), - "paused_duration_sec": ctx._budget_paused_sec, "threshold_refresh": refresh} + "grant_generation": int(grant.get("generation") or 0), + "paused_duration_sec": ctx._budget_paused_sec, "threshold_refresh": refresh, + "last_fit_relaxed": bool(graceful)} plan, mode = rebind_restored_route(tools, state, messages) - external = (state.get("external_runs") or {}).get("runs") or [] + # The grant re-observed custody at Resume time and wrote that observation + # on the row; the checkpoint's copy is the pause-time reading. Disclose the + # freshest one the durable rows carry. + fresh_observation = row.get("external_runs") if isinstance(row.get("external_runs"), dict) else None + external_observation = fresh_observation if fresh_observation is not None else ( + state.get("external_runs") or {}) + # Name WHICH reading this is: the grant's re-observation and the checkpoint's + # pause-time copy license different amounts of trust, and a list labelled as + # pause-time history reads as stale even when it is the fresh one. + external_basis = ("re-observed at this Resume" if fresh_observation is not None + else "as recorded at the pause, NOT re-observed") + external = external_observation.get("runs") or [] external_lines = "".join( f"\n- run {run.get('run_id')}: {run.get('state')} (stop outcome: {run.get('stop_outcome') or 'n/a'})" for run in external if isinstance(run, dict) @@ -946,7 +1216,7 @@ def resume_paused_loop(tools: Any, state: Dict[str, Any], messages: list, trace: "recorded; do not repeat completed effects. The pause ended the previous browser process and " "task-local services; their recorded results remain evidence, not proof they are still running. " "The workspace may have drifted while paused: re-read any file before building on it. " - f"Delegated runs held at the pause:{external_lines}\n" + f"Delegated runs this task holds ({external_basis}):{external_lines}\n" "An unknown or merely requested stop is NOT proof of termination: never start a second writer " "over such a run; inspect its custody first. " + (f"The last tool batch was interrupted: {len(pending)} call(s) have no recorded result and were " @@ -958,33 +1228,58 @@ def resume_paused_loop(tools: Any, state: Dict[str, Any], messages: list, trace: # --- restore-after-restart gate (supervisor side) ------------------------------------------- -def restore_budget_pause_allowed(root: Any, task: Dict[str, Any]) -> bool: - """A paused PENDING row survives a restart and any snapshot age WITHOUT waking: - the row is a locator, the task-result authority and the readable source make - it restorable. Panic / no-resume flags and a terminal row refuse.""" +RESTORE_REFUSAL_NOT_EXACT = "not_an_exact_pause" +RESTORE_REFUSAL_RECORD_UNREADABLE = "pause_record_unreadable" +RESTORE_REFUSAL_TASK_TERMINAL = "task_terminal" +RESTORE_REFUSAL_RECORD_MISSING = "pause_record_missing" +RESTORE_REFUSAL_IDENTITY_MISMATCH = "pause_identity_mismatch" +RESTORE_REFUSAL_SOURCE_MISSING = "pause_source_missing" +RESTORE_REFUSAL_SOURCE_UNREADABLE = "pause_source_unreadable" + + +def budget_pause_restore_refusal(root: Any, task: Dict[str, Any]) -> str: + """Why a paused PENDING locator cannot be restored as dispatch-eligible; "" when it can. + + A paused row survives a restart and any snapshot age WITHOUT waking: the + queue row is a locator, and the task-result authority plus the readable + source are what make it restorable. Every refusal here is a fact about the + durable authority behind the locator (unreadable, terminal, a different + pause id, no source, an unreadable source) — never a cancellation and never + a reason to drop the row: the caller HOLDS it, typed and visible, so the + saved pause stays where the owner can see it. Stop/Panic and the no-resume + flag are NOT restore refusals: they are Resume refusals, read at grant time + by the same gate that reads them live, so a flag that clears later never + leaves a perfectly restorable pause held for nothing. + """ from ouroboros.artifacts import read_actor_source_bytes from ouroboros.task_results import _TRULY_TERMINAL_STATUSES, load_task_result pause = task.get("_budget_pause") if isinstance(task, dict) else None if not isinstance(pause, dict) or not pause.get("exact_continuation"): - return False + return RESTORE_REFUSAL_NOT_EXACT checkpoint = pause.get("checkpoint") if isinstance(pause.get("checkpoint"), dict) else {} root = pathlib.Path(root) - if any((root / "state" / name).exists() for name in ("panic_stop.flag",)): - return False task_id = str(task.get("id") or "") try: - row = load_task_result(root, task_id, strict=True) or {} + row = load_task_result(pathlib.Path(task.get("budget_drive_root") or root), task_id, strict=True) or {} except Exception: - return False - current = row.get("budget_pause") or {} - if (row.get("status") in _TRULY_TERMINAL_STATUSES - or current.get("state") not in LIVE_PAUSE_STATES - or current.get("pause_id") != checkpoint.get("pause_id") - or not current.get("source_ref")): - return False + return RESTORE_REFUSAL_RECORD_UNREADABLE + current = row.get("budget_pause") if isinstance(row.get("budget_pause"), dict) else {} + if row.get("status") in _TRULY_TERMINAL_STATUSES: + return RESTORE_REFUSAL_TASK_TERMINAL + if not current or current.get("state") not in LIVE_PAUSE_STATES: + return RESTORE_REFUSAL_RECORD_MISSING + if current.get("pause_id") != checkpoint.get("pause_id"): + return RESTORE_REFUSAL_IDENTITY_MISMATCH + if not current.get("source_ref"): + return RESTORE_REFUSAL_SOURCE_MISSING try: - read_actor_source_bytes(root, task_id, current["source_ref"]) + read_actor_source_bytes(pathlib.Path(task.get("budget_drive_root") or root), task_id, current["source_ref"]) except Exception: - return False - return True + return RESTORE_REFUSAL_SOURCE_UNREADABLE + return "" + + +def restore_budget_pause_allowed(root: Any, task: Dict[str, Any]) -> bool: + """Boolean view of ``budget_pause_restore_refusal`` for callers that only gate.""" + return budget_pause_restore_refusal(root, task) == "" diff --git a/ouroboros/delegate_continuation.py b/ouroboros/delegate_continuation.py new file mode 100644 index 000000000..ba7d5c822 --- /dev/null +++ b/ouroboros/delegate_continuation.py @@ -0,0 +1,304 @@ +"""Finite delegated-leaf continuation after a CONFIRMED wall-clock expiry (#1196). + +A delegated run started with ``maxSeconds`` is cancelled by the engine when that +cap expires and settles ``cancelled`` with the engine's typed reason +``wall_clock_exceeded`` (recorded on the SETTLED custody row as +``outcome_reason`` and replayed as ``RunCustody.terminal_reason``). Such a run +did real work that the nanny may want finished. This module is the ONE gate a +``delegate_start(continue_from=)`` passes before a NEW run is started +for the remaining work, and the ONE author of the host block that binds that +run to its predecessor. + +It is deliberately NOT crash recovery and NOT a resume: ``delegate_recovery`` +keeps its ``NO_RESUME_CAUSES`` untouched, nothing is replayed under an old key, +no engine session state is transferred, and the host never decides what the +remaining work is — the model writes the continuation prompt with the prior +result and its explicit patch disposition in front of it. The gate admits only: + +* the caller's OWN settled run (custody says OWNED and SETTLED); +* whose terminal is ``cancelled`` with reason ``wall_clock_exceeded`` — an owner + deadline, Stop or Panic (``host_cancelled``/``owner_task_gone``), a user cancel, + a failure, an absent/unknown engine outcome or a pre-field settlement are + refused typed, each with the fact that refused it — AND whose cap was a + finite leaf cap the nanny ASKED for (``max_seconds_basis`` on the STARTED + row: ``requested``, at most narrowed by the engine's schema bound): the + engine's typed reason alone is not enough, because a cap the nanny's own + deadline or lifetime derived or narrowed expiring IS that deadline, and a + row with no recorded basis is unknown, never assumed; +* whose result is RETAINED (its terminal detail staged in full) AND READ to EOF + (a run nobody waited on, a partial staging or an unread staging refuses: + continuing work nobody has read is a blind resend); +* whose captured patch has an EXPLICIT apply/reject disposition (an undisposed + or apply-ambiguous patch refuses: the continuation must know what the tree + already contains, and nothing may be applied twice), and whose partial + work order, if any, has fully verified source coverage; +* under POSITIVELY the same executor and authority: the recorded actor, route, + configuration fingerprint, task authority fingerprint, access, mode and + isolation must each be recorded AND equal to this start's (an unrecorded + side never "does not contradict" — it refuses), a mutating run's authority + target likewise, and a configured session's canonical work order must be + the one the prior run was bound to; the continuation's own ``max_seconds`` + is narrowed by the nanny's remaining bounds exactly like any start. +""" + +from __future__ import annotations + +import logging +from typing import Any, Dict, Optional, Tuple + +from ouroboros import delegate_custody as custody + +log = logging.getLogger(__name__) + +# The engine's typed cancel reason for a maxSeconds expiry (Claudexor +# ``RunOutcomeReason``: a cancelled lifecycle with reason ``wall_clock_exceeded``). +CONTINUATION_CAUSE = "wall_clock_exceeded" +CONTINUATION_TERMINAL_STATE = "cancelled" + +REFUSAL_SOURCE_UNKNOWN = "continuation_source_unknown" +REFUSAL_SOURCE_NOT_OWNED = "continuation_source_not_owned" +REFUSAL_SOURCE_NOT_TERMINAL = "continuation_source_not_terminal" +REFUSAL_CAUSE_UNRECORDED = "continuation_cause_unrecorded" +REFUSAL_CAUSE_NOT_WALL_CLOCK = "continuation_cause_not_wall_clock" +REFUSAL_RESULT_UNREAD = "continuation_result_unread" +REFUSAL_PATCH_UNDISPOSED = "continuation_patch_undisposed" +REFUSAL_APPLY_AMBIGUOUS = "continuation_apply_ambiguous" +REFUSAL_EXECUTOR_MISMATCH = "continuation_executor_mismatch" +REFUSAL_AUTHORITY_MISMATCH = "continuation_authority_mismatch" +REFUSAL_TARGET_MISMATCH = "continuation_target_mismatch" +REFUSAL_CAP_BASIS_UNKNOWN = "continuation_cap_basis_unknown" +REFUSAL_CAP_NOT_FINITE_LEAF = "continuation_cap_not_finite_leaf" +REFUSAL_RESULT_UNRETAINED = "continuation_result_unretained" +REFUSAL_RESULT_INCOMPLETE = "continuation_result_incomplete" +REFUSAL_SOURCE_UNVERIFIED = "continuation_source_unverified" +REFUSAL_CONFIG_MISMATCH = "continuation_config_mismatch" +REFUSAL_TASK_AUTHORITY_MISMATCH = "continuation_task_authority_mismatch" +REFUSAL_WORK_ORDER_UNBOUND = "continuation_work_order_unbound" +REFUSAL_WORK_ORDER_MISMATCH = "continuation_work_order_mismatch" + + +def _needs_disposition(entry: custody.RunCustody) -> bool: + """Whether the prior run's work reaches the tree only through an explicit disposition.""" + ref = entry.resource_ref if isinstance(entry.resource_ref, dict) else {} + return bool(entry.snapshot_id or (ref.get("workspace_kind") == "directory" and ref.get("strategy") == "copy")) + + +def bind_continuation(ctx: Any, drive: Any, run_id: str, *, actor: Dict[str, Any], route: Any, + authority: Any, target_root: str, + canonical_work_order_fingerprint: str = "") -> Tuple[Dict[str, Any], str, str]: + """``(facts, refusal_code, detail)``: the typed gate, from durable custody only. + + ``facts`` is complete only when ``refusal_code`` is empty. Nothing here + reads the engine: the SETTLED row, the STARTED row's cap basis, the + retained output facts and the disposition rows are the whole authority, so + a daemon that is gone cannot turn an unknown cause into an admitted + continuation. Every binding fact is checked POSITIVELY: recorded on the + prior run, present on this start, and equal. + """ + from ouroboros.configured_subagents import SESSION_ACCESS_PROFILES + from ouroboros.delegate_registration_policy import FINITE_LEAF_CAP_BASES + + rid = str(run_id or "").strip() + task_id = str(getattr(ctx, "task_id", "") or "") + status, entry = custody.lookup(drive, task_id, rid) + if status == custody.UNKNOWN or entry is None: + return {}, REFUSAL_SOURCE_UNKNOWN, ( + f"No durable custody record names run {rid!r} on this drive; a continuation " + "binds to a run this task can prove it owns.") + if status == custody.FOREIGN: + return {}, REFUSAL_SOURCE_NOT_OWNED, ( + f"Run {rid} belongs to task {entry.task_id or 'unknown'}, not to this task; " + "only its owner may continue it.") + if not entry.settled or not entry.terminal_state: + return {}, REFUSAL_SOURCE_NOT_TERMINAL, ( + f"Run {rid} has no settled terminal on its custody rows (state {entry.terminal_state or 'unsettled'!r}): " + "it may still be live. Wait on it with delegate_wait or cancel it and verify the receipt; " + "never start a second writer over a run that may still be running.") + if not entry.terminal_reason: + return {}, REFUSAL_CAUSE_UNRECORDED, ( + f"Run {rid} settled {entry.terminal_state!r} without a recorded engine reason, so a wall-clock " + "expiry cannot be CONFIRMED; a continuation is admitted only over a confirmed maxSeconds expiry. " + "Start a plain new run if the work is still needed, stating what the prior run left.") + if entry.terminal_state != CONTINUATION_TERMINAL_STATE or entry.terminal_reason != CONTINUATION_CAUSE: + return {}, REFUSAL_CAUSE_NOT_WALL_CLOCK, ( + f"Run {rid} ended {entry.terminal_state!r} with reason {entry.terminal_reason!r}, which is not a " + f"maxSeconds expiry ({CONTINUATION_CAUSE}). An owner deadline, Stop or Panic, a user cancel or a " + "failure is not continued through this seam.") + started_ts, prior_max_seconds = custody.run_timing(drive, rid) + cap_basis = custody.run_cap_basis(drive, rid) + if not cap_basis: + return {}, REFUSAL_CAP_BASIS_UNKNOWN, ( + f"Run {rid}'s STARTED row records no basis for its maxSeconds cap, so its expiry cannot be told " + "apart from this task's own deadline or lifetime. The engine's typed reason alone does not admit " + "a continuation; start a plain new run for the remaining work.") + if cap_basis not in FINITE_LEAF_CAP_BASES or int(prior_max_seconds or 0) <= 0: + return {}, REFUSAL_CAP_NOT_FINITE_LEAF, ( + f"Run {rid}'s cap was {cap_basis!r} ({int(prior_max_seconds or 0)}s): derived from or narrowed by " + "this task's own deadline or lifetime, not a finite leaf cap the nanny asked for. Its expiry IS " + "that bound; it is not continued through this seam.") + output = custody.output_disposition(entry) + if not output: + return {}, REFUSAL_RESULT_UNRETAINED, ( + f"Run {rid} settled without its terminal detail being staged: nothing of its result is retained " + "on this drive. Wait on it with delegate_wait (which stages the full detail), read that to EOF, " + "then continue — continuing work nobody has retained is a blind resend.") + if not output.get("staged_output_complete"): + return {}, REFUSAL_RESULT_INCOMPLETE, ( + f"Run {rid}'s staged output at {entry.output_artifact} is not verified FULL content; a preview " + "or a cut staging is not the result. Re-wait so the complete detail is staged, read it, then continue.") + if custody.settled_output_unread(entry) or not output.get("staged_output_consumed"): + return {}, REFUSAL_RESULT_UNREAD, ( + f"Run {rid}'s full output is staged at {entry.output_artifact} and was never read to EOF; " + "read it with read_file(root='task_drive') first — continuing work nobody has read is a blind resend.") + if entry.patch_apply_pending: + return {}, REFUSAL_APPLY_AMBIGUOUS, ( + f"Run {rid} has a pending apply intent with no disposition: the tree MAY already carry its patch. " + "Resolve it through integrate_delegated_patch(acknowledge_ambiguous=true) before continuing.") + if _needs_disposition(entry) and not entry.patch_disposed: + return {}, REFUSAL_PATCH_UNDISPOSED, ( + f"Run {rid}'s captured changes have no explicit disposition yet. Apply or reject them with " + "integrate_delegated_patch(run_id=...) first, so the continuation knows what the tree contains " + "and nothing is applied twice.") + verification = custody.work_order_source_verification(entry) + if str(verification.get("status") or "") == "cannot_verify": + return {}, REFUSAL_SOURCE_UNVERIFIED, ( + f"Run {rid}'s external work order was only partially delivered and its canonical source ranges " + "are not fully verified; a continuation cannot bind to a brief nobody has proven complete.") + prior_actor = str(entry.selected_subagent_id or "") + this_actor = str((actor or {}).get("selected_subagent_id") or "") + route_id = str(getattr(route, "route_id", "") or "") + if not prior_actor or not this_actor or prior_actor != this_actor \ + or not entry.route_id or not route_id or entry.route_id != route_id: + return {}, REFUSAL_EXECUTOR_MISMATCH, ( + f"Run {rid} ran on actor {prior_actor or 'unrecorded'!r} via route {entry.route_id or 'unrecorded'!r}; " + f"this start resolves to actor {this_actor or 'unrecorded'!r} via route {route_id or 'unrecorded'!r}. " + "A continuation keeps the SAME recorded executor on both sides; select that actor or start a plain new run.") + this_config = str((actor or {}).get("config_fingerprint") or "") + if not entry.config_fingerprint or not this_config or entry.config_fingerprint != this_config: + return {}, REFUSAL_CONFIG_MISMATCH, ( + f"Run {rid} was started from actor configuration {entry.config_fingerprint or 'unrecorded'!r}; this " + f"start carries {this_config or 'unrecorded'!r}. A continuation runs the SAME recorded configuration.") + this_authority = str((actor or {}).get("authority_fingerprint") or "") + if not entry.authority_fingerprint or not this_authority or entry.authority_fingerprint != this_authority: + return {}, REFUSAL_TASK_AUTHORITY_MISMATCH, ( + f"Run {rid} was started under task authority {entry.authority_fingerprint or 'unrecorded'!r}; this " + f"start derives {this_authority or 'unrecorded'!r}. A continuation runs under the SAME recorded " + "task authority (contract, constraint, workspace); it cannot be re-derived.") + if not entry.work_order_fingerprint: + return {}, REFUSAL_WORK_ORDER_UNBOUND, ( + f"Run {rid}'s STARTED row binds no work order; a continuation follows a run whose assignment is " + "recorded. Start a plain new run stating the remaining work.") + canonical = str(canonical_work_order_fingerprint or "") + if canonical and canonical != entry.work_order_fingerprint: + return {}, REFUSAL_WORK_ORDER_MISMATCH, ( + f"This configured session's canonical work order ({canonical[:12]}…) is not the one run {rid} was " + f"bound to ({entry.work_order_fingerprint[:12]}…). A continuation stays inside the SAME assignment.") + shape = {key: str(getattr(authority, key, "") or "") for key in ("access", "mode", "isolation")} + prior_shape = {"access": entry.access, "mode": entry.mode, "isolation": entry.isolation} + if (not prior_shape["access"] or not prior_shape["mode"] or not shape["access"] or not shape["mode"] + or any(prior_shape[key] != shape[key] for key in shape)): + return {}, REFUSAL_AUTHORITY_MISMATCH, ( + f"Run {rid}'s authority shape was {prior_shape}; this start derives {shape}. A continuation runs " + "under the SAME recorded workspace authority; it cannot widen, reshape or leave it unrecorded.") + if entry.access in SESSION_ACCESS_PROFILES and ( + not entry.target_root or not target_root or entry.target_root != target_root): + return {}, REFUSAL_TARGET_MISMATCH, ( + f"Run {rid} held authority over {entry.target_root or 'an unrecorded target'}; this task's present " + f"target is {target_root or 'unrecorded'}. A continuation writes for the target the prior run held, " + "never a tree it has since moved to.") + facts: Dict[str, Any] = { + "continuation_of": rid, + "cause": entry.terminal_reason, + "prior_terminal_state": entry.terminal_state, + "prior_started_at": started_ts, + "prior_max_seconds": int(prior_max_seconds or 0) or None, + "prior_cap_basis": cap_basis, + "prior_patch_disposition": entry.patch_disposed or ("not_applicable" if not _needs_disposition(entry) else ""), + "prior_target_root": entry.target_root, + "prior_baseline_sha": entry.baseline_sha, + "prior_output": output, + "prior_invocation_id": entry.invocation_id, + "prior_actor": prior_actor, + "prior_route": entry.route_id, + "prior_config_fingerprint": entry.config_fingerprint, + "prior_authority_fingerprint": entry.authority_fingerprint, + "prior_work_order_fingerprint": entry.work_order_fingerprint, + "canonical_work_order_fingerprint": canonical, + "state_transfer": "none", + } + return facts, "", "" + + +def continuation_instruction(facts: Dict[str, Any]) -> str: + """The host block appended to the run's instructions: what is bound, what is not.""" + disposition = str(facts.get("prior_patch_disposition") or "") + if disposition == "applied": + tree_line = ("Its captured changes were explicitly APPLIED to the authority target: the tree you start " + "from already contains them. Do not redo or re-apply that work.") + elif disposition == "rejected": + tree_line = ("Its captured changes were explicitly REJECTED: the tree you start from does NOT contain " + "them, and only the assignment in the prompt says what is still wanted.") + else: + tree_line = "It captured no changes to dispose of (a read-only run)." + cap = facts.get("prior_max_seconds") + cap_line = f" after its {int(cap)}s wall-clock cap" if cap else " at its wall-clock cap" + return ( + f"\n\nCONTINUATION OF RUN {facts.get('continuation_of')}: that run was cancelled by the engine{cap_line} " + f"(reason {facts.get('cause')}); it is settled and is NOT running. {tree_line} " + "NOTHING of its session state is transferred to you: no transcript, no memory, no assumptions. " + "The prompt states the remaining work as the host decided it; verify what is already done from the " + "workspace as it is NOW before repeating any step, never assume a result you cannot see in the tree, " + "and never re-apply what was applied. This run has its own wall-clock cap; finish the remaining work " + "or report exactly what remains." + ) + + +def selector_refusal(continue_from: Any, retry_of: Any, selector_root: str) -> Tuple[str, Optional[Any]]: + """``(continuation_token, refusal)``: the argument shapes a continuation cannot share. + + A retry replays an old key byte-identically; a continuation is a NEW + intention over a settled run — one call cannot be both. A skill-payload + selector run keeps its own target semantics and is started plain. + """ + from ouroboros.delegate_shared import _fail + + token = str(continue_from or "").strip() + if token and str(retry_of or "").strip(): + return token, _fail("delegate_start", "continuation_selector_conflict", + "continue_from starts a NEW run bound to a settled predecessor; retry_of replays a " + "pending invocation. Supply one of them.", definitely_unrun=True) + if token and str(selector_root or "").strip(): + return token, _fail("delegate_start", "continuation_resource_conflict", + "continue_from applies to ordinary workspace delegation only; a skill-payload " + "selector run is started plain.", definitely_unrun=True) + return token, None + + +def start_binding(ctx: Any, drive: Any, token: str, *, actor: Dict[str, Any], route: Any, + authority: Any, target_root: str, + canonical_work_order_fingerprint: str = "") -> Tuple[Dict[str, Any], str, Optional[Any]]: + """``(facts, instruction, refusal)`` for ONE start: the gate plus its host block. + + Decided from durable custody before any snapshot or registration exists; a + refusal is a definite no-run recorded as a start-blocked evidence row. + """ + from ouroboros.delegate_evidence import record_start_blocked + from ouroboros.delegate_shared import _fail + + facts, code, detail = bind_continuation( + ctx, drive, token, actor=actor, route=route, authority=authority, target_root=target_root, + canonical_work_order_fingerprint=canonical_work_order_fingerprint) + if code: + record_start_blocked(ctx, str(getattr(ctx, "task_id", "") or ""), code) + return {}, "", _fail("delegate_start", code, detail, continue_from=token, definitely_unrun=True) + return facts, continuation_instruction(facts), None + + +__all__ = [ + "CONTINUATION_CAUSE", + "CONTINUATION_TERMINAL_STATE", + "bind_continuation", + "continuation_instruction", + "selector_refusal", + "start_binding", +] diff --git a/ouroboros/delegate_custody.py b/ouroboros/delegate_custody.py index aceee34b6..8c470a19a 100644 --- a/ouroboros/delegate_custody.py +++ b/ouroboros/delegate_custody.py @@ -152,6 +152,8 @@ class RunCustody: ledger_recorded: bool = False settled: bool = False terminal_state: str = "" # SETTLED row's state, replayed (empty pre-existing/CLOSED_ABSENT) + terminal_reason: str = "" # #1196: engine ``outcomeFacts.reason`` replayed from SETTLED ("" = none) + continuation_of: str = "" # #1196: the settled run this one explicitly continued (``continue_from``) containment_disclosed: bool = False # written once; a re-poll must not re-find unread_disclosed: bool = False # settled-never-read omission named durably # Staged-output half of the terminal story (D7). ``output_artifact``: @@ -530,6 +532,7 @@ def _apply(state: Dict[str, RunCustody], row: Dict[str, Any]) -> None: # emitted it before SETTLED, so replay is unaffected). custody.ledger_recorded = custody.settled = True custody.terminal_state = str(row.get("state") or "") or custody.terminal_state + custody.terminal_reason = str(row.get("outcome_reason") or "") or custody.terminal_reason elif kind == CLOSED_ABSENT: # Closed, not settled: custody is over, the run leaves ``open_runs``. # The registration survives independently (wholesale clearing here was @@ -713,6 +716,22 @@ def run_timing(drive_root: Any, run_id: str) -> Tuple[str, int]: return started_ts, max_seconds +def run_cap_basis(drive_root: Any, run_id: str) -> str: + """How a run's ``maxSeconds`` was decided, from its durable STARTED row + (``delegate_registration_policy.CAP_BASIS_*``); "" when the run is unknown + or the row predates the field — an absent basis stays absent (#1196).""" + rid = str(run_id or "").strip() + if not rid: + return "" + for row in custody_rows(drive_root): + if str(row.get("run_id") or "") != rid or str(row.get("type") or "") != STARTED: + continue + basis = str(row.get("max_seconds_basis") or "") + if basis: + return basis + return "" + + def idempotency_key(*parts: Any) -> str: """A deterministic IDENTITY for one logical start — the lookup key, not the wire key. @@ -810,6 +829,10 @@ def invocation_record(drive_root: Any, invocation_id: str, *, "work_order_coverage": str(row.get("work_order_coverage") or ""), "authority_fingerprint": str(row.get("authority_fingerprint") or ""), "processing": copy.deepcopy(row.get("processing")) if isinstance(row.get("processing"), dict) else {}, + # #1196: the cap and HOW it was decided replay with the body a + # retry re-POSTs, so the replayed STARTED row keeps the same basis. + "max_seconds": int(row.get("max_seconds") or 0) if str(row.get("max_seconds") or "").lstrip("-").isdigit() else 0, + "max_seconds_basis": str(row.get("max_seconds_basis") or ""), "work_order_source_request": ( copy.deepcopy(row.get("work_order_source_request")) if isinstance(row.get("work_order_source_request"), dict) else {} @@ -1048,9 +1071,12 @@ def settle_run(drive_root: Any, gateway: Any, custody: RunCustody, detail: Dict[ # ENGINE's own code ("" = it gave none) and the words it reported (opaque, never # branched on). A succeeded row is byte-identical to before. failure = summary.get("failure") if isinstance(summary.get("failure"), dict) else {} + outcome_facts = summary.get("outcomeFacts") if isinstance(summary.get("outcomeFacts"), dict) else {} failure_facts = {} if str(summary.get("state") or "") in SUCCEEDED_STATES else { "requested_model": custody.model, "failure_code": str(failure.get("code") or ""), - "reported_cause": run_failure_cause(failure)} + "reported_cause": run_failure_cause(failure), + # Engine TYPED reason (``wall_clock_exceeded`` = maxSeconds expiry): the continuation gate's one fact. + "outcome_reason": str(outcome_facts.get("reason") or "")} # Claudexor reports CASH in `spendUsd`, EXACTNESS in `spendEstimated`. A run # is only free when the amount is really zero AND really settled: expired # sessions, bill-by-construction routes and auth fallbacks all charge, and @@ -1466,6 +1492,7 @@ __all__ = [ "reconcile_task_runs", "retire_project", "review_owned_source", + "run_cap_basis", "run_timing", "settle_run", "settled_output_unread", diff --git a/ouroboros/delegate_registration_policy.py b/ouroboros/delegate_registration_policy.py index 1564cedad..eb6e65f70 100644 --- a/ouroboros/delegate_registration_policy.py +++ b/ouroboros/delegate_registration_policy.py @@ -83,6 +83,21 @@ def resolve_registration(gateway, scope_root: str, execution_root: str, access: return project_id, owned_project_id, persistent_registration(execution_root, access) +# How a run's ``maxSeconds`` was decided (#1196), recorded on the START_REQUESTED +# and STARTED rows as ``max_seconds_basis``. Only a cap the nanny ASKED for — an +# explicit finite leaf cap, at most narrowed by the engine's schema bound — makes +# a later ``wall_clock_exceeded`` a finite-leaf expiry a continuation may follow; +# a cap the nanny's own deadline or lifetime derived (or narrowed) expiring IS +# that deadline, and a row that predates the field is unknown, never assumed. +CAP_BASIS_REQUESTED = "requested" +CAP_BASIS_REQUESTED_CLAMPED_SCHEMA = "requested_clamped_by_schema" +CAP_BASIS_REQUESTED_CLAMPED_DEADLINE = "requested_clamped_by_deadline" +CAP_BASIS_REQUESTED_CLAMPED_LIFETIME = "requested_clamped_by_lifetime" +CAP_BASIS_DEADLINE_DERIVED = "deadline_derived" +CAP_BASIS_LIFETIME_DERIVED = "lifetime_derived" +CAP_BASIS_OPERATION_WINDOW = "operation_window" +FINITE_LEAF_CAP_BASES = frozenset({CAP_BASIS_REQUESTED, CAP_BASIS_REQUESTED_CLAMPED_SCHEMA}) + # The STARTED row's string facts as ``(RunCustody attribute, row key)`` pairs — # one table shared by the replay and the ``record_started`` emit. STARTED_STR_FIELDS: Tuple[Tuple[str, str], ...] = tuple( @@ -94,6 +109,9 @@ STARTED_STR_FIELDS: Tuple[Tuple[str, str], ...] = tuple( "authority_source", "access", "mode", "isolation", "selected_subagent_id", "config_fingerprint", "work_order_fingerprint", "work_order_coverage", "authority_fingerprint", + # #1196: the prior run this start explicitly continues (a confirmed + # wall-clock expiry), "" for every ordinary start. + "continuation_of", ) ) # None means an old row omitted the choice; '' is a captured default choice. diff --git a/ouroboros/delegate_source_coverage.py b/ouroboros/delegate_source_coverage.py index 5dc74d378..1766382ef 100644 --- a/ouroboros/delegate_source_coverage.py +++ b/ouroboros/delegate_source_coverage.py @@ -214,8 +214,17 @@ def record_started_custody( snapshot_id: str, execution_binding_fingerprint: str, target_root: str, baseline_sha: str, authority_source: str, resource_ref: Dict[str, Any], capture_mode: str, processing: Mapping[str, Any] | None = None, + continuation_of: str = "", max_seconds_basis: str = "", ) -> bool: - """Write the one STARTED custody row, including the source binding.""" + """Write the one STARTED custody row, including the source binding. + + ``continuation_of`` names the prior run this start explicitly continues + after that run's confirmed wall-clock expiry (#1196); it rides the STARTED + row so the lineage replays with every other start fact. ``max_seconds_basis`` + records HOW ``seconds`` was decided (``delegate_registration_policy.CAP_BASIS_*``) + beside the cap itself, so a later expiry can be told apart from the nanny's + own deadline or lifetime. + """ from ouroboros import delegate_custody as custody_module @@ -255,6 +264,7 @@ def record_started_custody( mode=authority.mode, isolation=authority.isolation, delegated=authority.delegated, + continuation_of=str(continuation_of or ""), ) return custody_module.record_started( drive, @@ -262,7 +272,8 @@ def record_started_custody( shape={ "effort": route.effort, "access": access, "mode": authority.mode, "isolation": authority.isolation, "delegated": authority.delegated, - "root": root, "max_seconds": seconds, "capture_mode": capture_mode, + "root": root, "max_seconds": seconds, "max_seconds_basis": str(max_seconds_basis or ""), + "capture_mode": capture_mode, }, ) diff --git a/ouroboros/domains.toml b/ouroboros/domains.toml index c259916be..3a4e2fb41 100644 --- a/ouroboros/domains.toml +++ b/ouroboros/domains.toml @@ -63,6 +63,7 @@ D20 = "Presence" [modules] "launcher.py" = "D18" +"ouroboros/budget_pause.py" = "D01" "ouroboros/launcher_windows_runtime.py" = "D18" "ouroboros/__init__.py" = "D18" "ouroboros/_outcome_receipts.py" = "D01" @@ -133,6 +134,7 @@ D20 = "Presence" "ouroboros/deadline_utils.py" = "D01" "ouroboros/deep_self_review.py" = "D06" "ouroboros/delegate_containment.py" = "D07" +"ouroboros/delegate_continuation.py" = "D07" "ouroboros/delegate_custody.py" = "D07" "ouroboros/delegate_custody_memo.py" = "D07" "ouroboros/delegate_state_sweep.py" = "D07" @@ -559,6 +561,7 @@ D20 = "Presence" "ouroboros/startup_historical_audit.py" = "D11" "supervisor/__init__.py" = "D08" "supervisor/active_activity.py" = "D08" +"supervisor/budget_resume.py" = "D08" "supervisor/cancel_publication.py" = "D09" "supervisor/cognitive_operations.py" = "D08" "supervisor/events.py" = "D08" diff --git a/ouroboros/gateway/contracts.py b/ouroboros/gateway/contracts.py index 6392a1b2f..f1c243efa 100644 --- a/ouroboros/gateway/contracts.py +++ b/ouroboros/gateway/contracts.py @@ -728,7 +728,10 @@ class ActiveChatActivity(ActiveDirectTurn): no new dispatch, not yet released; additive, #1196) | ``working`` | ``finalizing`` (answer stored, post-task synthesis open); a direct row whose live wait owner could not be read is - ``phase="unknown"``. Same shape as ``ActiveDirectTurn`` so one reducer hydrates + ``phase="unknown"``; a direct turn paused on a budget rail (#1196) is parked under + its SAME id and reports the managed phases as ``kind="direct_chat"`` (``budget_paused``, + then ``working``/``finalizing`` on a pooled worker after an explicit Resume). Same shape + as ``ActiveDirectTurn`` so one reducer hydrates both (managed rows: empty ``client_message_id``). ``required_question_unavailable``: a recorded owner-question wait whose detail could not be resolved — possibly blocked. """ diff --git a/ouroboros/gateway/tasks.py b/ouroboros/gateway/tasks.py index e1a4f0aa0..429490d9a 100644 --- a/ouroboros/gateway/tasks.py +++ b/ouroboros/gateway/tasks.py @@ -1358,7 +1358,12 @@ async def api_task_resume(request: Request) -> JSONResponse: (#1196) receives ONE single-use grant and continues under the same task id. Every refusal is typed: money still exhausted, a live cancel intent, a passed deadline, an exhausted finite lifetime, a root that is itself still - paused, or a missing/unreadable checkpoint all leave the task paused. + paused, or a missing/unreadable checkpoint all leave the task paused. A row + HELD beside its pause (an unrestorable source at restart, an unwritten + revocation, an acceptance fence at restore) is granted by the same call once + its durable authority validates again; a fence-lifted zero-dispatch sibling + is released by this same call as an explicit selection. A paused direct + owner-chat turn is resumed here too, under its own task id. """ try: task_id = validate_task_id(request.path_params.get("task_id")) @@ -1381,6 +1386,13 @@ async def api_task_resume(request: Request) -> JSONResponse: "restart_no_resume", "pause_record_missing", "pause_source_unreadable", "pause_record_unreadable", "grant_not_recorded", "snapshot_not_persisted", "monetary_authority_unavailable", "cancellation_authority_unavailable", "task_terminal", + # holds and root-grant refusals (#1196, owner Q9): the row stays paused/held + "root_resume_grant_missing", "root_resume_generation_stale", "root_replay_unsafe", + "root_accounting_unavailable", "root_accounting_degraded", "external_custody_unreadable", + "accounting_unavailable", "resume_grant_revocation_unwritten", + # fresh custody at grant (#1196, owner Q8): a delegated run not proven + # terminal keeps the task paused; a marker/attempt drift is typed too + "external_runs_unsettled", "pause_attempt_mismatch", } else 404 return json_error(error, status, task_id=task_id, **({"action": result["action"]} if result.get("action") else {})) diff --git a/ouroboros/loop_budget.py b/ouroboros/loop_budget.py index 9e60745b7..2fdf1d707 100644 --- a/ouroboros/loop_budget.py +++ b/ouroboros/loop_budget.py @@ -100,6 +100,14 @@ def _check_budget_limits( prompt_estimate = int(accumulated_usage.get("_context_prompt_estimate") or 0) global_remaining = _wrapup_global_remaining() if prompt_estimate > 0 and not ctx.active_use_local else None wrapup_fits = None + # Owner Q10 (#1196): after an explicit Resume of a GRACEFUL pause the + # last-fit rail no longer demands room for TWO reservations. The owner's + # Resume spent exactly that early margin, so the one call that fits in the + # already-authorized remainder is admitted instead of re-pausing on the + # very number that paused the task; the ledger fence at the full cap still + # arbitrates every send. A stop that needs no wrap-up room at all + # (``wrapup_fits is False``) is unchanged. + last_fit_relaxed = _last_fit_relaxed(ctx) if prompt_estimate > 0 and (global_remaining is not None or (cost_ceiling.root_cap_usd is not None and deciding is not None)): finish_reason = task_pacing.wrapup_last_fit_text(deciding, cost_ceiling, global_remaining) forced_prompt = f"[BUDGET LIMIT] {finish_reason} {_loop()._FORCED_BEST_EFFORT_TAIL}" @@ -109,7 +117,7 @@ def _check_budget_limits( global_remaining_usd=global_remaining) wrapup_args = dict(**request_args, **balances) wrapup_fits = task_pacing.wrapup_reservation_fits(**wrapup_args) - two_fit = task_pacing.wrapup_reservation_fits(**wrapup_args, reservation_count=2) if wrapup_fits is True else None + two_fit = _second_reservation_fits(ctx, wrapup_args, wrapup_fits, relaxed=last_fit_relaxed) server_web = _loop()._server_web_allowed_by_task(getattr(getattr(ctx, "tools", None), "_ctx", None)) if wrapup_fits is False or two_fit is False or ( wrapup_fits is True and messages_carry_native_images(ctx.messages) @@ -126,7 +134,7 @@ def _check_budget_limits( ) wrapup_args = dict(request=probe, **balances) wrapup_fits = task_pacing.wrapup_reservation_fits(**wrapup_args) - two_fit = task_pacing.wrapup_reservation_fits(**wrapup_args, reservation_count=2) if wrapup_fits is True else None + two_fit = _second_reservation_fits(ctx, wrapup_args, wrapup_fits, relaxed=last_fit_relaxed) if wrapup_fits is False or two_fit is False: # The exact probe confirmed a stop: finalize services and prepare the # candidate that will be dispatched (forced augmentations included). @@ -149,8 +157,8 @@ def _check_budget_limits( ctx, trace, task_pacing.wrapup_unaffordable_text(deciding, cost_ceiling, global_remaining), "budget_exhausted", source="budget_wrapup_unaffordable", ) - if wrapup_fits is True and task_pacing.wrapup_reservation_fits( - **wrapup_args, reservation_count=2, + if wrapup_fits is True and _second_reservation_fits( + ctx, wrapup_args, wrapup_fits, relaxed=last_fit_relaxed, ) is False: accumulated_usage["cost_stop_spend_basis"] = spend_basis accumulated_usage["cost_stop_rail"] = "wrapup_reservation_last_fit" @@ -197,6 +205,32 @@ def _check_budget_limits( return None +def _last_fit_relaxed(ctx: "_RoundLimitContext") -> bool: + """Whether the resumed loop's explicit owner Resume relaxed the last-fit rail (Q10).""" + tool_ctx = getattr(getattr(ctx, "tools", None), "_ctx", None) + return bool(getattr(tool_ctx, "_budget_resume_last_fit_relaxed", False)) + + +def _second_reservation_fits(ctx: "_RoundLimitContext", wrapup_args: Dict[str, Any], + wrapup_fits: Optional[bool], *, relaxed: bool) -> Optional[bool]: + """The two-reservation (last-fit) probe, or ``None`` when it does not decide. + + Priced only while ONE reservation fits (otherwise the harder stop already + decides). Under the Q10 relaxation the probe still runs, so the admitted + early call is DISCLOSED on the usage record, but it no longer stops the + task: the one affordable call is placed and the ledger fence keeps binding. + """ + if wrapup_fits is not True: + return None + second = task_pacing.wrapup_reservation_fits(**wrapup_args, reservation_count=2) + if relaxed and second is False: + ctx.accumulated_usage["budget_resume_last_fit_admitted"] = { + "round_idx": int(ctx.round_idx), "reservations_affordable": 1, + "basis": "owner_resume_relaxed_last_fit"} + return None + return second + + def _pause_scope(cost_ceiling: Optional["task_pacing.CostCeiling"]) -> str: """A tree-capped ceiling is the ROOT's money (its fence covers the tree); a global-share ceiling is global money.""" @@ -557,6 +591,13 @@ def _cleanup_loop_resources( """Release attempt-scoped executors, services, and delegated runs.""" if ctx.trace_ctx is not None: ctx.trace_ctx._execution_trace = ctx.previous_execution_trace + # This attempt's tool-future quiescence rows end with it (#1196): a pause + # only got here after they settled, and any other exit is on the task's own + # control rail; the registry prunes nothing across attempts by itself. + try: + budget_pause.forget_tool_scope(ctx.tools._ctx) + except Exception: + log.debug("Tool-future registry scope could not be dropped", exc_info=True) if stateful_executor: try: from ouroboros.tools.browser import cleanup_browser diff --git a/ouroboros/loop_tool_execution.py b/ouroboros/loop_tool_execution.py index a95674a69..db3fac074 100644 --- a/ouroboros/loop_tool_execution.py +++ b/ouroboros/loop_tool_execution.py @@ -1010,7 +1010,10 @@ def _execute_with_timeout( future = stateful_executor.submit( _execute_browser_tool_bound, tools, tc, drive_logs, task_id, submit_generation, ) - budget_pause.register_tool_future(tool_ctx, tool_call_id, fn_name, future) + # The registration PINS settlement ownership until this call's own + # handling is over (result in time, or the late hold claimed below): + # released in the finally, after either branch (#1196). + release_tool_custody = budget_pause.register_tool_future(tool_ctx, tool_call_id, fn_name, future) try: result = future_result(future, timeout_sec) result_meta = result.get("result_meta") or {} @@ -1082,12 +1085,15 @@ def _execute_with_timeout( "timeout_sec": timeout_sec, }, correlation, tool_call_id=tool_call_id)) return timeout_result + finally: + release_tool_custody() else: with abandoned_on_timeout(timeout_sec, bounded=not is_reviewed_mutative) as submit: future = submit(_execute_single_tool, tools, tc, drive_logs, task_id) # Registered before the wait, so a call abandoned at its timeout is - # already visible to budget-pause quiescence (#1196). - budget_pause.register_tool_future(tool_ctx, tool_call_id, fn_name, future) + # already visible to budget-pause quiescence (#1196); ownership is + # pinned until the finally below, after any late hold was claimed. + release_tool_custody = budget_pause.register_tool_future(tool_ctx, tool_call_id, fn_name, future) try: result = future.result() if is_reviewed_mutative else future_result(future, timeout_sec) result_meta = result.get("result_meta") or {} @@ -1157,6 +1163,8 @@ def _execute_with_timeout( "timeout_sec": timeout_sec, }, correlation, tool_call_id=tool_call_id)) return timeout_result + finally: + release_tool_custody() _PARALLEL_SAFE_TOOLS: frozenset[str] = READ_ONLY_PARALLEL_TOOLS | PARALLEL_SAFE_ENQUEUE_TOOLS diff --git a/ouroboros/model_wait.py b/ouroboros/model_wait.py index 8af8a4ca9..299a2ab1b 100644 --- a/ouroboros/model_wait.py +++ b/ouroboros/model_wait.py @@ -127,6 +127,31 @@ def quota_waited_seconds(meta: dict, now: float) -> float: return max(0.0, elapsed) + (max(0.0, now - observed) if clock.get("active") is True else 0.0) +def budget_paused_seconds(meta: dict) -> float: + """The ONE budget-paused carrier (#1196), read from a RUNNING row, a resume + handoff or a pause row alike: wall time a task spent PAUSED, which is not + execution and is NOT part of the quota union (that clock stays its own).""" + try: + paused = float(meta.get("budget_paused_sec") or meta.get("paused_duration_sec") or 0.0) + except (TypeError, ValueError): + return 0.0 + return max(0.0, paused) if math.isfinite(paused) else 0.0 + + +def execution_elapsed_seconds(meta: dict, now: float) -> float: + """Execution time of one task: wall clock minus the quota-wait union minus + the separate budget-paused interval. ``started_at`` is never moved, so every + finite-lifetime consumer (supervisor timeouts, owner stop, the exact-pause + grant, the live wait controls) subtracts the same two carriers.""" + try: + started = float(meta.get("started_at") or 0.0) + except (TypeError, ValueError): + return 0.0 + if started <= 0 or not math.isfinite(started): + return 0.0 + return max(0.0, now - started - quota_waited_seconds(meta, now) - budget_paused_seconds(meta)) + + _CURRENT: contextvars.ContextVar[TaskModelWait | None] = contextvars.ContextVar( "ouroboros_model_wait", default=None) _REPREPARE: contextvars.ContextVar[dict[str, Callable] | None] = contextvars.ContextVar( @@ -246,6 +271,11 @@ class TaskModelWait: self.waits: dict[str, dict] = {} self.clocks: dict[str, _QuotaClock] = {"": _QuotaClock()} self.started_monotonic = time.monotonic() + # The SAME budget-paused carrier the queue puts on the RUNNING row + # (#1196): a resumed task's finite lifetime excludes the paused wall + # time while its original start stays where it was. + self.budget_paused_sec = budget_paused_seconds( + task.get("_budget_pause_resume") if isinstance(task.get("_budget_pause_resume"), dict) else {}) self.revision = 0 self.auto_continue: dict[str, bool] = {} self.seen_controls: set[str] = set() @@ -270,6 +300,12 @@ class TaskModelWait: key: {k: copy.deepcopy(v) for k, v in row.items() if not k.startswith("_")} for key, row in self.waits.items()}} + def executed_seconds(self, *, now: float | None = None) -> float: + """Live execution time: elapsed minus the quota union minus budget pause.""" + stamp = time.monotonic() if now is None else now + return max(0.0, stamp - self.started_monotonic + - self.paused_seconds(now=stamp) - self.budget_paused_sec) + def execution_window_remaining(self) -> float | None: """A custom live owner supplies its own clock and a task without an absolute lifetime has none; None invents no deadline, and 0.0 means the window is spent.""" @@ -279,7 +315,7 @@ class TaskModelWait: ceiling = get_task_abs_ceiling_sec() if ceiling is None: return None - return max(0.0, ceiling - (time.monotonic() - self.started_monotonic - self.paused_seconds())) + return max(0.0, ceiling - self.executed_seconds()) def quota_clock_snapshot(self) -> dict: """The same task-wide clock fact for live publication and continuation.""" @@ -289,14 +325,29 @@ class TaskModelWait: "observed_at": time.time(), "active": bool(clock.active)} def continuation_state(self) -> dict: - """Keep completed-call choices and accrued quota time, never live waiters.""" + """Keep completed-call choices, accrued quota time and the paused carrier. + + The budget-paused interval (#1196) rides EVERY same-ID continuation, not + only a budget one: a task that was paused and later parks in an owner + wait must resume on the same execution clock, or its planned restart + would count the paused wall time as execution and shrink the finite + lifetime it already spent (F5). Live waiters are never carried. + """ with self.lock: return {"overrides": copy.deepcopy(self.overrides), "auto_continue": dict(self.auto_continue), + "budget_paused_sec": self.budget_paused_sec, "quota_clock": {**self.quota_clock_snapshot(), "active": False}} - def restore_continuation(self, saved: dict, *, started_at: float | None) -> None: - """Rebind one fresh task owner before Runtime context or new model work.""" + def restore_continuation(self, saved: dict, *, started_at: float | None, + budget_paused_sec: float | None = None) -> None: + """Rebind one fresh task owner before Runtime context or new model work. + + ``started_at`` is the ORIGINAL start, so the restored wall clock also + spans any budget pause; ``budget_paused_sec`` carries that interval + separately (#1196) and is subtracted by every finite-lifetime read. + ``None`` keeps whatever the task row already supplied. + """ with self.lock: self.overrides = copy.deepcopy(saved.get("overrides") or {}) self.auto_continue = dict(saved.get("auto_continue") or {}) @@ -304,6 +355,14 @@ class TaskModelWait: self.revision = int(clock.get("revision") or 0) elapsed = quota_waited_seconds({"model_wait_quota_clock": clock}, time.time()) self.clocks = {"": _QuotaClock(elapsed=elapsed)} + if budget_paused_sec is None: + # The carrier the serializer saved: an owner-wait continuation of + # a task that had been budget-paused keeps the SAME paused + # interval across its planned restart (#1196, F5). + budget_paused_sec = saved.get("budget_paused_sec") + if budget_paused_sec is not None: + self.budget_paused_sec = budget_paused_seconds( + {"budget_paused_sec": budget_paused_sec}) if started_at: self.started_monotonic = time.monotonic() - max(0.0, time.time() - float(started_at)) @@ -372,9 +431,8 @@ class TaskModelWait: return "execution_deadline" if self.owner_control is not None: return self.owner_control() - elapsed = time.monotonic() - self.started_monotonic - self.paused_seconds() - ceiling = get_task_abs_ceiling_sec() - if ceiling is not None and elapsed >= ceiling: + ceiling = get_task_abs_ceiling_sec() # None = unlimited; 0 = exhausted, never unlimited + if ceiling is not None and self.executed_seconds() >= ceiling: return "absolute_ceiling" # The loop owns delivery. A private seen copy leaves that ownership # intact and excludes an already-drained or superseded stop control. diff --git a/ouroboros/owner_wait.py b/ouroboros/owner_wait.py index 82433710b..55b334ee3 100644 --- a/ouroboros/owner_wait.py +++ b/ouroboros/owner_wait.py @@ -142,6 +142,10 @@ def checkpoint_owner_wait(ctx: Any, messages: list, trace: dict, usage: dict, "execution_drive_root": str(ctx.drive_root), "started_at": getattr(ctx, "task_started_at", None), "model_wait_quota_clock": model_state.get("quota_clock", {}), + # The SAME two carriers every finite-lifetime reader subtracts: the quota + # union above and the budget-paused interval (#1196, F5). Without it the + # row's own lifetime check would count a pause as execution. + "budget_paused_sec": float(model_state.get("budget_paused_sec") or 0.0), } @@ -176,7 +180,7 @@ def restore_owner_wait_allowed(root: Any, task: dict) -> bool: from ouroboros.deadline_utils import parse_deadline_ts, utc_now from ouroboros.delegate_recovery import _ack_direct_exec_successor, _read_restart_transaction from ouroboros.config import get_task_abs_ceiling_sec - from ouroboros.model_wait import quota_waited_seconds + from ouroboros.model_wait import execution_elapsed_seconds import time handoff = task.get("_owner_wait_resume") @@ -200,10 +204,23 @@ def restore_owner_wait_allowed(root: Any, task: dict) -> bool: deadline = parse_deadline_ts(task.get("deadline_at") or (task.get("task_contract") or {}).get("deadline_at")) if deadline is not None and deadline <= utc_now(): return False - started = float(handoff.get("started_at") or 0) + started = float(handoff.get("started_at") or wait.get("started_at") or 0) now = time.time() ceiling = get_task_abs_ceiling_sec() # None = no lifetime bound to have outlived - if started and ceiling is not None and now - started - quota_waited_seconds(wait, now) >= ceiling: + # ONE shared clock (``model_wait.execution_elapsed_seconds``): wall time minus + # the quota union minus the budget-paused carrier. A task that was paused and + # then parked in an owner wait must not have that paused time charged to its + # finite lifetime by this reader alone (#1196, F5). + # The durable row is the authority whenever it carries the field (0.0 included); + # the handoff is the fallback for a row written before it existed. + paused_carrier = wait.get("budget_paused_sec") + if paused_carrier is None: + paused_carrier = handoff.get("budget_paused_sec") or 0.0 + executed = execution_elapsed_seconds( + {"started_at": started, + "model_wait_quota_clock": wait.get("model_wait_quota_clock") or {}, + "budget_paused_sec": paused_carrier}, now) + if started and ceiling is not None and executed >= ceiling: return False read_actor_source_bytes(root, task_id, wait["source_ref"]) return True diff --git a/ouroboros/safety.py b/ouroboros/safety.py index 841f34cec..4af5723a6 100644 --- a/ouroboros/safety.py +++ b/ouroboros/safety.py @@ -125,6 +125,12 @@ TOOL_POLICY: Dict[str, str] = { # carries no authority the task lacks (same reasoning as the verbs above). "delegate_answer": POLICY_SKIP, "cancel_task": POLICY_SKIP, + # The other half of the same nanny authority (#1196, owner Q9): selecting ONE + # of this task's OWN budget-paused descendants to continue under its same id. + # The tool only REQUESTS; the supervisor re-checks lineage, the root's live + # owner Resume grant, money, Stop/cancel, deadline and lifetime through the + # seam the owner's own Resume uses, so it adds no reach the task lacks. + "resume_child_task": POLICY_SKIP, # Parent's explicit decision to abandon a child result: stamps parent_decision + # records the reason on the tree ledger; tree-scoped, no external effect (like cancel_task). "discard_child_result": POLICY_SKIP, diff --git a/ouroboros/size_ratchet_manifest.py b/ouroboros/size_ratchet_manifest.py index 38a075d3f..cae54be80 100644 --- a/ouroboros/size_ratchet_manifest.py +++ b/ouroboros/size_ratchet_manifest.py @@ -108,6 +108,7 @@ BAND_PATHS = { "devtools/benchmarks/terminal_bench/run_tb.py": None, "ouroboros/agent.py": "Subagent message identity now lives in a shared helper; keep agent.py below the giant-file threshold rather than re-expanding it.", "ouroboros/agent_task_pipeline.py": "Shrank INTO the band from 1599 lines: the post-task synthesis family moved byte-preserving into ouroboros/post_task_synthesis.py (D01 lane); no new content was added.", + "ouroboros/budget_pause.py": "Entered the band from 990 lines (#1196): the direct-turn pause, the typed restore refusal beside the boolean gate and the authoritative Q10 refresh belong with the one pause/resume owner they extend; the supervisor-side grant lifecycle moved out to supervisor/budget_resume.py instead of growing here.", "ouroboros/cancel_intents.py": "Entered the band from 929 lines: reciprocal timeout-retry lineage validation and physical-leaf/logical-root aliasing stay with the durable cancel-intent mutation authority so Stop-now hardens the same request across retry races.", "ouroboros/capability_evidence.py": "Grew INTO the band by the #284 fix: a fresh exact-model density witness may honestly undercut the cold floor \u2014 evidence logic belongs beside the witness store it reads.", "ouroboros/claudexor_daemon.py": "Installation daemon lifecycle owns marker and authenticated endpoint stop authority, confirmed self-started handles, and duplicate-start refusal; process signal and ledger mechanics remain in process_custody. No new lifecycle store or scheduler.", diff --git a/ouroboros/subagent_runtime.py b/ouroboros/subagent_runtime.py index ed8cd5c14..7a004764d 100644 --- a/ouroboros/subagent_runtime.py +++ b/ouroboros/subagent_runtime.py @@ -807,7 +807,7 @@ def exact_start(ctx: Any, prompt: str, spec: Optional[dict[str, Any]] = None) -> _canonical_work_order_fingerprint=canonical_work_order_fingerprint, _work_order_source_request=work_order_source_request, _coordination_context=coordination_context, - **{key: options.pop(key) for key in ("directory_strategy", "scope_paths") + **{key: options.pop(key) for key in ("directory_strategy", "scope_paths", "continue_from") if key in options}, ) # Every configured-session start lands here — the host's pre-start diff --git a/ouroboros/tool_capabilities.py b/ouroboros/tool_capabilities.py index f296b8bc9..ccaba293c 100644 --- a/ouroboros/tool_capabilities.py +++ b/ouroboros/tool_capabilities.py @@ -38,6 +38,11 @@ CORE_TOOL_NAMES: frozenset[str] = frozenset({ # D#7 soft-join child controls (siblings of steer_task): inspect/decide a child's fate # before finalizing (peek = pure read, discard = explicit abandon, cancel = real stop). "cancel_task", "peek_task", "discard_child_result", "override_delegation_constraint", + # The same family (#1196, owner Q9): a resumed parent selects each of its OWN + # budget-paused children explicitly. It belongs in the round-one envelope for + # the reason cancel_task does — a task that just came back from a pause must + # not need an enable_tools detour to continue the children it still needs. + "resume_child_task", # Task-tree coordination must be in the round-one envelope so a parent can publish the # shared frame BEFORE fanning out interdependent children (no enable_tools detour). "tree_note", "tree_read", @@ -83,6 +88,10 @@ LOCAL_READONLY_SUBAGENT_TOOL_NAMES: frozenset[str] = frozenset({ "chat_history", "recent_tasks", "get_task_result", "wait_task", "wait_tasks", "escalate", "forward_to_worker", "peek_task", "cancel_task", "discard_child_result", + # A recursive parent selects its OWN budget-paused children (#1196, Q9); the + # supervisor checks lineage and the root's live grant, so no authority the + # child lacks is widened — the same reasoning as cancel_task above. + "resume_child_task", "schedule_subagent", # Reading the schedule table is research: a child asked about what this mind # has standing can see it. The tool's own authority check refuses every @@ -128,6 +137,7 @@ ACTING_SUBAGENT_TOOL_NAMES: frozenset[str] = frozenset({ "schedule_subagent", "wait_task", "wait_tasks", "get_task_result", "escalate", "forward_to_worker", "peek_task", "cancel_task", "discard_child_result", + "resume_child_task", "verify_and_record", "knowledge_read", "knowledge_list", "tree_note", "tree_read", "override_delegation_constraint", @@ -314,7 +324,7 @@ OBSERVE_WORLD_MUTATION_TOOLS: frozenset[str] = frozenset({ # starting or steering work (steer_task stays: the nanny of a running campaign) "promote_chat_to_task", "schedule_subagent", "schedule_followup", "plan_task", "route_to_project", "ensure_project_scope", "delegate_start", "initiate_presence", - "cancel_task", "override_delegation_constraint", "request_deep_self_review", + "cancel_task", "resume_child_task", "override_delegation_constraint", "request_deep_self_review", # writing files, running processes, integrating patches "write_file", "edit_text", "apply_patch", "edit_batch", "run_command", "run_script", "start_service", "stop_service", "verify_and_record", diff --git a/ouroboros/tools/delegate.py b/ouroboros/tools/delegate.py index 1aadfd4f7..d531bda9c 100644 --- a/ouroboros/tools/delegate.py +++ b/ouroboros/tools/delegate.py @@ -33,7 +33,7 @@ import logging import time import uuid from pathlib import Path -from typing import TYPE_CHECKING, Any, Dict, List, Optional +from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple, NamedTuple from ouroboros import delegate_custody as custody from ouroboros import delegate_progress as progress @@ -324,10 +324,35 @@ def _processing_start_request(request, actor, gateway, route): "reason": "submitted" if "processingPreference" in request else "processing_not_submitted"} +def _start_argument_refusal(ctx: ToolContext, text: str, selector_root: str, retry_of: Any, + bucket: Any, skill_name: Any, continue_from: Any) -> Tuple[str, Optional[ToolResult]]: + """``(continuation_token, refusal)``: every refusal a start's ARGUMENTS earn before + the daemon is touched, in their historical order. Each is a definite no-run: an + empty prompt, a malformed exact-resource selector, a deadline already behind the + nanny (``definitely_unrun`` = the producer's own no-run verdict, P2), and the + continuation selector shapes one call cannot combine.""" + from ouroboros.delegate_continuation import selector_refusal + + if not text.strip(): + return "", _fail("delegate_start", "empty_prompt", "prompt is required") + refusal = _payload_selector_refusal(selector_root, retry_of, bucket, skill_name) + if refusal: + return "", refusal + if deadline_expired(ctx): + return "", _fail( + "delegate_start", "task_deadline_expired", + "This task's deadline has already passed, so a delegated run started now " + "would outlive it by design. Finalize with what you have — do not start " + "new work a deadline has already closed.", definitely_unrun=True, + ) + return selector_refusal(continue_from, retry_of, selector_root) + + def _delegate_start(ctx: ToolContext, prompt: str, max_seconds: Optional[int] = None, retry_of: Optional[str] = None, root: Optional[str] = None, bucket: Optional[str] = None, skill_name: Optional[str] = None, directory_strategy: Optional[str] = None, scope_paths: Optional[list] = None, + continue_from: Optional[str] = None, _resolved_binding: Any = None, _canonical_work_order_fingerprint: str = "", _work_order_source_request: Any = None, @@ -338,27 +363,26 @@ def _delegate_start(ctx: ToolContext, prompt: str, max_seconds: Optional[int] = from ouroboros.subagents import delegated_execution_workspace_root, resolve_subagent_executor, route_health text = str(prompt or "") - if not text.strip(): - return _fail("delegate_start", "empty_prompt", "prompt is required") selector_root = str(root or "").strip() - selector_refusal = _payload_selector_refusal(selector_root, retry_of, bucket, skill_name) - if selector_refusal: - return selector_refusal - if deadline_expired(ctx): - # EXPIRED pre-daemon; definitely_unrun = the producer's own no-run verdict (P2). - return _fail( - "delegate_start", "task_deadline_expired", - "This task's deadline has already passed, so a delegated run started now " - "would outlive it by design. Finalize with what you have — do not start " - "new work a deadline has already closed.", definitely_unrun=True, - ) + continuation_token, argument_refusal = _start_argument_refusal( + ctx, text, selector_root, retry_of, bucket, skill_name, continue_from) + if argument_refusal: + return argument_refusal + seconds_basis = "" + if not str(retry_of or "").strip(): + # Decided BEFORE the daemon is touched: a spent lifetime or a sub-second + # deadline is a definite no-run, and the basis rides both custody rows. + bound = bounded_max_seconds(ctx, max_seconds) + if bound.refusal_code: + return _fail("delegate_start", bound.refusal_code, bound.refusal_detail, definitely_unrun=True) + seconds_basis = bound.basis drive = custody.custody_root(ctx) owned_project_id, project_persistent = "", False invocation_id = snapshot_id = baseline_sha = target_root = authority_source = "" binding_fingerprint = "" processing_info: Dict[str, Any] = {} - resource_ref, directory_options = {}, {} + resource_ref, directory_options, continuation = {}, {}, {} retry_token = str(retry_of or "").strip() source_binding = prepare_work_order_start_binding( ctx, drive, retry_token, _canonical_work_order_fingerprint, text, @@ -387,6 +411,9 @@ def _delegate_start(ctx: ToolContext, prompt: str, max_seconds: Optional[int] = project_persistent, seconds, snapshot_id, target_root, baseline_sha, authority_source, resource_ref, processing_info, binding_fingerprint) = binding invocation_id = retry_token + # A replay presents the recorded body byte-identically, so its cap + # basis is the recorded one too — never re-derived from today's clocks. + seconds_basis = str((custody.invocation_record(drive, retry_token) or {}).get("max_seconds_basis") or "") if directory_strategy is not None or scope_paths is not None: return _fail("delegate_start", "retry_selector_conflict", "A retry replays its recorded directory strategy and scope; omit new geometry arguments.") @@ -453,6 +480,16 @@ def _delegate_start(ctx: ToolContext, prompt: str, max_seconds: Optional[int] = return root_error invocation_id = custody.new_invocation_id() root = record_auth["target_root"] + if continuation_token: # #1196: gated from durable custody, before any snapshot exists + from ouroboros.delegate_continuation import start_binding + + continuation, continuation_block, refusal = start_binding( + ctx, drive, continuation_token, actor=actor, route=route, authority=authority, + target_root=str(record_auth.get("target_root") or ""), + canonical_work_order_fingerprint=str(_canonical_work_order_fingerprint or "")) + if refusal: + return refusal + instructions += continuation_block if authority.access in SESSION_ACCESS_PROFILES: target_root = record_auth["target_root"] authority_source = record_auth["source"] @@ -494,7 +531,7 @@ def _delegate_start(ctx: ToolContext, prompt: str, max_seconds: Optional[int] = project_persistent = True if authority.access == "full": gateway.ensure_full_access(scope_root) - seconds = _bounded_max_seconds(ctx, max_seconds) + seconds = bound.seconds request_body = _start_request(ctx, route, authority, scope_root, text, seconds, instructions, execution_root, **({"directory_options": directory_options} if directory_options else {})) @@ -510,7 +547,7 @@ def _delegate_start(ctx: ToolContext, prompt: str, max_seconds: Optional[int] = actor_ctx=ctx, enforce_actor_idle=not recovering, run_id="", task_id=str(getattr(ctx, "task_id", "") or ""), idempotency_key=key, invocation_id=invocation_id, - max_seconds=seconds, request=request_body, project_id=project_id, + max_seconds=seconds, max_seconds_basis=seconds_basis, request=request_body, project_id=project_id, project_owned=bool(owned_project_id), project_persistent=project_persistent, route=route.route_id, root_task_id=str(lineage.get("root_task_id") or ""), parent_task_id=str(lineage.get("parent_task_id") or ""), snapshot_id=snapshot_id, execution_root=(root if snapshot_id or resource_ref.get("strategy") == "direct" else ""), @@ -598,6 +635,7 @@ def _delegate_start(ctx: ToolContext, prompt: str, max_seconds: Optional[int] = drive, run_id, ctx, route, authority, key=key, access=access, root=root, seconds=seconds, invocation_id=invocation_id, project_id=project_id, + continuation_of=str(continuation.get("continuation_of") or ""), project_owned=bool(owned_project_id), project_persistent=project_persistent, selected_subagent_id=selected_subagent_id, config_fingerprint=config_fingerprint, work_order_fingerprint=work_order_fingerprint, @@ -608,6 +646,7 @@ def _delegate_start(ctx: ToolContext, prompt: str, max_seconds: Optional[int] = authority_source=authority_source, resource_ref=resource_ref, processing=processing_info, capture_mode=("engine_directory" if resource_ref.get("workspace_kind") == "directory" else _CAPTURE_DELEGATED_SNAPSHOT if snapshot_id else ""), + max_seconds_basis=seconds_basis, ) from ouroboros.tools.control import maybe_emit_delegated_run_fanout maybe_emit_delegated_run_fanout(ctx, run_id=run_id, route_id=route.route_id, objective=text, durable=durable) @@ -618,13 +657,16 @@ def _delegate_start(ctx: ToolContext, prompt: str, max_seconds: Optional[int] = baseline_sha=baseline_sha, resource_ref=resource_ref, processing=processing_info, + continuation=continuation, + max_seconds=seconds, max_seconds_basis=seconds_basis, engine_version=str(getattr(gateway, "engine_version", "") or "")) def _started_payload(handle: Dict[str, Any], run_id: str, route: Any, access: str, authority: "DelegatedRunShape", root: str, *, durable: bool, recovering: bool, invocation_id: str, snapshot_id: str, target_root: str, - baseline_sha: str, engine_version: str = "", resource_ref=None, processing=None) -> ToolResult: + baseline_sha: str, engine_version: str = "", resource_ref=None, processing=None, + continuation=None, max_seconds: int = 0, max_seconds_basis: str = "") -> ToolResult: """The one author of delegate_start's started result (note + payload). The AUTHORITY guidance and the CUSTODY warning are independent facts about the same @@ -671,12 +713,21 @@ def _started_payload(handle: Dict[str, Any], run_id: str, route: Any, access: st "root": root, "custody_durable": durable, "invocation_id": str(invocation_id or ""), + # The cap and HOW it was decided (#1196): only a `requested` cap's expiry + # is a finite-leaf expiry a later continue_from may follow. + "max_seconds": int(max_seconds or 0), + "max_seconds_basis": str(max_seconds_basis or ""), "note": note, } if not durable: payload["pending_invocation_id"] = str(invocation_id or "") if processing: payload["processing"] = processing + if continuation: + # The binding facts, stated where the nanny reads them: which settled run + # this continues, its confirmed cause, its explicit disposition, and that + # NO session state was transferred (#1196). + payload["continuation"] = dict(continuation) if snapshot_id: # The C1 binding, stated where the nanny can read it: the run edits the # EXECUTION snapshot; the authority target receives nothing until apply. @@ -760,38 +811,126 @@ def _retire_orphaned_registration(ctx: ToolContext, gateway: Any, project_id: st "project_retention_reason": "start_outcome_unknown_run_may_exist"} -def _bounded_max_seconds(ctx: ToolContext, requested: Optional[int]) -> int: - """Narrow-only: the delegated run may never outlive the nanny's own deadline. +class MaxSecondsBound(NamedTuple): + """One decided ``maxSeconds``: the seconds, HOW they were decided + (``delegate_registration_policy.CAP_BASIS_*``), or the typed refusal.""" - A caller must ask ``deadline_expired`` FIRST: an expired deadline cannot produce an - honest bound at all, and this function's fallback is for a nanny that has NO - deadline, never for one whose deadline is behind it. + seconds: int + basis: str + refusal_code: str = "" + refusal_detail: str = "" + + +def _lifetime_remaining_sec(ctx: ToolContext) -> Optional[float]: + """Seconds of the task's FINITE lifetime still unspent, or ``None`` (unlimited). + + Cumulative EXECUTION time decides, never a reset clock: the live model-wait + owner's window (elapsed minus the quota union minus the budget-paused + carrier) when this task's is bound, else the same arithmetic over the + original ``task_started_at`` and the paused carrier the resume handed over + (quota waits unknown here count as execution — the narrower direction). + No recorded start uses the operation-bounded ceiling; a failed clock read + raises for a typed unknown-lifetime refusal, never a fresh finite window. """ - from ouroboros.deadline_utils import deadline_remaining_sec + from ouroboros.config import get_task_abs_ceiling_sec - remaining = int(max(0.0, deadline_remaining_sec(ctx))) + ceiling = get_task_abs_ceiling_sec() + if ceiling is None: + return None try: - asked = int(requested) if requested is not None else 0 + from ouroboros.model_wait import current_model_wait, execution_elapsed_seconds + + waiter = current_model_wait() + if (waiter is not None + and str(getattr(waiter, "task_id", "") or "") == str(getattr(ctx, "task_id", "") or "")): + remaining = waiter.execution_window_remaining() + if remaining is not None: + return float(remaining) + started = getattr(ctx, "task_started_at", None) + if started: + paused = getattr(ctx, "_budget_paused_sec", None) + if paused is None: + paused = (getattr(ctx, "budget_pause_resume", None) or {}).get("paused_duration_sec") + executed = execution_elapsed_seconds( + {"started_at": float(started), "budget_paused_sec": float(paused or 0.0), + "model_wait_quota_clock": {}}, time.time()) + return max(0.0, float(ceiling) - executed) + except Exception as exc: + raise ValueError("task_lifetime_unknown") from exc + return float(ceiling) + + +def bounded_max_seconds(ctx: ToolContext, requested: Optional[int]) -> MaxSecondsBound: + """Narrow-only: the delegated run may never outlive the nanny's own deadline + or its finite lifetime, and the decision is RECORDED beside the number. + + A caller must ask ``deadline_expired`` FIRST: an expired deadline cannot + produce an honest bound at all. Less than one second of deadline or of + lifetime is refused typed (``task_deadline_too_close`` / + ``task_lifetime_exhausted``): ``int()`` would truncate it to 0, which the + fallback branch read as "no bound" and answered with hours of delegated + work. An explicit ask is ``requested`` unless the deadline, the lifetime or + the engine's schema bound narrowed it (each named); an omitted ask derives + from the tighter of deadline and lifetime, else the operation window. + """ + from ouroboros.config import operation_window_sec + from ouroboros.deadline_utils import deadline_remaining_sec, has_deadline + from ouroboros.delegate_registration_policy import ( + CAP_BASIS_DEADLINE_DERIVED, CAP_BASIS_LIFETIME_DERIVED, CAP_BASIS_OPERATION_WINDOW, + CAP_BASIS_REQUESTED, CAP_BASIS_REQUESTED_CLAMPED_DEADLINE, CAP_BASIS_REQUESTED_CLAMPED_LIFETIME, + CAP_BASIS_REQUESTED_CLAMPED_SCHEMA, + ) + + try: + asked = max(0, int(requested)) if requested is not None else 0 except (TypeError, ValueError): asked = 0 - candidates = [value for value in (asked, remaining) if value > 0] + deadline_left: Optional[float] = None + if has_deadline(ctx): + deadline_left = float(deadline_remaining_sec(ctx)) + if deadline_left < 1.0: + return MaxSecondsBound(0, "", "task_deadline_too_close", + "Less than one second of this task's deadline remains: no delegated run " + "can be started under it. Finalize with what you have.") + try: + lifetime_left = _lifetime_remaining_sec(ctx) + except ValueError: + return MaxSecondsBound(0, "", "task_lifetime_unknown", + "This task's remaining finite lifetime could not be established; no delegated run started.") + if lifetime_left is not None and lifetime_left < 1.0: + return MaxSecondsBound(0, "", "task_lifetime_exhausted", + "This task's finite lifetime is spent (cumulative execution time, the " + "paused interval excluded): no delegated run can be started. Finalize.") + if asked > 0: + seconds, basis = asked, CAP_BASIS_REQUESTED + if deadline_left is not None and int(deadline_left) < seconds: + seconds, basis = int(deadline_left), CAP_BASIS_REQUESTED_CLAMPED_DEADLINE + if lifetime_left is not None and int(lifetime_left) < seconds: + seconds, basis = int(lifetime_left), CAP_BASIS_REQUESTED_CLAMPED_LIFETIME + if seconds > _CLAUDEXOR_MAX_SECONDS: + # Clamp HERE too, not only on the fallback below: `max_seconds` is a model-supplied + # tool argument with no maximum in its schema (control.ts `.max(604_800)`). + seconds, basis = _CLAUDEXOR_MAX_SECONDS, CAP_BASIS_REQUESTED_CLAMPED_SCHEMA + return MaxSecondsBound(max(1, seconds), basis) + candidates = [] + if deadline_left is not None: + candidates.append((int(deadline_left), CAP_BASIS_DEADLINE_DERIVED)) + if lifetime_left is not None: + candidates.append((int(lifetime_left), CAP_BASIS_LIFETIME_DERIVED)) if candidates: - # Clamp HERE too, not only on the fallback below: `max_seconds` is a model-supplied - # tool argument with no maximum in its schema, so an explicit ask sailed past the - # bound the fallback branch was careful about — the same defect, one branch over. - return min(_CLAUDEXOR_MAX_SECONDS, min(candidates)) - # No positive bound is knowable: either the nanny has no deadline, or its deadline - # has already passed. Omitting `maxSeconds` — the old behavior — handed the run - # Claudexor's 7-day schema bound; the cap is damage limitation, and custody (the + seconds, basis = min(candidates, key=lambda item: item[0]) + return MaxSecondsBound(max(1, min(_CLAUDEXOR_MAX_SECONDS, seconds)), basis) + # Neither a deadline nor a finite lifetime: the finite operation window. + # Omitting `maxSeconds` — the old behavior — handed the run Claudexor's + # 7-day schema bound; the cap is damage limitation, and custody (the # durable start row plus reconciliation) is what actually stops an orphan. - from ouroboros.config import get_task_abs_ceiling_sec, operation_window_sec + return MaxSecondsBound(min(_CLAUDEXOR_MAX_SECONDS, int(operation_window_sec(None))), CAP_BASIS_OPERATION_WINDOW) - # The task's operation window: its finite absolute lifetime, else the finite operation - # fallback (a task without a lifetime bound never sends an unbounded run). Claudexor - # bounds maxSeconds at 7 days (control.ts `.max(604_800)`), and the task ceiling clamps - # only from BELOW — an owner who raises it past a week would make every deadline-less - # start send an out-of-schema value. - return min(_CLAUDEXOR_MAX_SECONDS, int(operation_window_sec(get_task_abs_ceiling_sec()))) + +def _bounded_max_seconds(ctx: ToolContext, requested: Optional[int]) -> int: + """The seconds of ``bounded_max_seconds`` (0 on a typed refusal) — the + historical integer view the tests and the wait clamp still address.""" + return bounded_max_seconds(ctx, requested).seconds def _halt_breached_run(ctx: ToolContext, gateway: Any, entry: _RunCustody, @@ -1206,7 +1345,9 @@ def get_tools() -> List[ToolEntry]: "fresh start requires subagent_id. In a configured session the host already STARTED the exact " "leaf before your first round (the startup receipt carries its run id): never start a duplicate — " "supervise it; a replacement delegate_start(prompt='') is legal only after verified cancellation/" - "terminal settlement or a typed refusal proving no run exists. Recovery retries use retry_of without a new selector." + "terminal settlement or a typed refusal proving no run exists. Recovery retries use retry_of without a new selector. " + "A run the engine cancelled at its wall-clock cap is continued explicitly with continue_from once its " + "result is read and its patch disposed (see that argument)." " This ordinary call requests no extra Claudexor review panel; new ordinary " "runs on engine 3.9.8+ default to no panel. The started receipt names the serving " "engine_version; an older engine or a recovered historical run may retain its " @@ -1247,6 +1388,15 @@ def get_tools() -> List[ToolEntry]: "tight cap for what feels like a quick edit. While delegate_wait shows " "an advancing cursor the run is WORKING, and it enforces this cap " "itself — cancelling a progressing run discards the whole run's spend."}, + "continue_from": {"type": "string", "description": + "EXPLICIT continuation of ONE of your own settled runs that the engine cancelled " + "at its wall-clock cap (delegate_wait terminal: state=cancelled, " + "outcome_facts.reason=wall_clock_exceeded). Admitted only after that run's result " + "was read and its captured patch explicitly applied or rejected, on the same " + "actor/route and the same workspace authority; refused typed for any other ending " + "(deadline, Stop/Panic, failure, unknown). Starts a NEW run with a NEW cap: put the " + "REMAINING work in prompt — the prior result and disposition are your evidence of " + "what is done; nothing of the old session is transferred. Never combine with retry_of."}, "retry_of": {"type": "string", "description": "EXPLICIT retry token: the pending_invocation_id from a start whose " "outcome was unknown (transport failure, lost response). Replays THAT " diff --git a/ouroboros/tools/join_ledger.py b/ouroboros/tools/join_ledger.py index 7dffd74ee..b54804cc2 100644 --- a/ouroboros/tools/join_ledger.py +++ b/ouroboros/tools/join_ledger.py @@ -400,10 +400,11 @@ def _status_drive_root(ctx: ToolContext) -> Path: return Path(str(metadata.get("budget_drive_root") or getattr(ctx, "budget_drive_root", "") or ctx.drive_root)) -def _is_own_child(ctx: ToolContext, status_drive_root: Path, tid: str) -> bool: +def _is_own_child(ctx: ToolContext, status_drive_root: Path, tid: str, *, root_tree: bool = False) -> bool: """True if ``tid`` is a DIRECT child of the CURRENT task (D#7 safety): a parent decision may only describe the caller's OWN children, never an unrelated parent's - join ledger. Fail-CLOSED — any error returns False.""" + join ledger. Resume alone may also select the caller's root tree, using + stored lineage rather than a supplied root id. Fail-CLOSED on any error.""" try: from ouroboros.task_status import find_child_tasks @@ -412,7 +413,8 @@ def _is_own_child(ctx: ToolContext, status_drive_root: Path, tid: str) -> bool: if not my_id or not tid: return False children = find_child_tasks( - Path(status_drive_root), parent_task_id=my_id, root_task_id="", exclude_task_id=my_id + Path(status_drive_root), parent_task_id=my_id, root_task_id=my_id if root_tree else "", + exclude_task_id=my_id, materialize_artifacts=False, ) return any(str(c.get("task_id") or c.get("id") or "") == tid for c in children) except Exception: @@ -611,7 +613,7 @@ def _override_delegation_constraint(ctx: ToolContext, constraint_id: str, reason def _resume_child_task(ctx: ToolContext, task_id: str, reason: str = "") -> str: - """Owner Q9 (#1196): explicitly select ONE budget-paused descendant to continue. + """Owner Q9: select ONE paused root descendant or intermediate parent's direct child. The request rides the existing worker->supervisor control channel; the supervisor validates lineage and grants through the ONE resume seam, and @@ -624,7 +626,7 @@ def _resume_child_task(ctx: ToolContext, task_id: str, reason: str = "") -> str: return f"⚠️ TOOL_ARG_ERROR (resume_child_task): {exc}" reason_text = _clip(" ".join(str(reason or "").split()), 500) status_drive_root = _status_drive_root(ctx) - own = _is_own_child(ctx, status_drive_root, tid) + own = _is_own_child(ctx, status_drive_root, tid, root_tree=True) requester = str(getattr(ctx, "task_id", "") or "") if not own: return _publish_tool_result(ctx, ToolResult(status="blocked", code="ACCESS_BLOCKED", text=( @@ -777,8 +779,9 @@ def get_tools() -> list[ToolEntry]: return [ ToolEntry("resume_child_task", { "name": "resume_child_task", - "description": "Owner Q9 (#1196): after YOUR OWN task was resumed from a budget pause, " - "select ONE of your budget-paused descendants to continue under its same task id. " + "description": "After the owner resumes your root from a budget pause, select ONE paused task: " + "a root may select any of its descendants; an intermediate parent may select its " + "own direct children. Continuation keeps the same task id. " "Nothing resumes automatically and no fan-out happens: you name each child you still " "need, with a reason. The supervisor validates money, Stop/cancel intent, deadline and " "finite lifetime through the same seam the owner's Resume uses; the typed outcome is " diff --git a/ouroboros/usage_accounting.py b/ouroboros/usage_accounting.py index 77f6c010d..3e72f6669 100644 --- a/ouroboros/usage_accounting.py +++ b/ouroboros/usage_accounting.py @@ -104,6 +104,8 @@ def _stash_root_accounting( accounted_usd: Optional[float], root_limit_usd: Optional[float], reservation: Optional[Dict[str, Any]] = None, + *, + integrity_degraded: bool = False, ) -> None: """Refresh the process-local root snapshot. ``reservation`` is the identity of a row this call has just APPENDED (attempt id, task, category, review @@ -130,6 +132,10 @@ def _stash_root_accounting( _ROOT_ACCOUNTING_TELEMETRY[root_task_id] = { "accounted_usd": None if accounted_usd is None else float(accounted_usd), "root_limit_usd": None if root_limit_usd is None else float(root_limit_usd), + # The projection's own integrity verdict rides the snapshot (#1196): a + # money decision (the exact-pause grant, the Q10 refresh) refuses a + # degraded tree instead of reading its number as room. + "integrity_degraded": bool(integrity_degraded), "updated_monotonic": now, "reservations": kept, } @@ -172,6 +178,7 @@ def refresh_root_accounting( root_task_id, _number(projection.get("accounted_usd")), _number(projection.get("limit_usd")), + integrity_degraded=bool(projection.get("integrity_degraded")), ) return last_root_accounting(root_task_id) except Exception: diff --git a/supervisor/budget_resume.py b/supervisor/budget_resume.py new file mode 100644 index 000000000..d1a419de4 --- /dev/null +++ b/supervisor/budget_resume.py @@ -0,0 +1,530 @@ +"""Exact-continuation budget Resume grants and their revocation (#1196). + +The owner's explicit Resume of a task paused MID-RUN mints ONE single-use grant +bound to one pause id and one resume generation; the grant rides the queue row +as ``_budget_pause_resume`` until a worker consumes it, and returns to the +pause — never to a replay or a terminal — when money vanishes, a restart +intervenes, or its revocation cannot be written (then the row is HELD, typed, +with its pause marker retained). Split out of ``supervisor/queue_transitions.py`` +at that module's band ceiling: the grant lifecycle is one owner with its own +reason to change (owner Q7/Q9/Q10 semantics), and ``queue_transitions`` keeps +the general resume seam (``resume_budget_paused_task``) that calls into it. +Every call here runs with the queue lock held by that seam or by restore. +""" + +from __future__ import annotations + +import logging +import pathlib +import time +import uuid +from typing import Any, Dict, List, Optional + +from ouroboros.utils import utc_now_iso + +log = logging.getLogger(__name__) + + +def _queue_module(): + from supervisor import queue + + return queue + + +# A root-tree accounting snapshot older than this at grant time is the +# refresher's stale fallback (the ledger could not answer now), not authority. +_FRESH_ROOT_ACCOUNTING_MAX_AGE_SEC = 5.0 + + +def _root_budget_paused_locked(q: Any, root_task_id: str, *, except_task_id: str = "") -> bool: + """Whether the ROOT of a tree is itself still budget-paused (queue lock held). + + Owner Q9: a root's Resume makes its own budget-paused descendants ELIGIBLE; + a descendant cannot be resumed under a root that is still paused. + """ + root_task_id = str(root_task_id or "") + if not root_task_id or root_task_id == except_task_id: + return False + for row in q.PENDING: + if str(row.get("id") or "") == root_task_id and isinstance(row.get("_budget_pause"), dict): + return True + return False + + +def grant_exact_budget_resume(task: Dict[str, Any], pause: Dict[str, Any], + *, selected_by: str = "", + external: Optional[Dict[str, Any]] = None) -> Dict[str, Any]: + """Mint ONE pause/generation-bound grant under the queue lock. + + Refuse unknown/stale money, exhausted wallets, Stop, deadline, finite lifetime, + unreadable checkpoint or unsettled custody. The queue marker is a locator: + refresh it from a newer durable pause, never trust its stale identity. + ``paused_duration_sec`` carries the paused interval without moving the + original ``started_at`` or resetting the quota clock. ``external`` is THIS + grant's fresh custody observation, otherwise read here: request stops for + uncovered live runs; unknown/requested stops cannot license a second writer. + Model ``selected_by`` needs its root's live owner-derived grant (Q9). + Direct actors use the same seam after parking under their own id. + """ + from ouroboros.artifacts import read_actor_source_bytes + from ouroboros.budget_pause import ( + LIVE_PAUSE_STATES, STATE_PAUSED, STATE_RESUME_GRANTED, exact_pause_marker, + observe_task_runs, set_budget_pause, unsettled_external_runs, + ) + from ouroboros.cancel_intents import has_active_intent + from ouroboros.config import get_task_abs_ceiling_sec + from ouroboros.deadline_utils import parse_deadline_ts, utc_now + from ouroboros.model_wait import execution_elapsed_seconds + from ouroboros.task_results import _TRULY_TERMINAL_STATUSES, load_task_result + from supervisor.events_budget import ( + BUDGET_HOLD_KEY, HOLD_RESTART_REVOCATION_UNWRITTEN, HOLD_REVOCATION_UNWRITTEN, HOLD_MALFORMED_RESUME_IDENTITY, + hold_budget_row, live_root_resume_grant, hold_root_resume_descendants, + release_budget_hold, + ) + from supervisor.state import budget_remaining + + q = _queue_module() + task_id = str(task.get("id") or "") + checkpoint = pause.get("checkpoint") if isinstance(pause.get("checkpoint"), dict) else {} + pause_id = str(checkpoint.get("pause_id") or "") + if not pause_id.strip(): + return {"ok": False, "error": "malformed_pause_identity"} + result_root = pathlib.Path(task.get("budget_drive_root") or q.DRIVE_ROOT) + if any((pathlib.Path(q.DRIVE_ROOT) / "state" / name).exists() + for name in ("owner_restart_no_resume.flag", "panic_stop.flag")): + return {"ok": False, "error": "restart_no_resume", "action": "wait_or_cancel"} + try: + result_row = load_task_result(result_root, task_id, strict=True) or {} + except Exception: + return {"ok": False, "error": "pause_record_unreadable", "action": "cancel_or_new_run"} + row = result_row.get("budget_pause") if isinstance(result_row.get("budget_pause"), dict) else {} + if result_row.get("status") in _TRULY_TERMINAL_STATUSES: + return {"ok": False, "error": "task_terminal"} + if (row and row.get("state") in LIVE_PAUSE_STATES and row.get("source_ref") + and str(row.get("pause_id") or "") and str(row.get("pause_id") or "") != pause_id): + # Refresh the stale locator from the durable pause, retaining the queue's fence id. + refreshed = exact_pause_marker(row, default_root=str(pause.get("root_task_id") + or task.get("root_task_id") or task_id)) + if pause.get("fence_id"): + refreshed["fence_id"] = pause["fence_id"] + task["_budget_pause"] = refreshed + q.append_jsonl(q.DRIVE_ROOT / "logs" / "events.jsonl", + {"ts": utc_now_iso(), "type": "budget_pause_marker_refreshed", "task_id": task_id, + "stale_pause_id": pause_id, "pause_id": str(row.get("pause_id") or ""), + "pause_generation": int(row.get("pause_generation") or 0)}) + pause, checkpoint = refreshed, refreshed["checkpoint"] + pause_id = str(row.get("pause_id") or "") + if (not row or row.get("pause_id") != pause_id or row.get("state") not in LIVE_PAUSE_STATES + or not row.get("source_ref")): + return {"ok": False, "error": "pause_record_missing", "action": "cancel_or_new_run"} + if int(row.get("task_attempt") or 0) != int(task.get("_attempt") or 1): + # The queue row would dispatch another attempt than the one the + # checkpoint belongs to; the loop would refuse the grant on arrival. + return {"ok": False, "error": "pause_attempt_mismatch", "action": "cancel_or_new_run", + "row_attempt": int(row.get("task_attempt") or 0), "queue_attempt": int(task.get("_attempt") or 1)} + hold = task.get(BUDGET_HOLD_KEY) if isinstance(task.get(BUDGET_HOLD_KEY), dict) else {} + if hold.get("reason") == HOLD_MALFORMED_RESUME_IDENTITY: + return {"ok": False, "error": HOLD_MALFORMED_RESUME_IDENTITY} + live_grant = row.get("grant") if isinstance(row.get("grant"), dict) else {} + if "grant" in row and not str(live_grant.get("grant_id") or "").strip(): + return {"ok": False, "error": "malformed_grant_identity"} + if row.get("state") == STATE_RESUME_GRANTED and not live_grant.get("revoked_at"): + # Only a hold naming this undispatched grant, with no remaining handoff, + # proves it orphaned. Write its deferred revocation before minting again. + orphaned = bool( + hold and not hold.get("selected") + and str(hold.get("reason") or "") in {HOLD_REVOCATION_UNWRITTEN, HOLD_RESTART_REVOCATION_UNWRITTEN} + and str(hold.get("grant_id") or "") == str(live_grant.get("grant_id") or "") + and not any(isinstance(item.get("_budget_pause_resume"), dict) + and str(item["_budget_pause_resume"].get("grant_id") or "") == str(live_grant.get("grant_id") or "") + for item in list(q.PENDING) + [m.get("task") for m in q.RUNNING.values() if isinstance(m, dict)] + if isinstance(item, dict)) + ) + if not orphaned: + return {"ok": False, "error": "resume_already_granted", + "grant_id": live_grant.get("grant_id")} + revoked = {**live_grant, "revoked_at": utc_now_iso(), + "revoke_reason": f"deferred:{hold.get('reason')}:{str(hold.get('detail') or '')[:120]}"} + try: + set_budget_pause(result_root, task_id, {**row, "state": STATE_PAUSED, "grant": revoked}, + expected_pause_id=pause_id, expected_state=STATE_RESUME_GRANTED, + expected_grant_id=str(live_grant.get("grant_id") or "")) + except Exception as exc: + return {"ok": False, "error": HOLD_REVOCATION_UNWRITTEN, "detail": str(exc)[:200], + "grant_id": live_grant.get("grant_id"), "action": "retry_or_cancel"} + row = {**row, "state": STATE_PAUSED, "grant": revoked} + try: + read_actor_source_bytes(result_root, task_id, row["source_ref"]) + except Exception: + return {"ok": False, "error": "pause_source_unreadable", "action": "cancel_or_new_run"} + # The caller observed custody outside the queue lock; absent that observation, + # read it now. Preserve unknown/requested stops as refusals, never an empty set. + if not isinstance(external, dict): + external = observe_task_runs(result_root, task_id, reason="budget_resume_uncovered_cost") + if external.get("custody_read") != "ok": + return {"ok": False, "error": "external_custody_unreadable", + "detail": str(external.get("error") or ""), "action": "retry_or_cancel"} + unsettled = unsettled_external_runs(external) + if unsettled: + try: + set_budget_pause(result_root, task_id, {**row, "external_runs": external}, + expected_pause_id=pause_id, expected_state=str(row.get("state") or "")) + except Exception: + log.debug("Fresh custody observation could not be recorded on %s", task_id, exc_info=True) + return {"ok": False, "error": "external_runs_unsettled", + "runs": [{key: run.get(key) for key in ("run_id", "state", "stop_outcome")} for run in unsettled], + "action": "wait_for_delegated_runs_to_settle_or_cancel_them"} + row = {**row, "external_runs": external} + try: + if has_active_intent(pathlib.Path(q.DRIVE_ROOT), task_id, strict=True): + return {"ok": False, "error": "cancel_intent_active"} + except Exception: + return {"ok": False, "error": "cancellation_authority_unavailable"} + deadline = parse_deadline_ts(task.get("deadline_at") or (task.get("task_contract") or {}).get("deadline_at")) + if deadline is not None and deadline <= utc_now(): + return {"ok": False, "error": "deadline_passed"} + now = time.time() + started = float(row.get("started_at") or checkpoint.get("started_at") or 0.0) + paused_at = float(row.get("paused_at") or checkpoint.get("paused_at") or now) + prior_paused = float(row.get("paused_duration_sec") or 0.0) + # ONE shared clock: wall time minus the quota union minus the paused carrier. + executed_sec = execution_elapsed_seconds( + {"started_at": started, "budget_paused_sec": prior_paused, + "model_wait_quota_clock": row.get("model_wait_quota_clock") or {}}, paused_at) + ceiling = get_task_abs_ceiling_sec() # None = unlimited lifetime; 0 = exhausted + if ceiling is not None and started and executed_sec >= float(ceiling): + return {"ok": False, "error": "lifetime_exhausted", "executed_sec": round(executed_sec, 1)} + try: + # Authoritative, never the admit-only stale snapshot: a grant is money. + remaining = budget_remaining(q.load_state(), strict=True, allow_stale=False) + except Exception: + return {"ok": False, "error": "monetary_authority_unavailable"} + if remaining <= 0: + return {"ok": False, "error": "budget_still_exhausted", "action": "increase_budget_then_resume"} + root_task_id = str(pause.get("root_task_id") or task.get("root_task_id") or task_id) + root_grant = live_root_resume_grant(q, root_task_id, result_root) if root_task_id != task_id else {} + if selected_by and root_task_id != task_id: + # Q9: lineage alone grants nothing; model selection needs this root's live grant. + if not root_grant: + return {"ok": False, "error": "root_resume_grant_missing", + "root_task_id": root_task_id, "action": "resume_root_first"} + if str(pause.get("scope") or "") == "root": + from ouroboros.usage_accounting import refresh_root_accounting + + tree = refresh_root_accounting(result_root, root_task_id, max_age_sec=0.0) + if not isinstance(tree, dict): + # Unknown tree spend is not room: an unreadable ledger refuses typed. + return {"ok": False, "error": "root_accounting_unavailable", + "action": "retry_or_cancel"} + if float(tree.get("age_sec") or 0.0) > _FRESH_ROOT_ACCOUNTING_MAX_AGE_SEC: + # A stale fallback is not this grant's monetary authority. + return {"ok": False, "error": "root_accounting_unavailable", + "tree_age_sec": round(float(tree.get("age_sec") or 0.0), 1), + "action": "retry_or_cancel"} + if tree.get("integrity_degraded"): + return {"ok": False, "error": "root_accounting_degraded", + "action": "retry_or_cancel"} + limit, accounted = tree.get("root_limit_usd"), tree.get("accounted_usd") + if limit is not None: + if accounted is None: + return {"ok": False, "error": "root_accounting_degraded", + "action": "retry_or_cancel"} + if float(accounted) >= float(limit) - 1e-9: + return {"ok": False, "error": "root_hard_cap_exhausted", + "action": "increase_budget_then_resume"} + if _root_budget_paused_locked(q, root_task_id, except_task_id=task_id): + return {"ok": False, "error": "root_still_paused", "root_task_id": root_task_id, + "action": "resume_root_first"} + # A later Resume raises the generation; older grants cannot become live again. + generation = int(row.get("resume_generation") or 0) + 1 + pause_generation = int(row.get("pause_generation") or 0) + grant = { + "grant_id": uuid.uuid4().hex, "granted_at": utc_now_iso(), "granted_at_ts": now, + "single_use": True, "paused_duration_sec": prior_paused + max(0.0, now - paused_at), + "executed_sec_before_pause": round(executed_sec, 3), + "pause_id": pause_id, "pause_generation": pause_generation, "generation": generation, + "selected_by": str(selected_by or "owner"), + "root_grant_id": str(root_grant.get("grant_id") or ""), + "root_resume_generation": int(root_grant.get("generation") or 0), + "root_fence_id": (str((q.BUDGET_ROOT_FENCES.get(root_task_id) or {}).get("fence_id") or "") + if root_task_id != task_id else ""), + "refresh_planning_threshold": str(row.get("rail") or "") in { + "graceful_ceiling", "wrapup_last_fit", "soft_land"}, + } + prior_pause = dict(pause) + try: + # CAS against the state this grant was validated on: a concurrent + # writer (a late revocation, another grant) refuses here, typed. + set_budget_pause(result_root, task_id, + {**row, "state": STATE_RESUME_GRANTED, "grant": grant, + "resume_generation": generation}, + expected_pause_id=pause_id, expected_state=str(row.get("state") or "")) + except Exception as exc: + return {"ok": False, "error": "grant_not_recorded", "detail": str(exc)[:200]} + task.pop("_budget_pause", None) + task["_budget_pause_resume"] = { + **checkpoint, "grant_id": grant["grant_id"], "granted_at": grant["granted_at"], + "grant_generation": generation, "pause_id": pause_id, "pause_generation": pause_generation, + "paused_duration_sec": grant["paused_duration_sec"], "pause": prior_pause, + "external_runs": external, + **{key: grant[key] for key in ("selected_by", "root_grant_id", "root_resume_generation", "root_fence_id")}, + } + task["budget_resumed_at"] = grant["granted_at"] + # Release a restore/revocation hold only after re-validating its durable authority. + released_hold = release_budget_hold(task, released_by=grant["selected_by"], reason="exact_resume_granted") + fence = q.BUDGET_ROOT_FENCES.get(root_task_id) + fence_released = False + held_siblings: List[str] = [] + released_markers: Dict[str, Dict[str, Any]] = {} + rebound_holds: Dict[str, Dict[str, Any]] = {} + if (task_id == root_task_id and isinstance(fence, dict) and str(fence.get("fence_id") or "") + == str(prior_pause.get("fence_id") or fence.get("fence_id"))): + # The root's own Resume lifts its admission latch: exact-continuation + # descendants keep their OWN `_budget_pause` rows and are only ELIGIBLE + # now; the model selects each through this same control (Q9). + q.BUDGET_ROOT_FENCES.pop(root_task_id, None) + fence_released = True + held_siblings, released_markers, rebound_holds = hold_root_resume_descendants(q, root_task_id, fence, grant) + if not q.persist_queue_snapshot(reason="budget_exact_resume_granted"): + task.pop("_budget_pause_resume", None) + task["_budget_pause"] = prior_pause + if released_hold is not None: + task[BUDGET_HOLD_KEY] = released_hold + if fence_released: + q.BUDGET_ROOT_FENCES[root_task_id] = fence + for member in q.PENDING: + member_id = str(member.get("id") or "") + if member_id in set(held_siblings): + member.pop(BUDGET_HOLD_KEY, None) + if member_id in released_markers: + member["_budget_pause"] = released_markers[member_id] + elif member_id in rebound_holds: + member[BUDGET_HOLD_KEY] = rebound_holds[member_id] + try: + set_budget_pause(result_root, task_id, row, expected_pause_id=pause_id, + expected_state=STATE_RESUME_GRANTED, + expected_grant_id=str(grant["grant_id"])) + return {"ok": False, "error": "snapshot_not_persisted"} + except Exception as rollback_error: + detail = str(rollback_error)[:120] + log.warning("Exact resume grant rollback remains unpersisted for %s", task_id, exc_info=True) + # The rollback failed: retain this orphaned grant's identity in the existing + # hold so the next Resume can write its deferred revocation before minting. + hold_budget_row( + task, reason=HOLD_REVOCATION_UNWRITTEN, + detail=f"snapshot_not_persisted:{detail}", + extra={"pause_id": pause_id, "grant_id": str(grant["grant_id"]), + "root_task_id": root_task_id, + **({"prior_hold_reason": str(released_hold.get("reason") or "")} + if released_hold else {})}, + result_root=result_root) + task["_budget_pause"] = prior_pause + return {"ok": False, "error": "snapshot_not_persisted", + "held": HOLD_REVOCATION_UNWRITTEN, "grant_id": str(grant["grant_id"])} + try: + from ouroboros.task_results import STATUS_SCHEDULED, write_task_result + + write_task_result( + result_root, task_id, STATUS_SCHEDULED, reason_code="", + resource_limit={**prior_pause, "status": "resume_granted", "resumed_at": grant["granted_at"], + "grant_id": grant["grant_id"], "auto_resume": False}, + ) + except Exception: + log.debug("Failed to project exact budget resume for %s", task_id, exc_info=True) + eligible = [str(r.get("id") or "") for r in q.PENDING + if isinstance(r.get("_budget_pause"), dict) and r["_budget_pause"].get("exact_continuation") + and str(r.get("root_task_id") or "") == root_task_id and str(r.get("id") or "") != task_id] + q.append_jsonl( + q.DRIVE_ROOT / "logs" / "events.jsonl", + {"ts": utc_now_iso(), "type": "budget_task_explicitly_resumed", "task_id": task_id, + "root_task_id": root_task_id, "same_generation": True, "exact_continuation": True, + "grant_id": grant["grant_id"], "grant_generation": generation, + "selected_by": grant["selected_by"], + "paused_duration_sec": grant["paused_duration_sec"], + "eligible_descendants": eligible if task_id == root_task_id else [], + "held_siblings": held_siblings, "rebound_held_siblings": sorted(rebound_holds), + "released_hold": str((released_hold or {}).get("reason") or "")}, + ) + return {"ok": True, "task_id": task_id, "root_task_id": root_task_id, "exact_continuation": True, + "grant_id": grant["grant_id"], "grant_generation": generation, + "paused_duration_sec": round(grant["paused_duration_sec"], 1), + "eligible_descendants": eligible if task_id == root_task_id else [], + "held_siblings": held_siblings, "rebound_held_siblings": sorted(rebound_holds), + **({"released_hold": str(released_hold.get("reason") or "")} if released_hold else {})} + + +def revoke_exact_budget_resume(task: Dict[str, Any], reason: str) -> bool: + """Return a granted-but-undispatched task to its exact pause (queue lock held). + + Money can vanish between the grant and the dispatch (a sibling spent it) and + a restart may intervene; the grant is single-use and must not be dispatched + into a refused send. Identity decides what may be written: a revocation is + recorded ONLY against the pause, state and grant this handoff names + (compare-and-set on all three). A grant the durable row says was CONSUMED + is never re-armed: the task ran on, the queue row is stale — it takes a + typed hold, its handoff leaves, and no ``_budget_pause`` marker is + re-minted over a task that is not paused. If the durable row already + carries a NEWER pause (pauseA -> Resume -> pauseB), or a different grant, + nothing is written over it — the spent handoff simply leaves the queue + row, which re-reads the current pause or holds; at restore, a newer grant + that never reached a worker either is revoked too, so the next Resume finds + a pause, not a grant no row carries. A failed write leaves the row + un-dispatchable rather than carrying a stale grant. + """ + from ouroboros.budget_pause import ( + LIVE_PAUSE_STATES, STATE_PAUSED, STATE_RESUME_GRANTED, STATE_RESUMED, budget_pause_row, + exact_pause_marker, set_budget_pause, + ) + from supervisor.events_budget import ( + HOLD_GRANT_CONSUMED_STALE_ROW, HOLD_RECORD_UNREADABLE_AT_REVOKE, HOLD_RESTART_REVOCATION_UNWRITTEN, + HOLD_REVOCATION_UNWRITTEN, HOLD_STALE_GRANT_SUPERSEDED, HOLD_MALFORMED_RESUME_IDENTITY, hold_budget_row, + ) + + q = _queue_module() + handoff = task.get("_budget_pause_resume") if isinstance(task.get("_budget_pause_resume"), dict) else None + if handoff is None: + return False + task_id = str(task.get("id") or "") + result_root = pathlib.Path(task.get("budget_drive_root") or q.DRIVE_ROOT) + prior_pause = dict(handoff.get("pause") or {}) if isinstance(handoff.get("pause"), dict) else {} + expected_pause_id = str(handoff.get("pause_id") or "").strip() + handoff_grant_id = str(handoff.get("grant_id") or "") + if not expected_pause_id or not handoff_grant_id.strip() or not prior_pause: + task["_budget_pause"] = prior_pause + hold_budget_row(task, reason=HOLD_MALFORMED_RESUME_IDENTITY, detail="malformed_resume_identity", + extra={"pause_id": expected_pause_id, "grant_id": handoff_grant_id}, result_root=None) + return False + try: + row = budget_pause_row(result_root, task_id) + except Exception: + log.warning("Exact resume grant revocation could not read the pause row for %s", + task_id, exc_info=True) + # The saved pause is retained: the marker stays the locator the owner's + # next Resume validates; the hold keeps the row off the dispatch path. + task["_budget_pause"] = prior_pause + hold_budget_row(task, reason=HOLD_RECORD_UNREADABLE_AT_REVOKE, + detail=str(reason or ""), + extra={"pause_id": expected_pause_id, "grant_id": handoff_grant_id}, + result_root=result_root) + return False + current_pause_id = str(row.get("pause_id") or "") + row_state = str(row.get("state") or "") + grant = dict(row["grant"]) if isinstance(row.get("grant"), dict) else {} + if (not current_pause_id.strip() or (current_pause_id == expected_pause_id or row_state == STATE_RESUME_GRANTED) + and not str(grant.get("grant_id") or "").strip()): + task["_budget_pause"] = prior_pause + hold_budget_row(task, reason=HOLD_MALFORMED_RESUME_IDENTITY, detail="malformed_durable_resume_identity", + extra={"pause_id": expected_pause_id, "grant_id": handoff_grant_id}, result_root=None) + return False + same_pause = current_pause_id == expected_pause_id + same_grant = str(grant.get("grant_id") or "") == handoff_grant_id + consumed = bool(grant.get("consumed_at")) or row_state == STATE_RESUMED + if same_pause and same_grant and consumed: + # NEVER re-armed: the loop consumed this grant, so the task ran (or + # ran and ended). This queue row is a stale carrier; it is held typed, + # off the dispatch path, with no pause marker — and a restore fences it + # as the running work it names. + task.pop("_budget_pause_resume", None) + if isinstance(task.get("_owner_wait_resume"), dict): + # The task ran ON past this grant and LATER parked in an owner wait: + # the spent carrier is simply retired and the newer planned-restart + # handoff decides the row (its own restore gate re-validates it). + # Fencing the row here would drop that valid continuation (#1196, F3). + q.append_jsonl(q.DRIVE_ROOT / "logs" / "events.jsonl", + {"ts": utc_now_iso(), "type": "budget_resume_carrier_retired", + "task_id": task_id, "reason": str(reason or ""), + "grant_id": str(grant.get("grant_id") or ""), + "pause_id": current_pause_id, + "consumed_at": grant.get("consumed_at"), + "retained": "owner_wait_resume"}) + return False + task["_budget_pause_consumed"] = { + "pause_id": current_pause_id, "grant_id": str(grant.get("grant_id") or ""), + "consumed_at": grant.get("consumed_at"), "reason": str(reason or ""), + } + hold_budget_row(task, reason=HOLD_GRANT_CONSUMED_STALE_ROW, detail=str(reason or ""), + extra={"pause_id": current_pause_id, "grant_id": str(grant.get("grant_id") or "")}, + result_root=None) # the task's own status is not ours to rewrite + q.append_jsonl(q.DRIVE_ROOT / "logs" / "events.jsonl", + {"ts": utc_now_iso(), "type": "budget_resume_grant_revoke_refused_consumed", + "task_id": task_id, "reason": str(reason or ""), + "grant_id": str(grant.get("grant_id") or ""), "pause_id": current_pause_id, + "consumed_at": grant.get("consumed_at")}) + return False + superseded = not same_pause or not same_grant + if superseded: + log.warning("Stale exact-resume handoff for %s (grant %s, pause %s) is NOT written over the " + "current pause %s", task_id, handoff_grant_id, expected_pause_id, + current_pause_id or "") + task.pop("_budget_pause_resume", None) + newer_undispatched = ( + str(reason or "") == "restart_before_dispatch" + and row_state == STATE_RESUME_GRANTED + and str(grant.get("grant_id") or "") + and not grant.get("revoked_at") and not grant.get("consumed_at") + ) + if newer_undispatched: + # The snapshot lagged a NEWER grant; no worker survives a restart, so + # that grant never reached one either. Revoked under its OWN + # identity, or the row is held naming it (the next Resume writes + # the deferred revocation before minting again). + revoked = {**grant, "revoked_at": utc_now_iso(), + "revoke_reason": "restart_before_dispatch:superseding_grant_undispatched"} + try: + set_budget_pause(result_root, task_id, {**row, "state": STATE_PAUSED, "grant": revoked}, + expected_pause_id=current_pause_id, expected_state=STATE_RESUME_GRANTED, + expected_grant_id=str(grant.get("grant_id") or "")) + row = {**row, "state": STATE_PAUSED, "grant": revoked} + row_state = STATE_PAUSED + except Exception: + log.warning("Superseding grant of %s could not be revoked at restore; held", task_id, exc_info=True) + task["_budget_pause"] = exact_pause_marker( + row, default_root=str(task.get("root_task_id") or task_id)) + hold_budget_row(task, reason=HOLD_RESTART_REVOCATION_UNWRITTEN, + detail=str(reason or ""), + extra={"pause_id": current_pause_id, "grant_id": str(grant.get("grant_id") or "")}, + result_root=result_root) + return False + if row_state in LIVE_PAUSE_STATES and row.get("source_ref"): + task["_budget_pause"] = exact_pause_marker( + row, default_root=str(task.get("root_task_id") or task_id)) + else: + hold_budget_row(task, reason=HOLD_STALE_GRANT_SUPERSEDED, + detail=str(reason or ""), + extra={"handoff_pause_id": expected_pause_id, + "current_pause_id": current_pause_id}, + result_root=result_root) + q.append_jsonl(q.DRIVE_ROOT / "logs" / "events.jsonl", + {"ts": utc_now_iso(), "type": "budget_resume_grant_revoke_superseded", + "task_id": task_id, "reason": str(reason or ""), + "handoff_grant_id": handoff_grant_id, + "handoff_pause_id": expected_pause_id, "current_pause_id": current_pause_id, + "current_grant_id": grant.get("grant_id"), + "superseding_grant_revoked": bool(newer_undispatched)}) + return False + try: + grant.update(revoked_at=utc_now_iso(), revoke_reason=str(reason or "")) + set_budget_pause(result_root, task_id, {**row, "state": STATE_PAUSED, "grant": grant}, + expected_pause_id=current_pause_id, expected_state=row_state, + expected_grant_id=str(grant.get("grant_id") or "")) + except Exception: + log.warning("Exact resume grant revocation not recorded for %s", task_id, exc_info=True) + # Retained, typed, un-dispatchable: the marker stays so the owner's next + # Resume finds the exact pause; the grant named here is proven + # undispatched (its handoff leaves the row with this hold), so that + # Resume writes the deferred revocation before minting a new grant. + task["_budget_pause"] = prior_pause + hold_budget_row(task, reason=(HOLD_RESTART_REVOCATION_UNWRITTEN + if str(reason or "") == "restart_before_dispatch" + else HOLD_REVOCATION_UNWRITTEN), + detail=str(reason or ""), + extra={"pause_id": current_pause_id, "grant_id": grant.get("grant_id")}, + result_root=result_root) + return False + task["_budget_pause"] = prior_pause + task.pop("_budget_pause_resume", None) + q.append_jsonl(q.DRIVE_ROOT / "logs" / "events.jsonl", + {"ts": utc_now_iso(), "type": "budget_resume_grant_revoked", "task_id": task_id, + "reason": str(reason or ""), "grant_id": grant.get("grant_id"), + "pause_id": current_pause_id}) + return True diff --git a/supervisor/events_budget.py b/supervisor/events_budget.py index 6590720c8..15e1140c2 100644 --- a/supervisor/events_budget.py +++ b/supervisor/events_budget.py @@ -2,7 +2,9 @@ One owner for folding a worker's reported usage into the ledger and for the two budget fences a paused root raises: the pause itself and the admission fence -that keeps its descendants out of the queue. +that keeps its descendants out of the queue — including the per-row HOLD that +survives lifting that fence, and the explicit selection that releases one row +(#1196, owner Q9). """ from __future__ import annotations @@ -11,7 +13,7 @@ import logging import pathlib import time import uuid -from typing import Any, Dict +from typing import Any, Dict, Optional from ouroboros.utils import append_jsonl, utc_now_iso from ouroboros.task_results import STATUS_SCHEDULED, write_task_result @@ -179,13 +181,16 @@ def _set_root_budget_pause_locked(root_task_id: str, pause: Dict[str, Any]) -> D if not root_task_id: raise ValueError("root budget pause requires root_task_id") existing = queue_mod.BUDGET_ROOT_FENCES.get(root_task_id) + root_rows = list(queue_mod.PENDING) + [m.get("task", {}) for m in queue_mod.RUNNING.values()] + resumed = any(str(t.get("id") or "") == root_task_id and budget_fence_selected(t, existing) + for t in root_rows) row = { "status": "paused", "scope": "root", "root_task_id": root_task_id, "fence_id": str( pause.get("fence_id") - or (existing or {}).get("fence_id") + or (None if resumed else (existing or {}).get("fence_id")) or uuid.uuid4().hex ), "auto_resume": False, @@ -196,6 +201,18 @@ def _set_root_budget_pause_locked(root_task_id: str, pause: Dict[str, Any]) -> D ), } queue_mod.BUDGET_ROOT_FENCES[root_task_id] = row + if not existing or existing.get("fence_id") != row["fence_id"]: + from supervisor.budget_resume import revoke_exact_budget_resume + + for member in queue_mod.PENDING: + if str(member.get("root_task_id") or "") != root_task_id or member.get("id") == root_task_id: + continue + if isinstance(member.get("_budget_pause_resume"), dict): + revoke_exact_budget_resume(member, "new_root_budget_fence") + hold = member.get(BUDGET_HOLD_KEY) + if isinstance(hold, dict) and hold.get("selected"): + hold_budget_row(member, reason=HOLD_ROOT_FENCE_LIFTED, + extra={"root_task_id": root_task_id, "fence_id": row["fence_id"]}) return row @@ -282,14 +299,23 @@ def install_exact_budget_pause(ctx: Any, task_id: str, checkpoint: Dict[str, Any Order: durable row must already say ``pausing`` for this pause_id/attempt (the worker wrote it before raising) -> queue transition -> snapshot -> - row ``paused``. A snapshot that cannot be persisted leaves the row at - ``pausing``: the pause is real (the source exists) but not confirmed, and - the next persisted snapshot or a restore completes it — never a fake - ``paused``. Also the crash-during-pausing completion path (``source`` - names it), which is why nothing here reads the worker's event body. + row ``paused``. The WHOLE order — including the confirmation, the owner + projection and the event — runs under the queue lock a Resume grant also + holds, and the confirmation compare-and-sets the pause id AND its state, so + a park that completes late can never overwrite a grant or republish a + resumed task as paused (#1196, F1). A snapshot that cannot be persisted + leaves the row at ``pausing``: the pause is real (the source exists) but not + confirmed, and the next persisted snapshot or a restore completes it — + never a fake ``paused``. Also the crash-during-pausing completion path (``source`` + names it), which is why nothing here reads the worker's event body — with + ONE exception: a direct owner-chat turn was never in RUNNING, so its event + carries the turn's own task record (``task`` beside ``_is_direct_chat``), + and THAT record is what the SAME task id is parked under (#1196). The row + check above still decides; the record only supplies the queue row. """ from ouroboros.budget_pause import ( - STATE_PAUSED, STATE_PAUSING, budget_pause_row, exact_pause_marker, set_budget_pause, + STATE_PAUSED, STATE_PAUSING, budget_pause_row, exact_pause_marker, parkable_direct_task, + set_budget_pause, ) from supervisor import queue as queue_mod from supervisor.queue import _queue_lock @@ -305,6 +331,14 @@ def install_exact_budget_pause(ctx: Any, task_id: str, checkpoint: Dict[str, Any with _queue_lock: meta = ctx.RUNNING.get(task_id) task = meta.get("task") if isinstance(meta, dict) and isinstance(meta.get("task"), dict) else None + direct_turn = False + if task is None and evt.get("_is_direct_chat") and isinstance(evt.get("task"), dict) \ + and str(evt["task"].get("id") or "") == task_id: + # A direct turn ends its live actor on the way out (no second live + # actor for this id); the queue row is minted from its own record. + task = parkable_direct_task(evt["task"]) + meta = {"task": task, "attempt": int(task.get("_attempt") or 1), "worker_id": None} + direct_turn = True if task is None: raise RuntimeError(f"budget-paused task is not running: {task_id}") result_root = pathlib.Path(task.get("budget_drive_root") or ctx.DRIVE_ROOT) @@ -320,59 +354,101 @@ def install_exact_budget_pause(ctx: Any, task_id: str, checkpoint: Dict[str, Any ctx.RUNNING.pop(task_id, None) paused_task = dict(task) paused_task["_budget_pause"] = marker + if direct_turn: + stamp_direct_queue_row(paused_task, queue_mod) if not any(str(item.get("id") or "") == task_id for item in ctx.PENDING): ctx.PENDING.append(paused_task) sort_pending() worker_id = evt.get("worker_id") if evt else meta.get("worker_id") if worker_id in ctx.WORKERS and ctx.WORKERS[worker_id].busy_task_id == task_id: ctx.WORKERS[worker_id].busy_task_id = None - persisted = persist_snapshot(reason="budget_pause_exact_continuation") - if persisted: + # The confirmation, the status projection and the owner-visible event + # stay under the SAME queue lock as the park. Resume holds that lock for + # its whole grant, so a late park completion can no longer publish a + # stale ``paused`` over a live grant; the state CAS below is the second + # rail, for a writer this lock does not cover (#1196, F1). + persisted = persist_snapshot(reason="budget_pause_exact_continuation") + confirmed, superseded = False, "" + if persisted: + try: + set_budget_pause(result_root, task_id, {**row, "state": STATE_PAUSED, + "paused_confirmed_at": time.time(), + "pause_source": source}, + expected_pause_id=pause_id, + expected_state=(STATE_PAUSING, STATE_PAUSED)) + confirmed = True + except Exception: + log.warning("Exact budget pause row for %s was not confirmed 'paused'", + task_id, exc_info=True) + superseded = _pause_row_superseded(result_root, task_id, pause_id) + else: + log.error("Exact budget pause for %s parked in memory but its snapshot was not persisted; " + "the durable row stays 'pausing' until a later snapshot confirms it", task_id) + if not superseded: + # A row this park no longer owns keeps whatever the newer writer + # projected: a resumed task must never read as paused again. + try: + write_task_result( + result_root, task_id, STATUS_SCHEDULED, + reason_code="budget_paused", resource_limit=marker, + result=("Task paused exactly at a completed boundary (budget). Cumulative spend, rounds " + "and execution time are retained; an explicit owner Resume continues the same task."), + ) + except Exception: + log.warning("Failed to persist exact budget pause status for %s", task_id, exc_info=True) + event = { + "ts": (evt or {}).get("ts", utc_now_iso()), + "type": "budget_scope_paused", + "task_id": task_id, + "task_type": (evt or {}).get("task_type") or task.get("type"), + "owner_visible": True, + "toast_once": f"{task_id}:budget-paused:{pause_id}", + "pause_source": source, + "park_state": STATE_PAUSED if confirmed else STATE_PAUSING, + **{key: value for key, value in marker.items() if key != "checkpoint"}, + "pause_id": pause_id, + "external_runs": [ + {k: run.get(k) for k in ("run_id", "state", "stop_outcome")} + for run in ((row.get("external_runs") or {}).get("runs") or []) if isinstance(run, dict) + ], + } + if superseded: + # Not an owner-visible pause: the durable row moved on (a grant, a + # newer pause, an abandoned row). The anomaly is recorded as itself. + event = {key: value for key, value in event.items() if key != "toast_once"} + event.update(type="budget_pause_park_superseded", owner_visible=False, + park_state=superseded) + _address_task_event({task_id: meta} if isinstance(meta, dict) else None, ctx.DRIVE_ROOT, event) + append_jsonl(ctx.DRIVE_ROOT / "logs" / "events.jsonl", event) try: - set_budget_pause(result_root, task_id, {**row, "state": STATE_PAUSED, - "paused_confirmed_at": time.time(), - "pause_source": source}, - expected_pause_id=pause_id) + bridge = getattr(ctx, "bridge", None) + if bridge is not None: + bridge.push_log(event) except Exception: - log.warning("Exact budget pause row for %s stays 'pausing'", task_id, exc_info=True) - else: - log.error("Exact budget pause for %s parked in memory but its snapshot was not persisted; " - "the durable row stays 'pausing' until a later snapshot confirms it", task_id) - try: - write_task_result( - result_root, task_id, STATUS_SCHEDULED, - reason_code="budget_paused", resource_limit=marker, - result=("Task paused exactly at a completed boundary (budget). Cumulative spend, rounds " - "and execution time are retained; an explicit owner Resume continues the same task."), - ) - except Exception: - log.warning("Failed to persist exact budget pause status for %s", task_id, exc_info=True) - event = { - "ts": (evt or {}).get("ts", utc_now_iso()), - "type": "budget_scope_paused", - "task_id": task_id, - "task_type": (evt or {}).get("task_type") or task.get("type"), - "owner_visible": True, - "toast_once": f"{task_id}:budget-paused:{pause_id}", - "pause_source": source, - **{key: value for key, value in marker.items() if key != "checkpoint"}, - "pause_id": pause_id, - "external_runs": [ - {k: run.get(k) for k in ("run_id", "state", "stop_outcome")} - for run in ((row.get("external_runs") or {}).get("runs") or []) if isinstance(run, dict) - ], - } - _address_task_event({task_id: meta} if isinstance(meta, dict) else None, ctx.DRIVE_ROOT, event) - append_jsonl(ctx.DRIVE_ROOT / "logs" / "events.jsonl", event) - try: - bridge = getattr(ctx, "bridge", None) - if bridge is not None: - bridge.push_log(event) - except Exception: - log.warning("Failed to forward exact budget pause to Activity", exc_info=True) + log.warning("Failed to forward exact budget pause to Activity", exc_info=True) return marker +def _pause_row_superseded(result_root: pathlib.Path, task_id: str, pause_id: str) -> str: + """Why a park confirmation may NOT be published, or "" when it is only late. + + A write that failed transiently leaves the row at ``pausing`` for THIS pause + id — the pause is real, merely unconfirmed, and the ordinary projection + still belongs to it. A row that names another pause, carries a grant, or is + no longer live belongs to a newer writer and is never written over. + """ + from ouroboros.budget_pause import STATE_PAUSED, STATE_PAUSING, budget_pause_row + + try: + row = budget_pause_row(result_root, task_id) + except Exception: + return "pause_record_unreadable" + if str(row.get("pause_id") or "") != str(pause_id): + return "pause_identity_changed" + state = str(row.get("state") or "") + return "" if state in {STATE_PAUSING, STATE_PAUSED} else (state or "pause_record_missing") + + def _handle_budget_resume_child(evt: Dict[str, Any], ctx: Any) -> None: """Owner Q9: a resumed root's model SELECTS one budget-paused child to continue. @@ -396,7 +472,9 @@ def _handle_budget_resume_child(evt: Dict[str, Any], ctx: Any) -> None: if not lineage_ok: outcome: Dict[str, Any] = {"ok": False, "error": "not_a_budget_paused_descendant"} else: - outcome = resume_budget_paused_task(task_id) + # A MODEL-issued selection: the grant additionally requires the live + # owner-derived Resume grant of the target's own root (never lineage alone). + outcome = resume_budget_paused_task(task_id, selected_by=requester) event = { "ts": evt.get("ts", utc_now_iso()), "type": "budget_resume_child_outcome", @@ -440,3 +518,332 @@ def _handle_budget_root_fence(evt: Dict[str, Any], ctx: Any) -> None: ctx.bridge.push_log(event) except Exception: log.warning("Failed to forward root budget pause to Activity", exc_info=True) + + +# --- budget admission HOLDS (#1196) -------------------------------------------- +# +# The root fence above keeps a paused root's descendants out of the queue. When +# that root is resumed the fence is LIFTED, which must not by itself make its +# zero-dispatch siblings assignable: each one takes the durable hold below until +# the model selects it through the same resume control (owner Q9). The same +# marker carries a row whose exact continuation could not be restored and one +# whose spent grant could not be revoked, so an un-dispatchable row is always a +# typed, visible fact instead of a dropped or silently runnable task. + +BUDGET_HOLD_KEY = "_budget_pause_hold" + +# The typed hold reasons (one vocabulary for the queue row, the task result and +# the events log). A hold beside a retained ``_budget_pause`` marker is released +# by a successful exact grant (the grant re-validates the durable authority); +# a hold on a marker-less row is released only by an explicit selection. +HOLD_ROOT_FENCE_LIFTED = "root_fence_lifted_pending_selection" +# One MEMBER of a tree whose root latch is still up: the latch belongs to the +# root, so releasing this row must not lift it for every sibling (owner Q9). +# The selection recorded on the row names the fence it was granted against. +HOLD_ROOT_FENCE_MEMBER_SELECTION = "root_fence_member_pending_selection" +HOLD_RESTORE_REFUSED_PREFIX = "restore_refused:" +HOLD_ROOT_ACCEPTANCE_FENCED = "root_acceptance_fenced_at_restore" +HOLD_REVOCATION_UNWRITTEN = "resume_grant_revocation_unwritten" +HOLD_RECORD_UNREADABLE_AT_REVOKE = "pause_record_unreadable_at_revoke" +HOLD_STALE_GRANT_SUPERSEDED = "stale_resume_grant_superseded" +HOLD_MALFORMED_RESUME_IDENTITY = "malformed_resume_identity" +HOLD_RESTART_REVOCATION_UNWRITTEN = "restart_revocation_unwritten" +# A queue row still carrying a grant handoff whose grant the durable row says +# was CONSUMED: the task ran on. The row is stale, never re-armed as a pause +# and never dispatched; a restart fences it as the running work it names. +HOLD_GRANT_CONSUMED_STALE_ROW = "stale_queue_row_grant_consumed" +# Malformed acceptance-fence evidence in the snapshot fails the restore closed +# for ordinary rows; a saved exact pause is retained under this hold instead. +HOLD_INVALID_ACCEPTANCE_FENCE_SNAPSHOT = HOLD_RESTORE_REFUSED_PREFIX + "invalid_acceptance_fence_snapshot" +HOLD_INVALID_BUDGET_FENCE_SNAPSHOT = HOLD_RESTORE_REFUSED_PREFIX + "invalid_budget_fence_snapshot" + + +def stamp_direct_queue_row(task: Dict[str, Any], queue_mod: Any) -> None: + """Give a parked direct turn the queue-order facts ``enqueue_task`` would have. + + The direct lane never enqueued it, so the row has no sequence, priority or + ``queued_at``; without them the sort key and the census phase have nothing + to read. Called with the queue lock held. + """ + counter = getattr(queue_mod, "QUEUE_SEQ_COUNTER_REF", None) + if isinstance(counter, dict) and "_queue_seq" not in task: + counter["value"] = int(counter.get("value") or 0) + 1 + task["_queue_seq"] = counter["value"] + if "priority" not in task: + try: + task["priority"] = queue_mod.coerce_queue_order( + None, queue_mod._task_priority(str(task.get("type") or "task"))) + except Exception: + task["priority"] = 1 + task.setdefault("queued_at", utc_now_iso()) + task["_is_direct_chat"] = True + + +def budget_hold_fact(task) -> Optional[Dict[str, Any]]: + """The durable NON-dispatch hold on one queued row (#1196), or ``None``. + + Three shapes share it and none invents a checkpoint identity (no pause_id, + no ``exact_continuation``): a zero-dispatch sibling whose paused root's + admission fence was lifted by that root's Resume — lifting the fence must + not make it assignable, the model selects it explicitly (owner Q9) — a row + whose exact continuation could not be restored, and a row whose spent + grant could not be revoked. ``selected`` is the only release. + """ + hold = task.get(BUDGET_HOLD_KEY) if isinstance(task, dict) else None + return hold if isinstance(hold, dict) and not hold.get("selected") else None + + +def budget_fence_selected(task: Any, fence: Any) -> bool: + """Whether THIS row carries an explicit selection recorded against THIS fence. + + A root's admission latch keeps a whole tree out of the queue. The owner (or, + under the root's live grant, the model) may select ONE member of that tree + without lifting the latch for its siblings: the selection rides the row's own + hold and names the fence generation it was granted against, so a later fence + — a root that paused again — is never pre-released by an older selection + (#1196, owner Q9). + """ + hold = task.get(BUDGET_HOLD_KEY) if isinstance(task, dict) else None + fence_id = str((fence or {}).get("fence_id") or "") if isinstance(fence, dict) else "" + return bool(isinstance(hold, dict) and hold.get("selected") and fence_id + and str(hold.get("fence_id") or "") == fence_id) + + +def budget_resume_dispatch_allowed(q: Any, task: Dict[str, Any]) -> bool: + """An exact child selection belongs to the CURRENT root grant and fence only.""" + handoff = task.get("_budget_pause_resume") + if not isinstance(handoff, dict): + return True + root_id = str(task.get("root_task_id") or task.get("id") or "") + fence_id = str((q.BUDGET_ROOT_FENCES.get(root_id) or {}).get("fence_id") or "") + if fence_id != str(handoff.get("root_fence_id") or ""): + return False + if root_id == str(task.get("id") or ""): + return True + root_grant = live_root_resume_grant(q, root_id, pathlib.Path(task.get("budget_drive_root") or q.DRIVE_ROOT)) + return bool(handoff.get("selected_by") == "owner" and not handoff.get("root_grant_id") and not root_grant + or root_grant and handoff.get("root_grant_id") == root_grant["grant_id"] + and handoff.get("root_resume_generation") == root_grant["generation"]) + + +def hold_budget_row(task: Dict[str, Any], *, reason: str, detail: str = "", + extra: Optional[Dict[str, Any]] = None, + result_root: Optional[pathlib.Path] = None) -> Dict[str, Any]: + """Hold one queued row: typed, visible, never dropped and never cancelled. + + The row stays PENDING with its own identity; any spent resume handoff is + removed so no stale grant can dispatch, and the typed reason is projected + onto the task result so the owner and the model read the same fact. + """ + hold = {"reason": str(reason), "detail": str(detail or "")[:300], "held_at": utc_now_iso(), + "selected": False, "dispatchable": False, **(extra or {})} + task.pop("_budget_pause_resume", None) + task[BUDGET_HOLD_KEY] = hold + if result_root is not None: + try: + write_task_result( + result_root, str(task.get("id") or ""), STATUS_SCHEDULED, + reason_code="budget_paused", + resource_limit={"status": "budget_hold", "auto_resume": False, + "exact_continuation": False, + "resume_policy": "explicit_selection_same_seam", **hold}, + ) + except Exception: + log.debug("Budget hold projection failed for %s", task.get("id"), exc_info=True) + return hold + + +def hold_restored_budget_pause(task: Dict[str, Any], drive_root: Any, *, reason: str, + detail: str = "") -> Dict[str, Any]: + """Restore-time hold for a paused row the restart could not clear for dispatch. + + A corrupt, missing or refused checkpoint source, or an acceptance fence over + the root, does not drop or cancel the task: the row stays PENDING and + un-dispatchable with a typed reason, and it KEEPS its ``_budget_pause`` + marker — the marker is the locator the owner's Resume validates against the + durable authority. A later Resume re-reads that authority: a source that is + readable again is granted (releasing this hold), one that is not refuses + typed and leaves the saved pause where it is. + """ + prior = task.get("_budget_pause") if isinstance(task.get("_budget_pause"), dict) else {} + hold_budget_row( + task, reason=reason, + detail=detail or "the exact budget-pause continuation could not be cleared for dispatch after a restart", + extra={"pause_id": str((prior.get("checkpoint") or {}).get("pause_id") or ""), + "root_task_id": str(prior.get("root_task_id") or task.get("root_task_id") or "")}, + result_root=pathlib.Path(task.get("budget_drive_root") or drive_root), + ) + return task + + +def release_budget_hold(task: Dict[str, Any], *, released_by: str, reason: str) -> Optional[Dict[str, Any]]: + """Mark a hold RELEASED in place; returns the prior hold for rollback, or ``None``. + + Reached only from the exact grant, which has just re-validated the pause's + durable authority: the hold's own facts are kept on the row (``selected`` + flips, nothing is erased) so the release stays auditable. + """ + hold = task.get(BUDGET_HOLD_KEY) if isinstance(task.get(BUDGET_HOLD_KEY), dict) else None + if hold is None or hold.get("selected"): + return None + task[BUDGET_HOLD_KEY] = {**hold, "selected": True, "selected_at": utc_now_iso(), + "selected_by": str(released_by or "owner"), "released_reason": str(reason or "")} + return hold + + +def hold_root_resume_descendants(q: Any, root_id: str, fence: dict, grant: dict) -> tuple: + """Replace lifted-fence eligibility with per-child holds; retain exact pauses. + + A fence-derived marker is not a checkpoint: convert it to a selection hold + before removing its fence, or it would be stranded. Older exact child grants + are revoked and selected again under this root grant. Return hold/marker + changes for the Resume transaction's snapshot rollback; revocations stay safe. + """ + from supervisor.budget_resume import revoke_exact_budget_resume + + held, markers = [], {} + for member in q.PENDING: + member_id = str(member.get("id") or "") + if not member_id or member_id == root_id or str(member.get("root_task_id") or "") != root_id: + continue + if isinstance(member.get("_budget_pause_resume"), dict): + revoke_exact_budget_resume(member, "root_resume_generation_changed") + pause = member.get("_budget_pause") + fence_derived = bool(isinstance(pause, dict) and not pause.get("exact_continuation") + and pause.get("scope") == "root" and pause.get("root_task_id") == root_id + and pause.get("fence_id") == fence.get("fence_id")) + if (pause is not None and not fence_derived) or isinstance(member.get(BUDGET_HOLD_KEY), dict): + continue + if fence_derived: + markers[member_id] = member.pop("_budget_pause") + hold_budget_row( + member, reason=HOLD_ROOT_FENCE_LIFTED, + detail="root resumed; this zero-dispatch sibling awaits explicit selection", + extra={"root_task_id": root_id, "root_grant_id": grant["grant_id"], + "root_resume_generation": grant["generation"], + **({"replaced_fence_marker": True} if fence_derived else {})}, + result_root=pathlib.Path(member.get("budget_drive_root") or q.DRIVE_ROOT)) + held.append(member_id) + rebound = refresh_root_grant_holds(q.PENDING, root_id, grant_id=grant["grant_id"], generation=grant["generation"]) + return held, markers, rebound + + +def refresh_root_grant_holds(pending: Any, root_task_id: str, *, grant_id: str, + generation: int) -> Dict[str, Dict[str, Any]]: + """Re-bind every unselected fence-lifted hold of ``root_task_id`` to its NEW grant. + + A root that paused again and was resumed again carries a newer grant; a + sibling held from the earlier Resume must be selectable under the live one, + not refused forever as stale. Returns the prior holds keyed by task id for + rollback when the snapshot cannot be persisted. + """ + prior: Dict[str, Dict[str, Any]] = {} + for member in pending: + hold = member.get(BUDGET_HOLD_KEY) if isinstance(member, dict) and isinstance(member.get(BUDGET_HOLD_KEY), dict) else None + if (hold is None or hold.get("selected") or str(hold.get("reason") or "") != HOLD_ROOT_FENCE_LIFTED + or str(hold.get("root_task_id") or member.get("root_task_id") or "") != root_task_id + or str(hold.get("root_grant_id") or "") == grant_id): + continue + prior[str(member.get("id") or "")] = dict(hold) + member[BUDGET_HOLD_KEY] = {**hold, "root_grant_id": grant_id, "root_resume_generation": int(generation), + "rebound_at": utc_now_iso()} + return prior + + +def select_held_budget_row(q: Any, task: Dict[str, Any], hold: Dict[str, Any], + *, selected_by: str) -> Dict[str, Any]: + """Record an explicit selection on a held row (queue lock held). + + The hold is released ONLY here, and only for a row that still proves it + never dispatched. A model-issued selection additionally needs the live + owner-derived Resume grant of its root: lifting a fence granted eligibility, + never continuation. + """ + task_id = str(task.get("id") or "") + result_root = pathlib.Path(task.get("budget_drive_root") or q.DRIVE_ROOT) + root_task_id = str(hold.get("root_task_id") or task.get("root_task_id") or task_id) + if hold.get("reason") not in {HOLD_ROOT_FENCE_LIFTED, HOLD_ROOT_FENCE_MEMBER_SELECTION}: + return {"ok": False, "error": str(hold.get("reason") or "budget_hold_unresolved")} + if selected_by: + root_grant = live_root_resume_grant(q, root_task_id, result_root) + if not root_grant: + return {"ok": False, "error": "root_resume_grant_missing", + "root_task_id": root_task_id, "action": "resume_root_first"} + if (hold.get("root_grant_id") and str(hold["root_grant_id"]) != root_grant["grant_id"]): + return {"ok": False, "error": "root_resume_generation_stale", + "root_task_id": root_task_id, "action": "resume_root_first"} + from supervisor.queue_transitions import pending_member_replay_safe + + safe, unsafe_error = pending_member_replay_safe(q, task) + if not safe: + return {"ok": False, "error": unsafe_error, "action": "cancel_or_new_run"} + selection = {**hold, "selected": True, "selected_at": utc_now_iso(), + "selected_by": str(selected_by or "owner")} + if task_id == root_task_id: + selection.update(root_grant_id=uuid.uuid4().hex, root_resume_generation=1) + prior_pause = task.pop("_budget_pause", None) + task[BUDGET_HOLD_KEY] = selection + if not q.persist_queue_snapshot(reason="budget_hold_selected"): + task[BUDGET_HOLD_KEY] = dict(hold) + if prior_pause is not None: + task["_budget_pause"] = prior_pause + return {"ok": False, "error": "snapshot_not_persisted"} + try: + write_task_result(result_root, task_id, STATUS_SCHEDULED, reason_code="", + resource_limit={"status": "resumed", "auto_resume": False, + "exact_continuation": False, **selection}) + except Exception: + log.debug("Failed to project held-row selection for %s", task_id, exc_info=True) + q.append_jsonl( + q.DRIVE_ROOT / "logs" / "events.jsonl", + {"ts": utc_now_iso(), "type": "budget_hold_selected", "task_id": task_id, + "root_task_id": root_task_id, "selected_by": selection["selected_by"], + "hold_reason": str(hold.get("reason") or "")}, + ) + return {"ok": True, "task_id": task_id, "root_task_id": root_task_id, + "selection": "budget_hold_released", "same_generation": True} + + +def live_root_resume_grant(q: Any, root_task_id: str, result_root: pathlib.Path) -> Dict[str, Any]: + """The root's CURRENT owner-derived Resume grant, or ``{}`` (owner Q9). + + A descendant continues only under a root Resume that is still live: the + root's durable pause row must carry a granted-or-consumed, unrevoked grant. + A root that paused again after that grant opens a NEW generation, so a late + selection carrying the old one finds nothing live and is refused. + """ + from ouroboros.budget_pause import STATE_RESUME_GRANTED, STATE_RESUMED, budget_pause_row + + root_task_id = str(root_task_id or "").strip() + if not root_task_id: + return {} + rows = [row for row in q.PENDING + if isinstance(row, dict) and str(row.get("id") or "") == root_task_id] + rows += [meta.get("task") for meta in q.RUNNING.values() + if isinstance(meta, dict) and isinstance(meta.get("task"), dict) + and str(meta["task"].get("id") or "") == root_task_id] + fence = q.BUDGET_ROOT_FENCES.get(root_task_id) + if fence: + # Legacy zero-dispatch root Resume selects only the root. Its selection + # grants child eligibility while the fence continues holding siblings. + for task in rows: + hold = task.get(BUDGET_HOLD_KEY) or {} + if budget_fence_selected(task, fence) and hold.get("root_grant_id"): + return {"grant_id": hold["root_grant_id"], + "generation": hold["root_resume_generation"], "pause_id": ""} + return {} + root_drive = next((str(row.get("budget_drive_root")) for row in rows + if isinstance(row, dict) and row.get("budget_drive_root")), "") + try: + row = budget_pause_row(pathlib.Path(root_drive or result_root), root_task_id) + except Exception: + log.debug("Root resume grant unreadable for %s", root_task_id, exc_info=True) + return {} + grant = row.get("grant") if isinstance(row.get("grant"), dict) else {} + if (row.get("state") not in {STATE_RESUME_GRANTED, STATE_RESUMED} + or not str(row.get("pause_id") or "").strip() + or not str(grant.get("grant_id") or "").strip() or grant.get("revoked_at")): + return {} + return {"grant_id": str(grant["grant_id"]), + "generation": int(row.get("resume_generation") or 0), + "pause_id": str(row.get("pause_id") or "")} diff --git a/supervisor/log_addressing.py b/supervisor/log_addressing.py index 31df61af6..90cc6b8af 100644 --- a/supervisor/log_addressing.py +++ b/supervisor/log_addressing.py @@ -144,6 +144,11 @@ def address_task_event(running: Any, drive_root: Any, payload: Dict[str, Any]) - for key in ("parent_task_id", "root_task_id"): if not payload.get(key) and task_row.get(key): payload[key] = str(task_row[key]) + if task_row.get("_is_direct_chat"): + # A direct turn resumed from its exact budget pause runs on a pooled + # worker (#1196): its frames keep the lane fact the direct lane would + # have stamped, so the chat chrome reads the same host truth. + payload.setdefault("_is_direct_chat", True) bound_chat = resolve_project_chat( drive_root, task_id, payload.get("parent_task_id"), payload.get("root_task_id") ) diff --git a/supervisor/owner_stop.py b/supervisor/owner_stop.py index 422fda941..98df5ef3f 100644 --- a/supervisor/owner_stop.py +++ b/supervisor/owner_stop.py @@ -248,19 +248,24 @@ def sweep_owner_stop_hold(q: Any, task_id: str, intent: Dict[str, Any], *, now: def _task_hard_bound_reached(q: Any, task_id: str, *, now: float) -> bool: """Mirror the queue's two hard time axes without introducing a new SSOT.""" + from ouroboros.model_wait import execution_elapsed_seconds + try: with q._queue_lock: meta = q.RUNNING.get(task_id) if isinstance(q.RUNNING, dict) else None if not isinstance(meta, dict): return False task = meta.get("task") if isinstance(meta.get("task"), dict) else {} - started_at = float(meta.get("started_at") or 0.0) + # The SAME clock the queue's timeout rail reads: wall time minus the + # quota union minus the separate budget-paused interval (#1196). + clock = {key: meta.get(key) for key in + ("started_at", "model_wait_quota_clock", "budget_paused_sec")} deadline_ts = float(q._task_deadline_ts(task) or 0.0) if deadline_ts and now >= deadline_ts: return True absolute_ceiling = q.get_task_abs_ceiling_sec() # None = no lifetime bound - if started_at > 0 and absolute_ceiling is not None: - return max(0.0, now - started_at) >= float(absolute_ceiling) + if float(clock.get("started_at") or 0.0) > 0 and absolute_ceiling is not None: + return execution_elapsed_seconds(clock, now) >= float(absolute_ceiling) return False except Exception: # Unreadable hard-bound authority is not permission to extend a task. diff --git a/supervisor/queue_snapshot.py b/supervisor/queue_snapshot.py index e9b9fcc9d..452ec5b0a 100644 --- a/supervisor/queue_snapshot.py +++ b/supervisor/queue_snapshot.py @@ -159,9 +159,17 @@ def persist_queue_snapshot(reason: str = "") -> bool: "_cancel_intent_authority_hold": t.get("_cancel_intent_authority_hold"), "_owner_wait_resume": t.get("_owner_wait_resume"), "_budget_pause_resume": t.get("_budget_pause_resume"), + # The durable non-dispatch hold (#1196) and any recorded + # selection: a restart must not silently make a held row runnable. + "_budget_pause_hold": t.get("_budget_pause_hold"), + # A direct owner-chat turn parked under its exact budget pause + # keeps its lane fact: it is resumed under the same id and its + # frames, census kind and delivery read that fact (#1196). + "_is_direct_chat": t.get("_is_direct_chat"), }, }) running_rows = [] + from ouroboros.model_wait import execution_elapsed_seconds from supervisor.task_model_wait import quota_waited_seconds now = time.time() for task_id, meta in running_items: @@ -176,8 +184,7 @@ def persist_queue_snapshot(reason: str = "") -> bool: "runtime_sec": round(max(0.0, now - started), 2) if started > 0 else 0.0, "quota_wait_sec": quota_waited_seconds(meta, now), "budget_paused_sec": paused_sec, - "execution_sec": (max(0.0, now - started - quota_waited_seconds(meta, now) - paused_sec) - if started > 0 else 0.0), + "execution_sec": execution_elapsed_seconds(meta if isinstance(meta, dict) else {}, now), "heartbeat_lag_sec": round(max(0.0, now - hb), 2) if hb > 0 else None, "soft_sent": bool(meta.get("soft_sent")), "task": task, }) @@ -277,6 +284,288 @@ def _fence_snapshot_running_rows(rows: Any, *, restored_ids: "set[str]") -> "lis return fenced +def _exact_pause_row(task: Any) -> bool: + """Whether this snapshot row is an EXACT mid-run budget pause locator (#1196).""" + pause = task.get("_budget_pause") if isinstance(task, dict) else None + return isinstance(pause, dict) and bool(pause.get("exact_continuation")) + + +def _retain_snapshot_pending(snapshot_pending: list, running_rows: list, *, stale: bool, + parked_direct: Optional[list] = None) -> tuple: + """Which snapshot rows survive the restore, and the RUNNING rows parked beside them. + + A grant that never reached a worker before this restart is not carried into + the new generation: it returns to its exact pause and the owner re-issues + Resume (a revocation that cannot be written leaves the row HELD). A grant + the durable row says was CONSUMED is never re-armed: that row is stale + carrier of work that ran on, so it is not retained at all and is fenced as + the running work it names (its id is returned third). A RUNNING row whose + durable pause was already complete is parked, never fenced; so is a direct + root the stop caught in that state (``parked_direct``). An exact budget + pause is retained WITHOUT waking whatever the snapshot's age — a corrupt, + missing or refused source becomes a typed, visible HOLD beside its marker, + never a dropped or cancelled task. An owner-wait handoff needs its + acknowledged restart transaction; every other row needs a fresh snapshot. + Returns ``(retained_rows, parked_rows, consumed_task_ids)``. + """ + from ouroboros.budget_pause import budget_pause_restore_refusal + from ouroboros.owner_wait import restore_owner_wait_allowed + from supervisor.events_budget import HOLD_RESTORE_REFUSED_PREFIX, hold_restored_budget_pause + from supervisor.queue_transitions import revoke_exact_budget_resume + + for task in snapshot_pending: + if isinstance(task.get("_budget_pause_resume"), dict): + revoke_exact_budget_resume(task, "restart_before_dispatch") + parked = _park_pausing_running_rows(running_rows, snapshot_pending) + list(parked_direct or []) + retained = [] + consumed: list = [] + for task in list(snapshot_pending) + parked: + if isinstance(task.get("_budget_pause_consumed"), dict): + consumed.append(str(task.get("id") or "")) + continue + if _exact_pause_row(task): + refusal = budget_pause_restore_refusal(_queue().DRIVE_ROOT, task) + retained.append(task if not refusal else hold_restored_budget_pause( + task, _queue().DRIVE_ROOT, reason=HOLD_RESTORE_REFUSED_PREFIX + refusal)) + elif task.get("_owner_wait_resume"): + if restore_owner_wait_allowed(_queue().DRIVE_ROOT, task): + retained.append(task) + elif not stale: + retained.append(task) + if consumed: + _queue().append_jsonl( + _queue().DRIVE_ROOT / "logs" / "supervisor.jsonl", + {"ts": utc_now_iso(), "type": "queue_restore_stale_consumed_grant_rows", + "task_ids": consumed, "action": "fenced_as_running_work"}, + ) + return retained, parked, consumed + + +def _raise_parked_root_fences(parked_pausing: list) -> None: + """Raise the root admission latch of every root parked at restore — AFTER the + snapshot's fence map is restored, or the restore would erase it.""" + if not parked_pausing: + return + from supervisor.events_budget import _set_root_budget_pause_locked + + with _queue()._queue_lock: + for parked in parked_pausing: + marker = parked.get("_budget_pause") if isinstance(parked.get("_budget_pause"), dict) else {} + if str(marker.get("scope") or "") == "root" and marker.get("root_task_id"): + fence = _set_root_budget_pause_locked(str(marker["root_task_id"]), marker) + marker["fence_id"] = fence["fence_id"] + + +def _refuse_restore_invalid_fences(snapshot_pending: list, *, budget: bool = False) -> int: + """Malformed fences never restore ordinary work. Invalid acceptance evidence + cancels ordinary rows; invalid budget evidence leaves them unrestored. + + A saved EXACT budget pause is the exception (#1196): it is retained as a + typed, non-dispatchable HOLD with its ORIGINAL ``_budget_pause`` locator + (an earlier restore hold on the same row is kept), appended to PENDING + directly, never cancelled — the owner's next Resume re-validates it. A + pause whose durable result is already terminal is left alone. Returns the + number of retained rows. + """ + from ouroboros.task_results import ( + _TRULY_TERMINAL_STATUSES, STATUS_CANCELLED, load_task_result, write_task_result, + ) + from supervisor.events_budget import ( + HOLD_INVALID_ACCEPTANCE_FENCE_SNAPSHOT, HOLD_INVALID_BUDGET_FENCE_SNAPSHOT, + budget_hold_fact, hold_restored_budget_pause, + ) + + cancelled: list[str] = [] + retained: list[str] = [] + skipped_terminal: list[str] = [] + for task in snapshot_pending: + task_id = str(task.get("id") or "") + if not task_id: + continue + if not _exact_pause_row(task): + cancelled.append(task_id) + continue + try: + stored = load_task_result(_queue().DRIVE_ROOT, task_id, strict=True) or {} + if str(stored.get("status") or "") in _TRULY_TERMINAL_STATUSES: + skipped_terminal.append(task_id) + continue + except Exception: + log.debug("Result authority unreadable for paused row %s; retained held", task_id, exc_info=True) + held = dict(task) + if budget_hold_fact(held) is None: + held = hold_restored_budget_pause( + held, _queue().DRIVE_ROOT, reason=(HOLD_INVALID_BUDGET_FENCE_SNAPSHOT if budget + else HOLD_INVALID_ACCEPTANCE_FENCE_SNAPSHOT), + detail="the snapshot's fence evidence was malformed; the saved pause is " + "retained, not cancelled, until an explicit Resume re-validates it") + with _queue()._queue_lock: + _append_held_pending_row(held) + retained.append(task_id) + _queue().append_jsonl( + _queue().DRIVE_ROOT / "logs" / "supervisor.jsonl", + {"ts": utc_now_iso(), "type": ("queue_restore_invalid_budget_root_fences" if budget + else "queue_restore_invalid_acceptance_fences"), + "affected_task_ids": cancelled, "action": "fail_closed_no_restore", + "retained_budget_paused_task_ids": retained, "skipped_terminal_task_ids": skipped_terminal}, + ) + try: + for task in ([] if budget else snapshot_pending): + task_id = str(task.get("id") or "") + if task_id and task_id in cancelled: + existing = load_task_result(_queue().DRIVE_ROOT, task_id) or {} + write_task_result( + _queue().DRIVE_ROOT, task_id, STATUS_CANCELLED, + **_cancel_result_fields( + task, existing=existing, + result="Task was not restored because its acceptance-fence snapshot was invalid."), + ) + except Exception: + log.warning("Failed to terminalize tasks from invalid acceptance-fence snapshot", exc_info=True) + return len(retained) + + +def _hold_acceptance_fenced_pause(task: dict) -> dict: + """An acceptance fence over the root holds a saved mid-run pause; it never cancels it.""" + from supervisor.events_budget import HOLD_ROOT_ACCEPTANCE_FENCED, hold_restored_budget_pause + + return hold_restored_budget_pause( + dict(task), _queue().DRIVE_ROOT, reason=HOLD_ROOT_ACCEPTANCE_FENCED, + detail="the root entered acceptance review; the saved pause is retained, not cancelled") + + +def _append_held_pending_row(task: dict) -> None: + """Append one proven-revivable held row to PENDING with queue-order facts (lock held).""" + if "_queue_seq" not in task: + _queue().QUEUE_SEQ_COUNTER_REF["value"] += 1 + task["_queue_seq"] = _queue().QUEUE_SEQ_COUNTER_REF["value"] + task.setdefault("queued_at", utc_now_iso()) + _queue().PENDING.append(task) + _queue().sort_pending() + + +def _park_pausing_running_rows(running_rows: list, snapshot_pending: list) -> list: + """RUNNING rows whose durable budget pause is LIVE become parked PENDING rows (#1196). + + The worker wrote the ``budget_pause`` row and stored its source BEFORE the + park event left it; a shutdown in that window leaves a RUNNING row in the + snapshot whose exact continuation is complete on disk. Handing that row to + the shutdown cancel fence would cancel a saved pause, so it is parked under + its exact marker instead: ``pausing``/``paused`` rows directly (the row is + confirmed ``paused`` with ``pause_source=restart_during_pausing``), and a + ``resume_granted`` row whose grant was never consumed (the loop writes + ``consumed_at`` before any new effect) after its grant is revoked — a + revocation that cannot be written keeps the row parked under a typed hold. + A consumed grant is ordinary running work and keeps the ordinary fence; an + unreadable pause authority proves nothing and keeps it too. Rows already in + the pending snapshot are never duplicated. + """ + from ouroboros.budget_pause import ( + STATE_PAUSED, STATE_PAUSING, STATE_RESUME_GRANTED, budget_pause_row, exact_pause_marker, + set_budget_pause, + ) + from supervisor.events_budget import HOLD_RESTART_REVOCATION_UNWRITTEN, hold_budget_row + + pending_ids = {str(row.get("id") or "") for row in snapshot_pending if isinstance(row, dict)} + parked: list = [] + for row in running_rows if isinstance(running_rows, list) else []: + if not isinstance(row, dict) or not isinstance(row.get("task"), dict): + continue + task = dict(row["task"]) + task_id = str(task.get("id") or "") + if not task_id or task_id in pending_ids: + continue + attempt = int(row.get("attempt") or task.get("_attempt") or 1) + result_root = pathlib.Path(task.get("budget_drive_root") or _queue().DRIVE_ROOT) + try: + pause = budget_pause_row(result_root, task_id) + except Exception: + continue + if (not pause or int(pause.get("task_attempt") or 0) != attempt or not pause.get("source_ref") + or pause.get("state") not in {STATE_PAUSING, STATE_PAUSED, STATE_RESUME_GRANTED}): + continue + pause_id = str(pause.get("pause_id") or "") + grant = pause.get("grant") if isinstance(pause.get("grant"), dict) else {} + held_reason = "" + if pause.get("state") == STATE_RESUME_GRANTED: + if grant.get("consumed_at") or grant.get("revoked_at"): + continue + revoked = {**grant, "revoked_at": utc_now_iso(), "revoke_reason": "restart_before_consumption"} + try: + set_budget_pause(result_root, task_id, {**pause, "state": STATE_PAUSED, "grant": revoked}, + expected_pause_id=pause_id) + pause = {**pause, "state": STATE_PAUSED, "grant": revoked} + except Exception: + log.warning("Restart could not revoke the unconsumed grant of %s; parked under a hold", + task_id, exc_info=True) + held_reason = HOLD_RESTART_REVOCATION_UNWRITTEN + elif pause.get("state") == STATE_PAUSING: + try: + set_budget_pause(result_root, task_id, {**pause, "state": STATE_PAUSED, + "paused_confirmed_at": time.time(), + "pause_source": "restart_during_pausing"}, + expected_pause_id=pause_id) + except Exception: + log.warning("Parked pausing row %s stays 'pausing' (row unwritable at restore)", + task_id, exc_info=True) + task["_attempt"] = attempt + task.pop("_budget_pause_resume", None) + task["_budget_pause"] = exact_pause_marker(pause, default_root=str(task.get("root_task_id") or task_id)) + if held_reason: + hold_budget_row( + task, reason=held_reason, + detail="restart found an unconsumed Resume grant whose revocation could not be written", + extra={"pause_id": pause_id, "grant_id": str(grant.get("grant_id") or ""), + "root_task_id": str(task.get("root_task_id") or task_id)}, + result_root=result_root) + parked.append(task) + if parked: + _queue().append_jsonl( + _queue().DRIVE_ROOT / "logs" / "supervisor.jsonl", + {"ts": utc_now_iso(), "type": "queue_restore_parked_budget_pauses", + "task_ids": [str(task.get("id") or "") for task in parked]}, + ) + return parked + + +def _park_direct_root_pauses(task_ids: list, snapshot_pending: list) -> list: + """Direct roots the stop caught whose EXACT pause was already complete on disk. + + A direct turn is never a RUNNING row, so the roster is the only place the + stop names it; handing such an id to the shutdown cancel fence would + cancel a saved pause (#1196). Its queue record is rebuilt from the durable + result the pause wrote (chat, lane fact, origin, metadata, attempt) and + parked through the same path a RUNNING row takes; a turn with no complete + pause (no row, no source, another attempt) stays on the fence path. + """ + from ouroboros.budget_pause import budget_pause_row + from ouroboros.task_results import load_task_result + + rows: list = [] + for task_id in task_ids: + task_id = str(task_id or "") + if not task_id: + continue + try: + stored = load_task_result(_queue().DRIVE_ROOT, task_id, strict=True) or {} + pause = budget_pause_row(_queue().DRIVE_ROOT, task_id) + except Exception: + continue + if not (stored and pause and pause.get("source_ref") and pause.get("is_direct_chat")): + continue + attempt = int(pause.get("task_attempt") or 1) + record: dict = { + "id": task_id, "type": "task", "chat_id": stored.get("chat_id"), "_is_direct_chat": True, + "_attempt": attempt, "root_task_id": task_id, "depth": 0, + } + for key in ("origin_message_ref", "origin_message_text", "metadata", "project_id", + "task_contract", "title", "text", "budget_drive_root"): + if stored.get(key) not in (None, ""): + record[key] = stored[key] + rows.append({"task": record, "attempt": attempt}) + return _park_pausing_running_rows(rows, snapshot_pending) + + def _descends_from(task: Any, roots: "set[str]", pending_by_id: dict) -> bool: """Whether this row's lineage reaches ANY of ``roots``. @@ -414,41 +703,33 @@ def restore_pending_from_snapshot( for row in (snap.get("pending") or []) if isinstance(row, dict) and isinstance(row.get("task"), dict) ] - from ouroboros.owner_wait import restore_owner_wait_allowed - from ouroboros.budget_pause import restore_budget_pause_allowed - from supervisor.queue_transitions import revoke_exact_budget_resume - - for task in snapshot_pending: - # A grant that never reached a worker before this restart is not - # carried into the new generation: it returns to its exact pause - # and the owner re-issues Resume (no dispatch across a restart the - # grant did not see; the row's single-use identity stays honest). - if isinstance(task.get("_budget_pause_resume"), dict): - revoke_exact_budget_resume(task, "restart_before_dispatch") - snapshot_pending = [ - task for task in snapshot_pending - if (not task.get("_owner_wait_resume") and not stale) - or (task.get("_owner_wait_resume") and restore_owner_wait_allowed(_queue().DRIVE_ROOT, task)) - # An exact budget pause is retained WITHOUT waking, whatever the - # snapshot's age: the durable row and readable source authorize the - # locator; nothing dispatches it until an explicit owner Resume. - or (stale and restore_budget_pause_allowed(_queue().DRIVE_ROOT, task)) - ] + running_rows = snap.get("running") + running_rows = running_rows if isinstance(running_rows, list) else [] # The pre-restart RUNNING rows are read HERE, before the stale gate: this # is the last moment the list exists, and a stale snapshot is exactly the # case where nothing else will ever settle them. Direct-chat roots never # reach the queue, so the roster `queue.init` took over is the only place # they are named; both lists describe work the stop caught, and one fence - # call gives them the one cancel-intent path custody settles. + # call gives them the one cancel-intent path custody settles — except a + # direct root whose exact budget pause was complete on disk, which is + # parked under its own id like a paused RUNNING row (#1196). direct_roots = dict(_queue().PRIOR_DIRECT_ROOTS) # An in-process supervisor revival re-runs queue init while direct turns of THIS process are # alive: the roster then names live work, not what a stop caught. from supervisor.active_activity import get_direct_activity_registry live_direct = {str(row.get("activity_id") or "") for row in get_direct_activity_registry().snapshot()} - running_rows = snap.get("running") + pending_ids = {str(task.get("id") or "") for task in snapshot_pending} + direct_caught = [task_id for task_id in direct_roots.get("task_ids") or [] + if task_id not in live_direct and task_id not in pending_ids] + parked_direct = _park_direct_root_pauses(direct_caught, snapshot_pending) + snapshot_pending, parked_pausing, consumed_rows = _retain_snapshot_pending( + snapshot_pending, running_rows, stale=stale, parked_direct=parked_direct) fenced_running = _fence_snapshot_running_rows( - (running_rows if isinstance(running_rows, list) else []) - + [{"id": task_id} for task_id in direct_roots.get("task_ids") or [] if task_id not in live_direct], + running_rows + + [{"id": task_id} for task_id in direct_caught] + # A stale row whose grant was consumed names work that ran on: the + # ordinary fence, never a re-armed pause (revoke_exact_budget_resume). + + [{"id": task_id} for task_id in consumed_rows], restored_ids={str(task.get("id") or "") for task in snapshot_pending}, ) if terminalized is not None: @@ -462,59 +743,32 @@ def restore_pending_from_snapshot( queue_seq_counter_ref=_queue().QUEUE_SEQ_COUNTER_REF, sort_pending=_queue().sort_pending, ) fenced_roots, malformed_fences, malformed_budget_fences = restore_queue_fences(raw_fences, raw_budget_fences) - if malformed_budget_fences: - _queue().append_jsonl( - _queue().DRIVE_ROOT / "logs" / "supervisor.jsonl", - {"ts": utc_now_iso(), "type": "queue_restore_invalid_budget_root_fences", - "action": "fail_closed_no_restore"}, - ) - # The fence was already minted above, and the boot notice names it: - # a fail-closed exit may skip the restore, never the record of what - # it handed to cancellation custody. - _record_queue_restore(restored=restored, terminalized_running=fenced_running, - direct_roots_incomplete=direct_roots.get("incomplete", False)) - return restored - if malformed_fences: - affected = [str(task.get("id") or "") for task in snapshot_pending if task.get("id")] - _queue().append_jsonl( - _queue().DRIVE_ROOT / "logs" / "supervisor.jsonl", - { - "ts": utc_now_iso(), - "type": "queue_restore_invalid_acceptance_fences", - "affected_task_ids": affected, - "action": "fail_closed_no_restore", - }, - ) - try: - for task in snapshot_pending: - task_id = str(task.get("id") or "") - if task_id: - existing = load_task_result(_queue().DRIVE_ROOT, task_id) or {} - write_task_result( - _queue().DRIVE_ROOT, - task_id, - STATUS_CANCELLED, - **_cancel_result_fields( - task, - existing=existing, - result="Task was not restored because its acceptance-fence snapshot was invalid.", - ), - ) - except Exception: - log.warning("Failed to terminalize tasks from invalid acceptance-fence snapshot", exc_info=True) + if not malformed_budget_fences: + _raise_parked_root_fences(parked_pausing) + if malformed_budget_fences or malformed_fences: + restored += _refuse_restore_invalid_fences(snapshot_pending, budget=malformed_budget_fences) _record_queue_restore(restored=restored, terminalized_running=fenced_running, direct_roots_incomplete=direct_roots.get("incomplete", False)) + if restored > 0: + _queue().persist_queue_snapshot(reason="queue_restored") return restored skipped_terminal, invalid_depth_restore = 0, [] cancel_authority_holds: list[str] = [] skipped_fenced, blocked_restore, orphan_children = [], [], [] + acceptance_held: list[str] = [] interrupted = _interrupted_ancestors(fenced_running, snapshot_pending, pending_by_id) for task in snapshot_pending: chat_id = task.get("chat_id") if not task.get("id") or chat_id is None or chat_id == "": continue - if _descends_from(task, fenced_roots, pending_by_id): + if _descends_from(task, fenced_roots, pending_by_id) and _exact_pause_row(task): + # #1196: an acceptance fence over the root cancels NOTHING that was + # paused mid-run: retained under a typed hold, the ordinary revival + # checks below still apply, and the row is appended directly. + task = _hold_acceptance_fenced_pause(task) + acceptance_held.append(str(task.get("id") or "")) + elif _descends_from(task, fenced_roots, pending_by_id): task_id = str(task.get("id") or "") skipped_fenced.append(task_id) try: @@ -539,6 +793,26 @@ def restore_pending_from_snapshot( existing = load_task_result(_queue().DRIVE_ROOT, str(task.get("id")), strict=True) existing_status = str(existing.get("status") or "") if existing else "" except Exception: + if _exact_pause_row(task): + # #1196: an unreadable result authority never terminalizes a + # saved pause. The row stays PENDING under its ORIGINAL + # locator and a typed hold (the earlier restore hold, if + # any, is kept); the next Resume re-reads the authority. + from ouroboros.budget_pause import RESTORE_REFUSAL_RECORD_UNREADABLE + from supervisor.events_budget import ( + HOLD_RESTORE_REFUSED_PREFIX, budget_hold_fact, hold_restored_budget_pause, + ) + + if budget_hold_fact(task) is None: + task = hold_restored_budget_pause( + dict(task), _queue().DRIVE_ROOT, + reason=HOLD_RESTORE_REFUSED_PREFIX + RESTORE_REFUSAL_RECORD_UNREADABLE) + _append_held_pending_row(task) + restored += 1 + acceptance_held.append(str(task.get("id") or "")) + log.warning("Snapshot restore retained paused row %s under a hold: its " + "result authority is unreadable", task.get("id"), exc_info=True) + continue # Result-authority loss already has terminal custody: once # its retry can prove a writable result, the task is failed # rather than replayed over an unknown exact-id lifecycle. @@ -639,6 +913,12 @@ def restore_pending_from_snapshot( orphan_children.append(str(task.get("id") or "")) skipped_terminal += 1 continue + if str(task.get("id") or "") in acceptance_held: + # Proven revivable and unowned above; the fence gates DISPATCH + # through the hold (the enqueue fence would drop the row). + _append_held_pending_row(task) + restored += 1 + continue admitted = _queue().enqueue_task(task, restoring_snapshot=True) if isinstance(admitted, dict) and admitted.get("_admission_blocked"): _queue().restore_invalid_depth_admission(task, admitted, drive_root=_queue().DRIVE_ROOT, pending=_queue().PENDING, blocked=blocked_restore, terminalized=invalid_depth_restore, queue_seq_counter_ref=_queue().QUEUE_SEQ_COUNTER_REF) @@ -648,13 +928,14 @@ def restore_pending_from_snapshot( log.warning("Deferred snapshot sort failed; custody retained", exc_info=True) continue restored += 1 - if skipped_fenced: + if skipped_fenced or acceptance_held: _queue().append_jsonl( _queue().DRIVE_ROOT / "logs" / "supervisor.jsonl", { "ts": utc_now_iso(), "type": "queue_restore_skipped_acceptance_fence", "task_ids": skipped_fenced, + "held_budget_paused_task_ids": acceptance_held, "root_task_ids": sorted(fenced_roots), }, ) diff --git a/supervisor/queue_timeouts.py b/supervisor/queue_timeouts.py index f899f2fa6..ec62a1cdf 100644 --- a/supervisor/queue_timeouts.py +++ b/supervisor/queue_timeouts.py @@ -16,7 +16,8 @@ import uuid from typing import Any, Dict from supervisor.cognitive_operations import _active_operation_progressing -from supervisor.task_model_wait import model_waiting, quota_waited_seconds +from ouroboros.model_wait import execution_elapsed_seconds +from supervisor.task_model_wait import model_waiting from supervisor.task_reaper import ( resolve_grace_episode_for_spared_task as _resolve_grace_episode_for_spared_task, ) @@ -205,8 +206,7 @@ def _enforce_task_timeouts_locked( last_hb = float(meta.get("last_heartbeat_at") or started_at) # Execution time = wall clock minus quota waits minus the SEPARATE # budget-paused interval (#1196); started_at itself is never moved. - runtime_sec = max(0.0, now - started_at - quota_waited_seconds(meta, now) - - float(meta.get("budget_paused_sec") or 0.0)) + runtime_sec = execution_elapsed_seconds(meta, now) hb_lag_sec = max(0.0, now - last_hb) hb_stale = hb_lag_sec >= _queue().HEARTBEAT_STALE_SEC _wid = meta.get("worker_id") diff --git a/supervisor/queue_transitions.py b/supervisor/queue_transitions.py index 6652c76df..f78086d1a 100644 --- a/supervisor/queue_transitions.py +++ b/supervisor/queue_transitions.py @@ -9,7 +9,9 @@ QUIESCENCE state, driven entirely through the queue module: - the acceptance FENCE (open/inspect/seal a root so no new subtask is admitted while its acceptance review runs), -- explicit BUDGET resume of a zero-dispatch task and its root latch, +- explicit BUDGET resume of a zero-dispatch task and its root latch (the + exact mid-run continuation's single-use GRANT and its revocation live in + ``supervisor/budget_resume.py`` and are re-exported here), - fenced PROJECT deletion: cancel the project's tree, then tombstone only after the tree is provably quiescent, - shared live-subtree and timeout-retry lineage views used by cancellation and @@ -218,47 +220,89 @@ def budget_pause_fact(task, fences=None): pause = task.get("_budget_pause") if isinstance(task, dict) else None if isinstance(pause, dict): return pause + from supervisor.events_budget import BUDGET_HOLD_KEY, budget_fence_selected, budget_hold_fact + fence_map = _queue_module().BUDGET_ROOT_FENCES if fences is None else fences root_id = str((task or {}).get("root_task_id") or (task or {}).get("id") or "") fence = fence_map.get(root_id) - if isinstance(fence, dict) and str(fence.get("status") or "") in {"active", "paused"}: + if (isinstance(fence, dict) and str(fence.get("status") or "") in {"active", "paused"} + and not budget_fence_selected(task, fence)): + # An explicit selection recorded against THIS fence released exactly one + # row; its siblings stay fenced (owner Q9). return fence + if isinstance(task, dict) and budget_hold_fact(task) is not None: + return dict(task[BUDGET_HOLD_KEY]) return None -def resume_budget_paused_task(task_id: str) -> Dict[str, Any]: +def pending_member_replay_safe(q: Any, member: Dict[str, Any]) -> Tuple[bool, str]: + """Whether one PENDING row genuinely never dispatched (zero physical calls).""" + member_id = str(member.get("id") or "") + cost_fields = q.reconstruct_task_cost( + member_id, fields=True, + drive_root=pathlib.Path(member.get("budget_drive_root") or q.DRIVE_ROOT), + ) + if cost_fields.get("cost_accounting_status") != "available": + return False, "accounting_unavailable" + retry_lineage = bool( + int(member.get("_attempt") or 1) > 1 + or member.get("original_task_id") or member.get("timeout_retry_from") + ) + return bool( + int(cost_fields.get("total_rounds") or 0) == 0 + and not bool(cost_fields.get("ledger_integrity_degraded")) + and not retry_lineage + ), "replay_unsafe" + + +def resume_budget_paused_task(task_id: str, *, selected_by: str = "") -> Dict[str, Any]: """Explicitly resume one budget-paused task and, if needed, its root latch. A replay-safe ZERO-dispatch row is released as before. An EXACT continuation row (#1196) receives one single-use grant instead - (``grant_exact_budget_resume``); it is never re-run from scratch. + (``grant_exact_budget_resume``); it is never re-run from scratch. A row held + behind a lifted root fence is released only by recording the selection. + ``selected_by`` names the MODEL-issued request (owner Q9); an empty value is + the owner's own act, which needs no root grant above itself. """ q = _queue_module() task_id = str(task_id or "").strip() if not task_id: return {"ok": False, "error": "missing_task_id"} + external = None + with q._queue_lock: + located = next((item for item in q.PENDING if str(item.get("id") or "") == task_id), None) + exact = bool(located is not None and isinstance(located.get("_budget_pause"), dict) + and located["_budget_pause"].get("exact_continuation")) + result_root = pathlib.Path((located or {}).get("budget_drive_root") or q.DRIVE_ROOT) + if exact: + # Fresh custody at EVERY grant, read and stop-requested OUTSIDE the + # queue lock (a harness round trip is not a queue-lock tenant); the + # grant below re-locates the row and decides on this observation. + from ouroboros.budget_pause import observe_task_runs + + external = observe_task_runs(result_root, task_id, reason="budget_resume_uncovered_cost") with q._queue_lock: task = next((item for item in q.PENDING if str(item.get("id") or "") == task_id), None) if task is None: return {"ok": False, "error": "task_not_pending"} pause = task.get("_budget_pause") if isinstance(task.get("_budget_pause"), dict) else None if pause and pause.get("exact_continuation"): - return grant_exact_budget_resume(task, pause) + return grant_exact_budget_resume(task, pause, selected_by=selected_by, external=external) + from supervisor.events_budget import budget_hold_fact, select_held_budget_row + + hold = budget_hold_fact(task) + if hold is not None and not pause: + return select_held_budget_row(q, task, hold, selected_by=selected_by) if not pause: - # A root marker blocks every already-pending sibling without - # copying pause state onto each task. An explicit resume request - # may nominate any genuinely zero-dispatch member of that root. candidate_root = str(task.get("root_task_id") or task_id).strip() candidate_fence = q.BUDGET_ROOT_FENCES.get(candidate_root) if not isinstance(candidate_fence, dict): return {"ok": False, "error": "task_not_budget_paused"} - pause = { - **candidate_fence, - "status": "paused_before_dispatch", - "physical_calls": 0, - "replay_safe": True, - "resume_policy": "manual_same_generation", - } + if isinstance(task.get("_budget_pause_resume"), dict): + return {"ok": False, "error": "resume_already_granted", + "grant_id": str(task["_budget_pause_resume"].get("grant_id") or "")} + pause = {**candidate_fence, "physical_calls": 0, "replay_safe": True} root_scope = str(pause.get("scope") or "") == "root" root_task_id = str(pause.get("root_task_id") or "").strip() fence = q.BUDGET_ROOT_FENCES.get(root_task_id) if root_scope and root_task_id else None @@ -266,26 +310,7 @@ def resume_budget_paused_task(task_id: str) -> Dict[str, Any]: return {"ok": False, "error": "root_budget_fence_missing", "action": "cancel_or_new_run"} if root_scope and str(pause.get("fence_id") or "") != str(fence.get("fence_id") or ""): return {"ok": False, "error": "replay_unsafe", "action": "cancel_or_new_run"} - def _pending_member_is_replay_safe(member: Dict[str, Any]) -> tuple[bool, str]: - member_id = str(member.get("id") or "") - cost_fields = q.reconstruct_task_cost( - member_id, - fields=True, - drive_root=pathlib.Path(member.get("budget_drive_root") or q.DRIVE_ROOT), - ) - if cost_fields.get("cost_accounting_status") != "available": - return False, "accounting_unavailable" - retry_lineage = bool( - int(member.get("_attempt") or 1) > 1 - or member.get("original_task_id") or member.get("timeout_retry_from") - ) - return bool( - int(cost_fields.get("total_rounds") or 0) == 0 - and not bool(cost_fields.get("ledger_integrity_degraded")) - and not retry_lineage - ), "replay_unsafe" - - nominated_safe, nominated_error = _pending_member_is_replay_safe(task) + nominated_safe, nominated_error = pending_member_replay_safe(q, task) nominated_safe = bool( nominated_safe and pause.get("replay_safe") @@ -298,34 +323,21 @@ def resume_budget_paused_task(task_id: str) -> Dict[str, Any]: "action": "cancel_or_new_run", } if root_scope: - # Clearing one root latch makes every pending member assignable. Check - # those members together under the existing queue lock; completed - # historical siblings are deliberately irrelevant. - unsafe_members: list[str] = [] - for member in q.PENDING: - member_id = str(member.get("id") or "") - member_root = str(member.get("root_task_id") or member_id) - if member_root != root_task_id or member_id == task_id: - continue - member_safe, _member_error = _pending_member_is_replay_safe(member) - if not member_safe: - unsafe_members.append(member_id) - if unsafe_members: - return { - "ok": False, - "error": "root_replay_unsafe", - "unsafe_task_ids": unsafe_members, - "action": "cancel_or_new_run", - } + # Root Resume selects the root alone; children remain behind this + # same fence until individually selected under its root grant (Q9). + from supervisor.events_budget import HOLD_ROOT_FENCE_MEMBER_SELECTION, hold_budget_row + + hold = hold_budget_row( + task, reason=HOLD_ROOT_FENCE_MEMBER_SELECTION, + extra={"root_task_id": root_task_id, "fence_id": fence["fence_id"]}) + return select_held_budget_row(q, task, hold, selected_by=selected_by) resumed_at = utc_now_iso() prior_pause = dict(pause) task.pop("_budget_pause", None) task["budget_resumed_at"] = resumed_at - if root_scope: - q.BUDGET_ROOT_FENCES.pop(root_task_id, None) q.persist_queue_snapshot( - reason="budget_root_explicit_resume" if root_scope else "budget_pause_explicit_resume", + reason="budget_pause_explicit_resume", ) try: from ouroboros.task_results import STATUS_SCHEDULED, write_task_result @@ -357,226 +369,6 @@ def resume_budget_paused_task(task_id: str) -> Dict[str, Any]: return {"ok": True, "task_id": task_id, "same_generation": True} -def _root_budget_paused_locked(q: Any, root_task_id: str, *, except_task_id: str = "") -> bool: - """Whether the ROOT of a tree is itself still budget-paused (queue lock held). - - Owner Q9: a root's Resume makes its own budget-paused descendants ELIGIBLE; - a descendant cannot be resumed under a root that is still paused. - """ - root_task_id = str(root_task_id or "") - if not root_task_id or root_task_id == except_task_id: - return False - for row in q.PENDING: - if str(row.get("id") or "") == root_task_id and isinstance(row.get("_budget_pause"), dict): - return True - return False - - -def grant_exact_budget_resume(task: Dict[str, Any], pause: Dict[str, Any]) -> Dict[str, Any]: - """Validate money / Stop / deadline / finite lifetime / checkpoint, then mint ONE grant. - - Called with the queue lock held. The paused interval is carried as its own - ``paused_duration_sec`` (the original ``started_at`` is never moved; the - quota clock is not a pause clock). Refusals are typed and leave the task - paused: money still exhausted needs an owner increase (Q3/Q7/Q10), a live - cancel intent or a passed deadline or an exhausted finite lifetime is not - continued, a missing or unreadable source refuses (a stale locator revives - nothing). Direct actors never reach here (they were never queued). - """ - import time - import uuid - - from ouroboros.artifacts import read_actor_source_bytes - from ouroboros.budget_pause import ( - LIVE_PAUSE_STATES, STATE_RESUME_GRANTED, set_budget_pause, - ) - from ouroboros.cancel_intents import has_active_intent - from ouroboros.config import get_task_abs_ceiling_sec - from ouroboros.deadline_utils import parse_deadline_ts, utc_now - from ouroboros.model_wait import quota_waited_seconds - from ouroboros.task_results import _TRULY_TERMINAL_STATUSES, load_task_result - from supervisor.state import budget_remaining - - q = _queue_module() - task_id = str(task.get("id") or "") - checkpoint = pause.get("checkpoint") if isinstance(pause.get("checkpoint"), dict) else {} - pause_id = str(checkpoint.get("pause_id") or "") - result_root = pathlib.Path(task.get("budget_drive_root") or q.DRIVE_ROOT) - if any((pathlib.Path(q.DRIVE_ROOT) / "state" / name).exists() - for name in ("owner_restart_no_resume.flag", "panic_stop.flag")): - return {"ok": False, "error": "restart_no_resume", "action": "wait_or_cancel"} - try: - result_row = load_task_result(result_root, task_id, strict=True) or {} - except Exception: - return {"ok": False, "error": "pause_record_unreadable", "action": "cancel_or_new_run"} - row = result_row.get("budget_pause") if isinstance(result_row.get("budget_pause"), dict) else {} - if result_row.get("status") in _TRULY_TERMINAL_STATUSES: - return {"ok": False, "error": "task_terminal"} - if (not row or row.get("pause_id") != pause_id or row.get("state") not in LIVE_PAUSE_STATES - or not row.get("source_ref")): - return {"ok": False, "error": "pause_record_missing", "action": "cancel_or_new_run"} - if row.get("state") == STATE_RESUME_GRANTED and not (row.get("grant") or {}).get("revoked_at"): - return {"ok": False, "error": "resume_already_granted", - "grant_id": (row.get("grant") or {}).get("grant_id")} - try: - read_actor_source_bytes(result_root, task_id, row["source_ref"]) - except Exception: - return {"ok": False, "error": "pause_source_unreadable", "action": "cancel_or_new_run"} - external = row.get("external_runs") if isinstance(row.get("external_runs"), dict) else {} - if external.get("custody_read") != "ok": - # The pause could not read its delegated custody: UNKNOWN is held, not - # cleared. Re-read now (observation only — no stop is requested from the - # supervisor); refuse while it stays unreadable. - try: - from ouroboros import delegate_custody as custody - - held = [run for run in custody.replay(pathlib.Path(q.DRIVE_ROOT)).values() - if str(getattr(run, "task_id", "") or "") == task_id and not getattr(run, "settled", True)] - external = { - "runs": [{"run_id": str(getattr(run, "run_id", "") or ""), "route": str(getattr(run, "route", "") or ""), - "state": "running", "stop_outcome": "not_requested_at_grant", - "cost_coverage": "unproven_preterminal", "stop_policy": "request_stop"} - for run in held], - "observed_at": time.time(), "custody_read": "ok", - "coverage_basis": "reobserved_at_grant_after_pause_read_failure", - } - row = {**row, "external_runs": external} - except Exception as exc: - return {"ok": False, "error": "external_custody_unreadable", "detail": f"{type(exc).__name__}: {str(exc)[:200]}"} - try: - if has_active_intent(pathlib.Path(q.DRIVE_ROOT), task_id, strict=True): - return {"ok": False, "error": "cancel_intent_active"} - except Exception: - return {"ok": False, "error": "cancellation_authority_unavailable"} - deadline = parse_deadline_ts(task.get("deadline_at") or (task.get("task_contract") or {}).get("deadline_at")) - if deadline is not None and deadline <= utc_now(): - return {"ok": False, "error": "deadline_passed"} - now = time.time() - started = float(row.get("started_at") or checkpoint.get("started_at") or 0.0) - paused_at = float(row.get("paused_at") or checkpoint.get("paused_at") or now) - prior_paused = float(row.get("paused_duration_sec") or 0.0) - quota_waited = quota_waited_seconds( - {"model_wait_quota_clock": row.get("model_wait_quota_clock") or {}}, paused_at) - executed_sec = max(0.0, paused_at - started - quota_waited - prior_paused) if started else 0.0 - ceiling = get_task_abs_ceiling_sec() # None = unlimited lifetime; 0 = exhausted - if ceiling is not None and started and executed_sec >= float(ceiling): - return {"ok": False, "error": "lifetime_exhausted", "executed_sec": round(executed_sec, 1)} - try: - remaining = budget_remaining(q.load_state(), strict=True, allow_stale=True) - except Exception: - return {"ok": False, "error": "monetary_authority_unavailable"} - if remaining <= 0: - return {"ok": False, "error": "budget_still_exhausted", "action": "increase_budget_then_resume"} - root_task_id = str(pause.get("root_task_id") or task.get("root_task_id") or task_id) - if str(pause.get("scope") or "") == "root": - try: - from ouroboros.usage_accounting import refresh_root_accounting - - tree = refresh_root_accounting(result_root, root_task_id, max_age_sec=0.0) or {} - limit = tree.get("limit_usd") - if (limit is not None and tree.get("accounted_usd") is not None - and float(tree["accounted_usd"]) >= float(limit) - 1e-9): - return {"ok": False, "error": "root_hard_cap_exhausted", - "action": "increase_budget_then_resume"} - except Exception: - log.debug("Root accounting unavailable at exact resume for %s", task_id, exc_info=True) - if _root_budget_paused_locked(q, root_task_id, except_task_id=task_id): - return {"ok": False, "error": "root_still_paused", "root_task_id": root_task_id, - "action": "resume_root_first"} - grant = { - "grant_id": uuid.uuid4().hex, "granted_at": utc_now_iso(), "granted_at_ts": now, - "single_use": True, "paused_duration_sec": prior_paused + max(0.0, now - paused_at), - "executed_sec_before_pause": round(executed_sec, 3), - "refresh_planning_threshold": str(row.get("rail") or "") in { - "graceful_ceiling", "wrapup_last_fit", "soft_land"}, - } - prior_pause = dict(pause) - try: - set_budget_pause(result_root, task_id, {**row, "state": STATE_RESUME_GRANTED, "grant": grant}, - expected_pause_id=pause_id) - except Exception as exc: - return {"ok": False, "error": "grant_not_recorded", "detail": str(exc)[:200]} - task.pop("_budget_pause", None) - task["_budget_pause_resume"] = { - **checkpoint, "grant_id": grant["grant_id"], "granted_at": grant["granted_at"], - "paused_duration_sec": grant["paused_duration_sec"], "pause": prior_pause, - } - task["budget_resumed_at"] = grant["granted_at"] - fence = q.BUDGET_ROOT_FENCES.get(root_task_id) - fence_released = False - if (task_id == root_task_id and isinstance(fence, dict) and str(fence.get("fence_id") or "") - == str(prior_pause.get("fence_id") or fence.get("fence_id"))): - # The root's own Resume lifts its admission latch: exact-continuation - # descendants keep their OWN `_budget_pause` rows and are only ELIGIBLE - # now; the model selects each through this same control (Q9). - q.BUDGET_ROOT_FENCES.pop(root_task_id, None) - fence_released = True - if not q.persist_queue_snapshot(reason="budget_exact_resume_granted"): - task.pop("_budget_pause_resume", None) - task["_budget_pause"] = prior_pause - if fence_released: - q.BUDGET_ROOT_FENCES[root_task_id] = fence - try: - set_budget_pause(result_root, task_id, row, expected_pause_id=pause_id) - except Exception: - log.warning("Exact resume grant rollback remains unpersisted for %s", task_id, exc_info=True) - return {"ok": False, "error": "snapshot_not_persisted"} - try: - from ouroboros.task_results import STATUS_SCHEDULED, write_task_result - - write_task_result( - result_root, task_id, STATUS_SCHEDULED, reason_code="", - resource_limit={**prior_pause, "status": "resume_granted", "resumed_at": grant["granted_at"], - "grant_id": grant["grant_id"], "auto_resume": False}, - ) - except Exception: - log.debug("Failed to project exact budget resume for %s", task_id, exc_info=True) - eligible = [str(r.get("id") or "") for r in q.PENDING - if isinstance(r.get("_budget_pause"), dict) and r["_budget_pause"].get("exact_continuation") - and str(r.get("root_task_id") or "") == root_task_id and str(r.get("id") or "") != task_id] - q.append_jsonl( - q.DRIVE_ROOT / "logs" / "events.jsonl", - {"ts": utc_now_iso(), "type": "budget_task_explicitly_resumed", "task_id": task_id, - "root_task_id": root_task_id, "same_generation": True, "exact_continuation": True, - "grant_id": grant["grant_id"], "paused_duration_sec": grant["paused_duration_sec"], - "eligible_descendants": eligible if task_id == root_task_id else []}, - ) - return {"ok": True, "task_id": task_id, "root_task_id": root_task_id, "exact_continuation": True, - "grant_id": grant["grant_id"], "paused_duration_sec": round(grant["paused_duration_sec"], 1), - "eligible_descendants": eligible if task_id == root_task_id else []} - - -def revoke_exact_budget_resume(task: Dict[str, Any], reason: str) -> bool: - """Return a granted-but-undispatched task to its exact pause (queue lock held). - - Money can vanish between the grant and the dispatch (a sibling spent it); - the grant is single-use and must not be dispatched into a refused send. - """ - from ouroboros.budget_pause import STATE_PAUSED, budget_pause_row, set_budget_pause - - q = _queue_module() - handoff = task.get("_budget_pause_resume") if isinstance(task.get("_budget_pause_resume"), dict) else None - if not handoff or not isinstance(handoff.get("pause"), dict): - return False - task_id = str(task.get("id") or "") - result_root = pathlib.Path(task.get("budget_drive_root") or q.DRIVE_ROOT) - try: - row = budget_pause_row(result_root, task_id) - grant = dict(row.get("grant") or {}) - grant.update(revoked_at=utc_now_iso(), revoke_reason=str(reason or "")) - set_budget_pause(result_root, task_id, {**row, "state": STATE_PAUSED, "grant": grant}, - expected_pause_id=str(row.get("pause_id") or "")) - except Exception: - log.warning("Exact resume grant revocation not recorded for %s", task_id, exc_info=True) - return False - task["_budget_pause"] = dict(handoff["pause"]) - task.pop("_budget_pause_resume", None) - q.append_jsonl(q.DRIVE_ROOT / "logs" / "events.jsonl", - {"ts": utc_now_iso(), "type": "budget_resume_grant_revoked", "task_id": task_id, - "reason": str(reason or ""), "grant_id": grant.get("grant_id")}) - return True - - def _live_project_task_ids( drive_root: object, project_id: str, *, roots_only: bool = False, covering: Optional[set] = None, @@ -1353,3 +1145,13 @@ def reconcile_terminal_task_projections(drive_root, task_id: str) -> None: get_bridge().send_quiz_state(quiz_id, str(task_id), "expired_terminal") except Exception: log.debug("owner_quiz terminal reconcile failed for %s", task_id, exc_info=True) + + +# The exact-continuation grant lifecycle (#1196) lives in its own owner module; +# re-exported here because assignment, restore and the tests address these +# names on THIS surface (the same shape ``supervisor.queue`` uses for this file). +from supervisor.budget_resume import ( # noqa: E402, F401 -- intentional public re-exports + _root_budget_paused_locked, + grant_exact_budget_resume, + revoke_exact_budget_resume, +) diff --git a/supervisor/steering.py b/supervisor/steering.py index 17a1d5c96..4ff48d8f0 100644 --- a/supervisor/steering.py +++ b/supervisor/steering.py @@ -216,18 +216,21 @@ def _handle_steer_task(evt: Dict[str, Any], ctx: Any) -> None: direct_active = task is not None except Exception: direct_active = False + queue_resident = False if not direct_active: running = getattr(ctx, "RUNNING", None) meta = running.get(target) if isinstance(running, dict) else None task = meta.get("task") if isinstance(meta, dict) and isinstance(meta.get("task"), dict) else ( meta if isinstance(meta, dict) else None ) + queue_resident = isinstance(task, dict) if not isinstance(task, dict): pending = getattr(ctx, "PENDING", []) task = next( (row for row in list(pending or []) if isinstance(row, dict) and str(row.get("id") or "") == target), None, ) + queue_resident = isinstance(task, dict) target_label = "" if isinstance(task, dict): @@ -256,7 +259,11 @@ def _handle_steer_task(evt: Dict[str, Any], ctx: Any) -> None: # the two apart afterwards either. if not isinstance(task, dict): refusal = "target_unknown" - elif not (direct_active or not task.get("_is_direct_chat")): + elif task.get("_is_direct_chat") and not direct_active and not queue_resident: + # A direct turn that is neither a live in-process actor nor a queue row. + # A direct turn parked under its exact budget pause, or resumed on a + # pooled worker under the SAME id (#1196), is a queue row: the owner's + # follow-up reaches that actor's mailbox, never a second direct turn. refusal = "direct_chat_turn" elif str(task.get("delegation_role") or "") == "subagent": refusal = "subagent_target" diff --git a/supervisor/worker_assignment.py b/supervisor/worker_assignment.py index 16374eaa4..38e5f3424 100644 --- a/supervisor/worker_assignment.py +++ b/supervisor/worker_assignment.py @@ -13,6 +13,7 @@ import pathlib import time from typing import Any, Dict +from supervisor.events_budget import budget_fence_selected, budget_hold_fact from supervisor.queue import _queue_lock @@ -33,6 +34,20 @@ def _pool(): log = logging.getLogger(__name__) +def _direct_actor_still_registered(task_id: str) -> bool: + """Whether the in-process direct actor that paused ``task_id`` still holds its + registry entry. A parked direct turn releases the entry as its last act; a + grant dispatched before that would put a pooled worker beside a live actor + under the SAME id (#1196). Unreadable registry state fences (never admits).""" + try: + from supervisor.active_activity import get_direct_activity_registry + + return get_direct_activity_registry().get(str(task_id or "")) is not None + except Exception: + log.debug("Direct activity registry unreadable during assignment", exc_info=True) + return True + + def _evolution_assignment_error(task: Dict[str, Any]) -> str: """Return the exact authority error for an evolution task about to run.""" if str(task.get("type") or "") != "evolution": @@ -212,6 +227,8 @@ def assign_tasks() -> None: continue if isinstance(task.get("_budget_pause"), dict): continue + if budget_hold_fact(task) is not None: + continue # Held, not paused: no second pause row over the hold. if task.get("_owner_wait_resume"): continue # Restore the checkpoint; the loop still owns its budget stop. if isinstance(task.get("_budget_pause_resume"), dict): @@ -345,10 +362,34 @@ def assign_tasks() -> None: continue if isinstance(candidate.get("_budget_pause"), dict): continue + if budget_hold_fact(candidate) is not None: + # A durable budget hold (#1196): a sibling whose paused + # root's fence was lifted without an explicit selection, + # an unrestorable continuation, or a grant whose + # revocation could not be written. Never assignable + # until the selection is recorded on the row. + continue + if (candidate.get("_is_direct_chat") + and _direct_actor_still_registered(str(candidate.get("id") or ""))): + # The direct actor that parked this id has not released + # its registry entry yet: no second live actor for one id. + continue root_task_id = str(candidate.get("root_task_id") or "").strip() + from supervisor.events_budget import budget_resume_dispatch_allowed + + if not budget_resume_dispatch_allowed(queue, candidate): + from supervisor.budget_resume import revoke_exact_budget_resume + + revoke_exact_budget_resume(candidate, "root_resume_generation_stale") + queue.persist_queue_snapshot(reason="stale_child_resume_held") + continue if (root_task_id in queue.BUDGET_ROOT_FENCES and not candidate.get("_owner_wait_resume") - and not candidate.get("_budget_pause_resume")): + and not candidate.get("_budget_pause_resume") + # One member explicitly selected against THIS fence + # is admitted; the latch stays up for the rest (Q9). + and not budget_fence_selected( + candidate, queue.BUDGET_ROOT_FENCES.get(root_task_id))): continue if str(candidate.get("type") or "") == "evolution" and remaining < EVOLUTION_BUDGET_RESERVE: continue diff --git a/supervisor/worker_chat_lane.py b/supervisor/worker_chat_lane.py index 672bf33ab..55f7451ea 100644 --- a/supervisor/worker_chat_lane.py +++ b/supervisor/worker_chat_lane.py @@ -464,6 +464,10 @@ def _execute_chat_task(admitted: Dict[str, Any]) -> bool: events = agent.handle_task(task) finally: agent._event_queue = prev_queue + # An exact budget pause of THIS turn is parked here, in-process and + # synchronously, while the registry entry below still owns the id: + # the durable PENDING carrier exists before ownership is released. + events = _park_direct_budget_pause_inline(task, events) for e in events: _pool().get_event_q().put(turn_queue.stamp(e)) ok = True @@ -474,6 +478,59 @@ def _execute_chat_task(admitted: Dict[str, Any]) -> bool: return ok +def _park_direct_budget_pause_inline(task: Dict[str, Any], events: Any) -> list: + """Park a direct turn's exact budget pause BEFORE its registry entry is released (#1196). + + The turn's ``budget_pause`` event carries its own task record (a direct + turn was never in RUNNING). Handing that event to the supervisor loop and + unregistering the actor races: between the unregister and the park the + SAME id is nowhere — not live, not queued — and a restart in that window + fences a saved pause. The direct lane runs inside the supervisor process, + so it parks the record itself through the ONE park owner + (``events_budget.install_exact_budget_pause``) against the same queue + state, under the queue lock, and drops the event from the hand-off. A park + that fails leaves the event on the ordinary path (typed, logged), never a + silently lost pause. Every other event passes through unchanged. + """ + remaining: list = [] + for event in list(events or []): + checkpoint = ((event.get("resource_limit") or {}).get("checkpoint") + if isinstance(event, dict) and isinstance(event.get("resource_limit"), dict) else None) + if not (isinstance(event, dict) and str(event.get("type") or "") == "budget_pause" + and event.get("_is_direct_chat") and isinstance(checkpoint, dict)): + remaining.append(event) + continue + task_id = str(event.get("task_id") or task.get("id") or "") + try: + from types import SimpleNamespace + + from supervisor import queue as queue_mod + from supervisor.events_budget import install_exact_budget_pause + from supervisor.message_bus import get_bridge + + pool = _pool() + shim = SimpleNamespace( + RUNNING=pool.RUNNING, PENDING=pool.PENDING, WORKERS=pool.WORKERS, DRIVE_ROOT=pool.DRIVE_ROOT, + sort_pending=queue_mod.sort_pending, persist_queue_snapshot=queue_mod.persist_queue_snapshot, + bridge=get_bridge(), + ) + install_exact_budget_pause(shim, task_id, checkpoint, evt=event, source="direct_turn_inline_park") + from ouroboros.budget_pause import end_dispatch_fence + + end_dispatch_fence(task_id) # quiescent actor unwound; PENDING now owns the dispatch hold + append_jsonl( + pool.DRIVE_ROOT / "logs" / "supervisor.jsonl", + {"ts": utc_now_iso(), "type": "direct_turn_budget_pause_parked_inline", + "task_id": task_id, "chat_id": event.get("chat_id"), + "pause_id": str(checkpoint.get("pause_id") or "")}, + ) + except Exception: + log.error("Direct turn %s could not be parked inline; its pause event takes the ordinary path", + task_id, exc_info=True) + remaining.append(event) + return remaining + + def _report_direct_chat_error(admitted: Dict[str, Any], e: BaseException) -> None: """Record a direct-turn failure and conclude the turn in the chat.""" import traceback diff --git a/supervisor/worker_health.py b/supervisor/worker_health.py index 84eeb7738..bce809ce7 100644 --- a/supervisor/worker_health.py +++ b/supervisor/worker_health.py @@ -288,8 +288,23 @@ def _complete_exact_budget_pause_after_death(job: dict, root: pathlib.Path, task continuation (#1196). This is non-admission of the crash-retry path for exactly this window — a checkpointed task must never be replayed — not a general crash recovery: any other crash keeps its ordinary custody path. + + A worker that died holding an UNCONSUMED Resume grant belongs to the same + window (#1196, F4): the grant was minted but nothing consumed it — the loop + writes ``consumed_at`` before any new effect, and a refused continuation + load (a source that went unreadable, a grant a newer writer superseded) + kills the worker exactly here. The saved pause is not lost to a terminal + crash: the grant this dead process can no longer consume is revoked under + its own identity and the SAME task id is parked back on its exact pause for + the owner to Resume again. A CONSUMED grant is never reopened — that task + ran on, and its death takes terminal crash custody without ordinary retry. """ - from ouroboros.budget_pause import STATE_PAUSED, STATE_PAUSING, budget_pause_row + from ouroboros.budget_pause import ( + STATE_PAUSED, + STATE_PAUSING, + STATE_RESUME_GRANTED, + budget_pause_row, + ) from supervisor.events_budget import install_exact_budget_pause result_root = pathlib.Path(task.get("budget_drive_root") or root) @@ -297,19 +312,67 @@ def _complete_exact_budget_pause_after_death(job: dict, root: pathlib.Path, task row = budget_pause_row(result_root, task_id) except Exception: return False - if not (row and row.get("state") in {STATE_PAUSING, STATE_PAUSED} + state = str(row.get("state") or "") if row else "" + if not (row and state in {STATE_PAUSING, STATE_PAUSED, STATE_RESUME_GRANTED} and int(row.get("task_attempt") or 0) == int(attempt) and row.get("source_ref")): return False + source = "worker_death_during_pausing" with _queue_lock: if not _dead_job_is_current(job): return False + if state == STATE_RESUME_GRANTED: + grant = row.get("grant") if isinstance(row.get("grant"), dict) else {} + if grant.get("consumed_at"): + return False # the loop consumed it and ran on: ordinary custody + from supervisor.budget_resume import revoke_exact_budget_resume + from supervisor.events_budget import BUDGET_HOLD_KEY + from ouroboros.budget_pause import exact_pause_marker + + task.setdefault("_budget_pause_resume", { + "pause_id": row.get("pause_id"), "grant_id": grant.get("grant_id"), + "pause": exact_pause_marker(row, default_root=str(task.get("root_task_id") or task_id)), + }) + if not revoke_exact_budget_resume(task, "worker_death_before_consumption"): + if task.get("_budget_pause_consumed"): + return False # a concurrent consumption keeps crash custody, never re-arms + # Revocation failure is a nonterminal hold, never a crash retry + # or terminal. Preserve exact source and grant identity even if + # neither the result store nor the snapshot is writable. + held = dict(task) + held.setdefault("_budget_pause", exact_pause_marker(row, default_root=task_id)) + hold = held.get(BUDGET_HOLD_KEY) + if not isinstance(hold, dict): + from supervisor.events_budget import HOLD_REVOCATION_UNWRITTEN, hold_budget_row + + hold = hold_budget_row(held, reason=HOLD_REVOCATION_UNWRITTEN, + extra={"pause_id": row.get("pause_id"), "grant_id": grant.get("grant_id")}) + _pool().RUNNING.pop(task_id, None) + if not any(item.get("id") == task_id for item in _pool().PENDING): + _pool().PENDING.append(held) + from supervisor import queue + + queue.sort_pending() + hold["snapshot_persisted"] = False + try: + hold["snapshot_persisted"] = bool(queue.persist_queue_snapshot(reason="dead_worker_budget_hold")) + except Exception: + log.error("Budget hold snapshot failed for %s", task_id, exc_info=True) + log.warning("Dead worker %s retained in nonterminal hold; snapshot persisted=%s", + task_id, hold["snapshot_persisted"]) + return True + running = _pool().RUNNING.get(task_id) + if isinstance(running, dict) and isinstance(running.get("task"), dict): + running["task"].pop("_budget_pause_resume", None) + source = "worker_death_before_grant_consumed" try: install_exact_budget_pause(_pool(), task_id, {"pause_id": row.get("pause_id")}, - source="worker_death_during_pausing") + source=source) except Exception: log.error("Exact budget pause of %s could not be completed after worker death; " "leaving the row for the next reconciliation", task_id, exc_info=True) - return False + from supervisor.task_reaper import TerminalFileRecoveryPending + + raise TerminalFileRecoveryPending("exact budget pause parking remains pending") return True @@ -361,9 +424,9 @@ def _recover_crashed_task_without_terminal(job: dict, queue: Any) -> None: deep = task_type == "deep_self_review" if budget_pausing: result_text = ( - "Worker process died while entering a budget pause before its continuation " - "checkpoint existed (or the pause record is unreadable). Completed actions were " - "not retried; no exact continuation is available for this attempt." + "Worker process died with budget-continuation evidence. The checkpoint was " + "incomplete, unreadable, or already consumed. Completed actions were not retried; " + "no exact continuation is available for this attempt." ) reason_code = "worker_crash_budget_pausing" elif replay_unsafe: diff --git a/supervisor/worker_owner_wait.py b/supervisor/worker_owner_wait.py index da993d4c8..62c000406 100644 --- a/supervisor/worker_owner_wait.py +++ b/supervisor/worker_owner_wait.py @@ -13,6 +13,7 @@ import logging from typing import Any from ouroboros.owner_wait import set_owner_wait +from ouroboros.utils import append_jsonl, utc_now_iso from supervisor.queue import _queue_lock from supervisor.worker_pool_lifecycle import ( _serialized_worker_lifecycle, _spawn_worker_slot, kill_worker_tree, retire_worker, @@ -35,6 +36,51 @@ def has_owner_wait_checkpoint(meta: dict, task_attempt: int) -> bool: and bool(wait.get("source_ref"))) +def retire_consumed_budget_carrier(drive_root: Any, task_id: str, task: Any) -> str: + """Drop a SPENT exact-budget-resume handoff from one queue row; returns its id. + + The loop writes ``consumed_at`` on the durable grant — a compare-and-set on + the pause, the state and the grant it was handed — BEFORE any new effect, so + a queue row still carrying that handoff is only a stale carrier of work that + ran on. Left there, the next planned restart reads the row as a consumed + grant and drops it, taking the NEWER owner-wait continuation parked on the + same row with it (#1196, F3). Cumulative account fields are untouched: the + paused interval rides the owner-wait checkpoint and ``budget_resumed_at`` + stays on the row. A grant that is NOT consumed is left exactly where it is — + the restart path still has to revoke it — and an unreadable pause record + proves nothing, so it retires nothing. + """ + import pathlib + + from ouroboros.budget_pause import STATE_RESUMED, budget_pause_row + + handoff = task.get("_budget_pause_resume") if isinstance(task, dict) else None + if not isinstance(handoff, dict) or not str(handoff.get("grant_id") or ""): + return "" + root = pathlib.Path(task.get("budget_drive_root") or drive_root) + try: + row = budget_pause_row(root, task_id) + except Exception: + log.debug("Budget carrier retirement could not read the pause row for %s", task_id, exc_info=True) + return "" + grant = row.get("grant") if isinstance(row.get("grant"), dict) else {} + grant_id = str(grant.get("grant_id") or "") + if grant_id != str(handoff.get("grant_id") or ""): + return "" + if not (grant.get("consumed_at") or str(row.get("state") or "") == STATE_RESUMED): + return "" + task.pop("_budget_pause_resume", None) + try: + append_jsonl(pathlib.Path(drive_root) / "logs" / "events.jsonl", + {"ts": utc_now_iso(), "type": "budget_resume_carrier_retired", + "task_id": task_id, "grant_id": grant_id, + "pause_id": str(row.get("pause_id") or ""), + "consumed_at": grant.get("consumed_at"), "reason": "owner_wait_park"}) + except Exception: + log.debug("Failed to record budget carrier retirement for %s", task_id, exc_info=True) + return grant_id + + def _command(worker: Any, task_id: str, wait: dict, phase: str, **extra: Any) -> None: worker.in_q.put({"type": "owner_wait", "task_id": task_id, "task_attempt": wait["task_attempt"], "wait_id": wait["wait_id"], @@ -84,6 +130,10 @@ def handle_owner_wait(event: dict, ctx: Any) -> None: try: wait = set_owner_wait(ctx.DRIVE_ROOT, task_id, wait) meta["owner_wait"] = wait + # This park is the supervisor's proof that the task ran on past any + # exact budget grant it was dispatched with: a spent carrier is + # retired here, never carried into the snapshot beside this wait. + retire_consumed_budget_carrier(ctx.DRIVE_ROOT, task_id, meta.get("task")) worker.active_capacity = False if not queue.persist_queue_snapshot(reason="owner_wait_parked"): raise RuntimeError("owner wait queue snapshot was not persisted") diff --git a/tests/test_budget_pause_exact.py b/tests/test_budget_pause_exact.py index 2848a2fc4..8b65c74c5 100644 --- a/tests/test_budget_pause_exact.py +++ b/tests/test_budget_pause_exact.py @@ -121,14 +121,101 @@ def test_unanswered_tool_calls_are_execution_unknown_not_replayed(): # --------------------------------------------------------------------------- loop-side pause -def test_direct_actor_is_excluded_loudly_and_fence_stays_open(tmp_path): - from ouroboros import budget_pause +def _quiet_external(monkeypatch, budget_pause): + monkeypatch.setattr(budget_pause, "observe_external_runs", lambda _ctx, request_stop=True: { + "runs": [], "observed_at": time.time(), "custody_read": "ok", "coverage_basis": "test"}) - ctx, limit_ctx = _loop_ctx(tmp_path, direct=True) - assert budget_pause.request_pause(limit_ctx, rail=budget_pause.RAIL_GLOBAL_EXHAUSTED, - scope="global", reason_text="x") is None - assert limit_ctx.accumulated_usage["exact_pause_unavailable"] == "direct_actor" - assert not budget_pause.dispatch_fenced(ctx.task_id) + +def test_direct_actor_pauses_under_its_own_task_id_and_its_event_carries_its_record(tmp_path, monkeypatch): + """A direct owner-chat turn is ELIGIBLE (#1196): it never had an admission-written + RUNNING row, so the pause writes one first; its event carries the turn's own + record (minus inline image bytes) because RUNNING is not its carrier.""" + from ouroboros import budget_pause + from ouroboros.task_results import load_task_result + + ctx, limit_ctx = _loop_ctx(tmp_path, "direct-1", direct=True) + ctx.current_chat_id = 42 + _fast_hold(monkeypatch, budget_pause) + _quiet_external(monkeypatch, budget_pause) + assert budget_pause.pause_ineligibility(ctx) == "" + with pytest.raises(budget_pause.BudgetPauseRequested) as raised: + budget_pause.request_pause(limit_ctx, rail=budget_pause.RAIL_GLOBAL_EXHAUSTED, + scope="global", reason_text="x") + budget_pause.end_dispatch_fence("direct-1") + row = load_task_result(tmp_path, "direct-1", strict=True) + assert row["_is_direct_chat"] is True and row["chat_id"] == 42 + assert row["budget_pause"]["is_direct_chat"] is True and row["budget_pause"]["source_ref"] + task = {"id": "direct-1", "type": "task", "chat_id": 42, "text": "hello", "_is_direct_chat": True, + "image_base64": "AAAA", "origin_message_ref": {"chat_id": 42}, "metadata": {"k": "v"}} + event = budget_pause.pause_event(task, raised.value.pause) + assert event["_is_direct_chat"] is True and event["resource_limit"]["exact_continuation"] is True + carried = event["task"] + assert carried["id"] == "direct-1" and carried["_is_direct_chat"] is True + assert "image_base64" not in carried and carried["origin_message_ref"] == {"chat_id": 42} + assert carried["metadata"] == {"k": "v"} and carried["_attempt"] == 1 and carried["depth"] == 0 + # A pooled task's event carries no record: RUNNING is its carrier. + assert "task" not in budget_pause.pause_event({"id": "direct-1", "type": "task"}, raised.value.pause) + # A context with no continuation owner at all is still excluded, loudly. + ctx.owner_wait_callback = None + assert budget_pause.pause_ineligibility(ctx) == "no_continuation_owner" + + +def test_direct_turn_pause_event_parks_the_carried_record_in_pending(tmp_path, monkeypatch): + """The supervisor parks a direct turn from its OWN record: same task id, lane fact + kept, queue-order facts minted, snapshot persisted, census phase paused.""" + from ouroboros import budget_pause + from supervisor.events import _handle_budget_pause + from supervisor.queue_transitions import budget_pause_fact + + queue, _state, workers = _install_queue(tmp_path, monkeypatch) + ctx, limit_ctx = _loop_ctx(tmp_path, "direct-2", direct=True) + ctx.current_chat_id = 7 + _fast_hold(monkeypatch, budget_pause) + _quiet_external(monkeypatch, budget_pause) + with pytest.raises(budget_pause.BudgetPauseRequested) as raised: + budget_pause.request_pause(limit_ctx, rail=budget_pause.RAIL_GLOBAL_EXHAUSTED, + scope="global", reason_text="x") + budget_pause.end_dispatch_fence("direct-2") + task = {"id": "direct-2", "type": "task", "chat_id": 7, "text": "hello", "_is_direct_chat": True, + "metadata": {"origin_message_ref": {"chat_id": 7}}} + persisted, pushed = [], [] + sctx = _supervisor_ctx(tmp_path, workers, queue, persisted, pushed) + assert workers.RUNNING == {} # a direct turn is never in RUNNING + _handle_budget_pause(budget_pause.pause_event(task, raised.value.pause), sctx) + parked = workers.PENDING[0] + assert parked["id"] == "direct-2" and parked["_is_direct_chat"] is True + assert parked["_budget_pause"]["exact_continuation"] is True + assert parked["_queue_seq"] and parked["queued_at"] and "priority" in parked + assert budget_pause_fact(parked)["exact_continuation"] is True + assert budget_pause.budget_pause_row(tmp_path, "direct-2")["state"] == budget_pause.STATE_PAUSED + assert persisted == ["budget_pause_exact_continuation"] + assert pushed[0]["type"] == "budget_scope_paused" and pushed[0]["task_id"] == "direct-2" + # The snapshot keeps the lane fact, so a restart restores the same direct row. + queue.persist_queue_snapshot(reason="test") + snap = json.loads(queue.QUEUE_SNAPSHOT_PATH.read_text()) + assert snap["pending"][0]["task"]["_is_direct_chat"] is True + # The census reads it as a paused DIRECT activity under the same id. + from ouroboros.gateway import state as gw_state + + rows = gw_state._chat_activities_snapshot_safe(tmp_path, direct_turns=[]) + assert [(r["activity_id"], r["kind"], r["phase"]) for r in rows if r["activity_id"] == "direct-2"] == [ + ("direct-2", "direct_chat", "budget_paused")] + # An event without the record cannot park a turn that is not running. + bare = budget_pause.pause_event({"id": "direct-2", "type": "task"}, raised.value.pause) + workers.PENDING[:] = [] + with pytest.raises(RuntimeError): + _handle_budget_pause(bare, sctx) + + +def test_task_event_addressing_stamps_the_direct_lane_fact_from_the_running_row(tmp_path): + """A resumed direct turn runs on a pooled worker: its frames keep the lane fact.""" + from supervisor.log_addressing import address_task_event + + running = {"d-1": {"task": {"id": "d-1", "chat_id": 5, "_is_direct_chat": True}}} + payload = address_task_event(running, tmp_path, {"task_id": "d-1", "type": "tool_call_started"}) + assert payload["_is_direct_chat"] is True and payload["chat_id"] == 5 + managed = address_task_event({"m-1": {"task": {"id": "m-1", "chat_id": 5}}}, tmp_path, {"task_id": "m-1"}) + assert "_is_direct_chat" not in managed def test_pause_writes_source_and_row_before_raising_and_closes_fence(tmp_path, monkeypatch): @@ -279,7 +366,12 @@ def test_pending_review_attempt_blocks_release_then_pauses_once_it_settles(tmp_p def test_timed_out_tool_future_blocks_release_until_its_settlement_callback_finishes(tmp_path): """``future.done()`` is not callback-complete: a call abandoned at its own timeout keeps the task unquiescent until the late settlement callback that - owns its effects has finished.""" + owns its effects has finished. + + The registering OWNER pins the row until it releases, so the instant between + the caller's own timeout and its ``hold_tool_settlement`` claim can never + read as settled-and-unheld: the claim is taken while the pin still holds. + """ import threading from concurrent.futures import ThreadPoolExecutor @@ -290,15 +382,18 @@ def test_timed_out_tool_future_blocks_release_until_its_settlement_callback_fini executor = ThreadPoolExecutor(max_workers=1) try: future = executor.submit(gate.wait, 5.0) - budget_pause.register_tool_future(ctx, "call_slow", "run_command", future) + owner_release = budget_pause.register_tool_future(ctx, "call_slow", "run_command", future) drain = budget_pause.drain_local_tool_futures(ctx, timeout_sec=0.05) assert drain["drained"] is False assert drain["unsettled"] == [{"operation_id": "call_slow", "tool": "run_command", "state": "running"}] - # The timeout path claims the row BEFORE the worker settles. + # The timeout path claims the row BEFORE the worker settles, while the + # registering owner still pins it; only then does the owner hand over. release = budget_pause.hold_tool_settlement(ctx, "call_slow") + owner_release() gate.set() assert future.result(timeout=5.0) is True + # ``done()`` is true now, but the late callback still owns the effects. assert budget_pause.drain_local_tool_futures(ctx, timeout_sec=0.05)["drained"] is False release() settled = budget_pause.drain_local_tool_futures(ctx, timeout_sec=1.0) @@ -310,6 +405,44 @@ def test_timed_out_tool_future_blocks_release_until_its_settlement_callback_fini budget_pause.forget_tool_scope(ctx) +def test_neither_half_of_the_settlement_protocol_alone_is_quiescence(tmp_path): + """Negative: an owner release over a still-running future is not quiescence, and a + registration interleaved with a finished-but-unreleased row never prunes it into + false quiescence.""" + from concurrent.futures import Future + + from ouroboros import budget_pause + + ctx, _limit = _loop_ctx(tmp_path, "tool-3") + try: + running = Future() + owner_running = budget_pause.register_tool_future(ctx, "call_running", "run_command", running) + # The owner is done deciding, but the future has NOT finished: not quiescent. + owner_running() + assert budget_pause.drain_local_tool_futures(ctx, timeout_sec=0.02)["unsettled"] == [ + {"operation_id": "call_running", "tool": "run_command", "state": "running"}] + running.set_result("late") + assert budget_pause.drain_local_tool_futures(ctx, timeout_sec=1.0)["drained"] is True + # A FINISHED future whose owner has not released is unsettled, and the next + # registration must prune only the released row, never the pinned one. + pinned = Future() + pinned.set_result("x") + budget_pause.register_tool_future(ctx, "call_pinned", "read_file", pinned) # pin kept + third = Future() + third.set_result("y") + owner_third = budget_pause.register_tool_future(ctx, "call_third", "read_file", third) + drain = budget_pause.drain_local_tool_futures(ctx, timeout_sec=0.02) + assert drain["drained"] is False + assert sorted(row["operation_id"] for row in drain["unsettled"]) == ["call_pinned", "call_third"] + owner_third() + drain = budget_pause.drain_local_tool_futures(ctx, timeout_sec=0.5) + assert drain["drained"] is False + assert [row["operation_id"] for row in drain["unsettled"]] == ["call_pinned"] + assert [row["operation_id"] for row in drain["settled"]] == ["call_third"] + finally: + budget_pause.forget_tool_scope(ctx) + + def test_tool_future_quiescence_is_scoped_to_one_attempt_and_prunes_itself(tmp_path): from concurrent.futures import Future @@ -320,17 +453,21 @@ def test_tool_future_quiescence_is_scoped_to_one_attempt_and_prunes_itself(tmp_p done = Future() done.set_result("x") try: - budget_pause.register_tool_future(ctx, "call_a", "read_file", done) - assert budget_pause.drain_local_tool_futures(ctx, timeout_sec=0.1)["drained"] is True + release_a = budget_pause.register_tool_future(ctx, "call_a", "read_file", done) + # Finished, but the registering owner still pins it: not yet settled. + assert budget_pause.drain_local_tool_futures(ctx, timeout_sec=0.05)["drained"] is False + release_a() + assert budget_pause.drain_local_tool_futures(ctx, timeout_sec=0.5)["drained"] is True # A later attempt never inherits a previous attempt's observations. assert budget_pause.drain_local_tool_futures(later, timeout_sec=0.0) == { "drained": True, "registry": "ok", "settled": [], "unsettled": []} - # A settled, unheld row is pruned by the next registration. + # A settled, unheld AND released row is pruned by the next registration. second = Future() second.set_result("y") - budget_pause.register_tool_future(ctx, "call_b", "read_file", second) + release_b = budget_pause.register_tool_future(ctx, "call_b", "read_file", second) + release_b() assert [row["operation_id"] for row in - budget_pause.drain_local_tool_futures(ctx, timeout_sec=0.1)["settled"]] == ["call_b"] + budget_pause.drain_local_tool_futures(ctx, timeout_sec=0.5)["settled"]] == ["call_b"] finally: budget_pause.forget_tool_scope(ctx) budget_pause.forget_tool_scope(later) @@ -395,8 +532,6 @@ def test_light_extraction_is_not_dispatched_under_the_fence(monkeypatch): def test_local_review_drain_lists_unsettled_attempts_without_settling_them(): - import threading - from ouroboros import budget_pause, review_custody as rc settled = rc.ActiveReviewAttempt(key="triad|t9|r", operation_id="op-settled", wave_key="task_acceptance|t9|r") @@ -460,6 +595,55 @@ def test_exact_pause_event_parks_same_task_id_and_confirms_row(tmp_path, monkeyp assert budget_pause_fact(workers.PENDING[0])["exact_continuation"] is True +def test_late_park_confirmation_never_regresses_a_live_grant(tmp_path, monkeypatch): + """F1: the park confirmation is a compare-and-set on the pause AND its state. + + The park's row reading is taken before the queue transition; a writer that + moved the row on in between (an owner Resume granting it) must not be + overwritten by a late ``paused`` carrying the stale grant-less row, and the + owner-facing projection must not be rewritten to say paused either. The + supervisor publishes the anomaly as itself instead of a false pause. + """ + from ouroboros import budget_pause + from ouroboros.task_results import load_task_result + from supervisor.events import _handle_budget_pause + + queue, _state, workers = _install_queue(tmp_path, monkeypatch) + ctx, _limit, pause = _pause(tmp_path, monkeypatch, task_id="late-1") + budget_pause.end_dispatch_fence("late-1") + task = {"id": "late-1", "type": "task", "chat_id": 0, "root_task_id": "late-1", "_attempt": 1} + workers.RUNNING["late-1"] = {"task": task, "worker_id": 0, "attempt": 1} + workers.WORKERS[0] = SimpleNamespace(busy_task_id="late-1") + row = budget_pause.budget_pause_row(tmp_path, "late-1") + grant = {"grant_id": "g-live", "single_use": True, "generation": 1} + persisted, pushed = [], [] + sctx = _supervisor_ctx(tmp_path, workers, queue, persisted, pushed) + + def _snapshot_then_grant(reason=""): + # A writer this queue lock does not cover moves the row on mid-park. + budget_pause.set_budget_pause( + tmp_path, "late-1", {**row, "state": budget_pause.STATE_RESUME_GRANTED, + "grant": grant, "resume_generation": 1}, + expected_pause_id=str(row["pause_id"])) + persisted.append(reason) + return True + + sctx.persist_queue_snapshot = _snapshot_then_grant + _handle_budget_pause({**budget_pause.pause_event(task, pause), "worker_id": 0}, sctx) + + # The park itself still happened: the SAME task id is parked, never dropped. + assert workers.RUNNING == {} and workers.PENDING[0]["id"] == "late-1" + after = budget_pause.budget_pause_row(tmp_path, "late-1") + assert after["state"] == budget_pause.STATE_RESUME_GRANTED + assert after["grant"]["grant_id"] == "g-live" # the live grant is intact + assert "paused_confirmed_at" not in after + assert pushed[-1]["type"] == "budget_pause_park_superseded" + assert pushed[-1]["owner_visible"] is False and "toast_once" not in pushed[-1] + assert pushed[-1]["park_state"] == budget_pause.STATE_RESUME_GRANTED + # The owner-facing status projection belongs to the newer writer, not to us. + assert load_task_result(tmp_path, "late-1", strict=True).get("reason_code") != "budget_paused" + + def test_exact_pause_event_without_durable_row_is_refused(tmp_path, monkeypatch): from supervisor.events import _handle_budget_pause @@ -496,11 +680,81 @@ def test_worker_death_during_pausing_completes_the_park_not_a_retry(tmp_path, mo assert worker_health._complete_exact_budget_pause_after_death(job, tmp_path, task, ctx.task_id, 2) is False +def test_worker_death_holding_an_unconsumed_grant_reparks_instead_of_terminalizing(tmp_path, monkeypatch): + """F4: a grant nothing consumed is revoked and the SAME task id returns to its + exact pause. The loop writes ``consumed_at`` before any new effect, so an + unconsumed grant proves the continuation never started — a refused + continuation load kills the worker exactly here, and the saved pause must + survive it. A CONSUMED grant is ordinary crash custody and never reopened.""" + from ouroboros import budget_pause + from supervisor import worker_health + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 5.0) + task, _row = _parked(tmp_path, monkeypatch, task_id="death-1") + assert queue.resume_budget_paused_task("death-1")["ok"] is True + grant_id = task["_budget_pause_resume"]["grant_id"] + # Dispatched, then the worker dies before the loop could consume the grant. + workers.PENDING.remove(task) + meta = {"task": task, "worker_id": 0, "attempt": 1} + workers.RUNNING["death-1"] = meta + workers.WORKERS[0] = SimpleNamespace(busy_task_id="death-1") + monkeypatch.setattr(worker_health, "_dead_job_is_current", lambda job: True) + monkeypatch.setattr(queue, "persist_queue_snapshot", lambda reason="": True) + job = {"worker": workers.WORKERS[0], "task_id": "death-1", "task": dict(task), "meta": meta, + "worker_id": 0, "exitcode": 1, "drive_root": str(tmp_path)} + assert worker_health._complete_exact_budget_pause_after_death(job, tmp_path, task, "death-1", 1) is True + assert workers.RUNNING == {} + parked = workers.PENDING[0] + assert parked["id"] == "death-1" and parked["_budget_pause"]["exact_continuation"] is True + assert "_budget_pause_resume" not in parked # the spent handoff left with the park + row = budget_pause.budget_pause_row(tmp_path, "death-1") + assert row["state"] == budget_pause.STATE_PAUSED + assert row["grant"]["grant_id"] == grant_id + assert row["grant"]["revoke_reason"] == "worker_death_before_consumption" + assert row["pause_source"] == "worker_death_before_grant_consumed" + # The owner may Resume the same id again; the dead grant is dead for good. + ctx, _limit = _loop_ctx(tmp_path, "death-1") + with pytest.raises(ValueError): + budget_pause.load_budget_pause(ctx, {"pause_id": row["pause_id"], "grant_id": grant_id}) + assert queue.resume_budget_paused_task("death-1")["ok"] is True + + +def test_worker_death_after_a_consumed_grant_is_not_reopened(tmp_path, monkeypatch): + """The other half of F4: a consumed grant means the task RAN. Its death keeps + the ordinary custody path — the pause is not re-armed over running work.""" + from ouroboros import budget_pause + from supervisor import worker_health + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 5.0) + task, _row = _parked(tmp_path, monkeypatch, task_id="death-2") + assert queue.resume_budget_paused_task("death-2")["ok"] is True + granted = budget_pause.budget_pause_row(tmp_path, "death-2") + consumed = {**dict(granted["grant"]), "consumed_at": time.time()} + budget_pause.set_budget_pause( + tmp_path, "death-2", {**granted, "state": budget_pause.STATE_RESUMED, "grant": consumed}, + expected_pause_id=str(granted["pause_id"]), + expected_state=budget_pause.STATE_RESUME_GRANTED, + expected_grant_id=str(granted["grant"]["grant_id"])) + workers.PENDING.remove(task) + meta = {"task": task, "worker_id": 0, "attempt": 1} + workers.RUNNING["death-2"] = meta + workers.WORKERS[0] = SimpleNamespace(busy_task_id="death-2") + monkeypatch.setattr(worker_health, "_dead_job_is_current", lambda job: True) + job = {"worker": workers.WORKERS[0], "task_id": "death-2", "task": dict(task), "meta": meta, + "worker_id": 0, "exitcode": 1, "drive_root": str(tmp_path)} + assert worker_health._complete_exact_budget_pause_after_death(job, tmp_path, task, "death-2", 1) is False + assert "death-2" in workers.RUNNING and workers.PENDING == [] + after = budget_pause.budget_pause_row(tmp_path, "death-2") + assert after["state"] == budget_pause.STATE_RESUMED and not after["grant"].get("revoked_at") + + # --------------------------------------------------------------------------- supervisor: resume def _parked(tmp_path, monkeypatch, *, task_id="pause-task", scope="global", root_task_id=None, extra=None): from ouroboros import budget_pause - from supervisor import queue, workers + from supervisor import workers # ONE lineage: the durable row, its queue marker and the pending task row all # name the same root, or a descendant's resume cannot see its paused root. @@ -534,7 +788,6 @@ def test_resume_refuses_while_money_is_still_exhausted(tmp_path, monkeypatch): def test_resume_refuses_cancel_intent_and_paused_root(tmp_path, monkeypatch): queue, state, workers = _install_queue(tmp_path, monkeypatch) monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 5.0) - import supervisor.queue_transitions as qt task, _row = _parked(tmp_path, monkeypatch) monkeypatch.setattr("ouroboros.cancel_intents.has_active_intent", lambda *_a, **_k: True) @@ -592,6 +845,7 @@ def test_grant_is_revoked_when_money_vanishes_before_dispatch(tmp_path, monkeypa monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 5.0) task, _row = _parked(tmp_path, monkeypatch, task_id="revoke-1") assert queue.resume_budget_paused_task("revoke-1")["ok"] is True + assert task["_budget_pause_resume"]["grant_generation"] == 1 monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 0.0) sent = [] workers.WORKERS[0] = SimpleNamespace(wid=0, busy_task_id=None, reaping=False, @@ -603,6 +857,10 @@ def test_grant_is_revoked_when_money_vanishes_before_dispatch(tmp_path, monkeypa row = budget_pause.budget_pause_row(tmp_path, "revoke-1") assert row["state"] == budget_pause.STATE_PAUSED and row["grant"]["revoke_reason"] == "budget_exhausted_before_dispatch" assert revoke_exact_budget_resume(task, "again") is False # nothing granted now + # A stale copy of the revoked handoff can never revive the continuation. + ctx, _limit = _loop_ctx(tmp_path, "revoke-1") + with pytest.raises(ValueError): + budget_pause.load_budget_pause(ctx, {"pause_id": row["pause_id"], "grant_id": row["grant"]["grant_id"]}) def test_granted_task_dispatches_with_original_started_at_and_paused_carrier(tmp_path, monkeypatch): @@ -671,15 +929,56 @@ def test_resume_consumes_grant_restores_cognition_and_never_reexecutes(tmp_path, # The unanswered call is closed as UNKNOWN, not re-run, not declared un-run. unknown = [m for m in messages if m.get("role") == "tool" and m.get("tool_call_id") == "call_b"] assert len(unknown) == 1 and "UNKNOWN" in unknown[0]["content"] and "NOT re-executed" in unknown[0]["content"] - assert "budget pause" in messages[-1]["content"] and "run-1" in messages[-1]["content"] + notice = messages[-1]["content"] + assert "budget pause" in notice + # Custody is re-observed FRESH at the grant (owner Q8) and that reading rides the + # row: the disclosure names it, never the pause-time summary. This drive holds no + # delegated run at Resume time, so the pause row's stale "run-1" must NOT resurface. + assert "Delegated runs this task holds (re-observed at this Resume):\n- none" in notice + assert "run-1" not in notice + assert "never start a second writer" in notice + fresh = budget_pause.budget_pause_row(tmp_path, "loop-1")["external_runs"] + assert fresh["custody_read"] == "ok" and fresh["runs"] == [] consumed = budget_pause.budget_pause_row(tmp_path, "loop-1") assert consumed["state"] == budget_pause.STATE_RESUMED and consumed["grant"]["consumed_at"] assert not budget_pause.dispatch_fenced("loop-1") + # A hard rail keeps both wrap-up reservations; only a graceful rail relaxes (Q10). + assert ctx._budget_resume_last_fit_relaxed is False + assert usage["budget_pause_resume"]["last_fit_relaxed"] is False + assert usage["budget_pause_resume"]["grant_generation"] == 1 # The grant is single-use: a second load refuses. with pytest.raises(ValueError): budget_pause.load_budget_pause(ctx, handoff) +def test_resume_labels_a_pause_time_run_list_as_not_re_observed(tmp_path, monkeypatch): + """The other branch of the same disclosure: when the row carries no fresh reading, + the checkpoint's pause-time copy is disclosed and NAMED as un-re-observed history — + a stale list must never read as a current one.""" + from ouroboros import budget_pause, owner_wait + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 5.0) + task, row = _parked(tmp_path, monkeypatch, task_id="loop-2") + assert queue.resume_budget_paused_task("loop-2")["ok"] is True + ctx, _limit = _loop_ctx(tmp_path, "loop-2") + ctx.budget_pause_resume = task["_budget_pause_resume"] + state_blob = budget_pause.load_budget_pause(ctx) + # Drop the grant's fresh observation, keeping the checkpoint's pause-time copy. + state_blob["_pause_row"] = {k: v for k, v in state_blob["_pause_row"].items() + if k != "external_runs"} + state_blob["external_runs"] = {"runs": [{"run_id": "run-1", "state": "stop_requested", + "stop_outcome": "requested"}]} + monkeypatch.setattr(owner_wait, "rebind_restored_route", lambda *_a, **_k: (None, "max")) + messages = [] + budget_pause.resume_paused_loop(SimpleNamespace(_ctx=ctx), state_blob, messages, {}, {}, set(), + budget_remaining_usd=5.0) + notice = messages[-1]["content"] + assert "(as recorded at the pause, NOT re-observed)" in notice + assert "run-1: stop_requested" in notice + assert "never start a second writer" in notice + + def test_load_refuses_foreign_or_missing_grant(tmp_path, monkeypatch): from ouroboros import budget_pause @@ -697,16 +996,73 @@ def test_graceful_rail_refreshes_planning_threshold_within_authorized_money(tmp_ ctx._cost_ceiling = task_pacing.CostCeiling(state="active", ceiling_usd=7.0, root_cap_usd=10.0, planning_margin_usd=3.0, basis="root_cap_minus_margin") ctx._accumulated_usage = {"cost": 8.0} - monkeypatch.setattr(loop_budget, "_loop_tree_accounting", lambda **_k: {"accounted_usd": 8.0}) + # Every number is read from the AUTHORITATIVE ledger at Resume time: a fresh + # wallet observation and a fresh, undegraded root-accounting read. + monkeypatch.setattr(loop_budget, "_wrapup_global_remaining", lambda: 100.0) + monkeypatch.setattr(loop_budget, "_loop_tree_accounting", + lambda **_k: {"accounted_usd": 8.0, "age_sec": 0.0}) monkeypatch.setattr(task_pacing, "resolve_budget_profile", lambda _c: {"cost_hard_stop_pct": 50}) disclosure = budget_pause._refresh_planning_threshold(ctx, budget_remaining_usd=100.0) assert disclosure["refreshed"] is True + # The wallet is the ledger projection, never the dispatch-time number. + assert disclosure["wallet_basis"] == "ledger_projection" + assert disclosure["global_remaining_usd"] == 100.0 + # This tree read carries no cap of its own, so the start-of-task cap stands and + # its provenance is disclosed rather than assumed. + assert disclosure["root_cap_usd"] == 10.0 and disclosure["root_cap_basis"] == "start_of_task" # min(cap - spent = 2, 50% of global remaining = 50) added on top of spend: no immediate re-pause. assert ctx._cost_ceiling.ceiling_usd == pytest.approx(10.0) assert ctx._cost_ceiling.root_cap_usd == 10.0 and ctx._cost_ceiling.basis.startswith("owner_resume_refresh") + # The hard tree cap is untouched by the refresh: spend AT the cap leaves no + # authorized room, and the owner's explicit act cannot invent any. ctx._accumulated_usage = {"cost": 10.0} - monkeypatch.setattr(loop_budget, "_loop_tree_accounting", lambda **_k: {"accounted_usd": 10.0}) - assert budget_pause._refresh_planning_threshold(ctx, budget_remaining_usd=100.0)["refreshed"] is False + monkeypatch.setattr(loop_budget, "_loop_tree_accounting", + lambda **_k: {"accounted_usd": 10.0, "age_sec": 0.0}) + spent = budget_pause._refresh_planning_threshold(ctx, budget_remaining_usd=100.0) + assert spent["refreshed"] is False and spent["reason"] == "no_authorized_room" + + +def test_threshold_refresh_refuses_every_unknown_or_stale_money_fact(tmp_path, monkeypatch): + """Negative (owner Q10): unknown money is NOT room. A wallet the ledger cannot + answer, a degraded or stale tree read and unknown tree spend each REFUSE the + refresh, and the dispatch-time number is only ever DISCLOSED, never spent.""" + from ouroboros import budget_pause, loop_budget, task_pacing + + ctx, _limit = _loop_ctx(tmp_path) + + def _ceiling(): + return task_pacing.CostCeiling(state="active", ceiling_usd=7.0, root_cap_usd=10.0, + planning_margin_usd=3.0, basis="root_cap_minus_margin") + + monkeypatch.setattr(task_pacing, "resolve_budget_profile", lambda _c: {"cost_hard_stop_pct": 50}) + # No ceiling at all: there is no threshold to move. + ctx._cost_ceiling = None + assert budget_pause._refresh_planning_threshold(ctx, budget_remaining_usd=100.0) == { + "refreshed": False, "reason": "no_ceiling"} + # The ledger cannot answer the wallet: the dispatch-time value is DISCLOSED only. + ctx._cost_ceiling = _ceiling() + ctx._accumulated_usage = {"cost": 8.0} + monkeypatch.setattr(loop_budget, "_wrapup_global_remaining", lambda: None) + monkeypatch.setattr(loop_budget, "_loop_tree_accounting", + lambda **_k: {"accounted_usd": 8.0, "age_sec": 0.0}) + assert budget_pause._refresh_planning_threshold(ctx, budget_remaining_usd=100.0) == { + "refreshed": False, "reason": "wallet_unavailable", "wallet_basis": "ledger_unavailable", + "dispatch_time_remaining_usd": 100.0} + assert ctx._cost_ceiling.ceiling_usd == 7.0 # the paused threshold is untouched + # With a fresh wallet, every unusable TREE read still refuses. + monkeypatch.setattr(loop_budget, "_wrapup_global_remaining", lambda: 100.0) + for tree, reason in ( + (None, "tree_spend_unavailable"), + ({"accounted_usd": 8.0, "age_sec": 0.0, "integrity_degraded": True}, "tree_accounting_degraded"), + ({"accounted_usd": 8.0, "age_sec": budget_pause._FRESH_TREE_MAX_AGE_SEC + 1.0}, + "tree_accounting_stale"), + ({"accounted_usd": None, "age_sec": 0.0}, "tree_spend_unknown"), + ): + ctx._cost_ceiling = _ceiling() + monkeypatch.setattr(loop_budget, "_loop_tree_accounting", lambda _t=tree, **_k: _t) + refused = budget_pause._refresh_planning_threshold(ctx, budget_remaining_usd=100.0) + assert refused["refreshed"] is False and refused["reason"] == reason + assert ctx._cost_ceiling.ceiling_usd == 7.0 # --------------------------------------------------------------------------- gateway / UI facts @@ -728,7 +1084,7 @@ def test_resume_child_tool_only_targets_own_children(monkeypatch): ctx = SimpleNamespace(task_id="parent-1", task_metadata={}) monkeypatch.setattr(join_ledger, "_status_drive_root", lambda _c: pathlib.Path("/tmp")) - monkeypatch.setattr(join_ledger, "_is_own_child", lambda _c, _r, tid: tid == "child-x") + monkeypatch.setattr(join_ledger, "_is_own_child", lambda _c, _r, tid, **_kw: tid == "child-x") monkeypatch.setattr(join_ledger, "_publish_tool_result", lambda _c, result: result.text) monkeypatch.setattr(join_ledger, "_record_child_decision_beacon", lambda *_a, **_k: None) emitted = [] @@ -738,3 +1094,51 @@ def test_resume_child_tool_only_targets_own_children(monkeypatch): text = join_ledger._resume_child_task(ctx, "child-x", "still needed") assert "Resume requested" in text and "REQUEST" in text assert emitted[0]["type"] == "budget_resume_child" and emitted[0]["requested_by"] == "parent-1" + + +def test_resume_child_task_is_policy_covered_and_exposed_beside_its_own_family(): + """F7: the Q9 selection verb was registered but named nowhere else — it fell + through to the default LLM safety check, was invisible in the round-one + envelope and to delegated children (who must select their OWN paused + children), and was not withheld from a consciousness wake at Observe, which + may not start work. It is declared beside ``cancel_task``, the verb whose + authority it mirrors; the supervisor still re-checks lineage and the root's + live grant, so no owner authority is widened.""" + from ouroboros import safety + from ouroboros import tool_capabilities as caps + from ouroboros.consciousness_authority import disabled_tools_for + from ouroboros.tools import join_ledger + + assert any(entry.name == "resume_child_task" for entry in join_ledger.get_tools()) + assert safety.TOOL_POLICY["resume_child_task"] == safety.POLICY_SKIP + for names in (caps.CORE_TOOL_NAMES, caps.LOCAL_READONLY_SUBAGENT_TOOL_NAMES, + caps.ACTING_SUBAGENT_TOOL_NAMES): + assert "cancel_task" in names # the family it belongs to + assert "resume_child_task" in names + # It STARTS work, so an Observe-level wake does without it (В10'). + assert "resume_child_task" in caps.OBSERVE_WORLD_MUTATION_TOOLS + assert "resume_child_task" in disabled_tools_for("observe") + assert "resume_child_task" not in disabled_tools_for("full") + + +def _repo_file(*parts): + return pathlib.Path(__file__).resolve().parents[1].joinpath(*parts).read_text() + + +def test_activity_rows_show_a_held_budget_row_as_paused_not_queued(): + """Scope note (static pin; the browser check is the parent's): a row whose root + fence was lifted carries an unselected HOLD and nothing will dispatch it, so + listing it as plain "queued" promises work that cannot start.""" + source = _repo_file("web", "modules", "activity.js") + assert "_budget_pause_hold" in source and "heldRow" in source + assert "|| heldRow(t)" in source # consulted by the pending-row pause predicate + + +def test_runbook_does_not_promise_a_managed_outage_window_the_runtime_has_no_rail_for(): + """F8: with no deadline and an unlimited absolute ceiling, a managed task's + transport-outage episode has NO window of its own — the 6h operation-window + fallback belongs to other operations. The runbook names the optional rails an + operator can set instead of promising a timeout that does not exist.""" + source = _repo_file("devtools", "benchmarks", "continual_learning", "RUNBOOK.md") + assert "6h operation window from episode entry" not in source + assert "OUROBOROS_TASK_ABS_CEILING_SEC" in source and "idle reaper" in source diff --git a/tests/test_budget_pause_holds.py b/tests/test_budget_pause_holds.py new file mode 100644 index 000000000..c2fe80277 --- /dev/null +++ b/tests/test_budget_pause_holds.py @@ -0,0 +1,562 @@ +"""Durable budget-pause HOLDS, generation-bound grants, restart parking and the Q10 +last-fit relaxation (#1196) — the supervisor-side half beside ``test_budget_pause_exact``. + +Static authoring note: these tests were WRITTEN against the candidate but NOT +RUN by their author (no runtime imports were permitted in that lane); the +parent's isolated harness is the first execution. + +Every refusal to restore, grant or dispatch RETAINS the saved pause: a corrupt +source, an unwritten revocation or an acceptance fence over the root becomes a +typed hold beside the ``_budget_pause`` marker; a fence-lifted zero-dispatch +sibling is held until an explicit selection under its root's LIVE grant; a +RUNNING row whose pause completed before a shutdown is parked, never fenced. +""" + +from __future__ import annotations + +import pathlib +import time +from types import SimpleNamespace + +import pytest + +from tests.test_budget_pause_exact import ( # noqa: F401 -- shared fixtures of the exact-pause suite + _install_queue, + _loop_ctx, + _parked, + _pause, + _supervisor_ctx, +) + + +def test_explicit_resume_relaxes_the_last_fit_two_reservation_rail(monkeypatch): + """Owner Q10: after a graceful Resume the one affordable call is admitted (disclosed), + never re-paused on the number that paused it; a hard rail keeps both reservations.""" + from ouroboros import loop_budget, task_pacing + + calls = [] + monkeypatch.setattr(task_pacing, "wrapup_reservation_fits", + lambda **kw: calls.append(kw.get("reservation_count")) or False) + relaxed = SimpleNamespace(tools=SimpleNamespace(_ctx=SimpleNamespace(_budget_resume_last_fit_relaxed=True)), + accumulated_usage={}, round_idx=5) + assert loop_budget._last_fit_relaxed(relaxed) is True + assert loop_budget._second_reservation_fits(relaxed, {"x": 1}, True, relaxed=True) is None + assert relaxed.accumulated_usage["budget_resume_last_fit_admitted"] == { + "round_idx": 5, "reservations_affordable": 1, "basis": "owner_resume_relaxed_last_fit"} + assert calls == [2] + strict = SimpleNamespace(tools=SimpleNamespace(_ctx=SimpleNamespace()), accumulated_usage={}, round_idx=5) + assert loop_budget._last_fit_relaxed(strict) is False + assert loop_budget._second_reservation_fits(strict, {"x": 1}, True, relaxed=False) is False + assert "budget_resume_last_fit_admitted" not in strict.accumulated_usage + # A harder stop (one reservation does not fit) already decides: no probe at all. + assert loop_budget._second_reservation_fits(strict, {"x": 1}, False, relaxed=True) is None + assert calls == [2, 2] + + +def test_restore_refusal_is_typed_and_a_panic_flag_is_a_resume_refusal_not_a_restore_one(tmp_path, monkeypatch): + from ouroboros import budget_pause + + queue, _state, workers = _install_queue(tmp_path, monkeypatch) + task, _row = _parked(tmp_path, monkeypatch, task_id="typed-1") + assert budget_pause.budget_pause_restore_refusal(tmp_path, task) == "" + (tmp_path / "state").mkdir(exist_ok=True) + (tmp_path / "state" / "panic_stop.flag").write_text("panic") + assert budget_pause.restore_budget_pause_allowed(tmp_path, task) is True # restorable; not dispatchable + assert queue.resume_budget_paused_task("typed-1")["error"] == "restart_no_resume" + (tmp_path / "state" / "panic_stop.flag").unlink() + assert budget_pause.budget_pause_restore_refusal(tmp_path, {"id": "typed-1"}) == budget_pause.RESTORE_REFUSAL_NOT_EXACT + assert budget_pause.budget_pause_restore_refusal(tmp_path, { + "id": "typed-1", "_budget_pause": {"exact_continuation": True, "checkpoint": {"pause_id": "other"}}, + }) == budget_pause.RESTORE_REFUSAL_IDENTITY_MISMATCH + assert budget_pause.budget_pause_restore_refusal(tmp_path, { + "id": "absent", "_budget_pause": {"exact_continuation": True, "checkpoint": {"pause_id": "p"}}, + }) == budget_pause.RESTORE_REFUSAL_RECORD_MISSING + + +def _source_file(tmp_path, task_id, row): + from ouroboros.artifacts import task_artifact_dir_path + + return task_artifact_dir_path(tmp_path, task_id, create=False).joinpath( + *pathlib.PurePosixPath(str(row["source_ref"]["path"])).parts) + + +def test_unrestorable_source_holds_the_row_with_its_marker_and_a_later_grant_releases_it(tmp_path, monkeypatch): + """A corrupt/missing source never drops or cancels the saved pause: the row is + HELD beside its marker, refuses Resume typed while unreadable, and the grant + that re-validates a readable source releases the hold.""" + from ouroboros import budget_pause + from supervisor.events_budget import budget_hold_fact + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 5.0) + task, row = _parked(tmp_path, monkeypatch, task_id="corrupt-1") + source = _source_file(tmp_path, "corrupt-1", row) + original = source.read_bytes() + source.unlink() + assert budget_pause.budget_pause_restore_refusal(tmp_path, task) == budget_pause.RESTORE_REFUSAL_SOURCE_UNREADABLE + queue.persist_queue_snapshot(reason="test") + workers.PENDING[:] = [] + assert queue.restore_pending_from_snapshot() == 1 + held = workers.PENDING[0] + assert held["id"] == "corrupt-1" and held["_budget_pause"]["exact_continuation"] is True + hold = budget_hold_fact(held) + assert hold["reason"] == "restore_refused:pause_source_unreadable" and hold["dispatchable"] is False + sent = [] + workers.WORKERS[0] = SimpleNamespace(wid=0, busy_task_id=None, reaping=False, + in_q=SimpleNamespace(put=lambda t: sent.append(dict(t)))) + workers.assign_tasks() + assert sent == [] # held, not assignable + assert queue.resume_budget_paused_task("corrupt-1")["error"] == "pause_source_unreadable" + assert "_budget_pause" in held and budget_hold_fact(held) is not None # retained, still held + source.write_bytes(original) + granted = queue.resume_budget_paused_task("corrupt-1") + assert granted["ok"] is True and granted["released_hold"] == "restore_refused:pause_source_unreadable" + assert budget_hold_fact(held) is None and held["_budget_pause_hold"]["selected"] is True + workers.assign_tasks() + assert [t["id"] for t in sent] == ["corrupt-1"] + + +def test_orphaned_unwritten_revocation_is_written_by_the_next_resume_before_a_new_grant(tmp_path, monkeypatch): + """A revocation the queue could not write HOLDS the row with its marker (the + handoff leaves with the hold, so that grant is never dispatched); the next + Resume writes the deferred revocation, then mints generation 2.""" + from ouroboros import budget_pause + from supervisor.events_budget import budget_hold_fact + from supervisor.queue_transitions import revoke_exact_budget_resume + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 5.0) + task, _row = _parked(tmp_path, monkeypatch, task_id="orphan-1") + assert queue.resume_budget_paused_task("orphan-1")["ok"] is True + first_grant = task["_budget_pause_resume"]["grant_id"] + real_writer = budget_pause.set_budget_pause + monkeypatch.setattr(budget_pause, "set_budget_pause", + lambda *_a, **_k: (_ for _ in ()).throw(OSError("read-only file system"))) + assert revoke_exact_budget_resume(task, "budget_exhausted_before_dispatch") is False + monkeypatch.setattr(budget_pause, "set_budget_pause", real_writer) + hold = budget_hold_fact(task) + assert hold["reason"] == "resume_grant_revocation_unwritten" and hold["grant_id"] == first_grant + assert task["_budget_pause"]["exact_continuation"] is True and "_budget_pause_resume" not in task + assert budget_pause.budget_pause_row(tmp_path, "orphan-1")["state"] == budget_pause.STATE_RESUME_GRANTED + sent = [] + workers.WORKERS[0] = SimpleNamespace(wid=0, busy_task_id=None, reaping=False, + in_q=SimpleNamespace(put=lambda t: sent.append(dict(t)))) + workers.assign_tasks() + assert sent == [] # a stale grant never dispatches + second = queue.resume_budget_paused_task("orphan-1") + assert second["ok"] is True and second["grant_id"] != first_grant + assert second["grant_generation"] == 2 and second["released_hold"] == "resume_grant_revocation_unwritten" + row = budget_pause.budget_pause_row(tmp_path, "orphan-1") + assert row["grant"]["grant_id"] == second["grant_id"] and row["resume_generation"] == 2 + ctx, _limit = _loop_ctx(tmp_path, "orphan-1") + with pytest.raises(ValueError): # the orphaned grant is dead for good + budget_pause.load_budget_pause(ctx, {"pause_id": row["pause_id"], "grant_id": first_grant}) + workers.assign_tasks() + assert [t["_budget_pause_resume"]["grant_id"] for t in sent] == [second["grant_id"]] + + +def test_root_re_resume_rebinds_fence_lifted_sibling_holds_to_the_live_grant(tmp_path, monkeypatch): + """pauseA -> Resume -> pauseB -> Resume: a zero-dispatch sibling held from the first + Resume is re-bound to the live grant and selectable by the model under it.""" + from supervisor.events_budget import HOLD_ROOT_FENCE_LIFTED, budget_hold_fact + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 5.0) + root, _row = _parked(tmp_path, monkeypatch, task_id="root-3", scope="root") + sibling = {"id": "sib-3", "type": "task", "chat_id": 0, "root_task_id": "root-3", + "parent_task_id": "root-3", "_attempt": 1} + workers.PENDING.append(sibling) + first = queue.resume_budget_paused_task("root-3") + assert first["ok"] is True and first["held_siblings"] == ["sib-3"] + hold = budget_hold_fact(sibling) + assert hold["reason"] == HOLD_ROOT_FENCE_LIFTED and hold["root_grant_id"] == first["grant_id"] + sent = [] + workers.WORKERS[0] = SimpleNamespace(wid=0, busy_task_id=None, reaping=False, + in_q=SimpleNamespace(put=lambda t: sent.append(dict(t)))) + workers.assign_tasks() + assert [t["id"] for t in sent] == ["root-3"] # the sibling waits for an explicit selection + pause_a = sent[0]["_budget_pause_resume"]["pause_id"] + # The root pauses AGAIN (pause B, a new pause id) and the owner resumes it again. + workers.RUNNING.clear() + workers.PENDING[:] = [sibling] + _root_b, row_b = _parked(tmp_path, monkeypatch, task_id="root-3", scope="root") + assert row_b["pause_id"] != pause_a + second = queue.resume_budget_paused_task("root-3") + assert second["ok"] is True and second["grant_id"] != first["grant_id"] + assert second["rebound_held_siblings"] == ["sib-3"] + assert budget_hold_fact(sibling)["root_grant_id"] == second["grant_id"] + # The model's selection is granted under the LIVE root grant only. + selected = queue.resume_budget_paused_task("sib-3", selected_by="root-3") + assert selected["ok"] is True and selected["selection"] == "budget_hold_released" + assert budget_hold_fact(sibling) is None and sibling["_budget_pause_hold"]["selected_by"] == "root-3" + + +def test_model_issued_child_selection_needs_the_roots_live_grant(tmp_path, monkeypatch): + """Owner Q9: lineage alone never revives a paused descendant; the child is granted + only under its root's live owner-derived Resume, and the grant names the requester.""" + from ouroboros import budget_pause + from supervisor.events_budget import _handle_budget_resume_child + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 5.0) + root, _r = _parked(tmp_path, monkeypatch, task_id="root-6", scope="root") + child, _c = _parked(tmp_path, monkeypatch, task_id="child-6", root_task_id="root-6") + child["parent_task_id"] = "root-6" + pushed = [] + sctx = _supervisor_ctx(tmp_path, workers, queue, [], pushed) + request = {"type": "budget_resume_child", "task_id": "child-6", "requested_by": "root-6", "reason": "still needed"} + _handle_budget_resume_child(dict(request), sctx) + assert pushed[-1]["type"] == "budget_resume_child_outcome" + assert pushed[-1]["ok"] is False and pushed[-1]["error"] == "root_resume_grant_missing" + assert "_budget_pause" in child and "_budget_pause_resume" not in child + assert queue.resume_budget_paused_task("root-6")["ok"] is True + _handle_budget_resume_child(dict(request), sctx) + assert pushed[-1]["ok"] is True and child["_budget_pause_resume"]["grant_id"] + row = budget_pause.budget_pause_row(tmp_path, "child-6") + root_grant = budget_pause.budget_pause_row(tmp_path, "root-6")["grant"]["grant_id"] + assert row["grant"]["selected_by"] == "root-6" and row["grant"]["root_grant_id"] == root_grant + + +def test_restart_parks_a_running_row_whose_pause_was_already_complete(tmp_path, monkeypatch): + """A RUNNING row at shutdown whose durable pause (row + source) was complete is parked + under its exact marker, its root fence raised — never handed to the shutdown cancel fence.""" + from ouroboros import budget_pause + + queue, _state, workers = _install_queue(tmp_path, monkeypatch) + ctx, _limit, _pause_row = _pause(tmp_path, monkeypatch, task_id="park-1", scope="root") + budget_pause.end_dispatch_fence("park-1") + task = {"id": "park-1", "type": "task", "chat_id": 0, "root_task_id": "park-1", "_attempt": 1, "_queue_seq": 3} + workers.RUNNING["park-1"] = {"task": task, "worker_id": 0, "attempt": 1, "started_at": time.time(), + "last_heartbeat_at": time.time(), "last_progress_at": time.time()} + queue.persist_queue_snapshot(reason="test") + workers.RUNNING.clear() + workers.PENDING[:] = [] + terminalized = [] + assert queue.restore_pending_from_snapshot(terminalized=terminalized) == 1 + assert terminalized == [] + parked = workers.PENDING[0] + assert parked["id"] == "park-1" and parked["_budget_pause"]["exact_continuation"] is True + assert parked["_budget_pause"]["fence_id"] and queue.BUDGET_ROOT_FENCES["park-1"]["status"] == "paused" + row = budget_pause.budget_pause_row(tmp_path, "park-1") + assert row["state"] == budget_pause.STATE_PAUSED and row["pause_source"] == "restart_during_pausing" + + +def test_a_consumed_grant_is_never_re_armed_by_a_later_revocation(tmp_path, monkeypatch): + """Negative: money vanishing (or a restart) AFTER the loop consumed its grant must not + put the task back on the pause path. The queue row is a stale carrier: its handoff + leaves, no ``_budget_pause`` marker is re-minted over a task that is not paused, the + durable RESUMED row is not written over, and the task's own status is not rewritten.""" + from ouroboros import budget_pause + from ouroboros.task_results import load_task_result + from supervisor.budget_resume import revoke_exact_budget_resume + from supervisor.events_budget import HOLD_GRANT_CONSUMED_STALE_ROW, budget_hold_fact + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 5.0) + task, _row = _parked(tmp_path, monkeypatch, task_id="loop-4") + assert queue.resume_budget_paused_task("loop-4")["ok"] is True + assert "_budget_pause" not in task and isinstance(task["_budget_pause_resume"], dict) + granted = budget_pause.budget_pause_row(tmp_path, "loop-4") + grant_id = str((granted.get("grant") or {}).get("grant_id") or "") + # The worker consumed the grant: the durable row says RESUMED, as the loop writes it. + consumed_grant = {**dict(granted.get("grant") or {}), "consumed_at": time.time()} + budget_pause.set_budget_pause( + tmp_path, "loop-4", {**granted, "state": budget_pause.STATE_RESUMED, "grant": consumed_grant}, + expected_pause_id=str(granted.get("pause_id") or ""), + expected_state=budget_pause.STATE_RESUME_GRANTED, expected_grant_id=grant_id) + + assert revoke_exact_budget_resume(task, "money_vanished") is False + assert "_budget_pause_resume" not in task + assert "_budget_pause" not in task # NOT re-minted over a task that is running on + assert task["_budget_pause_consumed"]["grant_id"] == grant_id + assert budget_hold_fact(task)["reason"] == HOLD_GRANT_CONSUMED_STALE_ROW + after = budget_pause.budget_pause_row(tmp_path, "loop-4") + assert after["state"] == budget_pause.STATE_RESUMED and after["grant"]["consumed_at"] + assert load_task_result(tmp_path, "loop-4", strict=True)["status"] != "cancelled" + + +def test_malformed_acceptance_fence_evidence_retains_a_saved_pause_but_fails_closed_elsewhere( + tmp_path, monkeypatch): + """Corrupt acceptance-fence evidence fails CLOSED: an ordinary row that cannot prove + its root's review state is cancelled rather than started. A saved exact budget pause + is the one exception — retained as a typed, non-dispatchable hold carrying its + ORIGINAL locator, never cancelled, until the owner's next Resume re-validates it.""" + from ouroboros.task_results import load_task_result + from supervisor.events_budget import HOLD_INVALID_ACCEPTANCE_FENCE_SNAPSHOT, budget_hold_fact + + queue, _state, workers = _install_queue(tmp_path, monkeypatch) + _paused, row = _parked(tmp_path, monkeypatch, task_id="child-6", root_task_id="root-6") + workers.PENDING.append({"id": "plain-6", "type": "task", "chat_id": 0, + "root_task_id": "root-6", "_attempt": 1}) + # An ACTIVE fence with no root id is evidence nothing can be proven from. + queue.ACCEPTANCE_FENCES["root-6"] = {"status": "active", "token": "tok", "generation": 1} + queue.persist_queue_snapshot(reason="test") + workers.PENDING[:] = [] + try: + assert queue.restore_pending_from_snapshot() == 1 + finally: + queue.ACCEPTANCE_FENCES.pop("root-6", None) + assert [task["id"] for task in workers.PENDING] == ["child-6"] + held = workers.PENDING[0] + assert held["_budget_pause"]["exact_continuation"] is True + # The ORIGINAL locator survives: the checkpoint still names the exact saved pause. + assert held["_budget_pause"]["checkpoint"]["pause_id"] == row["pause_id"] + assert held["_budget_pause"]["checkpoint"]["source_ref"] == row["source_ref"] + assert budget_hold_fact(held)["reason"] == HOLD_INVALID_ACCEPTANCE_FENCE_SNAPSHOT + assert load_task_result(tmp_path, "child-6", strict=True)["status"] != "cancelled" + # The ordinary row failed closed: not restored, and terminalized as cancelled. + assert load_task_result(tmp_path, "plain-6", strict=True)["status"] == "cancelled" + + +def test_acceptance_fence_at_restore_holds_a_saved_pause_instead_of_cancelling(tmp_path, monkeypatch): + from supervisor.events_budget import HOLD_ROOT_ACCEPTANCE_FENCED, budget_hold_fact + from ouroboros.task_results import load_task_result + + queue, _state, workers = _install_queue(tmp_path, monkeypatch) + child, _row = _parked(tmp_path, monkeypatch, task_id="child-5", root_task_id="root-5") + child["parent_task_id"] = "root-5" + queue.ACCEPTANCE_FENCES["root-5"] = {"status": "active", "root_task_id": "root-5", "token": "tok", "generation": 1} + queue.persist_queue_snapshot(reason="test") + workers.PENDING[:] = [] + try: + assert queue.restore_pending_from_snapshot() == 1 + finally: + queue.ACCEPTANCE_FENCES.pop("root-5", None) + held = workers.PENDING[0] + assert held["id"] == "child-5" and held["_budget_pause"]["exact_continuation"] is True + assert budget_hold_fact(held)["reason"] == HOLD_ROOT_ACCEPTANCE_FENCED + assert load_task_result(tmp_path, "child-5", strict=True)["status"] != "cancelled" + + +def _fenced_member(workers, task_id, root_task_id): + member = {"id": task_id, "type": "task", "chat_id": 0, "root_task_id": root_task_id, + "parent_task_id": root_task_id, "_attempt": 1} + workers.PENDING.append(member) + return member + + +def _idle_worker(workers, sent): + workers.WORKERS[0] = SimpleNamespace(wid=0, busy_task_id=None, reaping=False, + in_q=SimpleNamespace(put=lambda t: sent.append(dict(t)))) + + +def test_selecting_one_member_of_a_fenced_root_never_lifts_the_latch(tmp_path, monkeypatch): + """F2: the zero-dispatch branch used to clear the ROOT's admission latch when a + CHILD was nominated — every sibling became assignable at once, and a + model-issued request was granted without the root's live Resume grant. The + nomination is now one explicit per-row selection: the latch stands, the + unselected siblings stay fenced, and only the selected row dispatches.""" + from supervisor.events_budget import _set_root_budget_pause_locked, budget_hold_fact + from supervisor.queue_transitions import budget_pause_fact + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 5.0) + fence = _set_root_budget_pause_locked("root-f2", {"scope": "root", "root_task_id": "root-f2"}) + first = _fenced_member(workers, "m1-f2", "root-f2") + second = _fenced_member(workers, "m2-f2", "root-f2") + + # A MODEL-issued selection needs the root's live owner Resume grant (Q9). + refused = queue.resume_budget_paused_task("m1-f2", selected_by="root-f2") + assert refused["error"] == "root_resume_grant_missing" and refused["action"] == "resume_root_first" + assert queue.BUDGET_ROOT_FENCES["root-f2"]["fence_id"] == fence["fence_id"] + assert budget_hold_fact(first) is not None # held, not released, not dropped + + # The owner's own explicit act selects exactly this row. + granted = queue.resume_budget_paused_task("m1-f2") + assert granted["ok"] is True and granted["selection"] == "budget_hold_released" + assert queue.BUDGET_ROOT_FENCES["root-f2"]["fence_id"] == fence["fence_id"] # the latch stands + assert first["_budget_pause_hold"]["selected"] is True + assert first["_budget_pause_hold"]["fence_id"] == fence["fence_id"] + # The UI truth follows the same rule: one row released, the other still paused. + assert budget_pause_fact(first) is None + assert budget_pause_fact(second)["fence_id"] == fence["fence_id"] + sent = [] + _idle_worker(workers, sent) + workers.assign_tasks() + assert [t["id"] for t in sent] == ["m1-f2"] # no fan-out of the fenced tree + + +def test_a_selection_bound_to_an_older_fence_does_not_pre_release_a_new_one(tmp_path, monkeypatch): + """The selection names the fence generation it was granted against: a root that + paused AGAIN raises a new latch, and the earlier selection is not a key to it.""" + from supervisor.events_budget import _set_root_budget_pause_locked + from supervisor.queue_transitions import budget_pause_fact + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 5.0) + _set_root_budget_pause_locked("root-f2b", {"scope": "root", "root_task_id": "root-f2b"}) + member = _fenced_member(workers, "m1-f2b", "root-f2b") + assert queue.resume_budget_paused_task("m1-f2b")["ok"] is True + assert budget_pause_fact(member) is None + queue.BUDGET_ROOT_FENCES.pop("root-f2b", None) + newer = _set_root_budget_pause_locked("root-f2b", {"scope": "root", "root_task_id": "root-f2b", + "fence_id": "fence-2"}) + assert newer["fence_id"] == "fence-2" + assert budget_pause_fact(member)["fence_id"] == "fence-2" + sent = [] + _idle_worker(workers, sent) + workers.assign_tasks() + assert sent == [] + + +def test_root_resume_holds_a_fence_bound_child_marker_instead_of_stranding_it(tmp_path, monkeypatch): + """F2: a child carrying a NON-exact marker minted from the root's own latch was + skipped by the root's Resume and then refused forever with + ``root_budget_fence_missing`` — the fence it named was gone. The marker is the + fence, so it becomes the same hold a fence-only sibling takes, selectable + under the root's live grant.""" + from supervisor.events_budget import HOLD_ROOT_FENCE_LIFTED, budget_hold_fact + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 5.0) + root, _row = _parked(tmp_path, monkeypatch, task_id="root-f3", scope="root") + fence = queue.BUDGET_ROOT_FENCES["root-f3"] + child = _fenced_member(workers, "child-f3", "root-f3") + child["_budget_pause"] = {"status": "paused_before_dispatch", "scope": "root", + "root_task_id": "root-f3", "fence_id": fence["fence_id"], + "replay_safe": True, "physical_calls": 0, "auto_resume": False} + + granted = queue.resume_budget_paused_task("root-f3") + assert granted["ok"] is True and granted["held_siblings"] == ["child-f3"] + assert "_budget_pause" not in child + hold = budget_hold_fact(child) + assert hold["reason"] == HOLD_ROOT_FENCE_LIFTED and hold["replaced_fence_marker"] is True + assert hold["root_grant_id"] == granted["grant_id"] + sent = [] + _idle_worker(workers, sent) + workers.assign_tasks() + assert [t["id"] for t in sent] == ["root-f3"] # eligibility is not continuation + + selected = queue.resume_budget_paused_task("child-f3", selected_by="root-f3") + assert selected["ok"] is True and selected["selection"] == "budget_hold_released" + assert budget_hold_fact(child) is None + workers.WORKERS[0].busy_task_id = None # the slot the root took is free again + workers.assign_tasks() + assert [t["id"] for t in sent] == ["root-f3", "child-f3"] + + +def test_a_consumed_carrier_never_drops_a_newer_owner_wait_continuation(tmp_path, monkeypatch): + """F3: a spent exact-resume handoff is retired when the supervisor's own park + event proves the durable grant was consumed. A stale carrier that reaches a + restore beside a NEWER owner-wait handoff is retired there too — fencing the + row as "consumed" would drop a valid planned-restart continuation.""" + from ouroboros import budget_pause, owner_wait + from supervisor.events_budget import budget_hold_fact + from supervisor.worker_owner_wait import retire_consumed_budget_carrier + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 5.0) + task, _row = _parked(tmp_path, monkeypatch, task_id="wait-3") + assert queue.resume_budget_paused_task("wait-3")["ok"] is True + carrier = dict(task["_budget_pause_resume"]) + granted = budget_pause.budget_pause_row(tmp_path, "wait-3") + grant_id = str(granted["grant"]["grant_id"]) + + # An UNCONSUMED grant is left exactly where it is: the restart must revoke it. + assert retire_consumed_budget_carrier(tmp_path, "wait-3", task) == "" + assert task["_budget_pause_resume"]["grant_id"] == grant_id + + consumed = {**dict(granted["grant"]), "consumed_at": time.time()} + budget_pause.set_budget_pause( + tmp_path, "wait-3", {**granted, "state": budget_pause.STATE_RESUMED, "grant": consumed}, + expected_pause_id=str(granted["pause_id"]), + expected_state=budget_pause.STATE_RESUME_GRANTED, expected_grant_id=grant_id) + task["_owner_wait_resume"] = {"wait_id": "w-3", "restart_transaction_id": "tx-3", + "task_attempt": 1, "source_ref": {"path": "owner-wait.json"}, + "started_at": time.time() - 10.0} + assert retire_consumed_budget_carrier(tmp_path, "wait-3", task) == grant_id + assert "_budget_pause_resume" not in task and task["_owner_wait_resume"]["wait_id"] == "w-3" + + # A snapshot taken BEFORE that retirement still restores the owner wait. + task["_budget_pause_resume"] = carrier + queue.persist_queue_snapshot(reason="test") + workers.PENDING[:] = [] + monkeypatch.setattr(owner_wait, "restore_owner_wait_allowed", lambda *_a, **_k: True) + assert queue.restore_pending_from_snapshot() == 1 + restored = workers.PENDING[0] + assert restored["id"] == "wait-3" and restored["_owner_wait_resume"]["wait_id"] == "w-3" + assert "_budget_pause_resume" not in restored and "_budget_pause_consumed" not in restored + assert budget_hold_fact(restored) is None + after = budget_pause.budget_pause_row(tmp_path, "wait-3") + assert after["state"] == budget_pause.STATE_RESUMED # never re-armed + + +def test_owner_wait_keeps_the_budget_paused_carrier_across_a_planned_restart(tmp_path, monkeypatch): + """F5: the ONE serializer carries the paused interval, so a task that was budget + paused and later parks in an owner wait resumes on the SAME execution clock. + Without it the restart's finite-lifetime reader charges the pause as execution.""" + from ouroboros import config, model_wait, owner_wait + from ouroboros.task_results import STATUS_RUNNING, write_task_result + + write_task_result(tmp_path, "wait-5", STATUS_RUNNING, result="running") + ctx, _limit = _loop_ctx(tmp_path, "wait-5") + clock = {"revision": 0, "elapsed_sec": 30.0, "observed_at": time.time(), "active": False} + ctx.model_wait_context = SimpleNamespace(continuation_state=lambda: { + "overrides": {}, "auto_continue": {}, "budget_paused_sec": 600.0, "quota_clock": clock}) + handoff = owner_wait.checkpoint_owner_wait(ctx, [], {}, {}, 3, [], set()) + assert handoff["budget_paused_sec"] == 600.0 + + now = time.time() + started = now - 1000.0 + meta = {"started_at": started, "model_wait_quota_clock": handoff["model_wait_quota_clock"], + "budget_paused_sec": handoff["budget_paused_sec"]} + # ONE shared clock: 1000s of wall time minus 30s of quota wait minus 600s paused. + assert model_wait.execution_elapsed_seconds(meta, now) == pytest.approx(370.0, abs=5.0) + + owner_wait.set_owner_wait(tmp_path, "wait-5", {**handoff, "state": "waiting", "started_at": started}) + monkeypatch.setattr("ouroboros.delegate_recovery._read_restart_transaction", + lambda _root, _tid: {"status": "normal_exit_acknowledged", "task_ids": ["wait-5"]}) + monkeypatch.setattr("ouroboros.delegate_recovery._ack_direct_exec_successor", lambda _root: None) + monkeypatch.setattr("ouroboros.cancel_intents.has_active_intent", lambda *_a, **_k: False) + monkeypatch.setattr(config, "get_task_abs_ceiling_sec", lambda: 500.0) + row = {"id": "wait-5", "_owner_wait_resume": {**handoff, "restart_transaction_id": "tx-5", + "started_at": started}} + assert owner_wait.restore_owner_wait_allowed(tmp_path, row) is True + # The negative control: the same wall clock WITHOUT the carrier is 970s and + # would read as an exhausted lifetime — the pause must not be charged twice. + owner_wait.set_owner_wait(tmp_path, "wait-5", {**handoff, "state": "waiting", + "started_at": started, "budget_paused_sec": 0.0}, + expected_wait_id=handoff["wait_id"]) + assert owner_wait.restore_owner_wait_allowed(tmp_path, row) is False + + +def test_an_unwritten_grant_rollback_holds_the_row_instead_of_granting_forever(tmp_path, monkeypatch): + """F6: when the snapshot cannot be persisted AND the durable rollback cannot be + written, the row keeps a grant the queue no longer carries. Recorded as the + EXISTING unwritten-revocation hold, the next Resume writes that revocation + before minting — instead of answering ``resume_already_granted`` forever.""" + from ouroboros import budget_pause + from supervisor.events_budget import HOLD_REVOCATION_UNWRITTEN, budget_hold_fact + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda _st, **_k: 5.0) + task, _row = _parked(tmp_path, monkeypatch, task_id="f6-1") + real_writer = budget_pause.set_budget_pause + real_snapshot = queue.persist_queue_snapshot + + def _no_rollback(root, task_id, row, expected_pause_id=None, **kwargs): + if kwargs.get("expected_state") == budget_pause.STATE_RESUME_GRANTED: + raise OSError("read-only file system") + return real_writer(root, task_id, row, expected_pause_id, **kwargs) + + monkeypatch.setattr(budget_pause, "set_budget_pause", _no_rollback) + monkeypatch.setattr(queue, "persist_queue_snapshot", lambda reason="": False) + refused = queue.resume_budget_paused_task("f6-1") + assert refused["error"] == "snapshot_not_persisted" and refused["held"] == HOLD_REVOCATION_UNWRITTEN + hold = budget_hold_fact(task) + assert hold["reason"] == HOLD_REVOCATION_UNWRITTEN and hold["grant_id"] == refused["grant_id"] + assert task["_budget_pause"]["exact_continuation"] is True and "_budget_pause_resume" not in task + assert budget_pause.budget_pause_row(tmp_path, "f6-1")["state"] == budget_pause.STATE_RESUME_GRANTED + sent = [] + _idle_worker(workers, sent) + workers.assign_tasks() + assert sent == [] # held: the orphaned grant never dispatches + + monkeypatch.setattr(budget_pause, "set_budget_pause", real_writer) + monkeypatch.setattr(queue, "persist_queue_snapshot", real_snapshot) + second = queue.resume_budget_paused_task("f6-1") + assert second["ok"] is True and second["grant_id"] != refused["grant_id"] + assert second["grant_generation"] == 2 + assert second["released_hold"] == HOLD_REVOCATION_UNWRITTEN + row = budget_pause.budget_pause_row(tmp_path, "f6-1") + assert row["grant"]["grant_id"] == second["grant_id"] and row["resume_generation"] == 2 diff --git a/tests/test_budget_pause_safety.py b/tests/test_budget_pause_safety.py new file mode 100644 index 000000000..c8e20d7aa --- /dev/null +++ b/tests/test_budget_pause_safety.py @@ -0,0 +1,322 @@ +"""Pause-generation, persistence-failure and control regressions at real consumers.""" + +import copy +import json +from types import SimpleNamespace + +import pytest + +from tests.test_budget_pause_exact import _install_queue, _loop_ctx, _parked, _supervisor_ctx +from tests.test_budget_pause_holds import _fenced_member, _idle_worker, _source_file + +pytestmark = pytest.mark.serial + + +@pytest.mark.parametrize("late_carrier", ["none", "before_assignment", "before_root_resume"]) +def test_new_root_pause_invalidates_child_selection_until_selected_again(tmp_path, monkeypatch, late_carrier): + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda *_a, **_k: 5.0) + root, _ = _parked(tmp_path, monkeypatch, task_id="root", scope="root") + child, _ = _parked(tmp_path, monkeypatch, task_id="child", root_task_id="root") + sibling = _fenced_member(workers, "sibling", "root") + first = queue.resume_budget_paused_task("root") + assert first["ok"] and queue.resume_budget_paused_task("child", selected_by="root")["ok"] + assert queue.resume_budget_paused_task("sibling", selected_by="root")["ok"] + old_child = copy.deepcopy(child) + workers.PENDING.remove(root) # root ran, then paused for a second time + root2, _ = _parked(tmp_path, monkeypatch, task_id="root", scope="root") + assert "_budget_pause_resume" not in child and "_budget_pause" in child + assert not sibling["_budget_pause_hold"]["selected"] + if late_carrier != "none": + child.clear() + child.update(old_child) # a stale queue carrier cannot bypass either consumer + sent = [] + _idle_worker(workers, sent) + if late_carrier != "before_root_resume": + workers.assign_tasks() + assert sent == [] + second = queue.resume_budget_paused_task("root") + assert second["ok"] and second["grant_id"] != first["grant_id"] + workers.assign_tasks() + assert [row["id"] for row in sent] == ["root"] + workers.WORKERS[0].busy_task_id = None + workers.assign_tasks() + assert [row["id"] for row in sent] == ["root"] # root Resume only made children eligible + assert queue.resume_budget_paused_task("child", selected_by="root")["ok"] + workers.assign_tasks() + assert [row["id"] for row in sent] == ["root", "child"] + assert sent[-1]["_budget_pause_resume"]["root_grant_id"] == second["grant_id"] + assert not sibling["_budget_pause_hold"]["selected"] + + +def test_assignment_rechecks_root_grant_even_when_no_fence_is_left(tmp_path, monkeypatch): + from ouroboros import budget_pause + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda *_a, **_k: 5.0) + root, _ = _parked(tmp_path, monkeypatch, task_id="root", scope="root") + child, _ = _parked(tmp_path, monkeypatch, task_id="child", root_task_id="root") + assert queue.resume_budget_paused_task("root")["ok"] + assert queue.resume_budget_paused_task("child", selected_by="root")["ok"] + workers.PENDING.remove(root) + row = budget_pause.budget_pause_row(tmp_path, "root") + row["grant"]["grant_id"] = "new-root-grant" + row["resume_generation"] += 1 + budget_pause.set_budget_pause(tmp_path, "root", row) + sent = [] + _idle_worker(workers, sent) + workers.assign_tasks() + assert sent == [] and "_budget_pause_resume" not in child + assert queue.resume_budget_paused_task("child", selected_by="root")["ok"] + workers.assign_tasks() + assert [task["id"] for task in sent] == ["child"] + + +@pytest.mark.parametrize("marker", [False, True]) +def test_legacy_root_resume_selects_only_root_then_one_child(tmp_path, monkeypatch, marker): + from supervisor.events_budget import _set_root_budget_pause_locked + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda *_a, **_k: 5.0) + root = _fenced_member(workers, "root", "root") + root.pop("parent_task_id") + child = _fenced_member(workers, "child", "root") + sibling = _fenced_member(workers, "sibling", "root") + fence = _set_root_budget_pause_locked("root", {}) + if marker: + root["_budget_pause"] = {**fence, "physical_calls": 0, "replay_safe": True} + assert queue.resume_budget_paused_task("root")["ok"] + assert queue.BUDGET_ROOT_FENCES["root"] == fence + sent = [] + _idle_worker(workers, sent) + workers.assign_tasks() + assert [task["id"] for task in sent] == ["root"] + assert queue.resume_budget_paused_task("child", selected_by="root")["ok"] + workers.WORKERS[0].busy_task_id = None + workers.assign_tasks() + assert [task["id"] for task in sent] == ["root", "child"] and sibling in workers.PENDING + # A subsequent pause creates a new fence even though the legacy latch stayed up. + assert _set_root_budget_pause_locked("root", {})["fence_id"] != fence["fence_id"] + + +def _dead_granted_task(tmp_path, monkeypatch): + from supervisor import worker_health + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda *_a, **_k: 5.0) + task, row = _parked(tmp_path, monkeypatch, task_id="dead") + assert queue.resume_budget_paused_task("dead")["ok"] + _idle_worker(workers, []) + workers.assign_tasks() + monkeypatch.setattr(worker_health, "_dead_job_is_current", lambda _job: True) + monkeypatch.setattr(workers, "send_with_budget", lambda *_a, **_k: None) + job = {"worker": workers.WORKERS[0], "worker_id": 0, "task_id": "dead", "task": task, + "meta": workers.RUNNING["dead"], "exitcode": 1, "drive_root": str(tmp_path)} + return queue, workers, job, row + + +@pytest.mark.parametrize("snapshot_ok", [True, False]) +def test_dead_unconsumed_grant_keeps_exact_hold_when_revocation_stores_fail(tmp_path, monkeypatch, snapshot_ok): + from ouroboros import budget_pause + from supervisor import events_budget, worker_health + + queue, workers, job, row = _dead_granted_task(tmp_path, monkeypatch) + original_grant = copy.deepcopy(job["task"]["_budget_pause_resume"]) + writer, snapshot = budget_pause.set_budget_pause, queue.persist_queue_snapshot + monkeypatch.setattr(budget_pause, "set_budget_pause", lambda *_a, **_k: (_ for _ in ()).throw(OSError("disk full"))) + monkeypatch.setattr(events_budget, "write_task_result", lambda *_a, **_k: (_ for _ in ()).throw(OSError("disk full"))) + if not snapshot_ok: + monkeypatch.setattr(queue, "persist_queue_snapshot", lambda **_k: False) + terminals = [] + monkeypatch.setattr(workers, "_emit_task_done_terminal", lambda *_a, **_k: terminals.append(True)) + worker_health._recover_crashed_task_without_terminal(job, queue) + assert terminals == [] and not workers.RUNNING + held = workers.PENDING[0] + hold = held["_budget_pause_hold"] + assert hold["grant_id"] == original_grant["grant_id"] and hold["pause_id"] == row["pause_id"] + assert hold["snapshot_persisted"] is snapshot_ok and "_budget_pause_resume" not in held + assert held["_budget_pause"]["checkpoint"]["source_ref"] == row["source_ref"] + sent = [] + _idle_worker(workers, sent) + workers.assign_tasks() + assert sent == [] + monkeypatch.setattr(budget_pause, "set_budget_pause", writer) + monkeypatch.setattr(queue, "persist_queue_snapshot", snapshot) + assert queue.resume_budget_paused_task("dead")["ok"] + workers.assign_tasks() + assert [task["id"] for task in sent] == ["dead"] + + +def test_consumed_grant_crash_never_enters_ordinary_retry(tmp_path, monkeypatch): + from ouroboros import budget_pause, delegate_recovery + from ouroboros.task_results import load_task_result + from supervisor import worker_health + + queue, workers, job, _ = _dead_granted_task(tmp_path, monkeypatch) + ctx, _ = _loop_ctx(tmp_path, "dead") + handoff = job["task"]["_budget_pause_resume"] + # The real consumption owner marks the grant before any subsequent effect. + saved = budget_pause.load_budget_pause(ctx, handoff) + from ouroboros import owner_wait + + monkeypatch.setattr(owner_wait, "rebind_restored_route", lambda *_a, **_k: (None, "max")) + budget_pause.resume_paused_loop(SimpleNamespace(_ctx=ctx), saved, [], {}, {}, set(), + budget_remaining_usd=5.0) + monkeypatch.setattr(delegate_recovery, "reconcile_unrecoverable_task", lambda *_a, **_k: None) + monkeypatch.setattr(queue, "enqueue_task", lambda *_a, **_k: pytest.fail("completed effects replayed")) + terminals = [] + monkeypatch.setattr(workers, "_emit_task_done_terminal", lambda *_a, **kw: terminals.append(kw)) + worker_health._recover_crashed_task_without_terminal(job, queue) + assert not workers.PENDING and not workers.RUNNING and len(terminals) == 1 + assert load_task_result(tmp_path, "dead", strict=True)["status"] == "failed" + assert budget_pause.budget_pause_row(tmp_path, "dead")["state"] == budget_pause.STATE_RESUMED + + +@pytest.mark.parametrize("fences", ["budget", "acceptance", "both"]) +@pytest.mark.parametrize("source_state", ["intact", "missing", "corrupt"]) +def test_invalid_fences_retain_exact_source_locators(tmp_path, monkeypatch, fences, source_state): + from supervisor.events_budget import budget_hold_fact + + queue, _, workers = _install_queue(tmp_path, monkeypatch) + task, row = _parked(tmp_path, monkeypatch, task_id="saved") + source = _source_file(tmp_path, "saved", row) + if source_state == "missing": + source.unlink() + elif source_state == "corrupt": + source.write_text("broken checkpoint") + queue.persist_queue_snapshot(reason="test") + snapshot = json.loads(queue.QUEUE_SNAPSHOT_PATH.read_text()) + for field, name in (("budget_root_fences", "budget"), ("acceptance_fences", "acceptance")): + if fences in {name, "both"}: + snapshot[field] = [{"status": "active"}] + queue.QUEUE_SNAPSHOT_PATH.write_text(json.dumps(snapshot)) + workers.PENDING.clear() + assert queue.restore_pending_from_snapshot() == 1 + held = workers.PENDING[0] + assert held["_budget_pause"]["checkpoint"] == task["_budget_pause"]["checkpoint"] + assert budget_hold_fact(held) and held["_budget_pause"]["checkpoint"]["source_ref"] == row["source_ref"] + sent = [] + _idle_worker(workers, sent) + workers.assign_tasks() + assert sent == [] + if source_state != "intact": + assert not queue.resume_budget_paused_task("saved")["ok"] + + +@pytest.mark.parametrize("surface,key", [("handoff", "pause_id"), ("handoff", "grant_id"), + ("row", "pause_id"), ("grant", "grant_id")]) +def test_empty_resume_identity_cannot_match_or_overwrite(tmp_path, monkeypatch, surface, key): + from ouroboros import budget_pause + from supervisor.budget_resume import revoke_exact_budget_resume + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda *_a, **_k: 5.0) + task, _ = _parked(tmp_path, monkeypatch, task_id="malformed") + assert queue.resume_budget_paused_task("malformed")["ok"] + row = budget_pause.budget_pause_row(tmp_path, "malformed") + if surface == "handoff": + task["_budget_pause_resume"][key] = "" + else: + (row if surface == "row" else row["grant"])[key] = "" + monkeypatch.setattr(budget_pause, "budget_pause_row", lambda *_a: copy.deepcopy(row)) + writes = [] + monkeypatch.setattr(budget_pause, "set_budget_pause", lambda *_a, **_k: writes.append(True)) + assert revoke_exact_budget_resume(task, "restart_before_dispatch") is False + assert not writes and "_budget_pause_resume" not in task and task["_budget_pause_hold"] + sent = [] + _idle_worker(workers, sent) + workers.assign_tasks() + assert sent == [] and not queue.resume_budget_paused_task("malformed")["ok"] and not writes + + +def test_resume_tool_selects_root_grandchild_and_parent_child_but_no_other_tree(tmp_path, monkeypatch): + from ouroboros.task_results import STATUS_SCHEDULED, write_task_result + from ouroboros.tools import control, join_ledger + from supervisor.events_budget import _handle_budget_resume_child + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda *_a, **_k: 5.0) + _parked(tmp_path, monkeypatch, task_id="root", scope="root") + _parked(tmp_path, monkeypatch, task_id="grand", root_task_id="root", + extra={"parent_task_id": "middle", "delegation_role": "subagent"}) + for tid, parent, root in (("middle", "root", "root"), ("grand", "middle", "root"), + ("foreign", "other", "other")): + write_task_result(tmp_path, tid, STATUS_SCHEDULED, parent_task_id=parent, + root_task_id=root, delegation_role="subagent") + events = [] + monkeypatch.setattr(control, "_emit_control_event", lambda _ctx, evt: events.append(evt) or "live") + monkeypatch.setattr(join_ledger, "_record_child_decision_beacon", lambda *_a, **_k: None) + ctx = SimpleNamespace(task_id="root", drive_root=tmp_path, task_metadata={}) + assert "Resume requested" in join_ledger._resume_child_task(ctx, "grand") + assert "not a child" in join_ledger._resume_child_task(ctx, "foreign") + assert "not a child" in join_ledger._resume_child_task(ctx, "unknown") + ctx.task_id = "middle" + assert "Resume requested" in join_ledger._resume_child_task(ctx, "grand") + assert "not a child" in join_ledger._resume_child_task(ctx, "root") + assert queue.resume_budget_paused_task("root")["ok"] + outcomes = [] + _handle_budget_resume_child(events[0], _supervisor_ctx(tmp_path, workers, queue, [], outcomes)) + assert outcomes[-1]["ok"] is True + + +@pytest.mark.parametrize("control", ["panic", "stop"]) +def test_hold_reads_controls_from_canonical_budget_root(tmp_path, monkeypatch, control): + from ouroboros import budget_pause, model_wait + from ouroboros.cancel_intents import request_cancel + + queue, _, _ = _install_queue(tmp_path, monkeypatch) + ctx, _ = _loop_ctx(tmp_path / "execution", "split") + ctx.budget_drive_root = tmp_path + monkeypatch.setattr(model_wait, "current_model_wait", lambda: None) + assert budget_pause._hold_control_reason(ctx) == "" + if control == "panic": + (tmp_path / "state" / "panic_stop.flag").write_text("stop") + else: + request_cancel(tmp_path, "split", reason="owner_stopped", source="test") + assert budget_pause._hold_control_reason(ctx) == ("panic" if control == "panic" else "cancelled") + + +def test_failed_finite_lifetime_read_refuses_delegation_without_reset(tmp_path, monkeypatch): + from ouroboros import config, model_wait + from ouroboros.tools import delegate + + monkeypatch.setattr(config, "get_task_abs_ceiling_sec", lambda: 100.0) + ctx = SimpleNamespace(task_id="bounded", task_started_at=1.0) + waiter = SimpleNamespace(task_id="bounded", execution_window_remaining=lambda: (_ for _ in ()).throw(OSError("unknown"))) + monkeypatch.setattr(model_wait, "current_model_wait", lambda: waiter) + assert delegate.bounded_max_seconds(ctx, 20).refusal_code == "task_lifetime_unknown" + waiter.execution_window_remaining = lambda: 8.5 + result = delegate.bounded_max_seconds(ctx, 20) + assert not result.refusal_code and result.seconds == 8 + + +def test_direct_actor_releases_local_fence_after_parking_same_id(tmp_path, monkeypatch): + from ouroboros import budget_pause + from supervisor import message_bus, worker_chat_lane + from tests.test_budget_pause_exact import _fast_hold, _quiet_external + + queue, state, workers = _install_queue(tmp_path, monkeypatch) + monkeypatch.setattr(state, "budget_remaining", lambda *_a, **_k: 5.0) + ctx, limit = _loop_ctx(tmp_path, "direct", direct=True) + ctx.current_chat_id = 7 + _fast_hold(monkeypatch, budget_pause) + _quiet_external(monkeypatch, budget_pause) + with pytest.raises(budget_pause.BudgetPauseRequested) as raised: + budget_pause.request_pause(limit, rail=budget_pause.RAIL_GLOBAL_EXHAUSTED, + scope="global", reason_text="budget") + task = {"id": "direct", "type": "task", "chat_id": 7, "text": "work", + "project_id": "fixture", "_is_direct_chat": True} + events, released = [], [] + monkeypatch.setattr(workers, "get_event_q", lambda: SimpleNamespace(put=events.append)) + monkeypatch.setattr(message_bus, "get_bridge", lambda: SimpleNamespace(push_log=lambda _evt: None)) + agent = SimpleNamespace(handle_task=lambda _task: [budget_pause.pause_event(task, raised.value.pause)]) + assert budget_pause.dispatch_fenced("direct") + try: + assert worker_chat_lane._execute_chat_task({"task": task, "agent": agent, "chat_id": 7, + "registry": SimpleNamespace(unregister=released.append)}) + assert released == ["direct"] and not budget_pause.dispatch_fenced("direct") + assert workers.PENDING[0]["id"] == "direct" and workers.PENDING[0]["_budget_pause"] + assert not events # inline park consumed the pause; no duplicate supervisor event + assert queue.resume_budget_paused_task("direct")["ok"] + finally: + budget_pause.end_dispatch_fence("direct") diff --git a/tests/test_budget_pause_v664.py b/tests/test_budget_pause_v664.py index 4d3ebdac4..a53dce3cd 100644 --- a/tests/test_budget_pause_v664.py +++ b/tests/test_budget_pause_v664.py @@ -224,7 +224,7 @@ def test_root_budget_fence_is_one_durable_marker_without_subtree_reclassificatio assert workers.PENDING[0]["id"] == pending["id"] -def test_root_budget_resume_checks_one_task_and_clears_marker(tmp_path, monkeypatch): +def test_root_budget_resume_selects_one_task_and_keeps_tree_marker(tmp_path, monkeypatch): queue, _state, workers = _install_queue(tmp_path, monkeypatch) fence_id = "root-fence-id" queue.BUDGET_ROOT_FENCES["safe-root"] = { @@ -243,12 +243,12 @@ def test_root_budget_resume_checks_one_task_and_clears_marker(tmp_path, monkeypa result = queue.resume_budget_paused_task("safe-child") - assert result == {"ok": True, "task_id": "safe-child", "same_generation": True} - assert "safe-root" not in queue.BUDGET_ROOT_FENCES - assert workers.PENDING[0]["budget_resumed_at"] + assert result["ok"] and result["task_id"] == "safe-child" and result["same_generation"] + assert "safe-root" in queue.BUDGET_ROOT_FENCES + assert workers.PENDING[0]["_budget_pause_hold"]["selected"] -def test_root_budget_resume_refuses_unsafe_pending_sibling(tmp_path, monkeypatch): +def test_root_budget_selection_leaves_unsafe_pending_sibling_held(tmp_path, monkeypatch): queue, _state, workers = _install_queue(tmp_path, monkeypatch) fence_id = "root-fence-id" queue.BUDGET_ROOT_FENCES["mixed-root"] = { @@ -269,11 +269,7 @@ def test_root_budget_resume_refuses_unsafe_pending_sibling(tmp_path, monkeypatch result = queue.resume_budget_paused_task("safe-child") - assert result == { - "ok": False, - "error": "root_replay_unsafe", - "unsafe_task_ids": ["retry-child"], - "action": "cancel_or_new_run", - } + assert result["ok"] and result["task_id"] == "safe-child" assert "mixed-root" in queue.BUDGET_ROOT_FENCES - assert "budget_resumed_at" not in safe + assert safe["_budget_pause_hold"]["selected"] + assert queue.resume_budget_paused_task("retry-child")["error"] == "replay_unsafe" diff --git a/tests/test_delegate_continuation.py b/tests/test_delegate_continuation.py new file mode 100644 index 000000000..3223e261e --- /dev/null +++ b/tests/test_delegate_continuation.py @@ -0,0 +1,396 @@ +"""Finite delegated-leaf continuation after a CONFIRMED wall-clock expiry (#1196). + +Static authoring note: these tests were WRITTEN against the candidate but NOT +RUN by their author (no runtime imports were permitted in that lane); the +parent's isolated harness is the first execution. + +``delegate_start(continue_from=)`` is admitted only over this task's +own SETTLED run that the engine cancelled with reason ``wall_clock_exceeded``, +after its result was read and its patch explicitly disposed, on the same +executor and workspace authority. It is not recovery: every other ending +refuses typed, nothing is replayed, and no session state is transferred. +""" + +from __future__ import annotations + +import json +from types import SimpleNamespace + +from ouroboros import delegate_continuation as continuation, delegate_custody as custody +from ouroboros.delegate_registration_policy import ( + CAP_BASIS_DEADLINE_DERIVED, + CAP_BASIS_LIFETIME_DERIVED, + CAP_BASIS_OPERATION_WINDOW, + CAP_BASIS_REQUESTED, + CAP_BASIS_REQUESTED_CLAMPED_DEADLINE, + CAP_BASIS_REQUESTED_CLAMPED_LIFETIME, + CAP_BASIS_REQUESTED_CLAMPED_SCHEMA, + FINITE_LEAF_CAP_BASES, +) +from tests._delegated_transport_shared import ( # noqa: F401 -- autouse transport fixture + _nanny_ctx, + _owned_gateway_uses_each_test_transport, + _started_request, +) + +ROUTE = "some-route" +CAUSE = continuation.CONTINUATION_CAUSE +# The binding facts an admissible predecessor RECORDS. Each is a separate seed +# argument so a test can withhold exactly one and name the refusal it earns. +ACTOR = "actor-1" +CFG = "cfg-fingerprint-1" +AUTH = "authority-fingerprint-1" +WORK_ORDER = "work-order-fingerprint-1" + + +def _seed(tmp_path, run_id, *, task_id="t-a", state="cancelled", reason=CAUSE, settled=True, + actor=ACTOR, route=ROUTE, access="readonly", mode="ask", isolation="", snapshot_id="", + target_root="", max_seconds=90, continuation_of="", cap_basis=CAP_BASIS_REQUESTED, + config_fingerprint=CFG, authority_fingerprint=AUTH, work_order_fingerprint=WORK_ORDER, + output="consumed"): + """Durable rows for one prior run; the memo is cleared so lookups REPLAY them. + + The defaults describe an ADMISSIBLE predecessor: a finite leaf cap the nanny + asked for, recorded executor/configuration/task-authority/work-order bindings, + and a terminal detail staged in full and read to EOF. ``output`` selects how + much of the result story exists: ``none`` (nothing retained), ``staged`` (full + content staged, never read) or ``consumed`` (staged and acknowledged). + """ + entry = custody.RunCustody( + run_id=run_id, task_id=task_id, route_id=route, model="m", selected_subagent_id=actor, + snapshot_id=snapshot_id, target_root=target_root, root_task_id=task_id, + continuation_of=continuation_of, config_fingerprint=config_fingerprint, + authority_fingerprint=authority_fingerprint, work_order_fingerprint=work_order_fingerprint, + ) + assert custody.record_started(tmp_path, entry, shape={ + "access": access, "mode": mode, "isolation": isolation, "delegated": bool(isolation), + "root": "/r", "max_seconds": max_seconds, "max_seconds_basis": cap_basis}) + if settled: + row = {"run_id": run_id, "task_id": task_id, "route": route, "state": state} + if reason is not None: + row["outcome_reason"] = reason + assert custody.emit(tmp_path, custody.SETTLED, row) + if output in ("staged", "consumed"): + assert custody.emit(tmp_path, custody.OUTPUT_SPILLED, { + "run_id": run_id, "task_id": task_id, "artifact": f"delegated_runs/{run_id}.json", + "sha256": f"sha-{run_id}", "bytes": 3, "staged": True, "full_content": True}) + if output == "consumed": + assert custody.emit(tmp_path, custody.OUTPUT_CONSUMED, { + "run_id": run_id, "task_id": task_id, "sha256": f"sha-{run_id}"}) + custody._CUSTODY.clear() + return entry + + +def _gate(tmp_path, run_id, *, task_id="t-a", actor=ACTOR, route=ROUTE, access="readonly", mode="ask", + isolation="", target_root="", config_fingerprint=CFG, authority_fingerprint=AUTH, + canonical=""): + ctx = SimpleNamespace(task_id=task_id) + return continuation.bind_continuation( + ctx, tmp_path, run_id, actor={"selected_subagent_id": actor, + "config_fingerprint": config_fingerprint, + "authority_fingerprint": authority_fingerprint}, + route=SimpleNamespace(route_id=route), + authority=SimpleNamespace(access=access, mode=mode, isolation=isolation), + target_root=target_root, canonical_work_order_fingerprint=canonical) + + +# --------------------------------------------------------------------------- custody facts + +def test_settlement_rows_replay_the_typed_cause_and_the_continuation_lineage(tmp_path): + _seed(tmp_path, "run-a", continuation_of="run-0") + replayed = custody.replay(tmp_path)["run-a"] + assert replayed.settled and replayed.terminal_state == "cancelled" + assert replayed.terminal_reason == CAUSE and replayed.continuation_of == "run-0" + # A settlement that predates the field replays as an UNRECORDED cause, never as one. + _seed(tmp_path, "run-legacy", reason=None) + assert custody.replay(tmp_path)["run-legacy"].terminal_reason == "" + + +def test_settle_run_records_the_engines_typed_outcome_reason(tmp_path, monkeypatch): + import ouroboros.usage_accounting as accounting + + monkeypatch.setattr(accounting, "record_subscription_session", lambda *_a, **_k: None) + custody._CUSTODY.pop("run-s", None) + row = custody.RunCustody(run_id="run-s", task_id="t-a", route_id=ROUTE, model="m") + assert custody.record_started(tmp_path, row) + assert custody.settle_run(tmp_path, None, row, {"summary": { + "state": "cancelled", "spendUsd": 0, "spendEstimated": False, + "outcomeFacts": {"lifecycle": "cancelled", "reason": CAUSE}, + }})["settled"] + settled = [r for r in custody.custody_rows(tmp_path) if r.get("type") == custody.SETTLED and r.get("run_id") == "run-s"] + assert settled[-1]["outcome_reason"] == CAUSE and settled[-1]["state"] == "cancelled" + custody._CUSTODY.clear() + assert custody.replay(tmp_path)["run-s"].terminal_reason == CAUSE + # A succeeded run keeps its byte-identical settlement row (no failure facts at all). + ok = custody.RunCustody(run_id="run-ok", task_id="t-a", route_id=ROUTE, model="m") + assert custody.record_started(tmp_path, ok) + assert custody.settle_run(tmp_path, None, ok, {"summary": {"state": "succeeded", "spendUsd": 0, + "spendEstimated": False}})["settled"] + row_ok = [r for r in custody.custody_rows(tmp_path) if r.get("type") == custody.SETTLED and r.get("run_id") == "run-ok"][-1] + assert "outcome_reason" not in row_ok and "failure_code" not in row_ok + + +# --------------------------------------------------------------------------- the gate + +def test_gate_admits_only_this_tasks_own_settled_wall_clock_cancelled_run(tmp_path): + assert _gate(tmp_path, "run-none")[1] == continuation.REFUSAL_SOURCE_UNKNOWN + _seed(tmp_path, "run-theirs", task_id="t-other") + assert _gate(tmp_path, "run-theirs")[1] == continuation.REFUSAL_SOURCE_NOT_OWNED + _seed(tmp_path, "run-live", settled=False) + facts, code, detail = _gate(tmp_path, "run-live") + assert code == continuation.REFUSAL_SOURCE_NOT_TERMINAL and "second writer" in detail + _seed(tmp_path, "run-legacy", reason=None) + assert _gate(tmp_path, "run-legacy")[1] == continuation.REFUSAL_CAUSE_UNRECORDED + for state, reason in (("cancelled", "user_cancelled"), ("cancelled", "host_cancelled"), + ("cancelled", "owner_task_gone"), ("failed", "harness_failed"), + ("interrupted", "crash_interrupted")): + _seed(tmp_path, f"run-{reason}", state=state, reason=reason) + facts, code, detail = _gate(tmp_path, f"run-{reason}") + assert code == continuation.REFUSAL_CAUSE_NOT_WALL_CLOCK and reason in detail + _seed(tmp_path, "run-ok") + facts, code, _detail = _gate(tmp_path, "run-ok") + assert code == "" and facts["continuation_of"] == "run-ok" and facts["cause"] == CAUSE + assert facts["prior_max_seconds"] == 90 and facts["state_transfer"] == "none" + assert facts["prior_patch_disposition"] == "not_applicable" + + +def test_gate_requires_a_read_result_and_an_explicit_disposition(tmp_path): + # Staged full output never read to EOF: continuing it is a blind resend. + _seed(tmp_path, "run-unread", output="none") + assert custody.emit(tmp_path, custody.OUTPUT_SPILLED, { + "run_id": "run-unread", "task_id": "t-a", "artifact": "delegated_runs/run-unread.json", + "sha256": "s", "bytes": 3, "staged": True, "full_content": True}) + custody._CUSTODY.clear() + assert _gate(tmp_path, "run-unread")[1] == continuation.REFUSAL_RESULT_UNREAD + assert custody.emit(tmp_path, custody.OUTPUT_CONSUMED, {"run_id": "run-unread", "task_id": "t-a", "sha256": "s"}) + custody._CUSTODY.clear() + assert _gate(tmp_path, "run-unread")[1] == "" + # A snapshot run's captured patch must be explicitly applied or rejected first. + _seed(tmp_path, "run-snap", snapshot_id="snap-1", access="workspace_write", mode="agent", isolation="live", + target_root="/target") + facts, code, _d = _gate(tmp_path, "run-snap", access="workspace_write", mode="agent", isolation="live", + target_root="/target") + assert code == continuation.REFUSAL_PATCH_UNDISPOSED + assert custody.emit(tmp_path, custody.PATCH_APPLY_STARTED, {"run_id": "run-snap", "task_id": "t-a", + "snapshot_id": "snap-1", "apply_idempotency_key": "k"}) + custody._CUSTODY.clear() + assert _gate(tmp_path, "run-snap", access="workspace_write", mode="agent", isolation="live", + target_root="/target")[1] == continuation.REFUSAL_APPLY_AMBIGUOUS + assert custody.emit(tmp_path, custody.PATCH_DISPOSED, {"run_id": "run-snap", "task_id": "t-a", + "snapshot_id": "snap-1", "disposition": "rejected"}) + custody._CUSTODY.clear() + facts, code, _d = _gate(tmp_path, "run-snap", access="workspace_write", mode="agent", isolation="live", + target_root="/target") + assert code == "" and facts["prior_patch_disposition"] == "rejected" and facts["prior_target_root"] == "/target" + assert "REJECTED" in continuation.continuation_instruction(facts) + + +def test_gate_keeps_the_same_executor_and_workspace_authority(tmp_path): + _seed(tmp_path, "run-actor", actor="actor-a") + assert _gate(tmp_path, "run-actor", actor="actor-b")[1] == continuation.REFUSAL_EXECUTOR_MISMATCH + assert _gate(tmp_path, "run-actor", actor="actor-a")[1] == "" + # An UNRECORDED side never "does not contradict" — it refuses, on either side. + assert _gate(tmp_path, "run-actor", actor="")[1] == continuation.REFUSAL_EXECUTOR_MISMATCH + _seed(tmp_path, "run-noactor", actor="") + assert _gate(tmp_path, "run-noactor", actor="actor-a")[1] == continuation.REFUSAL_EXECUTOR_MISMATCH + _seed(tmp_path, "run-route", route="other-route") + assert _gate(tmp_path, "run-route")[1] == continuation.REFUSAL_EXECUTOR_MISMATCH + _seed(tmp_path, "run-shape", access="workspace_write", mode="agent", isolation="live", target_root="/t") + assert _gate(tmp_path, "run-shape")[1] == continuation.REFUSAL_AUTHORITY_MISMATCH + assert _gate(tmp_path, "run-shape", access="workspace_write", mode="agent", isolation="live", + target_root="/elsewhere")[1] == continuation.REFUSAL_TARGET_MISMATCH + # A mutating run WITHOUT a snapshot (nothing to dispose) passes on the same target. + facts, code, _d = _gate(tmp_path, "run-shape", access="workspace_write", mode="agent", isolation="live", + target_root="/t") + assert code == "" and facts["prior_target_root"] == "/t" and facts["prior_patch_disposition"] == "not_applicable" + + +def test_gate_admits_only_a_finite_leaf_cap_the_nanny_asked_for(tmp_path): + """A cap this task's own deadline or lifetime derived (or narrowed) expiring IS that + bound, so it is not a leaf expiry a continuation may follow; a row predating the + field is UNKNOWN, never assumed. Only what the nanny asked for is admitted.""" + # A legacy row with no recorded basis is unknown, not a leaf cap. + _seed(tmp_path, "run-nobasis", cap_basis="") + facts, code, detail = _gate(tmp_path, "run-nobasis") + assert code == continuation.REFUSAL_CAP_BASIS_UNKNOWN and "no basis" in detail + for basis in (CAP_BASIS_DEADLINE_DERIVED, CAP_BASIS_LIFETIME_DERIVED, + CAP_BASIS_OPERATION_WINDOW, CAP_BASIS_REQUESTED_CLAMPED_DEADLINE, + CAP_BASIS_REQUESTED_CLAMPED_LIFETIME): + assert basis not in FINITE_LEAF_CAP_BASES + _seed(tmp_path, f"run-{basis}", cap_basis=basis) + facts, code, detail = _gate(tmp_path, f"run-{basis}") + assert code == continuation.REFUSAL_CAP_NOT_FINITE_LEAF and basis in detail + # An absent number is not a finite bound either, whatever the basis claims. + _seed(tmp_path, "run-zerocap", max_seconds=0) + assert _gate(tmp_path, "run-zerocap")[1] == continuation.REFUSAL_CAP_NOT_FINITE_LEAF + # The two bases the nanny genuinely asked for are admitted, and disclosed as facts. + for basis in (CAP_BASIS_REQUESTED, CAP_BASIS_REQUESTED_CLAMPED_SCHEMA): + _seed(tmp_path, f"run-ok-{basis}", cap_basis=basis) + facts, code, _detail = _gate(tmp_path, f"run-ok-{basis}") + assert code == "" and facts["prior_cap_basis"] == basis and facts["prior_max_seconds"] == 90 + + +def test_gate_requires_each_recorded_binding_fact_positively(tmp_path): + """Configuration, task authority and the bound work order are each checked + POSITIVELY: recorded on the prior run, present on this start, and equal. An + unrecorded side on EITHER half refuses; it never passes for lack of a + contradiction, and the authority cannot be re-derived.""" + _seed(tmp_path, "run-cfg") + assert _gate(tmp_path, "run-cfg", config_fingerprint="other")[1] == continuation.REFUSAL_CONFIG_MISMATCH + assert _gate(tmp_path, "run-cfg", config_fingerprint="")[1] == continuation.REFUSAL_CONFIG_MISMATCH + _seed(tmp_path, "run-nocfg", config_fingerprint="") + assert _gate(tmp_path, "run-nocfg")[1] == continuation.REFUSAL_CONFIG_MISMATCH + + _seed(tmp_path, "run-auth") + assert _gate(tmp_path, "run-auth", + authority_fingerprint="other")[1] == continuation.REFUSAL_TASK_AUTHORITY_MISMATCH + assert _gate(tmp_path, "run-auth", + authority_fingerprint="")[1] == continuation.REFUSAL_TASK_AUTHORITY_MISMATCH + _seed(tmp_path, "run-noauth", authority_fingerprint="") + assert _gate(tmp_path, "run-noauth")[1] == continuation.REFUSAL_TASK_AUTHORITY_MISMATCH + + # A run whose STARTED row binds no work order is not followed at all. + _seed(tmp_path, "run-nowo", work_order_fingerprint="") + assert _gate(tmp_path, "run-nowo")[1] == continuation.REFUSAL_WORK_ORDER_UNBOUND + # A configured session's canonical brief must be the one the prior run was bound to; + # an absent canonical brief is not a mismatch (an ordinary start carries none). + _seed(tmp_path, "run-wo") + assert _gate(tmp_path, "run-wo", + canonical="a-different-brief")[1] == continuation.REFUSAL_WORK_ORDER_MISMATCH + facts, code, _detail = _gate(tmp_path, "run-wo", canonical=WORK_ORDER) + assert code == "" and facts["canonical_work_order_fingerprint"] == WORK_ORDER + assert facts["prior_work_order_fingerprint"] == WORK_ORDER + assert facts["prior_config_fingerprint"] == CFG and facts["prior_authority_fingerprint"] == AUTH + + +def test_gate_refuses_a_result_that_was_never_retained_or_only_partly_staged(tmp_path): + """Continuing work nobody retained or read is a blind resend: a run with no staged + detail, and one whose staging does not POSITIVELY claim full content, each refuse.""" + _seed(tmp_path, "run-noout", output="none") + facts, code, detail = _gate(tmp_path, "run-noout") + assert code == continuation.REFUSAL_RESULT_UNRETAINED and "nothing of its result" in detail.lower() + _seed(tmp_path, "run-partial", output="none") + assert custody.emit(tmp_path, custody.OUTPUT_SPILLED, { + "run_id": "run-partial", "task_id": "t-a", "artifact": "delegated_runs/run-partial.json", + "sha256": "p", "bytes": 3, "staged": True, "full_content": False}) + custody._CUSTODY.clear() + assert _gate(tmp_path, "run-partial")[1] == continuation.REFUSAL_RESULT_INCOMPLETE + # A staged-in-full result read to EOF is the admissible shape. + _seed(tmp_path, "run-read", output="consumed") + facts, code, _detail = _gate(tmp_path, "run-read") + assert code == "" and facts["prior_output"]["staged_output_consumed"] is True + + +def test_requested_cap_is_clamped_by_the_deadline_and_the_remaining_lifetime(tmp_path, monkeypatch): + """The producer side of the cap basis: ``bounded_max_seconds`` is narrow-only and + RECORDS which bound decided the number, so a deadline/lifetime narrowing can never + later be read as a finite leaf cap the nanny asked for.""" + import time + from datetime import datetime, timedelta, timezone + + from ouroboros import config + from ouroboros.tools.delegate import bounded_max_seconds + + def _ctx(*, deadline_in=None, started_ago=100.0): + meta = {} + if deadline_in is not None: + meta["deadline_at"] = (datetime.now(timezone.utc) + + timedelta(seconds=deadline_in)).isoformat() + return SimpleNamespace(task_id="t-a", task_metadata=meta, + task_started_at=time.time() - started_ago, + _budget_paused_sec=0.0, budget_pause_resume=None) + + # No deadline and no finite lifetime: an explicit ask is exactly what was asked. + monkeypatch.setattr(config, "get_task_abs_ceiling_sec", lambda: None) + plain = bounded_max_seconds(_ctx(), 120) + assert (plain.seconds, plain.basis) == (120, CAP_BASIS_REQUESTED) + # Omitting the ask derives from the operation window — never a leaf cap. + assert bounded_max_seconds(_ctx(), None).basis == CAP_BASIS_OPERATION_WINDOW + # A finite lifetime with 200s left narrows a larger ask and NAMES the lifetime. + monkeypatch.setattr(config, "get_task_abs_ceiling_sec", lambda: 300.0) + clamped = bounded_max_seconds(_ctx(), 1000) + assert clamped.basis == CAP_BASIS_REQUESTED_CLAMPED_LIFETIME + assert 150 <= clamped.seconds <= 200 + # An ask that already fits inside the remaining lifetime is untouched. + assert bounded_max_seconds(_ctx(), 30).basis == CAP_BASIS_REQUESTED + # A nearer deadline decides instead, and is named instead. + near = bounded_max_seconds(_ctx(deadline_in=50), 1000) + assert near.basis == CAP_BASIS_REQUESTED_CLAMPED_DEADLINE and 40 <= near.seconds <= 50 + # Omitting the ask under a finite lifetime derives from it — still not a leaf cap. + assert bounded_max_seconds(_ctx(), None).basis == CAP_BASIS_LIFETIME_DERIVED + # A spent lifetime is a typed definite no-run, not a zero-second cap. + monkeypatch.setattr(config, "get_task_abs_ceiling_sec", lambda: 10.0) + spent = bounded_max_seconds(_ctx(started_ago=100.0), 60) + assert spent.refusal_code == "task_lifetime_exhausted" and spent.seconds == 0 + # Every basis this producer can record is either a leaf cap or explicitly not one. + assert FINITE_LEAF_CAP_BASES == frozenset({CAP_BASIS_REQUESTED, CAP_BASIS_REQUESTED_CLAMPED_SCHEMA}) + + +def test_instruction_block_names_the_predecessor_and_transfers_no_state(): + text = continuation.continuation_instruction({ + "continuation_of": "run-x", "cause": CAUSE, "prior_max_seconds": 120, + "prior_patch_disposition": "applied"}) + assert "CONTINUATION OF RUN run-x" in text and "120s wall-clock cap" in text and CAUSE in text + assert "APPLIED" in text and "NOTHING of its session state is transferred" in text + assert "never re-apply" in text + + +# --------------------------------------------------------------------------- delegate_start wiring + +def test_continue_from_conflicts_are_refused_before_the_daemon(tmp_path): + from ouroboros.delegate_shared import delegate_payload + from ouroboros.tools import delegate + + ctx = _nanny_ctx(tmp_path) + clash = delegate_payload(delegate._delegate_start(ctx, "finish it", continue_from="run-x", retry_of="tok")) + assert clash["status"] == "refused" and clash["reason"] == "continuation_selector_conflict" + assert clash["definitely_unrun"] is True + payload_run = delegate_payload(delegate._delegate_start( + ctx, "finish it", continue_from="run-x", root="skill_payload", bucket="external", skill_name="s")) + assert payload_run["reason"] == "continuation_resource_conflict" + # An empty prompt is still the first refusal, ahead of every continuation check. + assert delegate_payload(delegate._delegate_start(ctx, " ", continue_from="run-x"))["reason"] == "empty_prompt" + + +def test_continue_from_binds_the_started_run_to_its_settled_predecessor(tmp_path, monkeypatch): + """End to end through the stubbed transport: the gate passes for this nanny's own + wall-clock-cancelled run on the same route/actor/authority, the host block rides + the instructions, the STARTED row carries the lineage and the payload states it.""" + from ouroboros.subagent_work_order import start_binding_fingerprints + from tests._delegated_transport_shared import _delegating_ctx + + # The predecessor's bindings are the ones this start genuinely DERIVES (the + # transport actor's configuration plus the production task-authority and + # work-order fingerprints), so the gate is exercised against real equality + # rather than against a hand-invented matching pair. + work_order, authority = start_binding_fingerprints( + _delegating_ctx(tmp_path, acting=False, task_id="t-nanny-read"), "edit the README") + _seed(tmp_path, "run-prev", task_id="t-nanny-read", actor="transport-fixture", access="readonly", + mode="ask", isolation="", config_fingerprint="transport-fixture-v1", + authority_fingerprint=authority, work_order_fingerprint=work_order) + request, payload = _started_request(tmp_path, acting=False, monkeypatch=monkeypatch, + start_kwargs={"continue_from": "run-prev"}) + assert "CONTINUATION OF RUN run-prev" in request["instructions"] + assert payload["continuation"]["continuation_of"] == "run-prev" + assert payload["continuation"]["cause"] == CAUSE and payload["continuation"]["state_transfer"] == "none" + custody._CUSTODY.clear() + assert custody.replay(tmp_path)["run-read"].continuation_of == "run-prev" + + +def test_continue_from_over_an_unknown_run_is_a_typed_definite_no_run(tmp_path, monkeypatch): + request, payload = _started_request(tmp_path, acting=False, monkeypatch=monkeypatch, expect="refused", + start_kwargs={"continue_from": "run-missing"}) + assert request is None + assert payload["reason"] == continuation.REFUSAL_SOURCE_UNKNOWN and payload["definitely_unrun"] is True + assert payload["continue_from"] == "run-missing" + + +def test_no_resume_causes_are_untouched_by_the_continuation_seam(): + """The continuation is not crash recovery: the recovery veto list keeps every cause.""" + from ouroboros.delegate_recovery import NO_RESUME_CAUSES + + assert NO_RESUME_CAUSES == ( + "owner_restart", "panic", "external_signal", "worker_signal", + "deadline", "timeout", "explicit_cancellation", "abrupt_whole_app_loss", + ) + assert json.dumps(sorted(NO_RESUME_CAUSES)) # a tuple of plain strings, nothing hidden diff --git a/tests/test_delegated_run_profile.py b/tests/test_delegated_run_profile.py index 1014c5cc4..eb4ac4241 100644 --- a/tests/test_delegated_run_profile.py +++ b/tests/test_delegated_run_profile.py @@ -142,11 +142,13 @@ def test_the_model_has_no_argument_that_could_widen_the_profile(): entry = next(e for e in delegate.get_tools() if e.name == "delegate_start") properties = set(entry.schema["parameters"]["properties"]) # `retry_of` names an INVOCATION, not authority (ownership-checked replay); - # root/bucket/skill_name are a SELECTOR resolved through the same - # ResolvedResourceBinding authorizer as ordinary writes (R1 item 9). + # `continue_from` names this task's OWN settled run (custody-checked, same + # executor and authority, #1196); root/bucket/skill_name are a SELECTOR + # resolved through the same ResolvedResourceBinding authorizer as ordinary + # writes (R1 item 9). assert properties == { - "prompt", "subagent_id", "max_seconds", "retry_of", "root", "bucket", "skill_name", - "directory_strategy", "scope_paths", "access", + "prompt", "subagent_id", "max_seconds", "retry_of", "continue_from", "root", "bucket", + "skill_name", "directory_strategy", "scope_paths", "access", } assert entry.schema["parameters"]["properties"]["root"]["enum"] == ["skill_payload"] assert entry.schema["parameters"]["properties"]["access"]["enum"] == ["readonly", "workspace_write"] diff --git a/tests/test_reference_book_budgets.py b/tests/test_reference_book_budgets.py index 62a3b9c92..88a6f8dc4 100644 --- a/tests/test_reference_book_budgets.py +++ b/tests/test_reference_book_budgets.py @@ -26,7 +26,9 @@ CHAPTER_BYTE_BUDGETS: dict[str, int] = { # extension_isolated_deps.py barrier and the widget_list.js request seam land # beside the handoff/schedule rows the base added; none displaces older text. # +400 (#1213): two new module rows (focus.py, room_consolidation.py) in the tree map. - "docs/architecture/01-high-level-architecture.md": 164800, + # 164800 -> 165300 (#1196): two new owner modules gain their map rows + # (supervisor/budget_resume.py, ouroboros/delegate_continuation.py). + "docs/architecture/01-high-level-architecture.md": 165300, # 15517 -> 16200 (#1195): the session-custodied startup historical audit is a # new node of the startup flow (readiness no longer waits for the historical # seal diagnostic); the chapter had no older description of that pass to replace. @@ -59,7 +61,9 @@ CHAPTER_BYTE_BUDGETS: dict[str, int] = { # 106400 -> 106600 (#1195 merge of 32d8dfc6): the base's settings_catalog.js # paragraph (#1214, +319 bytes) landed in the same window; both additions stand, # neither displaces the other's text. - "docs/architecture/03-web-ui-pages-and-buttons.md": 106600, + # 106600 -> 106800 (#1196): a paused direct turn reports the managed census + # phases; the phase sentence is extended, nothing older describes it. + "docs/architecture/03-web-ui-pages-and-buttons.md": 106800, "docs/architecture/04-server-api-endpoints.md": 26833, # 27137 -> 30400: the schedule table gains a documented write contract the # chapter had no text for — one transaction owning the lock ORDER, the strict @@ -75,7 +79,10 @@ CHAPTER_BYTE_BUDGETS: dict[str, int] = { # 30900 -> 31700 (#1196): the exact-continuation `_budget_pause` marker, its grant # carrier and the separate paused-interval carrier are new snapshot/assignment # facts the chapter lacked; nothing older describes them. - "docs/architecture/05-supervisor-loop.md": 31700, + # 31700 -> 32200 (#1196): restart parking of a completed pause, the typed + # restore/acceptance holds and the parked direct turn replace the restore + # sentence they grew from. + "docs/architecture/05-supervisor-loop.md": 32200, # 286850 -> 287600: "an answer that has not arrived is a gap" is a new invariant of # plan review and task acceptance (the slot census vocabulary, the `awaiting` # projection, the only-awaited task outcome); the in-flight sentence it grew from is @@ -102,7 +109,14 @@ CHAPTER_BYTE_BUDGETS: dict[str, int] = { # 297250 -> 301400 (#1196): the exact budget pause is a new mechanism of the budget # section (fence order, drain, external stop requests, program counter, park, grant, # revoke, Q9/Q10 rules); the wrap-up rails it sits beside keep their own text. - "docs/architecture/06-agent-core.md": 301400, + # 301400 -> 304500 (#1196, continued): the base sat 680 bytes OVER the previous + # budget (the slice-1 pause paragraph landed without raising it). The pause + # paragraph is REPLACED and grows ~1130 bytes for mechanisms it lacked (the + # direct-turn pause, the durable holds, the generation-bound grant, the + # restart park, the Q10 last-fit relaxation); the wall-clock continuation of a + # delegated leaf (+~1000) is a new seam of the delegation section with no + # older text to displace; the delegate_start argument sentence grows by one clause. + "docs/architecture/06-agent-core.md": 304500, "docs/architecture/07-configuration.md": 36991, # 18947 -> 19287: CI failure collection now documents diagnostic desktop builds while release remains gated. "docs/architecture/08-git-branching-ci-and-build.md": 19287, diff --git a/tests/test_smoke.py b/tests/test_smoke.py index 40911fd96..1d64e8ede 100644 --- a/tests/test_smoke.py +++ b/tests/test_smoke.py @@ -107,7 +107,7 @@ EXPECTED_TOOLS = [ "integrate_subagent_patch", "compare_subagent_patches", # C1: the explicit acceptance seam for a delegated run's captured patch — # a first-class tool, so the registry contract must name it. - "integrate_delegated_patch", "cancel_task", + "integrate_delegated_patch", "cancel_task", "resume_child_task", "peek_task", "discard_child_result", "override_delegation_constraint", "request_deep_self_review", "chat_history", "update_scratchpad", "send_user_message", "update_identity", "toggle_evolution", diff --git a/web/modules/activity.js b/web/modules/activity.js index bfc8d6021..00117b11f 100644 --- a/web/modules/activity.js +++ b/web/modules/activity.js @@ -66,8 +66,13 @@ export function initActivity({ mount, ws } = {}) { ((queue && queue.budget_root_fences) || []) .filter((f) => f && ['active', 'paused'].includes(String(f.status || ''))) .map((f) => String(f.root_task_id || ''))); + // #1196: a row whose root fence was lifted keeps a durable HOLD instead — + // nothing dispatches it until an explicit selection is recorded, so + // showing it as plain "queued" would promise work that cannot start. + const heldRow = (t) => Boolean(t && t._budget_pause_hold && !t._budget_pause_hold.selected); const rowBudgetPaused = (q, t, kind) => kind === 'pending' && Boolean( (t && t._budget_pause) + || heldRow(t) || fencedRoots.has(String((t && (t.root_task_id || t.id)) || q.id || ''))); const row = (q, kind) => { const t = (q && q.task) || {}; diff --git a/web/modules/api_types.js b/web/modules/api_types.js index e31cd97cf..30f057e46 100644 --- a/web/modules/api_types.js +++ b/web/modules/api_types.js @@ -83,7 +83,7 @@ * @property {string} project_id * @property {string} client_message_id // empty for managed queue rows * @property {string} kind // direct_chat | managed_task — presentational label; membership in this census, not kind, decides liveness - * @property {string} phase // managed rows: queued | budget_paused | budget_pausing (RUNNING, writing its exact pause record; additive, #1196) | working | finalizing; direct rows: thinking, or unknown when the live wait owner could not be read + * @property {string} phase // managed rows: queued | budget_paused | budget_pausing (RUNNING, writing its exact pause record; additive, #1196) | working | finalizing; direct rows: thinking, or unknown when the live wait owner could not be read; a direct turn paused on its budget rail is parked in the queue under the same id and reports the managed phases (budget_paused, then working/finalizing after an explicit Resume) * @property {number} started_at */