Merge remote-tracking branch 'managed/ouroboros' into claude/steer-20260926

# Conflicts:
#	docs/inventories/DATA_LAYOUT_INVENTORY.md
#	ouroboros/task_status.py
#	tests/test_reference_book_budgets.py
This commit is contained in:
Ouroboros 2026-09-26 15:29:27 +03:00
commit 7009dd5fd1
89 changed files with 6752 additions and 1081 deletions

View file

@ -18,17 +18,17 @@ The manifest is the SSOT of the module→domain assignment (1:1, complete over t
| D08 | Supervisor: queue, workers, events & runtime control | 48 | 0 |
| D09 | Cancellation, owner control & process custody | 13 | 0 |
| D10 | Git, update & release machinery | 28 | 0 |
| D11 | Gateway, server & Web UI | 56 | 0 |
| D11 | Gateway, server & Web UI | 57 | 0 |
| D12 | Settings & configuration | 15 | 0 |
| D13 | Safety, guards & runtime mode | 9 | 0 |
| D14 | Skills & extensions | 56 | 0 |
| D15 | Memory, knowledge, consciousness & self-evolution | 23 | 0 |
| D16 | Observability, usage accounting & cost | 11 | 0 |
| D17 | Projects, workspaces & task results | 23 | 0 |
| D17 | Projects, workspaces & task results | 24 | 0 |
| D18 | Launcher, packaging, platform & shared substrate | 15 | 0 |
| D19 | Frozen contracts (ABI) | 10 | 0 |
| D20 | Presence | 10 | 0 |
| **total** | | **575** | **0** |
| **total** | | **577** | **0** |
## Dependency direction matrix (strict, pinned)
@ -596,6 +596,7 @@ No function body (≥ 10 normalized lines) is shared verbatim across domains. Ne
- `ouroboros/gateway/settings.py`
- `ouroboros/gateway/skill_publish.py`
- `ouroboros/gateway/state.py`
- `ouroboros/gateway/task_archive.py`
- `ouroboros/gateway/task_decision.py`
- `ouroboros/gateway/task_events.py`
- `ouroboros/gateway/task_hurry.py`
@ -762,6 +763,7 @@ No function body (≥ 10 normalized lines) is shared verbatim across domains. Ne
- `ouroboros/projects_registry.py`
- `ouroboros/retention.py`
- `ouroboros/routing_wait.py`
- `ouroboros/task_custody.py`
- `ouroboros/task_result_schema.py`
- `ouroboros/task_results.py`
- `ouroboros/task_status.py`

View file

@ -134,7 +134,7 @@ scanned data-relative path to be covered by a row here (count-anchored both ways
| `logs/chat_annotations.jsonl` | `ouroboros/project_dialogue.py` (`append_jsonl` under the shared sidecar append lock) | `type: chat_annotation` keyed by `(client_message_id, routing_token)`; latest row per key wins (an owner message keeps one row per routing act; task-authored acts ride synthetic `agent-steer:<token>` ids) | self-compacting at 800 KB under the append lock: the latest row of every message still present in the chat chain (live `chat.jsonl` + the 3 newest `archive/chat_*.jsonl`) is rewritten, the rest is DROPPED — presentation state, so it is not rotated into `archive/` | annotation cards fall back to their plain chat rows; the one named exception (#198) also loses the durable picker decision-card token, so a pending manual routing choice must be made again |
| `logs/tasks/task_<id>.txt` | `ouroboros/utils.py` log sanitization (`write_text`, best-effort) | raw oversized task text, no envelope; the log row keeps `text_full_path` | unbounded: no retention sweep names `logs/tasks/` | the truncated text in the log row stays; only the spilled full text of oversized task prompts is lost |
| `logs/agent_stdout.log` | `launcher.py` pipe-copy thread | unstructured text | bounded ~8 MB (2 MB × `.1..3` backups, rotated by the copy thread) | pre-logging crash output lost; nothing parses it |
| `logs/server.log` (+`.1..3`), `logs/launcher.log` | stdlib `RotatingFileHandler` (`server.py`, `launcher.py`) with secret-redacting filter | text | bounded ~8 MB (2 MB × 4) — the model citizen | stdlib log history lost; nothing parses it |
| `logs/server.log` (+`.1..3`), `logs/launcher.log` | stdlib `RotatingFileHandler` (`server.py`, `launcher.py`) with secret-redacting filter; a spawn/forkserver worker importing `server.py` as `__mp_main__` gets a stream handler only (one rotator per file; `OUROBOROS_WORKER_START_METHOD=fork` inherits the parent's handlers unchanged) | text | bounded ~8 MB (2 MB × 4) — the model citizen | stdlib log history lost; nothing parses it |
## 6. `memory/` (Ouroboros cognition — operator read-only)
@ -148,7 +148,7 @@ scanned data-relative path to be covered by a row here (count-anchored both ways
| `memory/dialogue_summary.md` | none — legacy read-only (reader in context.py) | none | frozen | legacy artifact; nothing writes it |
| `memory/knowledge/**` (topic .md + `index-full.md` + `patterns.md`) | `ouroboros/tools/knowledge.py`, `consolidator.py` (index rebuild), `reflection.py` (patterns CAS rewrite) | none | topic files unbounded — accepted (curated by consolidation); backlog topic merge-only fail-closed | recreated lazily; knowledge lost |
| `memory/*_journal.jsonl`, `memory/knowledge_history.jsonl`, `memory/knowledge/patterns_history.jsonl` | `ouroboros/memory.py`, `tools/control_runtime.py`, `tools/knowledge.py`, `reflection.py` — every append through the `append_jsonl` sidecar-lock seam | scratchpad journal: `type` rows; others unversioned full-text snapshots; `knowledge_history.jsonl` `source_capture` rows carry the host stamp `writer`/`route`/`writer_input_ref`/`old_chars`/`new_chars` (rows older than the stamp read `unknown`), an automatic anchored edit also its authored `edits` (with `basis`) and, only when supplied, its `summary`; historical digested rows retain `content_digested: true` | complete new old+new snapshots are retained indefinitely; `memory_journal_compaction.py` is a read-only compatibility entry point, not a source rewriter; existing digest-only rows cannot be restored; the `memory_journal_observation` startup event gives byte sizes (or missing/unreadable) for the three named journals; scratchpad keeps its eviction journal | deleting the journals loses undo/provenance; eviction/rewrite paths fail closed when journal append fails; historically digested content remains irrecoverable |
| `memory/owner_mailbox/<task>.jsonl` + `.acks.jsonl` | `ouroboros/owner_mailbox.py` (append-only; revocation appends, reader resolves) | `kind` discriminator | lifecycle-bounded: unlinked at task terminal; a startup sweep unlinks mailboxes whose task has a SETTLED durable result (no result / non-terminal keeps the mailbox fail-closed) | undelivered owner directives + restart-surviving hurry latch lost; acks lost ⇒ re-delivery |
| `memory/owner_mailbox/<task>.jsonl` + `.acks.jsonl` | `ouroboros/owner_mailbox.py` (append-only; revocation appends, reader resolves) | `kind` discriminator | lifecycle-bounded: unlinked, under the per-task mail lock appends and acknowledgements share, only once the SETTLED canonical row with post-work closed holds every exact unread row AND a verified canonical projection of every input-bearing row's attachments (`unread_mailbox.rows`/`.inputs`, `task_custody.settle_task_mailbox`; the loop thread's seam carries no inputs, the off-loop drive-custody pass and the drive settlement do); a torn read, no result, open post-work, a non-terminal row or an uncarried input keeps it | undelivered owner directives + restart-surviving hurry latch lost; acks lost ⇒ re-delivery |
## 7. Skills payloads, tasks, uploads, projects, services
@ -159,8 +159,9 @@ scanned data-relative path to be covered by a row here (count-anchored both ways
| `state/skills/<name>/go/`, `state/skills/<name>/go-cache/` | Go compiler launched by `tools/skill_exec.py:_run_go_skill`, with GOPATH/GOCACHE bound to skill_state_dir | Go-owned cache formats | reused across script runs; follows existing skill-state cleanup | compiler recreates caches; skill payload and review remain unchanged |
| `task_results/<id>.json` | `ouroboros/task_results.py` (locked merge) | `_schema_version: 1` (ABI-2); unstamped/future/malformed → quarantine, no conversion — with one carve-out: the boot latch migration re-stamps unstamped rows still at `cancel_requested` in place (same status, one typed `task_result_cancel_latch_admitted` event) so a wedged task still reaches its `cancelled` terminal | UNBOUNDED — one file per task forever, no GC; lifecycle authority is retained deliberately | lifecycle authority lost; drive prunes degrade to age-only; strict authority reads break |
| `task_results/quarantine/` | `ouroboros/task_result_schema.py` (same-dir rename); `presence_runner.py` reads the retained id before a transport retry | quarantined bytes unchanged; a Presence retry with quarantined authority refuses regeneration | NEVER GC'd (pinned); recovery is manual owner re-stamp | quarantined evidence lost; a previous Presence attempt could be mistaken for a new event |
| `task_results/<id>.custody.lock`, `task_results/<id>.mail.lock`, `state/custody_staging/<id>-<token>/`, `state/custody_trash/<id>-<token>/` | `ouroboros/task_custody.py` (the custody lock every canonical-store publisher takes — copy-back, ref retry, host artifact finalization, mailbox input carry and drive settlement; the mail lock mailbox appends, acknowledgements, mailbox cleanup and the settlement's final check take; private staging copied and verified before publication, then placed create-only; a settled drive moved under the supervisor's queue interlock plus both locks, then deleted) | none — transient | locks released per phase; staging deleted by its settlement; leftovers older than an hour removed by the off-loop drive-custody pass | none: a lost staging copy or trash entry is re-prepared or was already fully in canonical custody |
| `task_results/artifacts/<id>/**` (+`verification_receipts.jsonl`, `.directory.*.tmp`, `.directory.*.json.tmp`), `task_results/artifact_versions/` | `ouroboros/artifacts.py`, `headless.py`, `outcome_receipt_store.py` | artifact and complete directory manifests `schema_version: 1`; scratch manifest 2 | artifact versions bounded (5 per name); artifacts live with their result; directory capture stages ZIP and manifest beside the result, removing owned temporaries on caught failures | deliverable bytes lost; results keep dangling manifests |
| `task_drives/<id>/**` (+`tmp_scripts/`) | `ouroboros/headless.py`, `tools/tool_context.py`, `tools/shell.py` | child stamps as above | GC-retention prune at startup (terminal + age, default 7 d); the `data/tmp_scripts` fallback's hard-kill orphans are in `sweep_stale_temp_files` scope (startup-only when no script can be live) | scratch lost; canonical artifacts survive |
| `task_drives/<id>/**` (+`tmp_scripts/`) | `ouroboros/headless.py`, `tools/tool_context.py`, `tools/shell.py` | child stamps as above | settled off the loop thread by the drive-custody pass (terminal + age, default 7 d; `task_custody.settle_child_drive`); the `data/tmp_scripts` fallback's hard-kill orphans are swept at startup (when no script can be live), the whole-tree walk for orphaned atomic temps by the first reconcile pass of the generation (`sweep_stale_temp_files`) | scratch lost; canonical artifacts survive |
| `task_trees/<root>/blackboard.jsonl` | `ouroboros/task_tree_ledger.py` | rows unversioned; snapshot digest `schema_version: 1` | GC-retention prune at startup (root terminal + age) | swarm coordination facts lost for live trees |
| `state/subagent_worktrees.json` (registry; checkouts live OUTSIDE data root) | `ouroboros/subagent_worktrees.py` | none — malformed → typed refusal (absent = empty is the designed asymmetry) — accepted; file_baseline retains copied binary input identities | prune_orphans (age + missing checkout; skips delegated_exec) + custody-cross-checked snapshot prune (fail-closed on unreadable custody) | permanent leak of checkouts + pinned refs (nothing else names them) |
| `task_results/artifacts/<task>/source_handles/context_checkpoints` (`focus_source_<reader>-<sha256>.md`) | `ouroboros/tools/project_journal.py` (`_retain_focus_source`, through `artifacts.store_actor_source_bytes`) on the canonical root; read by `task_finalization.focus_source_projection` (digest glob for a historical selector) | native `task_source` handle on `focus.source_handle` (size + sha256, verified on read) | write-once, digest-named; lives with the task's artifacts (no separate prune) | a roster's `retained_source` reads `source_unavailable`; the focus text itself survives on the task result |

View file

@ -212,6 +212,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de
├── context_health.py ← Health invariants for the reading task (`build_health_invariants`, ONCE per task attempt — a task-start snapshot); delegated-run obligations stay globally visible — a preserved-and-invisible result is how work rots on disk — while the instruction is ownership-aware (`delegate_shared.orphan_apply_target_ok`) (§6 Context fitting, retry, and compaction; Delegated subagents)
├── context_runtime_facts.py ← The runtime section's FACT builders: what the host can honestly say it knows about this turn
├── headless.py ← Child-drive isolation, workspace patch artifacts, memory export helpers; typed `sensitive_blocked` exclusions (§6 Headless finalization and workspace patch capture)
├── task_custody.py ← The one child-drive deletion owner (`settle_child_drive`), the per-task custody lock, unread-mail capture and the pure store view (§6 Headless finalization)
├── headless_status.py ← Artifact and task lifecycle vocabulary shared by the headless owners
├── workspace_patch_rules.py ← Pure patch-exclusion rules (env/cache sets, junk regex, lockfiles, credential-shaped names); the I/O checks + `untracked_capture_veto_reason` stay in headless
├── workspace_patch_capture.py ← Workspace patch capture: the patch artifact, its manifest, and its git plumbing
@ -378,6 +379,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de
│ ├── ws.py ← WS manager, extension WS dispatch (a synchronous in-process handler runs in a worker thread like the HTTP dispatcher, so one skill's blocking callback never stalls the ASGI loop), broadcast (§4 WebSocket protocol)
│ ├── state.py ← /api/health + /api/state
│ ├── tasks.py ← Headless task create/list/get/cancel/events; cancel accepts `stop_policy` (empty = immediate; `finalize_then_cancel` → 202 + open intent → supervisor/owner_stop.py; unknown → 400)
│ ├── task_archive.py ← Confined single-file and directory-ZIP reads of a task's own stores; a typed 503 without directory-relative opens
│ ├── task_events.py ← Task-event SSE endpoint: legacy GET ranks plus read-only POST v2 physical-chain cursors (§3 History reads and the SSE v2 transport)
│ ├── task_hurry.py ← POST hurry ingress: exact one-field `{request_id}` body — extra fields refused, because hurry carries no text by design and a smuggled field must not become a side channel; queue-owned admission initializes only an absent pooled lifecycle (`write_task_result(create_only=True)`), direct turns excluded (semantics: owner_hurry.py)
│ ├── task_decision.py ← ONE `POST /api/decisions` ingress with family-parsed ids (`quiz:` here, `routing:` → routing_decision.py, `interaction:` reserved); writes `KIND_QUIZ_ANSWER`, broadcasts `quiz_state` (lifecycle: owner_quiz.py; ABI: §11.1)
@ -522,7 +524,7 @@ Frontend calls go through `web/modules/api_client.js` with the JSDoc mirror `web
`ouroboros.cli` is a client of the same gateway/queue — no second task engine. Its parser is the command-surface SSOT (server, run, tasks, chat, logs, evolve, schedule, settings, skills, marketplace, local-model, MCP); streaming commands reserve stdout for the final answer/patch/result/JSONL and send progress to stderr.
`POST /api/tasks` creates an ordinary managed root; `GET /api/tasks` is a non-materializing list; `GET /api/tasks/<id>` returns the effective durable result; `/events` is the archive-aware SSE stream (§3 History reads and the SSE v2 transport); `/artifacts/<name>` serves simple filenames confined to `data/task_results/artifacts/<task_id>/` — a stored arbitrary path is not a download capability. The CLI refuses any `delegation_role` other than `root`, the gateway rejects caller lineage/subagent labels, and only `schedule_subagent` creates children; reserved service metadata is written after caller metadata. Admission reserves the task id plus a worker-pool slot under one queue lock, and a failure rolls back only the token-owned row with a loud typed refusal; blocking admission and materialization run off the HTTP event loop (`gateway._helpers.run_sync_to_completion`), so a cancelled HTTP waiter never cancels the admitted task. Attachments are copied into the effective task drive before enqueue; artifact-store references are not host-path authority.
`POST /api/tasks` creates an ordinary managed root; `GET /api/tasks` is a non-materializing list; `GET /api/tasks/<id>` returns the effective durable result; `/events` is the archive-aware SSE stream (§3 History reads and the SSE v2 transport); `/artifacts/<name>` serves one recorded file of the task's own stores through one confined descriptor (`?relpath=` a nested file, `?archive=` a directory ZIP) — a stored arbitrary path is not a download capability. The CLI refuses any `delegation_role` other than `root`, the gateway rejects caller lineage/subagent labels, and only `schedule_subagent` creates children; reserved service metadata is written after caller metadata. Admission reserves the task id plus a worker-pool slot under one queue lock, and a failure rolls back only the token-owned row with a loud typed refusal; blocking admission and materialization run off the HTTP event loop (`gateway._helpers.run_sync_to_completion`), so a cancelled HTTP waiter never cancels the admitted task. Attachments are copied into the effective task drive before enqueue; artifact-store references are not host-path authority.
Workspace tasks default `memory_mode=forked`; `shared` is rejected for an external workspace and materialized on a forked child drive for project scope — the stored `memory_mode` reports what was requested while `drive_root` reports where the task executes, so isolation does not depend on relabelling the request. The mode isolates the execution drive and the knowledge seed; identity and scratchpad writes still land on the canonical root the next context reads.
@ -616,7 +618,7 @@ Bundled resources use the CLI / Headless Boundary lookup order rather than assum
│ ├── settings.json ← user settings (API keys, models, budget; §7)
│ ├── task_results/ ← durable task results (task_results/<id>.json, every write stamped `_schema_version: 1`; an inadmissible row is QUARANTINED and keeps its id occupied — `task_result_schema.py`); artifacts/<task_id>/ holds .artifact_manifest.json (private metadata) + artifact files; .scratch_manifest.json declares ephemeral scratch {abs_path: sha256} excluded from patch capture only while content matches
│ │ └── artifact_versions/<task_id>/ ← artifact recovery history, last 5 versions per name (`artifacts.py`)
│ ├── task_drives/<task_id>/ ← task-scoped scratch, including live per-call manifests; startup prunes terminal tasks after the headless retention window
│ ├── task_drives/<task_id>/ ← task-scoped scratch, including live per-call manifests; the off-loop drive-custody pass settles terminal tasks after the retention window
│ ├── task_trees/<root>/blackboard.jsonl ← append-only swarm blackboard + beacons; tree-scoped and ephemeral (task_tree_ledger.py), pruned on root terminal
│ ├── state/
│ │ ├── state.json ← runtime state + compatibility cost projection; never the monetary authority
@ -654,6 +656,8 @@ Bundled resources use the CLI / Headless Boundary lookup order rather than assum
│ │ ├── review_continuations/ ← durable blocked-review continuations (+ corrupt/ quarantine; archived/ holds settled un-resumed rows ≥7 days, never deleted)
│ │ ├── workspace_executor_processes/ ← durable local/docker executor cleanup records
│ │ ├── headless_tasks/<task_id>/data ← forked/empty child execution drives whose live per-call manifests are promoted at terminal; until then the canonical reader cannot resolve their refs (issue #805) (CLI / Headless Boundary above)
│ │ ├── custody_staging/ ← unserved copies a drive settlement prepares (`task_custody.py`)
│ │ ├── custody_trash/ ← a fully custodied drive awaiting deletion (`task_custody.py`)
│ │ ├── pycache/ ← embedded-interpreter bytecode (packaged builds; CLI / Headless Boundary above)
│ │ ├── python-userbase/ ← embedded-interpreter user installs (packaged builds)
│ │ ├── betterleaks/ ← versioned scanner runtime + archive cache, created only by the explicit source-checkout installer

View file

@ -94,6 +94,7 @@ Shared frontend primitives keep pages from acquiring competing contracts — fro
- `chat_markdown.js` ← marked+DOMPurify, chat URLs (http(s), mailto, `/api/files/download`), KaTeX (no single `$`), lazy Mermaid and bounded chart blocks. `mountChatMarkdown` binds HTML to `.ui-rich-content` on existing hosts; bubble/card owners retain enhancement/teardown and density. `chat.js` disposes on removal. Vendors: `web/vendor/VENDOR-MANIFEST.md`.
- `log_events.js` ← event classification, shared technical outcome reducers and the one factual task-presentation projection for Chat and Logs: task truth maps only to `Working` / `Done` / `Done with warnings` / `Failed` / `Cancelled`; it owns no actions, incidents or notifications, compact headlines never expose raw reason codes, and a task-authored message is visible in the receiving task's timeline (`task_message_injected`, sender by value) and in the sender's (`task_message_routed`, written or refused with the host reason).
- `toast.js`, `masonry.js`, `widget_frame.js`, `widget_job.js`, CSS tokens ← notifications, layout, framed-widget bootstrap/lifecycle, bounded widget request/job policy.
- `result_files.js` ← a settled task card's one Files row: root files plus one folder record per nested tree with its member tooltip, links only from the host's `artifacts`/`artifact_archives`.
- `task_control_menu.js` ← the shared task stop/hurry dropdown, verbatim on Chat live cards and the Activity tab: "Wrap up" / "Hurry up" / "Stop now" (frozen owner wording); a host-attested budget-paused member swaps the working pair for "Resume" (`POST /api/tasks/{id}/resume`, refusals surfaced verbatim). Eligibility gates differ per surface; actions, endpoint bindings (`stop_policy` mapping, stable per-task `request_id` retry), locking and typed refusals do not. Choosing an action executes it immediately, dismissing continues the run, a pending cancel leaves only the hard escalation, and "Hurry up" acknowledges via local toast, never a chat message.
`confirm_dialog.js::openConfirmDialog` is the one browser-dialog authority: confirm mode resolves a strict boolean, input mode `{confirmed, value}` (empty on cancellation), alert mode one acknowledgement button; Cancel, Close, backdrop, Escape and supersession all resolve as non-confirmation. Native `window.prompt`/`confirm`/`alert` are forbidden in `web/modules`: inconsistent across shells, event-loop blocking, and `window.prompt` silently returns `null` in the macOS PyWebView shell. Critical controls act only on the exact confirmed result — Panic's confirm-and-send is one testable operation. Confirm and New Project dialogs use `bindDialogFocus` without sharing result semantics: callers bind after mounting and dispose before removal, restoration does not take focus from a different active surface, and menus close before their actions open a dialog. `model_chooser.js` supplies the shared editable suggestions for model roles and actor/reviewer route editors using popup geometry alone, so menu focus behavior cannot interfere with native text editing: the input keeps its draft/caret/composition through discovery updates, Arrow keys highlight and Enter or pointer selection assigns, Escape/blur dismiss without assignment; fixed source/account choices remain native selects, and unknown saved or arbitrary API model ids stay editable without an inventory allowlist.

View file

@ -85,7 +85,7 @@ Every `/api/files/*` operation resolves its requested path and refuses the opera
| GET | `/api/tasks/{task_id}` | `gateway.tasks.api_task_get` |
| GET | `/api/tasks/{task_id}/events` | `gateway.tasks.api_task_events` (legacy integer rank) |
| POST | `/api/tasks/{task_id}/events` | `gateway.tasks.api_task_events` (read-only v2 cursor) |
| GET | `/api/tasks/{task_id}/artifacts/{name}` | `gateway.tasks.api_task_artifact` |
| GET | `/api/tasks/{task_id}/artifacts/{name}` | `gateway.tasks.api_task_artifact` (the task's own stores via `task_archive`: a bare name is a top-level file, `?relpath=` a nested one, `?archive=<dir>` a directory ZIP; the detail's `artifact_archives` says what each ZIP holds. A row that records a digest is served only when its bytes still match it — a changed mutable file is 409 `artifact_identity_changed` naming the recorded digest, a failed capture 404 `artifact_unverified` — and the response says `x-ouroboros-artifact-identity: verified` or `unmeasured`; a ZIP member follows the same rule) |
| POST | `/api/tasks/{task_id}/cancel` | `gateway.tasks.api_task_cancel` |
| POST | `/api/tasks/{task_id}/hurry` | `gateway.tasks.api_task_hurry` |
| POST | `/api/tasks/{task_id}/resume` | `gateway.tasks.api_task_resume` |
@ -146,7 +146,7 @@ Rationale: `server.py` owns process startup/lifespan/static mounting, while `gat
The browser constructs one socket for the whole SPA. Feature modules subscribe before connection, and the initial complete Project chat-id set is fetched before the first open so an early Project frame cannot be mistaken for Main traffic. `ws.on(type, listener)` stores listeners in insertion-ordered sets and returns a disposer; emission uses a listener snapshot, so a listener added during dispatch does not receive the current frame and disposing one listener cannot skip its neighbor. Every decoded frame first reaches the generic `message` event and then its type-specific event, which lets Widgets consume reviewed namespaced events without duplicating the socket.
A browser `chat` frame contains the owner text and may add `sender_session_id`, `client_message_id`, `force_plan`, uploaded attachment references, `chat_id`, `project_id`, and `client_surface` — raw sending-surface observables measured at SEND time because the pywebview bridge appears asynchronously after load. The gateway normalizes that payload through `client_surface.normalize_client_surface`, stamps host `received_at`, and persists it on the canonical inbound row — distinct from the `transport` dict (transport is chat-scoped reply routing; the surface fact is per-message provenance). The fact is assembled at its PRODUCER, never inferred at render (the per-producer stamp catalog and closed-key bound: `ouroboros/client_surface.py`); synthetic A2A chats stamp no owner surface (machine traffic never wears one); machine producers stamp nothing (`client_surface` is a reserved schedule-template key rejected at admission); promotion/steering CARRY the originating owner turn's fact. The loop injects a surface note only when sending-surface identity changes within an attempt (viewport excluded — a resize is not a device change). Absence is an honest gap. The client generates a message id when absent and uses it to reconcile its pending bubble, the echoed canonical user row, routing annotations, and mailbox retries; a successful browser `send()` means only that the current socket accepted the frame, not that a task was durably admitted.
A browser `chat` frame contains the owner text and may add `sender_session_id`, `client_message_id`, `force_plan`, uploaded attachment references, `chat_id`, `project_id`, and `client_surface` — raw sending-surface observables measured at SEND time because the pywebview bridge appears asynchronously after load. The gateway normalizes that payload through `client_surface.normalize_client_surface`, stamps host `received_at` at socket receipt, and persists it on the canonical inbound row. Every transport passes the one common enqueue (`LocalChatBridge.enqueue_local_message`), which keeps that earlier host stamp or an accepted row's time, else stamps now; each update carries it, and so does a host channel fact that lacked one, so a row written at dequeue measures intake lag as `ts − client_surface.received_at` (no `logged_at`). The fact is distinct from the `transport` dict (transport is chat-scoped reply routing; the surface fact is per-message provenance). The fact is assembled at its PRODUCER, never inferred at render (the per-producer stamp catalog and closed-key bound: `ouroboros/client_surface.py`); synthetic A2A chats stamp no owner surface (machine traffic never wears one); machine producers stamp nothing (`client_surface` is a reserved schedule-template key rejected at admission); promotion/steering CARRY the originating owner turn's fact. The loop injects a surface note only when sending-surface identity changes within an attempt (viewport excluded — a resize is not a device change). Absence is an honest gap. The client generates a message id when absent and uses it to reconcile its pending bubble, the echoed canonical user row, routing annotations, and mailbox retries; a successful browser `send()` means only that the current socket accepted the frame, not that a task was durably admitted.
Ordinary frames sent while disconnected enter a process-local queue capped at 100 entries (oldest dropped), flushed in order after reconnect and lost on page reload — not a second durable outbox. Attachment messages deliberately set `queue:false`: uploads occur immediately before send, so retaining only the socket frame would leave unowned temporary files; on socket loss Chat refuses the message, cleans uploaded temporaries best-effort, and retains the staged files for explicit retry.

View file

@ -39,7 +39,7 @@ Cancellation is intent-then-custody; intent and outcome are separate fields, bec
5. **Reconcile and capture.** The task's open delegated runs are reconciled from durable custody rows and always re-audited and disclosed. Workspace artifacts are captured from the real tree; a failed or owed-but-unrunnable capture is `failed`, never `missing`, and a shared-tree capture carries `attribution: shared_unproven`.
6. **Settle.** The settled result carries reconstructed-or-honestly-unknown cost, never a fabricated final `$0`; `parent_decision` is stamped only at this outcome. `cancel_publication._intent_outcome_fields` preserves recorded cancellation provenance as `cancel_origin`: source, scope, reason, `request_id`/`requested_at`, `requested_by` when present, and the typed observation `request_origin`. These facts survive removal of the active intent and travel through the terminal event, task detail, history and result-tool reads, including conditional reads whose answer body is unchanged. HTTP proves transport, never a personal owner; absent actor evidence remains absent. The existing `requested_by` condition for parent-decision semantics is unchanged.
7. **Owe, then publish.** The owner's terminal answer is registered as OWED in the durable outbox (or a typed no-chat handoff row) BEFORE the intent settles and before `task_done` publishes, so a crash between settle and send replays the answer instead of losing it. A cascade delivers one root message with a children digest under the deterministic delivery id `cascade:<root_tid>:<request_id>`, each child's line rebuilt from its current durable status.
8. **Watchdog.** The supervisor tick runs the cancel/delivery/ref sweep off drain (`server_maintenance._run_cancel_delivery_ref_sweep`, ~20 s cadence); `sweep_cancel_intents` re-feeds unclaimed or abandoned-claim intents into custody — a cascade replayed as a cascade — so a lost control event or a custody attempt that died mid-teardown cannot wedge a cancellation. Only the physical no-live check settles a cascade's coordination intent, after the tree's summary is registered as owed.
8. **Watchdog.** The supervisor tick runs the cancel/delivery/usage sweep off drain (`server_maintenance._run_cancel_delivery_ref_sweep`, ~20 s cadence); `sweep_cancel_intents` re-feeds unclaimed or abandoned-claim intents into custody — a cascade replayed as a cascade — so a lost control event or a custody attempt that died mid-teardown cannot wedge a cancellation. Only the physical no-live check settles a cascade's coordination intent, after the tree's summary is registered as owed.
Readers see the typed projection `cancel_state: "pending"` (with `cancel_reason`) on effective results until the settle; the UI shows "Cancelling…" and restores the Cancel button only when a fetched live non-pending task detail proves the intent is gone. Steering writes — `steer_task`, mailbox follow-ups, `forward_to_worker` — are refused typed while a cancel is pending (that fence is what makes the owner-stop single-turn rail safe), and queue restore and pre-assignment consult the projection under the queue lock, so a cancelled pending task never starts. `task_done` asserts a SETTLED outcome and is validated against the DURABLE result for every event: a non-settled event status, or a settled or blank status over a non-settled or absent durable row, is a lifecycle fault — left to custody when a cancellation is pending, otherwise published with a typed infrastructure-failure axis that preserves an existing sticky terminal status.
@ -59,7 +59,7 @@ A spawned or respawned slot is not assignable until its child's PID-bound `worke
Crash storms suppress respawn while terminal sources settle, then fence pooled admission; direct chat remains. Startup runs the same recovery after process custody, before `_startup_prune_sweeps`, also in no-provider lifespan without spawning. Unknown/live ownership defers to avoid racing writers. Unresolved/protected sources or ownership/read errors set `preserve_task_sources`, skipping drive deletion. `_recover_terminal_task_files` may restore an older canonical scheduled row's start binding only from a known non-direct child's positive running/started-at record, with fresh-queue/later-boot orphan proof, no pending owner or active cancel. Existing orphan/terminal owners remain authoritative; nothing resumes. The report records `rebound`.
Startup and throttled maintenance reconcile three residue classes, the ~600 s pass off the loop thread (§10). Process custody checks strict PID, start-time, command, owner-task, session and generation evidence before it reaps. Delegated-run reconciliation applies the same owner-gone reasoning to harness rows (§6 Delegated subagents). Task, review and project reconciliation repair records whose producer no longer exists; task reconciliation decides on a status-only read, materializes artifacts and settles quiz/owner-wait only for the row it heals, and stamps its cadence when the pass ends. None of these are command-line-class kill sweeps, and one instance never reaps another. The dedicated watchdog separately observes phase-stamped loop liveness and every native actor; a wedged chat turn alerts with a `/restart` hint, a loop stall only journals, and neither kills a thread. Other owner conversations run on independent native actors, without a second scheduler.
Startup and throttled maintenance reconcile three residue classes; the ~600 s custody and ~300 s reconcile passes each run off the loop thread on their own latch (§10). Process custody checks strict PID, start-time, command, owner-task, session and generation evidence before it reaps. Delegated-run reconciliation applies the same owner-gone reasoning to harness rows (§6 Delegated subagents). Task, review and project reconciliation repair records whose producer no longer exists; task reconciliation decides and persists on a status-only read, fenced by the row's attempt basis and queue ownership, settles quiz/owner-wait only for the row it heals, then the drive-custody pass (child-ref retry, then bounded drive settlements under the queue interlock; startup copies no child store); the pass stamps its cadence when it ends; the generation fences each item, commit and write (`task_custody.publication_fence`). None of these are command-line-class kill sweeps, and one instance never reaps another. The dedicated watchdog, started with the startup phase, separately observes phase-stamped loop liveness (a stall row carries the loop thread's bounded stack) and every native actor; a wedged chat turn alerts with a `/restart` hint, a loop stall only journals, and neither kills a thread. Other owner conversations run on independent native actors, without a second scheduler.
Cooperative project checkpointing has two equivalent quiescence triggers: a host-minted genesis or cooperative tree is checked when its root settles with no live descendants, and again when the last child settles beneath an already-terminal root — a root-scope budget stop terminalizes the root before its children, so a root-only trigger would see a live tree once and never return. The bounded git chain runs on a daemon thread, revalidates quiescence under the queue lock immediately before mutation, and replays a trigger that arrives during an in-flight check. Only host-minted project roots are eligible: owner-attached folders are never auto-committed, credential-shaped files stay excluded and disclosed, and every material success, skip or error receives a durable receipt.

View file

@ -62,7 +62,7 @@ A forced turn sends the round's exact tool envelope — same schemas, same serve
#### Headless finalization and workspace patch capture
A workspace task's completion compares against the captured preflight base — task-local commits stay in the delta, not `git diff HEAD` — and the patch is bound to `task_constraint.base_sha`; a moved HEAD fails closed only for `self_worktree` (a shared tree relies on reverse-patch verification), and an unborn repo diffs against the canonical empty tree. `workspace_patch_capture.py` streams the tracked binary diff plus admitted untracked files under the pure rules of `workspace_patch_rules.py` (5 MiB per untracked file; `untracked_capture_veto_reason` is the one composite the patch and the delegated-run execution snapshot both ask), excluding each vetoed entry with a per-file reason; otherwise eligible oversized or binary untracked outputs ride complete manifest+zip file artifacts, and tracked files whose old or current size exceeds 50 MiB stay in the same file-reference manifest instead of a giant Git patch. Generated output (`dist/`, `build/`) is governed by the project's own `.gitignore`, honoured through `--exclude-standard`, not by a host name rule — git-ignored files are outside the capture universe and are not listed as exclusions, because a project whose deliverable IS its build output must not have it silently dropped; a sensitive-looking untracked credential is excluded per-file and disclosed as `sensitive_blocked`. `workspace_patch.json` is written for EVERY workspace finalization, no-change and failed included, and is the truth source for CLI strict-patch (it distinguishes omitted, no-op and failed); `workspace.patch` exists only for `ready_with_changes`. A forked or empty child drive under `data/state/headless_tasks/<task_id>/data` is execution state: the result copies back to the canonical root, declared artifacts rebase to `data/task_results/artifacts/<task_id>/`, and once the canonical result is terminal a late copy-back cannot overwrite the parent-owned terminal marker or the cost/round/token fields. The startup prune (`headless.prune_headless_task_drives`, after prior-process custody) removes a child drive only when the canonical parent is terminal, artifact finalization is terminal, retention has elapsed, the recorded child path matches the expected directory and no child-ref promotion is pending — everything needed after deletion must cross the canonical handoff before a task is presented as settled. The capture manifest explains its acting/admission/empty-tree/capture base and current branch/upstream observations; these do not establish task authorship or change application-patch bytes. An auxiliary comparison uses explicit `vcs_diff(base=..., head=...)` inputs rather than guessing a target. The CLI contract stays in §1 CLI / Headless Boundary.
A workspace task's completion compares against the captured preflight base — task-local commits stay in the delta, not `git diff HEAD` — and the patch is bound to `task_constraint.base_sha`; a moved HEAD fails closed only for `self_worktree` (a shared tree relies on reverse-patch verification), and an unborn repo diffs against the canonical empty tree. `workspace_patch_capture.py` streams the tracked binary diff plus admitted untracked files under the pure rules of `workspace_patch_rules.py` (5 MiB per untracked file; `untracked_capture_veto_reason` is the one composite the patch and the delegated-run execution snapshot both ask), excluding each vetoed entry with a per-file reason; otherwise eligible oversized or binary untracked outputs ride complete manifest+zip file artifacts, and tracked files whose old or current size exceeds 50 MiB stay in the same file-reference manifest instead of a giant Git patch. Generated output (`dist/`, `build/`) is governed by the project's own `.gitignore`, honoured through `--exclude-standard`, not by a host name rule — git-ignored files are outside the capture universe and are not listed as exclusions, because a project whose deliverable IS its build output must not have it silently dropped; a sensitive-looking untracked credential is excluded per-file and disclosed as `sensitive_blocked`. `workspace_patch.json` is written for EVERY workspace finalization, no-change and failed included, and is the truth source for CLI strict-patch (it distinguishes omitted, no-op and failed); `workspace.patch` exists only for `ready_with_changes`. A forked or empty child drive under `data/state/headless_tasks/<task_id>/data` is execution state: the result copies back to the canonical root, declared artifacts rebase to `data/task_results/artifacts/<task_id>/` at their store relpath, and once the canonical result is terminal a late copy-back cannot overwrite the parent-owned terminal marker or the cost/round/token fields. Every child-drive deletion (the off-loop drive-custody pass past retention, a cancelled subagent's at once, never the cancel path; admission rollback) is `task_custody.settle_child_drive`'s decision: per occupant the supervisor's probe proves no live owner (unknown keeps it), the DURABLE row is settled with post-work closed, copy-back adopted or blocked by cancellation, no ref pending, receipts unioned, and every obligation (each recorded deliverable, derived before the store is read; unrecorded files; the input closure of unread mail and the contract; every exact unread mailbox line) is held by the canonical row. Staged unserved, published by create-only rename under the custody lock copy-back shares (different bytes publish beside a canonical file, never over it), written from the row-locked CURRENT and re-read; the drive moves only under the queue interlock after a fresh census, probe and mailbox check. Missing, unreadable or mismatched material keeps the drive until a restore converges. Effective reads only list stores (stat, `measured: false`), rank identity apart from location and refuse a conflict; they never copy, hash or register. The capture manifest explains its acting/admission/empty-tree/capture base and current branch/upstream observations; these do not establish task authorship or change application-patch bytes. An auxiliary comparison uses explicit `vcs_diff(base=..., head=...)` inputs rather than guessing a target. The CLI contract stays in §1 CLI / Headless Boundary.
#### Owner routing verbs
@ -264,7 +264,7 @@ An ordinary managed task whose provider outcome becomes unknown stays in the tra
Ordinary delegation requests no extra engine review panel; new ordinary runs on Claudexor 3.9.8+ default to no panel — a version boundary of the behavior, not a second release pin. The start receipt's `engine_version` is the handshaken serving version, distinct from the release pin; engine review outcome, execution success, the parent's integration decision and Ouroboros review gates remain separate facts. Delegated snapshot capture stays relative to its recorded baseline: committed bytes can be captured with a disclosed `head_moved`, the instruction still forbids committing, and the distinct `self_worktree` unchanged-HEAD check is preserved.
Children coordinate through `tree_note` and `tree_read`; read-only children can read/list the project-scoped knowledge store without writing its index; only the parent may use `override_delegation_constraint`, and a `review_requested` note carries an exact evidence reference/hash and wakes the parent without starting a paid cycle. Read-only and acting children alike hold the descendant-scoped `forward_to_worker`, `peek_task`, `cancel_task` and `discard_child_result` controls, and recursive delegation never widens filesystem, budget, depth, deadline, commit or owner authority. `forward_to_worker` also reaches the child's parent or a sibling (same parent and root) as `peer_task` provenance, its prefix naming the stamped `relation` (`sibling`|`parent`): context-only, relay refused, 8000-char bound. `delegation_budget` governs descendants (`may_delegate`, `may_fan_out`, additive depth provenance; a free-form intent note is never authority); persisted admission facts outrank later Settings changes, and a lower permitted depth is reported `capability_reduced`, never a silent flat tree.
Children coordinate through `tree_note` and `tree_read`; read-only children can read/list the project-scoped knowledge store without writing its index; only the parent may use `override_delegation_constraint`, and a `review_requested` note carries an exact evidence reference/hash and wakes the parent without starting a paid cycle. Read-only and acting children alike hold the descendant-scoped `forward_to_worker`, `peek_task`, `cancel_task` and `discard_child_result` controls, and recursive delegation never widens filesystem, budget, depth, deadline, commit or owner authority. `forward_to_worker` also reaches the child's parent or a sibling (same parent and root) as `peer_task` provenance, its prefix naming the stamped `relation` (`sibling`|`parent`): context-only, relay refused, 8000-char bound; a queued target gets the row with a `queued` receipt (not started, not read). An accepted terminal write keeps mail no attempt read as `unread_mailbox` (exact rows, no ACK, union-only), shown by `get_task_result`. `delegation_budget` governs descendants (`may_delegate`, `may_fan_out`, additive depth provenance; a free-form intent note is never authority); persisted admission facts outrank later Settings changes, and a lower permitted depth is reported `capability_reduced`, never a silent flat tree.
**Registry.** `configured_subagents.py` owns `OUROBOROS_SUBAGENTS`: strict `{enabled, items}`, at most ten rows with hidden `subagent_id`, route-handle name (§3 Available subagents), any-language `recommended_use`, normalized API/session route, optional effort, account pin, access and row `enabled`. Session access defaults to `full` or the owner's `workspace_write`; snapshots lacking access retain workspace_write. Legacy env inputs fail closed. Row `enabled` defaults true, rejects nonbooleans and serializes only false, preserving fingerprints/receipts. Off rows retain configuration but leave catalog, alternatives, legacy matching, local autostart and reviewer resolution. Explicit selection returns `subagent_disabled`, distinct from global `subagents_disabled` and live unavailability. Tools take handle or stored id; roster edits refuse twins; records name their own engine. Captured snapshots never consult live settings. Descriptions guide the LLM, never host ranking/keyword routing; exact selection starts that route or refuses without substitution.

View file

@ -11,7 +11,7 @@ This chapter is the short list of properties the rest of the book must not contr
7. **This book is the present-tense map.** Structural owners, APIs, durable data, UI surfaces, and the rationale for non-obvious guards update in the owning chapter (entrypoint `docs/ARCHITECTURE.md`) in the same commit as the code (documentation contract: docs/DEVELOPMENT.md; residue ratchet: `tests/test_docs_sync.py`); release chronology lives in git and README.
8. **Skill gates do not collapse.** Discovery, deterministic preflight, content-hash-bound critic or qualified Advisory author authority, owner grants, dependency readiness, enablement, and execution remain separate. A PASS does not install dependencies, and `enabled=true` does not prove executable readiness.
9. **Startup rescue has one mutation owner.** Supervisor recovery writes rescue evidence before reset or blocks while preserving the tree. Worker or agent construction remains warning-only and never stages or commits inherited dirt.
10. **Projection over replay.** Interactive status, history, and cost reads are bounded, non-materializing projections; durable owners perform the one authoritative replay or terminal materialization. A process-local fingerprint memo (`_usage_rows_memo.py`, `delegate_custody_memo.py`) serves warm reads only while its store fingerprint holds, refolds on any doubt, and never touches disk.
10. **Projection over replay.** Interactive status, history, cost and task-file reads are bounded, pure projections (no copy, hash or registration); durable owners perform the one authoritative replay or terminal materialization. A process-local fingerprint memo (`_usage_rows_memo.py`, `delegate_custody_memo.py`) serves warm reads only while its store fingerprint holds, refolds on any doubt, and never touches disk.
11. **UI resources carry a disposer.** Every subscription, listener, observer, timer, stream, and live page instance has explicit teardown; navigation does not leave hidden instances mutating visible or durable state.
12. **Frozen contracts extend explicitly.** `ouroboros/contracts/` is a versioned, backward-compatible ABI — typed shapes together with their parsing/normalization/policy helpers (§11). New capability extends the frozen shape or ships an explicitly versioned successor; existing consumers keep working.
13. **Provider wire adaptation stays exact-route and success-confirmed.** Canonical history remains provider-neutral; typed physical projections may change values, fields, or a registered dialect on one provider/endpoint/API/model only. Failed candidates teach nothing durable, task-local cognition degradation never becomes future dispatch authority, and the physical-attempt ledger remains distinct from terminal request-wire history.
@ -21,11 +21,11 @@ This chapter is the short list of properties the rest of the book must not contr
17. **Review spend has one ceiling.** Every paid review gate shares `OUROBOROS_REVIEW_MAX_CYCLES`; the per-gate meanings live once in §6 Review stack, the SSOT in `review_cycles.py`. This paid ceiling does not remove the last author reaction. Blocking correction or stop grants no approval; informed Advisory finish keeps current-author authority separate from the critic.
18. **A typed permanent engine refusal discharges a custody duty once.** Durably, under the engine's own code — never retried on a timer, never recorded as a deletion. Owner: `delegate_custody._retire_project_locked`.
19. **An interrupted parent leaves no orphan.** A child whose parent was interrupted is cancelled with the parent's cause through any door (Restart, crash, window close; pooled or direct): the planned path settles it in `kill_workers(preserve_pending=True)`, snapshot restore marks the same custody on boot (`pending_parent_interrupted`). Owners: `supervisor/workers.py`, `supervisor/queue_snapshot.py`.
20. **A stalled loop says where it went silent.** `supervisor_loop_stall` carries the tick phase, the loop thread's CPU against the wall gap, the worst worker-stamped event lag and `daemon_pin_matched`; `supervisor_loop_stall_end` closes every alerted stall. The loop publishes, the watchdog only reads, and the owner's chat stays silent — the journal is the record.
20. **A stalled loop says where it went silent.** `supervisor_loop_stall` carries the tick phase, the loop thread's CPU against the wall gap, the worst worker-stamped event lag and `daemon_pin_matched`; `supervisor_loop_stall_end` closes every alerted stall with stack samples; `host_duty_stall` is its off-loop twin. The loop publishes, the watchdog only reads, and the owner's chat stays silent — the journal is the record.
21. **Source acknowledgement is a pre-check over known facts.** The queue's compare-and-seal stays the single fail-closed authority; an unknown queue state is disclosed, never read as a change, and every refusal carries its typed cause in the durable worker-side row `acceptance_source_ack` (owner `ouroboros/loop_messages.py`).
22. **A host-owed round never parks.** A turn parks behind a review panel only when the panel is the sole thing it waits for; a turn in which the host has just spoken to the model never parks (`loop._finalize_loop_candidate`).
23. **Every call has a bound, and a recorder speaks only for what it collected.** A reviewer's tool call runs under the loop's per-tool timeout narrowed by the inherited dispatch deadline; a call that outlives it is abandoned: its late value sources no receipt, no coverage. A deadline recorder reconciles the turn's own panel at $0 before it writes a terminal reason. Owners: `review_native_episode.py`, `loop_tool_execution.py`, `acceptance_settlement.py`.
24. **The thread that answers workers runs only queue-bounded work.** Work that scales with history, the daemon or the network runs off-thread, reads candidates before liveness (one in-memory live source under `_queue_lock`), stops mutating when its loop generation ends and is attach-only to the daemon once a stop is in flight. Residuals on the loop thread: the 300-s zombie reconcile (it decides on status-only reads and materializes only a row it heals) and the exact ledger reads of invariant 28. Owner: `ouroboros/server_maintenance.py`.
24. **The thread that answers workers runs only queue-bounded work.** Work that scales with history, the daemon or the network runs off-thread, reads candidates before liveness (one in-memory live source under `_queue_lock`), stops mutating when its loop generation ends and is attach-only to the daemon once a stop is in flight. The residual on the loop thread: the exact ledger reads of invariant 28. Owner: `ouroboros/server_maintenance.py`.
25. **An answer that has not arrived is a gap — never a refusal, a failure, a verdict or an owner message.** A direct turn applies its acceptance fence in-process (admission lock, then `_queue_lock`, never the reverse); a pooled request is idempotent by token and acknowledged per request (`<token>.<req>.json`); a transition is re-sent once, a read never; only `sealed` is a seal, an absent row is not one; a fence that did not answer buys no model round: the panel advises on `admission_fence_available=false`, delivery seals again, and a blocking install accepts a reviewer-approved answer with the typed note `admission_close_unconfirmed`. Owners: `ouroboros/agent.py`, `supervisor/queue_transitions.py`, `ouroboros/loop_delivery.py`.
26. **A cross-process guard derives from the durable artifact it guards, never from process memory.** The usage-ledger compaction floor is the `source_size_bytes` the last committed pass stamped into the live header, so one pass throttles every process, fresh ones included; the per-process memo throttles only a pass that changed no bytes. Owner: `ouroboros/usage_compaction.py`.
27. **The predecessor list is a hint; the door is a predicate on the result.** Any actor holding a routing verb continues any settled, readable result (a live root is `steer_task`, not a second root; a pending promote is not a result), whichever project it belongs to, wherever the continuation lands and whether it is a root's or a helper's: the pointer is rebuilt from the task id, the successor's ceiling, origin and contract come from the caller and admission, and the predecessor's project or helper origin is disclosed in the receipt, never an admission condition. Only a ROOT stamps the pointer and only roots are offered. An emitted promote is durably `promotion_admission{status:"emitted"}`: pending, granting no schedule, owning no id, never `unknown`. Owners: `tools/control_routing.py`, `tools/control_events.py`.

View file

@ -209,14 +209,14 @@ successor-parity and artifact-transport rules are review-only.
that reason).
- Durable artifacts are NOT age-pruned: genesis projects
(`OUROBOROS_SUBAGENT_PROJECTS_ROOT`) and forensic observability blobs (kept
compressed indefinitely by contract; startup runs a census, never a
deletion).
compressed indefinitely by contract; blobs are never deleted or
counted).
- Review continuations are recovery state, not disposable GC: archive a record
(collision-safe move, never delete) only when its owner task is settled, it
stayed un-resumed past the seven-day threshold and no recorded obligation
remains open; any uncertainty or move error leaves the live record intact.
Enforcement: `tests/test_phase3c_observability_gc.py` (the unified knob and the cutoff math) and `tests/test_observability_retention.py` (the census and preserve-indefinitely contract); the review-continuation archive rule has no automated surface — review-only.
Enforcement: `tests/test_phase3c_observability_gc.py` (the unified knob and the cutoff math) and `tests/test_observability_retention.py` (the preserve-indefinitely contract); the review-continuation archive rule has no automated surface — review-only.
### Live subagents
@ -315,7 +315,7 @@ and 23 (`delegated_transport`), both critical. The imperatives:
stale-replica regression at BOTH seams
(`tests/test_available_subagents_runtime_review_fixes.py`). Do not broaden
generic data-tool behavior while fixing isolation (`forward_to_worker`
writes only to validated running tasks in the current task/root lineage).
writes only to validated running or queued tasks in the current lineage).
- A custody row carries its owner's kind; every sweep, audit and counter over
custody rows states which kinds it covers. A review-owned run
(`RunCustody.review_owned`) belongs to its panel — never the task's open

View file

@ -2,9 +2,9 @@
Machine extraction of the `docs/ARCHITECTURE.md` "Data layout (`~/Ouroboros/`)" tree — the durable-file orientation carrier (this tree's counterpart of the reference PERSISTENCE_OWNERS derivation checklist) — regenerated by `python scripts/regenerate_inventories.py`. Do not edit. Every entry is probed against reality: repo entries must exist as tracked paths; data-plane entries must appear as a literal in the runtime sources that construct them. A durable file renamed or removed in code while its tree row survives = red (`tests/test_generated_inventories.py`).
Source: `docs/architecture/01-high-level-architecture.md`, physical LF lines 603-693; UTF-8 SHA-256 `7d323eb43a78d3b53cd3a2e2262c8177e024a82bbc164c5689357f58b56edac2`.
Source: `docs/architecture/01-high-level-architecture.md`, physical LF lines 605-697; UTF-8 SHA-256 `0905012b562019ff78192e1e427b75ad7b0d4dfad6e30c700018e46c5eb69858`.
- entries: **80** (code-ref: 73, repo-dir: 6, repo-path: 1)
- entries: **82** (code-ref: 75, repo-dir: 6, repo-path: 1)
| entry | probe | resolution |
|---|---|---|
@ -56,6 +56,8 @@ Source: `docs/architecture/01-high-level-architecture.md`, physical LF lines 603
| `review_continuations/` | `review_continuations` | code-ref |
| `workspace_executor_processes/` | `workspace_executor_processes` | code-ref |
| `headless_tasks/<task_id>/data` | `data` | code-ref |
| `custody_staging/` | `custody_staging` | code-ref |
| `custody_trash/` | `custody_trash` | code-ref |
| `pycache/` | `pycache` | code-ref |
| `python-userbase/` | `python-userbase` | code-ref |
| `betterleaks/` | `betterleaks` | code-ref |

View file

@ -2,7 +2,7 @@
AST-derived inventory of compatibility facades, regenerated by `python scripts/regenerate_inventories.py`. Do not edit. A facade row is any runtime module whose top-level `from <population module> import ...` statements carry the `noqa: F401` re-export marker — the codebase's declared "this binding exists for its binding, not for this module's own use" convention (reference FACADE_CONSUMERS method). Leaf domains come from `ouroboros/domains.toml`; a leaf outside the facade's domain is marked ✗ (that edge also appears in the manifest's pinned direction matrix). `tests/test_generated_inventories.py` pins byte-identity, so any re-export surface change must regenerate this file.
- facade modules: **63**; marked re-export bindings: **2353**; cross-domain facade→leaf pairs: **133**
- facade modules: **63**; marked re-export bindings: **2355**; cross-domain facade→leaf pairs: **133**
| facade | domain | bindings | leaves |
|---|---|---:|---|
@ -62,7 +62,7 @@ AST-derived inventory of compatibility facades, regenerated by `python scripts/r
| `server.py` | D11 | 56 | `ouroboros/server_liveness.py` (6)<br>`ouroboros/server_maintenance.py` (14)<br>`ouroboros/server_owner_routing.py` (5)<br>`ouroboros/server_process.py` (9)<br>`ouroboros/server_restart.py` (8)<br>`ouroboros/server_routing_context.py` (14) |
| `supervisor/events.py` | D08 | 96 | `ouroboros/config.py` (1 ✗D12)<br>`ouroboros/contracts/task_constraint.py` (1 ✗D19)<br>`ouroboros/cost_projection.py` (3 ✗D16)<br>`ouroboros/subagent_messages.py` (1 ✗D07)<br>`ouroboros/task_results.py` (2 ✗D17)<br>`ouroboros/tool_capabilities.py` (2 ✗D04)<br>`ouroboros/utils.py` (4 ✗D18)<br>`supervisor/cognitive_operations.py` (2)<br>`supervisor/events_budget.py` (5)<br>`supervisor/events_chat_delivery.py` (9)<br>`supervisor/events_coop_checkpoint.py` (6)<br>`supervisor/events_evolution_done.py` (1)<br>`supervisor/events_project_routing.py` (10)<br>`supervisor/events_runtime_controls.py` (6)<br>`supervisor/events_schedule_task.py` (4)<br>`supervisor/events_subagent_admission.py` (15)<br>`supervisor/events_task_done.py` (8)<br>`supervisor/events_worker_reports.py` (8)<br>`supervisor/log_addressing.py` (4)<br>`supervisor/queue_transitions.py` (1)<br>`supervisor/steering.py` (2 ✗D09)<br>`supervisor/task_dispatch.py` (1) |
| `supervisor/git_ops.py` | D10 | 37 | `ouroboros/utils.py` (1 ✗D18)<br>`supervisor/git_ops_remotes.py` (4)<br>`supervisor/git_ops_rescue.py` (8)<br>`supervisor/git_ops_reset.py` (10)<br>`supervisor/git_ops_updates.py` (8)<br>`supervisor/state.py` (4 ✗D08)<br>`supervisor/update_recovery.py` (2) |
| `supervisor/queue.py` | D08 | 112 | `ouroboros/config.py` (6 ✗D12)<br>`ouroboros/contracts/task_contract.py` (3 ✗D19)<br>`ouroboros/schedule_contract.py` (2)<br>`ouroboros/skill_loader.py` (1 ✗D14)<br>`ouroboros/utils.py` (3 ✗D18)<br>`supervisor/evolution_lifecycle.py` (12 ✗D15)<br>`supervisor/message_bus.py` (3)<br>`supervisor/queue_schedules.py` (19)<br>`supervisor/queue_snapshot.py` (5)<br>`supervisor/queue_timeouts.py` (8)<br>`supervisor/queue_transitions.py` (3)<br>`supervisor/schedule_lifecycle.py` (3)<br>`supervisor/schedule_time.py` (7)<br>`supervisor/state.py` (6)<br>`supervisor/task_admission.py` (9)<br>`supervisor/task_lifecycle.py` (17 ✗D09)<br>`supervisor/task_reaper.py` (5 ✗D09) |
| `supervisor/queue.py` | D08 | 114 | `ouroboros/config.py` (6 ✗D12)<br>`ouroboros/contracts/task_contract.py` (3 ✗D19)<br>`ouroboros/schedule_contract.py` (2)<br>`ouroboros/skill_loader.py` (1 ✗D14)<br>`ouroboros/utils.py` (3 ✗D18)<br>`supervisor/evolution_lifecycle.py` (12 ✗D15)<br>`supervisor/message_bus.py` (3)<br>`supervisor/queue_schedules.py` (19)<br>`supervisor/queue_snapshot.py` (5)<br>`supervisor/queue_timeouts.py` (8)<br>`supervisor/queue_transitions.py` (5)<br>`supervisor/schedule_lifecycle.py` (3)<br>`supervisor/schedule_time.py` (7)<br>`supervisor/state.py` (6)<br>`supervisor/task_admission.py` (9)<br>`supervisor/task_lifecycle.py` (17 ✗D09)<br>`supervisor/task_reaper.py` (5 ✗D09) |
| `supervisor/queue_transitions.py` | D08 | 2 | `supervisor/budget_resume.py` (2) |
| `supervisor/state.py` | D08 | 4 | `ouroboros/utils.py` (4 ✗D18) |
| `supervisor/task_lifecycle.py` | D09 | 32 | `supervisor/cancel_publication.py` (20)<br>`supervisor/queue_transitions.py` (11 ✗D08)<br>`supervisor/task_admission.py` (1 ✗D08) |

View file

@ -526,42 +526,21 @@ def check_budget(env: Any) -> Tuple[dict, int]:
def check_review_continuations(env: Any) -> Tuple[dict, int]:
try:
from ouroboros.task_continuation import list_review_continuations
from ouroboros.task_results import (
STATUS_CANCELLED,
STATUS_COMPLETED,
STATUS_FAILED,
STATUS_INTERRUPTED,
STATUS_REJECTED_DUPLICATE,
STATUS_REQUESTED,
STATUS_RUNNING,
STATUS_SCHEDULED,
list_task_results,
)
from ouroboros.task_results import STATUS_INTERRUPTED, load_task_result
continuations, corrupt = list_review_continuations(env.drive_root)
task_rows = list_task_results(
env.drive_root,
statuses=[
STATUS_REQUESTED,
STATUS_SCHEDULED,
STATUS_RUNNING,
STATUS_INTERRUPTED,
STATUS_COMPLETED,
STATUS_FAILED,
STATUS_CANCELLED,
STATUS_REJECTED_DUPLICATE,
],
)
task_by_id = {
str(item.get("task_id") or ""): item
for item in task_rows
if str(item.get("task_id") or "").strip()
}
def _status(task_id: str) -> str:
# One row read per continuation (TZ-1 A): a startup check never walks the store.
try:
return str((load_task_result(env.drive_root, task_id) or {}).get("status") or "")
except (OSError, ValueError):
return ""
rows = []
interrupted = []
for item in continuations:
task_status = str((task_by_id.get(item.task_id) or {}).get("status") or "")
task_status = _status(item.task_id)
row = {
"task_id": item.task_id,
"task_status": task_status or "missing",
@ -982,9 +961,8 @@ def verify_system_state(env: Any, git_sha: str) -> None:
"git_sha": git_sha,
}
append_jsonl(drive_logs / "events.jsonl", event)
if issues > 0:
log.warning(f"Startup verification found {issues} issue(s): {checks}")
# No stdlib WARNING beside the durable ``startup_verification`` row: the row and the Logs
# panel carry every check (#1184); a fact with a durable row gets no second line.
def _reconcile_review_attempts_on_startup(env: Any) -> Dict[str, Any]:

View file

@ -20,6 +20,7 @@ from typing import Any, Dict, Iterable, List, Optional, Union
from ouroboros.utils import atomic_write_json, read_json_dict, update_json_locked, write_bytes_atomic
from ouroboros.headless import ARTIFACT_STATUS_FAILED, ARTIFACT_STATUS_READY, SCRATCH_MANIFEST_NAME, task_artifacts_dir
from ouroboros.outcome_receipt_store import is_verification_receipts_path
from ouroboros.task_custody import fence_publication
from ouroboros.task_results import validate_task_id
log = logging.getLogger(__name__)
@ -707,6 +708,7 @@ def store_actor_source_bytes(
except OSError:
already_stored = False
if not already_stored:
fence_publication() # a closed publication generation writes no new source
write_bytes_atomic(target, bytes(data))
return {
"kind": "task_source",
@ -1226,11 +1228,13 @@ def stream_artifact_file(path: Any, sink: Any = None, *, expected: Any = None) -
def copy_artifact_file(source: Any, destination: pathlib.Path, *, expected: Any = None) -> Dict[str, Any]:
"""Publish a verified file copy atomically; preserve any prior bytes on failure."""
"""Publish a verified file copy atomically; preserve any prior bytes on failure. Inside a
``task_custody.publication_fence`` a closed generation starts no copy."""
source_path = pathlib.Path(source) if isinstance(source, (str, os.PathLike)) else None
destination = pathlib.Path(destination)
if source_path is not None and not destination.is_symlink() and source_path.resolve(strict=False) == destination.resolve(strict=False):
return stream_artifact_file(source, expected=expected)
fence_publication()
destination.parent.mkdir(parents=True, exist_ok=True)
temporary = destination.with_name(f".{uuid.uuid4().hex}.tmp")
try:
@ -1340,6 +1344,7 @@ def _archive_previous_artifact_version(drive_root: pathlib.Path, task_id: str, d
copy_artifact_file(dest, version_path, expected=previous)
versions = sorted((p for p in version_dir.iterdir() if p.is_file()), key=lambda p: p.name)
for stale in versions[:-_ARTIFACT_VERSION_RETENTION]:
fence_publication() # a closed generation deletes no retained version, even after its backup landed
try:
stale.unlink()
except OSError:
@ -1502,9 +1507,12 @@ def copy_directory_to_task_artifacts(
return records
def collect_task_artifact_records(drive_root: Union[pathlib.Path, str], task_id: str) -> List[Dict[str, Any]]:
"""Collect deliverables while excluding internal task metadata and source handles."""
def collect_task_artifact_records(
drive_root: Union[pathlib.Path, str], task_id: str, *, measure: bool = True, strict: bool = False,
) -> List[Dict[str, Any]]:
"""List one store's deliverables (nested ones carry ``relpath``), never its metadata or
inputs. ``measure=False`` is the pure view: size from lstat, ``measured: False``, only a
registration's recorded identity. ``strict`` raises on unreadable material instead."""
try:
artifact_dir = task_artifact_dir_path(pathlib.Path(drive_root), validate_task_id(task_id), create=False)
except ValueError:
@ -1512,41 +1520,44 @@ def collect_task_artifact_records(drive_root: Union[pathlib.Path, str], task_id:
records: List[Dict[str, Any]] = []
if not artifact_dir.exists():
return records
data = read_json_dict(artifact_dir / _ARTIFACT_MANIFEST) or {}
raw_manifest = data.get("artifacts") if isinstance(data.get("artifacts"), dict) else {}
manifest_path = artifact_dir / _ARTIFACT_MANIFEST
data = read_json_dict(manifest_path)
if data is None and strict and (manifest_path.exists() or manifest_path.is_symlink()):
raise OSError(f"artifact registration is unreadable: {manifest_path}")
raw_manifest = (data or {}).get("artifacts") if isinstance((data or {}).get("artifacts"), dict) else {}
manifest = {str(key): dict(value) for key, value in raw_manifest.items() if isinstance(value, dict)}
artifact_root = artifact_dir.resolve(strict=False)
for path in sorted(p for p in artifact_dir.rglob("*") if p.is_file() and not p.is_symlink()):
# Internal task-metadata files (the artifact manifest and the v6.52.2 scratch manifest)
# are NOT deliverables — never record them as produced artifacts.
if path.name in (_ARTIFACT_MANIFEST, SCRATCH_MANIFEST_NAME):
continue
if path == artifact_dir / (_ARTIFACT_MANIFEST + ".lock"):
continue # an in-flight registration lock is not a deliverable
# Verification receipts live beside artifacts for durable custody, but
# they are an append-only authority stream, not a deliverable. Letting
# generic materialization register/copy this file can replace a newer
# canonical-only lifecycle row with a stale child replica.
if is_verification_receipts_path(drive_root, task_id, path):
members = iter_artifact_tree(artifact_dir) if strict else artifact_dir.rglob("*")
for path in sorted(p for p in members if p.is_file() and not p.is_symlink()):
# Metadata (manifests, the registration lock) and the receipt stream (its own
# union writer) are not deliverables.
if (path.name in (_ARTIFACT_MANIFEST, SCRATCH_MANIFEST_NAME) or path == artifact_dir / (_ARTIFACT_MANIFEST + ".lock")
or is_verification_receipts_path(drive_root, task_id, path)):
continue
try:
rel_parts = path.resolve(strict=False).relative_to(artifact_root).parts
except (OSError, ValueError):
continue
# v6.52.0 (P1): staged INPUT attachments live under attachments/ and are NOT
# task deliverables — never record them as produced artifacts.
if rel_parts and rel_parts[0] in {
_ATTACHMENTS_SUBDIR, _CHAT_MEDIA_SUBDIR, _SOURCE_HANDLES_SUBDIR,
}:
continue # reached through a link: not this store's material
# Staged inputs, chat media and source handles are not deliverables.
if rel_parts and rel_parts[0] in {_ATTACHMENTS_SUBDIR, _CHAT_MEDIA_SUBDIR, _SOURCE_HANDLES_SUBDIR}:
continue
manifest_record = manifest.get(path.name) if path.parent == artifact_dir else None
nested = {"relpath": "/".join(rel_parts)} if len(rel_parts) > 1 else {}
try:
record = artifact_record(path)
if not measure: # an immutable registration keeps its recorded identity; nothing else is claimed
registered = manifest_record or {}
records.append({"kind": str(registered.get("kind") or "task_artifact"), "name": path.name,
"path": str(path), **nested, "size": path.lstat().st_size, "measured": False,
**({key: registered.get(key) for key in ("immutable", "size", "sha256")}
if registered.get("immutable") else {})})
continue
record = artifact_record(path) | nested
if manifest_record:
record = merge_artifact_records([{**manifest_record, "path": str(path)}], [record])[0]
records.append(record)
except OSError:
continue
if strict:
raise
return records

View file

@ -211,6 +211,7 @@ D20 = "Presence"
"ouroboros/gateway/skill_publish.py" = "D11"
"ouroboros/gateway/state.py" = "D11"
"ouroboros/gateway/task_events.py" = "D11"
"ouroboros/gateway/task_archive.py" = "D11"
"ouroboros/gateway/task_hurry.py" = "D11"
"ouroboros/gateway/task_list_scan.py" = "D11"
"ouroboros/gateway/tasks.py" = "D11"
@ -423,6 +424,7 @@ D20 = "Presence"
"ouroboros/task_results.py" = "D17"
"ouroboros/task_result_schema.py" = "D17"
"ouroboros/task_status.py" = "D17"
"ouroboros/task_custody.py" = "D17"
"ouroboros/terminal_projection.py" = "D17"
"ouroboros/task_tree_ledger.py" = "D07"
"ouroboros/tool_access.py" = "D04"

View file

@ -1131,15 +1131,17 @@ class TaskDetailResponse(TypedDict, total=False):
"""``GET /api/tasks/{task_id}`` — the public task-result envelope (open shape;
stored task-result keys pass through) plus additive typed projections."""
artifacts: List[Dict[str, Any]] # Open result rows; a nested file carries additive ``relpath``.
# Per top-level result dir: {name, files, size, excluded, available} of its on-demand ``?archive=`` ZIP.
artifact_archives: Dict[str, Dict[str, Any]]
cost_breakdown: TaskCostBreakdown
model_waits: Dict[str, Any]
# Cancel projection (additive-optional): ``"pending"`` while a durable cancel intent is open and the
# supervisor teardown has not settled — the status itself honestly stays running/scheduled; absent on
# settled results and on tasks nobody asked to cancel. The UI's interim "Cancelling…" reads this, never a status.
cancel_state: str
# Rides beside ``cancel_state`` when the intent carries a reason (GR2-11):
# the WHY of the pending cancellation (owner text, "subtree cancellation of
# <root>", "evolution stopped", …). Absent when no reason was recorded.
# Beside ``cancel_state`` when the intent carries a reason (GR2-11): the WHY of the pending
# cancellation (owner text, "subtree cancellation of <root>", …); absent when none was recorded.
cancel_reason: str
# S3 (Q1, additive-optional): rides beside a pending ``cancel_state`` when
# the open intent is the SOFT stop ("finalize_then_cancel") — the UI shows

View file

@ -0,0 +1,407 @@
"""Confined reads of a task's recorded files: one file, or one directory as a ZIP (TZ-1 V12).
A task's files live in its own stores only (``task_custody.task_artifact_stores``: the
canonical store, then each own child drive's); a recorded path is attributed to one of
them once (``_task_artifact_location``), and the bytes that leave are then read from ONE
open descriptor reached by a confined descent: from the resolved canonical drive root
(when the store lies under it, so a swapped drive, store or ``task_results`` component
is refused too) one ``O_DIRECTORY | O_NOFOLLOW`` directory-relative open per path
segment, and the file itself opened ``O_NOFOLLOW | O_NONBLOCK`` from its parent's
descriptor. A component swapped for a symlink after attribution is ELOOP, a planted FIFO
never blocks, and the open descriptor must fstat as a regular file. Nothing re-opens a
pathname after the check. A captured file (an immutable row, or content-addressed chat
media whose name is its sha256) is verified INTO a private spool and only the spool is
served, so the bytes on the wire are exactly the verified ones. The same holds for any
row that records a digest: a mutable file's bytes that no longer match it are refused
typed (409 ``artifact_identity_changed``, naming the recorded digest) rather than served
under an identity a parent's disposition may have bound; a listing without a digest
streams its current bytes and says so (``x-ouroboros-artifact-identity: unmeasured``).
A directory archive holds exactly the recorded rows under that directory that resolve
inside the stores, are regular files and carry no failed/missing capture; one relpath
is one member (canonical store first), named relative to the directory's parent. It
spools in 1 MiB chunks through one anonymous temporary file (bounded memory, exact
``Content-Length``, status settled before the first byte); a member that records a
digest must match it while read, or the answer is a typed refusal.
Where the platform lacks directory-relative no-follow opens (Windows today) no file or
archive is served: a typed HTTP 503, never an unconfined fallback (owner decision,
issue #1297).
"""
from __future__ import annotations
import errno
import hashlib
import mimetypes
import os
import pathlib
import re
import stat
import tempfile
import time
import zipfile
from email.utils import formatdate
from typing import Any, Dict, Iterator, List, Optional, Tuple
from urllib.parse import quote
import anyio
from starlette.responses import Response, StreamingResponse
from ouroboros import artifacts as artifact_store
from ouroboros.gateway._helpers import json_error
from ouroboros.task_custody import task_artifact_stores
from ouroboros.task_status import load_effective_task_result
ARCHIVE_SUFFIX = ".zip"
_CHUNK = 1024 * 1024
# A recorded status that still offers bytes: ``ready`` or none (legacy and unmeasured
# listings state nothing; their bytes are what is on disk now).
_SERVABLE_STATUSES = frozenset({"", "ready"})
# Gone, lost a directory, or turned into a link (ELOOP; EMLINK on some BSDs): not an I/O fault.
_MEMBER_GONE = frozenset({errno.ENOENT, errno.ENOTDIR, errno.ELOOP, errno.EMLINK})
CONFINED = bool(os.open in os.supports_dir_fd and os.stat in os.supports_dir_fd
and hasattr(os, "O_NOFOLLOW") and hasattr(os, "O_DIRECTORY"))
_DIR_FLAGS = (os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) | getattr(os, "O_NOFOLLOW", 0)
| getattr(os, "O_CLOEXEC", 0))
_FILE_FLAGS = (os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0)
| getattr(os, "O_NONBLOCK", 0) | getattr(os, "O_NOCTTY", 0))
_CHAT_MEDIA_DIGEST_RE = re.compile(r"chat-media-([0-9a-f]{64})\.[a-z0-9]+")
Route = Tuple[pathlib.Path, List[str]] # (directory opened by absolute path, segments below it)
def task_artifact_location(stores: List[pathlib.Path], raw_path: Any) -> Optional[tuple]:
"""``(store index, resolved file, POSIX relpath)`` of a recorded path in one of the task's
own ``stores`` (symlinks resolve first, so an escaping link or a sibling's store is None)."""
text = str(raw_path or "").strip()
path = pathlib.Path(text).resolve(strict=False) if text else None
return next(((index, path, path.relative_to(store).as_posix()) for index, store in enumerate(stores)
if path and path != store and path.is_relative_to(store)), None)
def plain_segments(relpath: Any) -> List[str]:
"""The segments of a store-relative path, or [] when any is empty/``.``/``..`` or the text
carries a backslash or NUL."""
text = str(relpath or "")
parts = text.split("/")
return [] if "\\" in text or "\x00" in text or {"", ".", ".."} & set(parts) else parts
def recorded_identity(row: Any) -> Optional[Dict[str, Any]]:
"""The identity a row records for its bytes (a measured digest, immutable or not); None for
an unmeasured listing, which states nothing about them."""
return row if isinstance(row, dict) and row.get("sha256") and row.get("measured") is not False else None
def _eligible(row: Dict[str, Any]) -> bool:
"""A row whose capture still offers bytes (an immutable one only with the identity to verify)."""
if row.get("immutable") and (not isinstance(row.get("size"), int) or not row.get("sha256")):
return False
return not row.get("errors") and not row.get("copy_status") \
and str(row.get("status") or "").strip().lower() in _SERVABLE_STATUSES
def _route(anchor: Optional[pathlib.Path], store: pathlib.Path, relpath: str) -> Optional[Route]:
segments = plain_segments(relpath)
if not segments:
return None
if anchor is not None and store != anchor and store.is_relative_to(anchor):
return anchor, [*store.relative_to(anchor).parts, *segments]
return store, segments
class _Parents:
"""One confined directory descriptor at a time; consecutive members under the same
directory reuse it instead of repeating the descent. A failed descent is not kept."""
def __init__(self) -> None:
self._key: Optional[tuple] = None
self._fd = -1
def __enter__(self) -> "_Parents":
return self
def __exit__(self, *_exc: Any) -> None:
self.close()
def get(self, root: pathlib.Path, directories: List[str]) -> int:
key = (root, tuple(directories))
if key != self._key:
self.close()
fd = os.open(root, _DIR_FLAGS)
try:
for name in directories:
child = os.open(name, _DIR_FLAGS, dir_fd=fd)
os.close(fd)
fd = child
except BaseException:
os.close(fd)
raise
self._fd, self._key = fd, key
return self._fd
def close(self) -> None:
if self._fd >= 0:
fd, self._fd, self._key = self._fd, -1, None
os.close(fd)
def _member_stat(parents: _Parents, route: Optional[Route]) -> Optional[os.stat_result]:
"""No-follow stat through the confined descent; None unless a regular file is there."""
if route is None or not CONFINED:
return None
root, segments = route
try:
observed = os.stat(segments[-1], dir_fd=parents.get(root, segments[:-1]), follow_symlinks=False)
except OSError:
return None
return observed if stat.S_ISREG(observed.st_mode) else None
def _open_member(parents: _Parents, route: Route) -> Tuple[Any, os.stat_result]:
"""A binary handle on the member itself plus the fstat of that descriptor; anything but a
regular file is refused (an OSError without errno, like a failed verification)."""
root, segments = route
fd = os.open(segments[-1], _FILE_FLAGS, dir_fd=parents.get(root, segments[:-1]))
try:
observed = os.fstat(fd)
if not stat.S_ISREG(observed.st_mode):
raise OSError(f"task file is not a regular file: {'/'.join(segments)}")
except BaseException:
os.close(fd)
raise
return os.fdopen(fd, "rb"), observed
def directory_archives(stores: List[pathlib.Path], rows: Any, *, anchor: Any = None) -> Dict[str, Dict[str, Any]]:
"""Per top-level result directory, what ``?archive=<dir>`` would stream now: ``files``/``size``
of its members from one confined no-follow stat each (no hashing), the recorded rows it leaves
out (``excluded``) and ``available`` iff it has a member (never where the platform cannot confine)."""
anchor = pathlib.Path(anchor).resolve(strict=False) if anchor is not None else None
view: Dict[str, Dict[str, Any]] = {}
picked: Dict[str, Tuple[tuple, int]] = {}
with _Parents() as parents:
for order, row in enumerate(rows if isinstance(rows, list) else []):
if not isinstance(row, dict):
continue
location = task_artifact_location(stores, row.get("path"))
relpath = location[2] if location else str(row.get("relpath") or "")
parts = plain_segments(relpath)
if len(parts) < 2:
continue # a root file (or an unattributable row) belongs to no directory
entry = view.setdefault(parts[0], {"name": parts[0] + ARCHIVE_SUFFIX, "files": 0, "size": 0,
"excluded": 0, "available": False})
observed = (_member_stat(parents, _route(anchor, stores[location[0]], relpath))
if location and _eligible(row) else None)
if observed is None:
entry["excluded"] += 1
elif relpath not in picked or (location[0], order) < picked[relpath][0]:
picked[relpath] = ((location[0], order), observed.st_size)
for relpath, (_rank, size) in picked.items():
entry = view[relpath.split("/", 1)[0]]
entry["files"] += 1
entry["size"] += size
for entry in view.values():
entry["available"] = entry["files"] > 0
return view
def _archive_members(stores: List[pathlib.Path], rows: Any, directory: str, *,
anchor: Any) -> List[Tuple[str, Route, Dict[str, Any]]]:
"""``(member name, route, row)`` of every eligible recorded file under DIRECTORY that a
confined stat finds, once per relpath (canonical store first), sorted by name."""
anchor = pathlib.Path(anchor).resolve(strict=False)
prefix, base = directory + "/", (directory.rsplit("/", 1)[0] + "/" if "/" in directory else "")
picked: Dict[str, Tuple[tuple, Route, Dict[str, Any]]] = {}
with _Parents() as parents:
for order, row in enumerate(rows if isinstance(rows, list) else []):
location = task_artifact_location(stores, row.get("path")) if isinstance(row, dict) else None
if location is None or not location[2].startswith(prefix) or not _eligible(row):
continue
route = _route(anchor, stores[location[0]], location[2])
if route is not None and _member_stat(parents, route) is not None and (
location[2] not in picked or (location[0], order) < picked[location[2]][0]):
picked[location[2]] = ((location[0], order), route, row)
return [(relpath[len(base):], route, row) for relpath, (_rank, route, row) in sorted(picked.items())]
def _zip_info(name: str, observed: os.stat_result) -> zipfile.ZipInfo:
"""What ``ZipInfo.from_file(strict_timestamps=False)`` records, from the open handle's fstat."""
date_time = time.localtime(observed.st_mtime)[:6]
date_time = max((1980, 1, 1, 0, 0, 0), min(date_time, (2107, 12, 31, 23, 59, 59)))
info = zipfile.ZipInfo(name, date_time)
info.external_attr = (observed.st_mode & 0xFFFF) << 16
info.file_size = observed.st_size
info.compress_type = zipfile.ZIP_DEFLATED
return info
def _drain(spool: Any) -> Iterator[bytes]:
try:
spool.seek(0)
yield from iter(lambda: spool.read(_CHUNK), b"")
finally:
spool.close()
def serve_directory_archive(drive_root: Any, task_id: str, name: str, directory: str, *,
other_selectors: bool) -> Response:
"""``GET /api/tasks/{task_id}/artifacts/{basename}.zip?archive=<directory>``: the recorded
directory's ZIP, or a typed refusal - 400 ``artifact_archive_invalid``, 404 ``task not found``
/ ``artifact_archive_empty`` / ``artifact_archive_unverified`` (a member changed, vanished,
became a link or failed its capture while read), 503 ``artifact_archive_unavailable`` (no
confinement on this platform, or the spool or a read failed)."""
parts = plain_segments(directory)
refusal = ("archive must name a store-relative directory in plain segments, without relpath or source"
if other_selectors or not parts else
f"archive name must be the directory's basename plus {ARCHIVE_SUFFIX}"
if name != parts[-1] + ARCHIVE_SUFFIX else "")
if refusal:
return json_error(refusal, 400, reason_code="artifact_archive_invalid", task_id=task_id, artifact=name)
unavailable = {"reason_code": "artifact_archive_unavailable", "task_id": task_id, "artifact": name,
"directory": directory}
if not CONFINED:
return json_error("archive confinement needs directory-relative no-follow opens, which this platform "
"lacks (issue #1297)", 503, **unavailable)
result = load_effective_task_result(drive_root, task_id) or {}
if not result:
return json_error("task not found", 404)
members = _archive_members(task_artifact_stores(drive_root, task_id), result.get("artifacts"), directory,
anchor=drive_root)
if not members:
return json_error("no recorded ready file of this task lies in that directory", 404,
reason_code="artifact_archive_empty", task_id=task_id, artifact=name, directory=directory)
member = ""
try:
spool = tempfile.TemporaryFile()
except OSError:
return json_error("archive could not be built", 503, member=member, **unavailable)
try:
with zipfile.ZipFile(spool, "w", compression=zipfile.ZIP_DEFLATED, allowZip64=True) as archive, \
_Parents() as parents:
for member, route, row in members:
handle, observed = _open_member(parents, route) # the handle, never the path, from here on
with handle, archive.open(_zip_info(member, observed), "w") as sink:
artifact_store.stream_artifact_file(handle, sink, expected=recorded_identity(row))
except OSError as exc:
spool.close()
if exc.errno is None or exc.errno in _MEMBER_GONE: # verification failures carry no errno
return json_error("archive member is missing, changed while read, or failed its capture verification",
404, reason_code="artifact_archive_unverified", task_id=task_id, artifact=name,
directory=directory, member=member)
return json_error("archive could not be built", 503, member=member, **unavailable)
except BaseException:
spool.close()
raise
size = spool.tell()
quoted = quote(name)
disposition = f"attachment; filename*=utf-8''{quoted}" if quoted != name else f'attachment; filename="{name}"'
return StreamingResponse(_drain(spool), media_type="application/zip",
headers={"Content-Length": str(size), "Content-Disposition": disposition})
def chat_media_identity(name: str) -> Dict[str, Any]:
"""The capture identity a content-addressed chat-media name promises."""
match = _CHAT_MEDIA_DIGEST_RE.fullmatch(str(name or ""))
if match is None:
raise ValueError(f"not a content-addressed chat media name: {name!r}")
return {"sha256": match.group(1)}
class _DescriptorResponse(Response):
"""The bytes of ONE already-open binary handle: GET/HEAD, ``Accept-Ranges`` and a single
``Range`` (206; an unsatisfiable one 416; several are served whole), headers from the fstat
of that descriptor. Never opens a path; the handle closes with the response."""
def __init__(self, handle: Any, name: str, observed: os.stat_result, verified: Optional[str] = None) -> None:
super().__init__(content=None, media_type=mimetypes.guess_type(name)[0] or "application/octet-stream")
self._handle, self._size = handle, observed.st_size
etag = hashlib.md5(f"{observed.st_mtime}-{observed.st_size}".encode(), usedforsecurity=False).hexdigest()
self.headers.update({"content-length": str(self._size), "accept-ranges": "bytes", "etag": f'"{etag}"',
"last-modified": formatdate(observed.st_mtime, usegmt=True),
"x-ouroboros-artifact-identity": "verified" if verified else "unmeasured"})
if verified:
self.headers["x-ouroboros-artifact-sha256"] = verified
def _range(self, header: str) -> Optional[Tuple[int, int]]:
spec = header.strip().lower()
if not spec.startswith("bytes=") or "," in spec:
return None
first, _, last = spec[6:].strip().partition("-")
try:
start, end = ((int(first), int(last) + 1 if last else self._size) if first
else (max(0, self._size - int(last)), self._size))
except ValueError:
return None
return (start, min(end, self._size)) if start < min(end, self._size) else (-1, -1)
async def __call__(self, scope: Any, receive: Any, send: Any) -> None:
try:
start, end, status = 0, self._size, 200
wanted = self._range(dict(scope.get("headers") or []).get(b"range", b"").decode("latin-1"))
if wanted == (-1, -1):
await Response(status_code=416, headers={"content-range": f"bytes */{self._size}"})(scope, receive, send)
return
if wanted is not None:
(start, end), status = wanted, 206
self.headers["content-range"] = f"bytes {start}-{end - 1}/{self._size}"
self.headers["content-length"] = str(end - start)
await send({"type": "http.response.start", "status": status, "headers": self.raw_headers})
if scope.get("method") == "HEAD":
await send({"type": "http.response.body", "body": b"", "more_body": False})
return
await anyio.to_thread.run_sync(self._handle.seek, start)
sent = False
while start < end:
chunk = await anyio.to_thread.run_sync(self._handle.read, min(_CHUNK, end - start))
if not chunk:
break # a short file leaves the declared length unmet, never padded
start += len(chunk)
sent = True
await send({"type": "http.response.body", "body": chunk, "more_body": start < end})
if start < end or not sent: # an empty or short body still completes the response
await send({"type": "http.response.body", "body": b"", "more_body": False})
finally:
self._handle.close()
def serve_task_file(drive_root: Any, store: pathlib.Path, relpath: str, name: str,
expected: Optional[Dict[str, Any]], *, task_id: str, mutable: bool = False) -> Response:
"""The ONE file the handler attributed to ``relpath`` of ``store``, read through the confined
descent: the descriptor response (``x-ouroboros-artifact-identity`` verified/unmeasured), else
409 ``artifact_identity_changed`` (a ``mutable`` row's bytes no longer match the digest it
records: the recorded identity rides the refusal, the row wants re-recording), 404
``artifact_unverified`` (missing, swapped for a link or FIFO, changed while read, or a capture
that did not verify) or 503 ``artifact_unavailable`` (including a platform without
confinement, issue #1297)."""
try:
if not CONFINED:
raise OSError(errno.ENOTSUP, "confined file open unavailable (issue #1297)")
route = _route(pathlib.Path(drive_root).resolve(strict=False), store, relpath)
if route is None:
raise OSError("artifact relpath is not plain segments")
with _Parents() as parents:
handle, observed = _open_member(parents, route)
if expected is not None:
with handle: # closed whatever the spool allocation or verification does
spool = tempfile.TemporaryFile()
try:
measured = artifact_store.stream_artifact_file(handle, spool, expected=expected)
if measured["size"] != observed.st_size:
raise OSError("task file changed between its open and its verification")
except BaseException:
spool.close()
raise
handle = spool
except OSError as exc:
if mutable and exc.errno is None and expected and "verification" in str(exc):
return json_error("artifact bytes no longer match the identity the task result records for this file",
409, reason_code="artifact_identity_changed", task_id=task_id, artifact=name,
recorded_sha256=str(expected.get("sha256") or ""), recorded_size=expected.get("size"))
if exc.errno is None or exc.errno in _MEMBER_GONE:
return json_error("artifact file is missing, changed while read, or failed its capture verification",
404, reason_code="artifact_unverified", task_id=task_id, artifact=name)
return json_error("artifact could not be read", 503, reason_code="artifact_unavailable",
task_id=task_id, artifact=name)
return _DescriptorResponse(handle, name, observed, str((expected or {}).get("sha256") or "") or None)

View file

@ -13,7 +13,7 @@ from datetime import datetime, timezone
from typing import Any, Dict, List, Optional
from starlette.requests import Request
from starlette.responses import FileResponse, JSONResponse, Response
from starlette.responses import JSONResponse, Response
from ouroboros.gateway._helpers import coerce_int, json_error, json_exception, request_drive_root, request_json_or, request_repo_dir, run_sync_to_completion, stage_initial_task_attachments
from ouroboros.gateway.cost_breakdown import _task_cost_breakdown_view # noqa: F401
@ -36,6 +36,11 @@ from ouroboros.gateway.task_events import ( # noqa: F401
# wiring and tests address gateway.tasks.api_task_hurry.
from ouroboros.gateway.task_hurry import api_task_hurry # noqa: F401
from ouroboros.gateway.task_decision import api_decision_answer # noqa: F401
from ouroboros.gateway.task_archive import (
chat_media_identity, directory_archives, plain_segments, serve_directory_archive, serve_task_file,
task_artifact_location, recorded_identity,
)
from ouroboros.task_custody import task_artifact_stores
from ouroboros.headless import (
ARTIFACTS_DIR,
ARTIFACT_STATUS_FAILED,
@ -125,8 +130,9 @@ def _cleanup_api_admission_attempt(
if child_drive is not None:
try:
from ouroboros.headless import remove_subagent_task_drive
remove_subagent_task_drive(drive_root, task_id)
from supervisor.queue import task_settlement_interlock, task_settlement_liveness
remove_subagent_task_drive(drive_root, task_id, live=task_settlement_liveness,
guard=task_settlement_interlock, admission_rollback=True)
except Exception:
log.warning("Failed to clean child drive for rejected task %s", task_id, exc_info=True)
try:
@ -241,8 +247,9 @@ def _admission_rejection_response(
)
if child_drive is not None:
from ouroboros.headless import remove_subagent_task_drive
removed = remove_subagent_task_drive(drive_root, task_id)
from supervisor.queue import task_settlement_interlock, task_settlement_liveness
removed = remove_subagent_task_drive(drive_root, task_id, live=task_settlement_liveness,
guard=task_settlement_interlock, admission_rollback=True)
write_task_result(
drive_root,
task_id,
@ -902,7 +909,7 @@ async def api_task_get(request: Request) -> JSONResponse:
def _task_get_response(request: Request) -> JSONResponse:
"""Materialize the complete detail and ledger projection off the HTTP loop."""
"""Project the complete detail and ledger view off the HTTP loop (a pure read)."""
try:
task_id = validate_task_id(request.path_params.get("task_id"))
except ValueError as exc:
@ -918,6 +925,9 @@ def _task_get_response(request: Request) -> JSONResponse:
pass
return json_error("task result is unavailable", 503)
payload = public_task_result(data)
if isinstance(payload.get("artifacts"), list): # what ``?archive=<dir>`` would stream now, per top-level dir
payload["artifact_archives"] = directory_archives(task_artifact_stores(drive_root, task_id),
payload["artifacts"], anchor=drive_root)
breakdown_view = _task_cost_breakdown_view(drive_root, data)
if breakdown_view is not None:
payload["cost_breakdown"] = breakdown_view
@ -925,6 +935,10 @@ def _task_get_response(request: Request) -> JSONResponse:
def api_task_artifact(request: Request):
"""Serve one task file read-only from its canonical or OWN child store (``task_archive``): a bare
name selects only a top-level file (several nested matches: 409 ``artifact_name_ambiguous`` naming
their ``relpaths``), ``?relpath=a/b/{name}`` is exact, ``?archive=<dir>`` with ``{name}`` =
``<basename>.zip`` streams a recorded directory; bytes leave only through ``serve_task_file``."""
try:
task_id = validate_task_id(request.path_params.get("task_id"))
except ValueError as exc:
@ -932,45 +946,50 @@ def api_task_artifact(request: Request):
name = str(request.path_params.get("name") or "").strip()
if not name or "/" in name or "\\" in name or name in {".", ".."} or ".." in pathlib.PurePosixPath(name).parts:
return json_error("artifact name must be a simple filename", 400)
source, relpath = request.query_params.get("source"), request.query_params.get("relpath")
if relpath is not None and (source or plain_segments(relpath)[-1:] != [name]):
return json_error("relpath must be store-relative plain segments ending in the name, without source",
400, reason_code="artifact_relpath_invalid", task_id=task_id, artifact=name)
drive_root = request_drive_root(request)
path = artifact_store.resolve_chat_media_path(drive_root, task_id, name)
if path is None:
registered = artifact_store.registered_task_artifact(drive_root, task_id, name)
source = request.query_params.get("source")
# Registered immutable bytes need one identity check below, not a
# materialization/hash of the whole result before that same check.
result = (load_effective_task_result(drive_root, task_id) or {}
if source or not registered or not registered.get("immutable") else {})
if not result and not registered:
return json_error("task not found", 404)
if source:
try:
return Response(artifact_store.read_task_result_source_bytes(drive_root, result, name, source), media_type="application/json")
except (OSError, ValueError, RuntimeError):
return json_error("task source is unavailable or does not match its recorded identity", 404)
artifact = registered if registered and registered.get("immutable") else next(
(row for row in result.get("artifacts") or []
if isinstance(row, dict)
and str(row.get("name") or pathlib.Path(str(row.get("path") or "")).name) == name),
None) or registered
if artifact is None:
return json_error("artifact not found", 404, task_id=task_id, artifact=name)
base = task_artifacts_dir(drive_root, task_id).resolve(strict=False)
path = pathlib.Path(str(artifact.get("path") or "")).resolve(strict=False)
if path.name != name:
return json_error("artifact metadata path does not match requested name", 500)
if (archive := request.query_params.get("archive")) is not None:
return serve_directory_archive(drive_root, task_id, name, archive,
other_selectors=source is not None or relpath is not None)
stores = task_artifact_stores(drive_root, task_id)
path = None if relpath is not None else artifact_store.resolve_chat_media_path(drive_root, task_id, name)
if path is not None: # content-addressed chat media: the canonical store, bytes that hash to the name
media = task_artifact_location(stores[:1], path)
return (serve_task_file(drive_root, stores[0], media[2], name, chat_media_identity(name), task_id=task_id)
if media else json_error("artifact not found", 404, task_id=task_id, artifact=name))
registered = artifact_store.registered_task_artifact(drive_root, task_id, name) if relpath in (None, name) else None
# A registered immutable top-level file needs one identity check, not a result projection.
fast = bool(not source and registered and registered.get("immutable")
and (at := task_artifact_location(stores, registered.get("path"))) and at[2] == name)
result = {} if fast else (load_effective_task_result(drive_root, task_id) or {})
if not result and not registered:
return json_error("task not found", 404)
if source:
try:
path.relative_to(base)
except ValueError:
return Response(artifact_store.read_task_result_source_bytes(drive_root, result, name, source), media_type="application/json")
except (OSError, ValueError, RuntimeError):
return json_error("task source is unavailable or does not match its recorded identity", 404)
rows = [] if fast else [row for row in result.get("artifacts") or [] if isinstance(row, dict) and (
relpath is not None or str(row.get("name") or pathlib.Path(str(row.get("path") or "")).name) == name)]
located = [(at, order, row) for order, row in enumerate(rows + ([registered] if registered else []))
if (at := task_artifact_location(stores, row.get("path")))]
matches = [item for item in located if item[0][2] == (relpath or name)]
nested = sorted({item[0][2] for item in located if "/" in item[0][2]})
if relpath is None and not matches and len(nested) > 1:
return json_error("artifact name matches several nested files; select one with ?relpath=", 409,
reason_code="artifact_name_ambiguous", task_id=task_id, artifact=name, relpaths=nested)
if not matches:
if relpath is None and any(at[2].rsplit("/", 1)[-1] != name for at, _order, _row in located):
return json_error("artifact metadata path does not match requested name", 500)
if (rows or registered) and not located and relpath is None:
return json_error("artifact path is outside task artifact directory", 500)
if not path.is_file():
return json_error("artifact file is missing", 404, task_id=task_id, artifact=name)
if artifact.get("immutable"):
try:
artifact_store.stream_artifact_file(path, expected=artifact)
except OSError:
return json_error("captured artifact failed byte verification", 404)
return FileResponse(path)
return json_error("artifact not found", 404, task_id=task_id, artifact=name)
(index, _path, relative), _order, artifact = min(matches, key=lambda item: (item[0][0], item[1]))
return serve_task_file(drive_root, stores[index], relative, name, recorded_identity(artifact), task_id=task_id,
mutable=not artifact.get("immutable"))
def _record_cascade_incident(task_id: str, kind: str, detail: str = "") -> None:

View file

@ -193,32 +193,64 @@ def _effective_task_result(parent: pathlib.Path, task_id: str) -> Dict[str, Any]
return load_task_result(parent, task_id) or {}
def _live_unpromoted_child_refs(
promotion: Any, expected_child: pathlib.Path,
) -> List[Dict[str, Any]]:
if not isinstance(promotion, dict):
return []
try:
version = int(promotion.get("schema_version") or 0)
except (TypeError, ValueError):
return []
if version < 1:
return []
if str(promotion.get("status") or "") == "complete":
return []
child_root = expected_child.resolve(strict=False)
live: List[Dict[str, Any]] = []
for row in promotion.get("pending_refs") or []:
if not isinstance(row, dict) or not row.get("path"):
continue
# How many drive settlements one off-loop pass attempts before yielding (each may copy and
# hash a whole child store); the caller carries the cursor so later passes continue.
DRIVE_SETTLEMENTS_PER_PASS = 16
def _prompt_settlement(result: Dict[str, Any]) -> bool:
"""A cancelled subagent's drive settles without waiting out retention: the cancel path
used to delete it at once and now leaves that work to the off-loop pass."""
return (str(result.get("status") or "").lower() == "cancelled"
and str(result.get("delegation_role") or "") == "subagent")
def _prune_drives(base: pathlib.Path, parent: pathlib.Path, *, drive_of: Any, not_terminal: str,
retention_days: Optional[int], now: Optional[float], live: Any, guard: Any, stop: Any,
budget: Optional[int], after: str, extra_checks: Any) -> Dict[str, Any]:
"""The prune both drive layouts share: candidates from the DURABLE canonical row (a projection is not
custody), settled through ``task_custody.settle_child_drive`` in name order after ``after`` (wrapping), at
most ``budget`` attempts per call; ``cursor`` is the last attempted drive, ``deferred`` the unreached rest."""
from ouroboros.retention import age_cutoff
from ouroboros.task_custody import settle_child_drive
days = _resolve_retention_days(retention_days)
cutoff = age_cutoff(days, now)
report: Dict[str, Any] = {"retention_days": days, "scanned": 0, "pruned": [], "skipped": [], "errors": [],
"deferred": [], "cursor": after}
if not base.is_dir():
return report
names = sorted(entry.name for entry in base.iterdir() if entry.is_dir())
names = [name for name in names if name > after] + [name for name in names if name <= after]
attempts = 0
for task_id in names:
task_dir = base / task_id
report["scanned"] += 1
try:
path = pathlib.Path(str(row["path"])).resolve(strict=False)
path.relative_to(child_root)
except (OSError, ValueError):
continue
if path.is_file():
live.append(dict(row))
return live
validate_task_id(task_id)
result = load_task_result(parent, task_id, strict=True) or {}
status = str(result.get("status") or "").lower()
if status not in _FINAL_STATUSES:
report["skipped"].append({"task_id": task_id, "reason": not_terminal, "status": status})
continue
if not _prompt_settlement(result) and _timestamp_from_result(result, task_dir.stat().st_mtime) > cutoff:
report["skipped"].append({"task_id": task_id, "reason": "younger_than_retention"})
continue
skip = extra_checks(task_dir, result)
if skip is not None:
report["skipped"].append({"task_id": task_id, **skip})
continue
if (budget is not None and attempts >= budget) or (stop is not None and stop()):
report["deferred"].append(task_id)
continue
attempts += 1
report["cursor"] = task_id
if settle_child_drive(parent, task_id, drive_of(task_dir), live=live, guard=guard, stop=stop,
report=report)["status"] == "removed":
report["pruned"].append({"task_id": task_id, "path": str(task_dir)})
except Exception as exc:
report["errors"].append({"task_id": task_id, "error": f"{type(exc).__name__}: {exc}"})
return report
def prune_headless_task_drives(
@ -226,74 +258,30 @@ def prune_headless_task_drives(
*,
retention_days: Optional[int] = None,
now: Optional[float] = None,
live: Any = None,
guard: Any = None,
stop: Any = None,
budget: Optional[int] = None,
after: str = "",
) -> Dict[str, Any]:
"""Best-effort startup prune for copied-back terminal child drives."""
from ouroboros.retention import age_cutoff
"""Prune terminal child drives past retention (a cancelled subagent's at once). Removal is
``task_custody.settle_child_drive``'s decision alone: ``live`` is the supervisor's probe (without it nothing
is removed), ``guard`` its interlock, ``stop`` the generation's close, ``budget``/``after`` the pass bound/cursor."""
parent = pathlib.Path(parent_drive_root)
base = parent / HEADLESS_TASKS_DIR
days = _resolve_retention_days(retention_days)
cutoff = age_cutoff(days, now)
report: Dict[str, Any] = {
"retention_days": days,
"scanned": 0,
"pruned": [],
"skipped": [],
"errors": [],
"promotion_retry": {},
}
if not base.is_dir():
return report
from ouroboros.observability import retry_pending_child_ref_promotions
report["promotion_retry"] = retry_pending_child_ref_promotions(parent)
for task_dir in sorted(base.iterdir()):
if not task_dir.is_dir():
continue
task_id = task_dir.name
report["scanned"] += 1
try:
validate_task_id(task_id)
dir_mtime = task_dir.stat().st_mtime
result = _effective_task_result(parent, task_id)
status = str(result.get("status") or "").lower()
if status not in _FINAL_STATUSES:
report["skipped"].append({"task_id": task_id, "reason": "parent_not_terminal", "status": status})
continue
artifact_status = str(result.get("artifact_status") or "").lower()
if artifact_status and artifact_status not in ARTIFACT_TERMINAL_STATUSES:
report["skipped"].append({"task_id": task_id, "reason": "artifacts_not_terminal", "artifact_status": artifact_status})
continue
retention_ts = _timestamp_from_result(result, dir_mtime)
if retention_ts > cutoff:
report["skipped"].append({"task_id": task_id, "reason": "younger_than_retention"})
continue
expected_child = str((task_dir / "data").resolve(strict=False))
known_child = str(
result.get("child_drive_root")
or result.get("headless_child_drive_root")
or result.get("drive_root")
or ""
).strip()
if known_child and str(pathlib.Path(known_child).resolve(strict=False)) != expected_child:
report["skipped"].append({"task_id": task_id, "reason": "child_drive_mismatch"})
continue
live_unpromoted = _live_unpromoted_child_refs(
result.get("child_ref_promotion"), pathlib.Path(expected_child),
)
if live_unpromoted:
report["skipped"].append({
"task_id": task_id,
"reason": "child_refs_unpromoted",
"pending_ref_count": len(live_unpromoted),
})
continue
shutil.rmtree(task_dir)
report["pruned"].append({"task_id": task_id, "path": str(task_dir)})
except Exception as exc:
report["errors"].append({"task_id": task_id, "error": f"{type(exc).__name__}: {exc}"})
return report
def checks(task_dir: pathlib.Path, result: Dict[str, Any]) -> Optional[Dict[str, Any]]:
artifact_status = str(result.get("artifact_status") or "").lower()
if artifact_status and artifact_status not in ARTIFACT_TERMINAL_STATUSES:
return {"reason": "artifacts_not_terminal", "artifact_status": artifact_status}
known_child = str(result.get("child_drive_root") or result.get("headless_child_drive_root")
or result.get("drive_root") or "").strip()
if known_child and pathlib.Path(known_child).resolve(strict=False) != (task_dir / "data").resolve(strict=False):
return {"reason": "child_drive_mismatch"}
return None
return _prune_drives(parent / HEADLESS_TASKS_DIR, parent, drive_of=lambda task_dir: task_dir / "data",
not_terminal="parent_not_terminal", retention_days=retention_days, now=now, live=live,
guard=guard, stop=stop, budget=budget, after=after, extra_checks=checks)
def prune_task_drives(
@ -301,44 +289,18 @@ def prune_task_drives(
*,
retention_days: Optional[int] = None,
now: Optional[float] = None,
live: Any = None,
guard: Any = None,
stop: Any = None,
budget: Optional[int] = None,
after: str = "",
) -> Dict[str, Any]:
"""Best-effort startup prune for direct-task scratch drives."""
from ouroboros.retention import age_cutoff
"""Prune direct-task scratch drives past retention (``settle_child_drive`` decides;
same knobs as ``prune_headless_task_drives``)."""
parent = pathlib.Path(parent_drive_root)
base = parent / TASK_DRIVES_DIR
days = _resolve_retention_days(retention_days)
cutoff = age_cutoff(days, now)
report: Dict[str, Any] = {"retention_days": days, "scanned": 0, "pruned": [], "skipped": [], "errors": []}
if not base.is_dir():
return report
from ouroboros.observability import retry_pending_child_ref_promotions
retry_pending_child_ref_promotions(parent)
for task_dir in sorted(base.iterdir()):
if not task_dir.is_dir():
continue
task_id = task_dir.name
report["scanned"] += 1
try:
validate_task_id(task_id)
dir_mtime = task_dir.stat().st_mtime
result = _effective_task_result(parent, task_id)
status = str(result.get("status") or "").lower()
if status not in _FINAL_STATUSES:
report["skipped"].append({"task_id": task_id, "reason": "task_not_terminal", "status": status})
continue
if _live_unpromoted_child_refs(result.get("child_ref_promotion"), task_dir):
report["skipped"].append({"task_id": task_id, "reason": "unpromoted_child_refs"})
continue
if _timestamp_from_result(result, dir_mtime) > cutoff:
report["skipped"].append({"task_id": task_id, "reason": "younger_than_retention"})
continue
shutil.rmtree(task_dir)
report["pruned"].append({"task_id": task_id, "path": str(task_dir)})
except Exception as exc:
report["errors"].append({"task_id": task_id, "error": f"{type(exc).__name__}: {exc}"})
return report
return _prune_drives(parent / TASK_DRIVES_DIR, parent, drive_of=lambda task_dir: task_dir,
not_terminal="task_not_terminal", retention_days=retention_days, now=now, live=live,
guard=guard, stop=stop, budget=budget, after=after, extra_checks=lambda *_a: None)
def prune_task_trees(
@ -381,44 +343,32 @@ def prune_task_trees(
return report
def remove_subagent_task_drive(parent_drive_root: pathlib.Path, task_id: str) -> bool:
"""Remove scratch after settled copyback/salvage, preserving completion wins.
Pending live refs retain their source. Returns whether any drive was removed.
"""
parent = pathlib.Path(parent_drive_root)
def remove_subagent_task_drive(parent_drive_root: pathlib.Path, task_id: str, *, live: Any = None,
guard: Any = None, admission_rollback: bool = False) -> bool:
"""Remove TASK's own drives through ``task_custody.settle_child_drive`` (the one deletion owner):
``live`` is the supervisor's probe, ``guard`` its interlock, ``admission_rollback`` frees a never-started drive."""
from ouroboros.task_custody import own_child_drives, settle_child_drive
try:
validate_task_id(task_id)
drives = own_child_drives(parent_drive_root, validate_task_id(task_id))
except Exception:
return False
headless_base = parent / HEADLESS_TASKS_DIR / task_id
task_drive_base = parent / TASK_DRIVES_DIR / task_id
try:
result = load_task_result(parent, task_id) or {}
promotion = result.get("child_ref_promotion")
if (
_live_unpromoted_child_refs(promotion, headless_base / "data")
or _live_unpromoted_child_refs(promotion, task_drive_base)
):
return False
except Exception:
# Legacy/no-metadata cleanup behavior remains fail-soft. Newly produced
# valid metadata is parsed by the helper without raising.
pass
bases = (headless_base, task_drive_base)
removed = False
for base in bases:
try:
if base.is_dir():
shutil.rmtree(base)
removed = True
except Exception:
log.debug("Failed to remove subagent task drive %s", base, exc_info=True)
return removed
return any([settle_child_drive(parent_drive_root, task_id, drive, live=live, guard=guard,
admission_rollback=admission_rollback)["status"] == "removed"
for drive in drives])
# How long a publisher waits for the task's custody lock before answering ``CustodyBusy``.
PUBLICATION_LOCK_SEC = 30.0
def copy_child_task_result(parent_drive_root: pathlib.Path, task: Dict[str, Any]) -> Optional[Dict[str, Any]]:
"""Copy a child-drive task result back to the parent data root."""
"""Copy a child-drive task result back to the parent data root under the task's custody lock
(``task_custody.task_custody_lock``) from the child read to the row write, so a settlement never moves
the drive under a copy in flight; a settled (gone) drive publishes nothing; a busy lock is ``CustodyBusy``."""
from ouroboros.observability import child_ref_promotion_scope
from ouroboros.task_custody import CustodyBusy, task_custody_lock
with child_ref_promotion_scope():
task_id = str(task.get("id") or "")
if not task_id:
@ -430,52 +380,76 @@ def copy_child_task_result(parent_drive_root: pathlib.Path, task: Dict[str, Any]
child_drive = _child_drive_from_task(task)
if child_drive is None:
return None
child_result = load_task_result(child_drive, task_id)
if not isinstance(child_result, dict):
return None
# Bulk refs precede publication/GC; review refs first select CURRENT below.
from ouroboros.observability import promote_child_task_refs
with task_custody_lock(parent_drive_root, task_id, timeout_sec=PUBLICATION_LOCK_SEC) as locked:
if not locked:
raise CustodyBusy(f"custody lock of {task_id} is held by another publisher")
canonical_existing = load_task_result(parent_drive_root, task_id) or {}
if cancellation_blocks_child_result(canonical_existing):
return canonical_existing # cancelled while this publisher waited: the child's authority is declined
return _copy_child_task_result_locked(parent_drive_root, task, task_id, canonical_existing, child_drive)
review_fields = {key: value for key, value in child_result.items() if key == "review_projection"}
child_result, ref_promotion = promote_child_task_refs(
pathlib.Path(parent_drive_root), child_drive, task_id,
{key: value for key, value in child_result.items() if key != "review_projection"})
child_result.update(review_fields)
_publish_child_verification_receipts(parent_drive_root, task_id, child_drive)
child_status = str(child_result.pop("status", None) or "completed")
child_result.pop("task_id", None)
child_result["child_ref_promotion"] = ref_promotion
if isinstance(child_result.get("artifacts"), list):
try:
from ouroboros.outcomes import artifact_bundle_from_result
child_result["artifact_bundle"] = artifact_bundle_from_result(child_result)
except Exception:
child_result.pop("artifact_bundle", None)
child_result.setdefault("headless_child_drive_root", str(child_drive))
if (child_status in _FINAL_STATUSES and _workspace_root_from_task(task) is not None
and not task_is_readonly_subagent(task)):
artifact_status = str(canonical_existing.get("artifact_status") or "").strip().lower()
if artifact_status in ARTIFACT_TERMINAL_STATUSES | {ARTIFACT_STATUS_PENDING, ARTIFACT_STATUS_FINALIZING}:
child_result["artifacts"] = _merge_artifacts(
list(canonical_existing.get("artifacts") or []), list(child_result.get("artifacts") or []))
child_result.update({key: canonical_existing[key] for key in _ARTIFACT_LIFECYCLE_FIELDS
if key in canonical_existing})
else:
child_result["artifact_status"] = ARTIFACT_STATUS_FINALIZING
child_result["child_status"] = child_status
return retry_child_task_refs(parent_drive_root, child_drive, task_id,
replica={**child_result, "status": child_status})
def _copy_child_task_result_locked(parent_drive_root: pathlib.Path, task: Dict[str, Any], task_id: str,
canonical_existing: Dict[str, Any], child_drive: pathlib.Path) -> Optional[Dict[str, Any]]:
child_result = load_task_result(child_drive, task_id)
if not isinstance(child_result, dict):
return None
# Bulk refs precede publication/GC; review refs first select CURRENT below.
from ouroboros.observability import promote_child_task_refs
review_fields = {key: value for key, value in child_result.items() if key == "review_projection"}
child_result, ref_promotion = promote_child_task_refs(
pathlib.Path(parent_drive_root), child_drive, task_id,
{key: value for key, value in child_result.items() if key != "review_projection"})
child_result.update(review_fields)
_publish_child_verification_receipts(parent_drive_root, task_id, child_drive)
child_status = str(child_result.pop("status", None) or "completed")
child_result.pop("task_id", None)
child_result["child_ref_promotion"] = ref_promotion
if isinstance(child_result.get("artifacts"), list):
try:
from ouroboros.outcomes import artifact_bundle_from_result
child_result["artifact_bundle"] = artifact_bundle_from_result(child_result)
except Exception:
child_result.pop("artifact_bundle", None)
child_result.setdefault("headless_child_drive_root", str(child_drive))
if (child_status in _FINAL_STATUSES and _workspace_root_from_task(task) is not None
and not task_is_readonly_subagent(task)):
artifact_status = str(canonical_existing.get("artifact_status") or "").strip().lower()
if artifact_status in ARTIFACT_TERMINAL_STATUSES | {ARTIFACT_STATUS_PENDING, ARTIFACT_STATUS_FINALIZING}:
child_result["artifacts"] = _merge_artifacts(
list(canonical_existing.get("artifacts") or []), list(child_result.get("artifacts") or []))
child_result.update({key: canonical_existing[key] for key in _ARTIFACT_LIFECYCLE_FIELDS
if key in canonical_existing})
else:
child_result["artifact_status"] = ARTIFACT_STATUS_FINALIZING
child_result["child_status"] = child_status
return _retry_child_task_refs_locked(parent_drive_root, child_drive, task_id,
replica={**child_result, "status": child_status})
class _GenerationClosed(Exception):
"""The maintenance generation closed before this publication's commit."""
def retry_child_task_refs(parent: pathlib.Path, child: pathlib.Path, task_id: str,
*, replica: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
"""One optimistic publisher for pending CURRENT refs and prepared copyback.
*, replica: Optional[Dict[str, Any]] = None, stop: Any = None) -> Dict[str, Any]:
"""One optimistic publisher for pending CURRENT refs and prepared copyback, under the task's custody
lock (``CustodyBusy`` when held). ``stop()`` fences every file it promotes (``publication_fence``) and is
re-asked at the commit: a closed generation starts no further write and returns CURRENT."""
from ouroboros.task_custody import CustodyBusy, publication_fence, task_custody_lock
Normal copyback supplies its already-copied replica; only its selected review
still needs I/O. Retry has no replica and never reads an old child body. A
changed CURRENT basis repeats preparation under the same file-I/O memo.
"""
with task_custody_lock(parent, task_id, timeout_sec=PUBLICATION_LOCK_SEC) as locked, publication_fence(stop):
if not locked:
raise CustodyBusy(f"custody lock of {task_id} is held by another publisher")
return _retry_child_task_refs_locked(parent, child, task_id, replica=replica, stop=stop)
def _retry_child_task_refs_locked(parent: pathlib.Path, child: pathlib.Path, task_id: str,
*, replica: Optional[Dict[str, Any]] = None, stop: Any = None) -> Dict[str, Any]:
"""Normal copyback supplies its already-copied replica (only its selected review needs I/O); retry has
no replica and never reads an old child body; a changed CURRENT basis repeats preparation."""
from ouroboros.observability import (
_has_pending_ref_promotion, _rewrite_child_ref_tree,
child_ref_promotion_scope, promote_child_task_ref_patch,
@ -483,6 +457,8 @@ def retry_child_task_refs(parent: pathlib.Path, child: pathlib.Path, task_id: st
with child_ref_promotion_scope():
while True:
source = load_task_result(parent, task_id, strict=True) or {}
if stop is not None and stop():
return source # a closed generation starts no file promotion, not only no commit
if replica is None and not source:
raise ValueError("pending child-ref authority is missing")
if replica is None:
@ -502,10 +478,19 @@ def retry_child_task_refs(parent: pathlib.Path, child: pathlib.Path, task_id: st
patch["review_projection"] = prepared
basis = {"review_projection": review}
def project(current: dict, _incoming: dict) -> dict:
def project(current: dict, _incoming: dict) -> Optional[dict]:
if stop is not None and stop():
raise _GenerationClosed() # re-asked at the commit, not only before the walk
if replica is not None and cancellation_blocks_child_result(current):
return None # cancelled under the row lock: the child enriches nothing, not even by replica
selected = {**current, **project_replica_task_result_fields(current, replica)} if replica is not None else current
if {key: selected.get(key) for key in basis} != basis:
raise _RefPublicationChanged()
if replica is not None and current.get("status") in _FINAL_STATUSES \
and selected.get("status") != current["status"]:
# A settled canonical row keeps its outcome; the child's is child_status.
selected = {**selected, **{key: current[key] for key in ("status", "result", "error", "ts")
if key in current}, "child_status": replica.get("status")}
return {**(selected if replica is not None else {}), **patch, "status": selected["status"]}
try:
@ -514,6 +499,8 @@ def retry_child_task_refs(parent: pathlib.Path, child: pathlib.Path, task_id: st
**{key: value for key, value in (replica or {}).items() if key != "status"})
except _RefPublicationChanged:
continue
except _GenerationClosed:
return source
def _child_result_adopted(child: pathlib.Path, result: Dict[str, Any]) -> bool:
@ -561,6 +548,8 @@ def prepare_terminal_task_files(canonical_root: pathlib.Path, task: Dict[str, An
"""
task_id = str(task.get("id") or task.get("task_id") or "")
report: Dict[str, Any] = {"task_id": task_id, "result": None, "error": "", "terminal_source_present": None}
from ouroboros.task_custody import CustodyBusy
try:
root = pathlib.Path(canonical_root)
current = load_task_result(root, task_id, strict=True) or {}
@ -582,6 +571,11 @@ def prepare_terminal_task_files(canonical_root: pathlib.Path, task: Dict[str, An
if not task_is_readonly_subagent(task) and current.get("artifact_status") not in ARTIFACT_TERMINAL_STATUSES:
finalize_task_artifacts(root, {**task, "id": task_id})
report["result"] = load_task_result(root, task_id, strict=True)
except CustodyBusy as exc:
# Another publisher (a settlement or copy-back) holds the store: the next attempt
# completes the same work; nothing failed and no failure is stamped.
report["error"] = f"CustodyBusy: {exc}"
log.warning("Terminal file preparation for %s waits: %s", task_id, exc)
except Exception as exc:
from ouroboros.observability import redact_projection
report["error"] = str(redact_projection(f"{type(exc).__name__}: {exc}").value)
@ -630,11 +624,16 @@ def _copy_child_artifacts_to_parent(
artifacts: List[Dict[str, Any]],
*, promotion: Dict[str, Any] | None = None,
) -> List[Dict[str, Any]]:
"""Verify/rebase files; failed copy-back shares the existing ref custody/GC."""
from ouroboros.artifacts import copy_artifact_file
"""Copy-back's file publication: a child-store file keeps its store relpath (a nested
row gains ``relpath``); an immutable capture publishes only its recorded bytes and never
replaces different canonical bytes, a mutable one publishes its current bytes after the
differing prior copy is versioned; a failed copy keeps its row plus a pending ref."""
from ouroboros.artifacts import _archive_previous_artifact_version, copy_artifact_file, stream_artifact_file
from ouroboros.outcome_receipt_store import is_verification_receipts_path
parent_dir = task_artifacts_dir(parent_drive_root, task_id)
parent_base = parent_dir.resolve(strict=False)
child_base = task_artifacts_dir(child_drive, task_id, create=False).resolve(strict=False)
rebased: List[Dict[str, Any]] = []
for artifact in artifacts:
item = dict(artifact)
@ -643,27 +642,25 @@ def _copy_child_artifacts_to_parent(
rebased.append(item)
continue
src = pathlib.Path(raw_path)
if not src.is_absolute():
src = (child_drive / raw_path).resolve(strict=False)
src = (src if src.is_absolute() else child_drive / raw_path).resolve(strict=False)
if is_verification_receipts_path(child_drive, task_id, src):
# Receipt union has its own locked writer; never replace its rows.
continue
try:
src.resolve(strict=False).relative_to(parent_dir.resolve(strict=False))
dest = src.resolve(strict=False)
except ValueError:
dest = parent_dir / src.name
if dest.exists() and dest.resolve(strict=False) != src.resolve(strict=False):
from ouroboros.artifacts import stream_artifact_file
expected = item if item.get("immutable") else None
if src.is_relative_to(parent_base):
dest = src
else:
dest = parent_dir / (src.relative_to(child_base) if src.is_relative_to(child_base) else src.name)
if expected is not None and dest.exists() and dest.resolve(strict=False) != src:
try:
if not item.get("sha256"):
raise OSError("legacy artifact has no captured digest")
stream_artifact_file(dest, expected=item)
src = dest # Exact canonical bytes already survive this copy-back.
except OSError:
dest = parent_dir / f"{src.stem}_{sha256(str(src).encode('utf-8')).hexdigest()[:8]}{src.suffix}"
dest = dest.with_name(f"{src.stem}_{sha256(str(src).encode('utf-8')).hexdigest()[:8]}{src.suffix}")
try:
measured = copy_artifact_file(src, dest, expected=item)
if expected is None and dest != src and dest.is_file() and not dest.is_symlink():
_archive_previous_artifact_version(pathlib.Path(parent_drive_root), task_id, dest, src)
measured = copy_artifact_file(src, dest, expected=expected)
except OSError as exc:
item.update(copy_status="failed", copy_error=f"{type(exc).__name__}: {exc}")
if promotion is not None:
@ -674,6 +671,9 @@ def _copy_child_artifacts_to_parent(
item.pop("copy_status", None)
item.pop("copy_error", None)
item.update(path=str(dest), name=str(item.get("name") or dest.name), **measured)
relpath = dest.resolve(strict=False).relative_to(parent_base).as_posix() \
if dest.resolve(strict=False).is_relative_to(parent_base) else ""
item.update({"relpath": relpath} if "/" in relpath else {})
rebased.append(item)
return rebased
@ -735,13 +735,22 @@ def _file_artifact(kind: str, path: pathlib.Path, **facts: Any) -> Dict[str, Any
def finalize_task_artifacts(parent_drive_root: pathlib.Path, task: Dict[str, Any]) -> List[Dict[str, Any]]:
"""Write patch/memory-export artifacts for a completed headless task."""
"""Write patch/memory-export artifacts for a completed headless task under the task's custody lock, the
one lock every canonical-store publisher takes (copy-back, ref retry, settlement, mailbox cleanup), so no
two publishers place or list one task's files at once; a held lock is ``CustodyBusy`` (nothing failed)."""
from ouroboros.task_custody import CustodyBusy, task_custody_lock
artifacts: List[Dict[str, Any]] = []
task_id = str(task.get("id") or "")
if not task_id:
return artifacts
return []
with task_custody_lock(parent_drive_root, task_id, timeout_sec=PUBLICATION_LOCK_SEC) as locked:
if not locked:
raise CustodyBusy(f"custody lock of {task_id} is held by another publisher")
return _finalize_task_artifacts_locked(parent_drive_root, task, task_id)
def _finalize_task_artifacts_locked(parent_drive_root: pathlib.Path, task: Dict[str, Any], task_id: str) -> List[Dict[str, Any]]:
artifacts: List[Dict[str, Any]] = []
existing = load_task_result(parent_drive_root, task_id) or {}
# A cancellation latch wins before artifact creation or surviving-root reads.
if cancellation_blocks_child_result(existing):

View file

@ -73,6 +73,9 @@ def _memo_publish(key: tuple, payload: Any, writer: Callable[[], dict], *, ref_r
return dict(cached[1])
except OSError:
pass
from ouroboros.task_custody import fence_publication
fence_publication() # every promotion write goes through here: a closed generation starts none
ref = writer()
if memo is not None:
memo[key] = (copy.deepcopy(payload), dict(ref), _file_identity(ref_root / ref["path"]))
@ -1041,38 +1044,34 @@ def _has_pending_ref_promotion(promotion: Any) -> bool:
)
def _retry_pending_child_ref_promotion(
parent: pathlib.Path,
child: pathlib.Path,
task_id: str,
loaded_result: Dict[str, Any],
) -> Dict[str, Any]:
"""Retry CURRENT refs through the headless publication owner, then cleanup."""
def _retry_pending_child_ref_promotion(parent: pathlib.Path, child: pathlib.Path, task_id: str,
loaded_result: Dict[str, Any], *, stop: Any = None) -> Dict[str, Any]:
"""Retry CURRENT refs through the publication owner, then the off-loop mailbox cleanup
(it may carry inputs); a closed generation declined the publication and cleans nothing."""
from ouroboros.headless import retry_child_task_refs
settled = retry_child_task_refs(parent, child, task_id)
settled = retry_child_task_refs(parent, child, task_id, stop=stop)
if stop is not None and stop():
return settled
from supervisor.terminal_delivery import cleanup_settled_owner_mailbox
cleanup_settled_owner_mailbox(parent, task_id, {"drive_root": str(child)})
cleanup_settled_owner_mailbox(parent, task_id, {"drive_root": str(child)}, carry_inputs=True, stop=stop)
return settled
def retry_pending_child_ref_promotions(
parent_drive_root: pathlib.Path,
*, stop: Any = None,
) -> Dict[str, Any]:
"""Retry only newly ledgered pending refs, never the stale child result."""
"""Retry only newly ledgered pending refs, never the stale child result. ``stop()`` is
the maintenance generation's close, asked before every item and again at each
publication's commit: a closed generation leaves the rest ``deferred``."""
from ouroboros.headless import HEADLESS_TASKS_DIR, TASK_DRIVES_DIR
from ouroboros.task_status import SETTLED_STATUSES
from ouroboros.task_results import load_task_result, validate_task_id
parent = pathlib.Path(parent_drive_root)
report: Dict[str, Any] = {
"scanned": 0,
"retried": [],
"completed": [],
"pending": [],
"errors": [],
}
report: Dict[str, Any] = {"scanned": 0, "retried": [], "completed": [], "pending": [], "errors": [], "deferred": []}
directories = [(path, path / suffix) for base, suffix in
((parent / HEADLESS_TASKS_DIR, "data"), (parent / TASK_DRIVES_DIR, ""))
if base.is_dir() for path in sorted(base.iterdir()) if path.is_dir()]
@ -1086,8 +1085,11 @@ def retry_pending_child_ref_promotions(
continue
if not _has_pending_ref_promotion(result.get("child_ref_promotion")):
continue
if stop is not None and stop():
report["deferred"].append(task_id)
continue
settled = _retry_pending_child_ref_promotion(
parent, child_root, task_id, result
parent, child_root, task_id, result, stop=stop,
)
report["retried"].append(task_id)
promotion = settled.get("child_ref_promotion") or {}
@ -1467,7 +1469,7 @@ def preserve_salvaged_output(preserve_root: pathlib.Path, task_id: str, text: st
"""Write the FULL salvaged text durably under ``preserve_root``; return its path.
The observability root is the drive's durable forensic area
(``prune_observability_blobs`` deliberately never deletes it), so a copy
(nothing deletes it), so a copy
landed here survives the child-drive removal that follows a cancel/timeout
publication. Returns "" when nothing could be written.
"""
@ -1545,42 +1547,6 @@ def salvaged_output_note(
return f"\n\n{label}):\n" + salvaged
def prune_observability_blobs(drive_root: pathlib.Path) -> Dict[str, Any]:
"""Startup observability census — counts only, never deletion.
Forensic call manifests and CAS blobs are durable replay evidence,
preserved indefinitely BY CONTRACT. The retirable half of this surface —
``OUROBOROS_OBSERVABILITY_RETENTION_DAYS``, a knob that was parsed,
clamped and reported while deleting nothing — is GONE (CPL4-C22, owner
7A): a documented no-op was a misleading operator surface. The key sits
in ``RETIRED_SETTING_KEYS`` so stored ghosts drop on settings load.
"""
root = pathlib.Path(drive_root) / OBSERVABILITY_DIR
calls_root = root / "calls"
blobs_root = root / "blobs"
report: Dict[str, Any] = {
"preserved_indefinitely": True,
"manifest_count": 0,
"blob_count": 0,
"errors": [],
}
if not root.exists():
return report
for manifest_path in list(calls_root.glob("*/*.json")) if calls_root.exists() else []:
try:
manifest_path.stat()
report["manifest_count"] += 1
except Exception as exc:
report["errors"].append(f"{manifest_path}: {type(exc).__name__}: {exc}")
if blobs_root.exists():
for blob_path in list(blobs_root.glob("*.gz")):
try:
blob_path.stat()
report["blob_count"] += 1
except Exception as exc:
report["errors"].append(f"{blob_path}: {type(exc).__name__}: {exc}")
return report
class SecretRedactingLogFilter(logging.Filter):
"""Mask secret-shaped values in every line of a stdlib logging handler.
Root loggers propagate third-party INFO lines verbatim — httpx printed the

View file

@ -235,6 +235,9 @@ def publish_verification_receipt_union(
content = "".join(
json.dumps(row, ensure_ascii=False) + "\n" for row in merged
)
from ouroboros.task_custody import fence_publication
fence_publication() # after the lock wait: a closed publication generation replaces nothing
write_text_atomic(dest, content)
return True
finally:

View file

@ -1357,7 +1357,7 @@ def artifact_bundle_from_result(result: Dict[str, Any]) -> Dict[str, Any]:
"errors": (list(item.get("errors") or []) if isinstance(item.get("errors"), list) else [])
+ ([str(item["copy_error"])] if item.get("copy_error") else []),
}
records.append(record)
records.append(record | ({"relpath": str(item["relpath"])} if item.get("relpath") else {}))
if old_status == ARTIFACT_STATUS_FAILED or any(item["status"] == ARTIFACT_STATUS_FAILED for item in records):
status = ARTIFACT_STATUS_FAILED
elif status != ARTIFACT_STATUS_FAILED and any(item["status"] == "missing" for item in records):

View file

@ -70,10 +70,29 @@ def _mailbox_path(drive_root: pathlib.Path, task_id: str) -> pathlib.Path:
return pathlib.Path(drive_root) / _MAILBOX_DIR / f"{validate_task_id(task_id)}.jsonl"
def mailbox_lines(content: str) -> List[str]:
"""The non-blank rows of a mailbox or ack file, for EVERY reader of them: ``append_jsonl``
ends each row with "\n" and escapes "\r", so only "\n" separates rows. ``str.splitlines``
would also cut at a literal U+2028/U+2029 inside owner text, and lose that row."""
return [line for line in content.split("\n") if line.strip()]
def _ack_path(drive_root: pathlib.Path, task_id: str) -> pathlib.Path:
return pathlib.Path(drive_root) / _MAILBOX_DIR / f"{validate_task_id(task_id)}.acks.jsonl"
def _append_mail(drive_root: pathlib.Path, task_id: str, path: pathlib.Path, entry: Dict[str, Any]) -> bool:
"""Append one mailbox row under the task's mail lock (``task_custody.task_mail_lock``,
on the canonical side), so a row can never land between a settlement's custody read and
the unlink or drive move it permits (a sender after that recreates the mailbox, which
the next settlement holds). Never held across a copy: a publisher's custody lock is a
different lock, so a long copy-back never blocks a sender."""
from ouroboros.task_custody import task_mail_lock
with task_mail_lock(drive_root, task_id) as locked:
return bool(locked and append_jsonl(path, entry))
def acknowledged_task_message_ids(
drive_root: pathlib.Path,
task_id: str,
@ -104,7 +123,7 @@ def acknowledged_task_message_ids(
try:
content = _mailbox_path(drive_root, task_id).read_text(encoding="utf-8")
complete = not content or content.endswith("\n")
for line in content.splitlines():
for line in mailbox_lines(content):
try:
entry = json.loads(line)
except (TypeError, ValueError):
@ -128,7 +147,7 @@ def acknowledged_task_message_ids(
try:
content = path.read_text(encoding="utf-8")
complete = complete and (not content or content.endswith("\n"))
for line in content.splitlines():
for line in mailbox_lines(content):
try:
row = json.loads(line)
except (TypeError, ValueError):
@ -159,23 +178,29 @@ def acknowledge_task_messages(
wake_id: str,
attempt_key: Any = None,
) -> bool:
"""Acknowledge messages only after their full content entered a transcript."""
"""Acknowledge messages only after their full content entered a transcript; the rows
land under the task's mail lock, so a custody read under it sees the mailbox and its
acknowledgements as one state."""
from ouroboros.task_custody import task_mail_lock
path = _ack_path(drive_root, task_id)
path.parent.mkdir(parents=True, exist_ok=True)
existing = acknowledged_task_message_ids(
drive_root, task_id, attempt_key=attempt_key,
)
for msg_id in [str(item) for item in msg_ids if str(item) and str(item) not in existing]:
row = {
"ts": utc_now_iso(), "type": "task_message_acknowledged",
"task_id": str(task_id), "msg_id": msg_id, "wake_id": str(wake_id or ""),
}
if attempt_key is not None:
row["attempt_key"] = str(attempt_key)
row["settled"] = False
if not append_jsonl(path, row):
with task_mail_lock(drive_root, task_id) as locked:
if not locked:
return False
existing = acknowledged_task_message_ids(
drive_root, task_id, attempt_key=attempt_key,
)
for msg_id in [str(item) for item in msg_ids if str(item) and str(item) not in existing]:
row = {
"ts": utc_now_iso(), "type": "task_message_acknowledged",
"task_id": str(task_id), "msg_id": msg_id, "wake_id": str(wake_id or ""),
}
if attempt_key is not None:
row["attempt_key"] = str(attempt_key)
row["settled"] = False
if not append_jsonl(path, row):
return False
return True
@ -243,7 +268,7 @@ def write_owner_message(
if isinstance(attachment_manifest, list):
from ouroboros.artifacts import attachment_manifest_projection
entry.update(attachment_manifest_projection(drive_root, task_id, attachment_manifest))
if not append_jsonl(path, entry):
if not _append_mail(drive_root, task_id, path, entry):
log.warning("Failed to durably append owner message for task %s", task_id)
return False
return True
@ -294,7 +319,7 @@ def write_task_message(
if provenance == "system" and isinstance(review_feedback, dict):
entry["review_feedback"] = dict(review_feedback)
try:
return bool(append_jsonl(path, entry))
return _append_mail(drive_root, task_id, path, entry)
except Exception:
log.warning("Failed to write task message for task %s", task_id, exc_info=True)
return False
@ -303,7 +328,9 @@ def write_task_message(
def owner_attachment_manifest(
drive_root: pathlib.Path, task_id: str, *, _source_refs: Optional[List[Dict[str, Any]]] = None,
) -> List[Dict[str, Any]]:
"""Read all owner input history, including ACKed rows, with optional exact refs."""
"""Read all owner input history, including ACKed rows, with optional exact refs. A row
that is not JSON (torn or unreadable, so possibly owner text with inputs) fails the
read closed with ``OSError``, as an unreadable file does."""
path = _mailbox_path(drive_root, task_id)
if not path.exists():
@ -311,11 +338,11 @@ def owner_attachment_manifest(
manifests: List[Dict[str, Any]] = []
seen_ids: set[str] = set()
try:
for line in path.read_text(encoding="utf-8").splitlines():
for line in mailbox_lines(path.read_text(encoding="utf-8")):
try:
entry = json.loads(line)
except (TypeError, ValueError):
continue
except ValueError as exc:
raise OSError(f"owner mailbox of {task_id} holds an unreadable row") from exc
if not isinstance(entry, dict) or str(entry.get("kind") or KIND_OWNER_TEXT) != KIND_OWNER_TEXT:
continue
msg_id = str(entry.get("msg_id") or "")
@ -479,7 +506,7 @@ def reset_attempt_controls_for_retry(
try:
rows: List[dict] = []
revoked: set[str] = set()
for line in path.read_text(encoding="utf-8").splitlines():
for line in mailbox_lines(path.read_text(encoding="utf-8")):
try:
row = json.loads(line)
except (TypeError, ValueError):
@ -531,17 +558,9 @@ def copy_owner_mailbox_for_retry(
if not source.exists():
continue
try:
source_rows = [
json.loads(line)
for line in source.read_text(encoding="utf-8").splitlines()
if line.strip()
]
source_rows = [json.loads(line) for line in mailbox_lines(source.read_text(encoding="utf-8"))]
target_rows = (
[
json.loads(line)
for line in target.read_text(encoding="utf-8").splitlines()
if line.strip()
]
[json.loads(line) for line in mailbox_lines(target.read_text(encoding="utf-8"))]
if target.exists() else []
)
except (OSError, TypeError, ValueError):
@ -588,7 +607,7 @@ def copy_owner_mailbox_for_retry(
)
if fingerprint in fingerprints:
continue
if not append_jsonl(target, row):
if not _append_mail(drive_root, retry_task_id, target, row):
return False
fingerprints.add(fingerprint)
return True
@ -637,17 +656,13 @@ def drain_owner_entries(
try:
content = path.read_text(encoding="utf-8")
complete = ack_status.get("complete", False) and (not content or content.endswith("\n"))
content = content.strip()
if not content:
if not content.strip():
if _read_status is not None:
_read_status["complete"] = complete
return []
parsed: List[dict] = []
revoked: set = set()
for line in content.splitlines():
line = line.strip()
if not line:
continue
for line in mailbox_lines(content):
try:
entry = json.loads(line)
except Exception:
@ -775,14 +790,30 @@ def drain_owner_messages(
]
def cleanup_task_mailbox(drive_root: pathlib.Path, task_id: str) -> None:
"""Remove a task's mailbox file after task completes."""
for path in (_mailbox_path(drive_root, task_id), _ack_path(drive_root, task_id)):
try:
if path.exists():
path.unlink()
except Exception:
log.debug("Failed to cleanup mailbox for task %s", task_id, exc_info=True)
def cleanup_task_mailbox(drive_root: pathlib.Path, task_id: str, *, canonical_root: Any = None,
carry_inputs: bool = True, stop: Any = None) -> bool:
"""Remove a settled task's mailbox and acks once post-work and input copy are closed
(``settled_mailbox_cleanup_allowed``) and its canonical result holds every unread row with
a verified canonical closure of its inputs (``task_custody.settle_task_mailbox``); returns
whether they are gone. ``canonical_root`` is the result root, by default the drive's
host-layout owner. ``carry_inputs=False`` copies and hashes nothing (the loop thread): a
mailbox with inputs to carry or verify is kept for an off-loop owner, whose generation
``stop()`` fences every copy, write and unlink."""
from ouroboros.task_custody import custody_anchor, settle_task_mailbox
return settle_task_mailbox(canonical_root or custody_anchor(drive_root, task_id), task_id, drive_root,
carry_inputs=carry_inputs, stop=stop)
def discard_mailbox_copy(drive_root: pathlib.Path, task_id: str) -> None:
"""Drop a by-value mailbox copy made for a retry id that will never run: the original
id keeps every row, so no custody is owed; appends still serialize with the unlink."""
from ouroboros.task_custody import task_mail_lock
with task_mail_lock(drive_root, task_id) as locked:
if locked:
for path in (_mailbox_path(drive_root, task_id), _ack_path(drive_root, task_id)):
path.unlink(missing_ok=True)
def mailbox_drain_ended(task_drive: pathlib.Path, task_id: str) -> bool:
@ -792,10 +823,9 @@ def mailbox_drain_ended(task_drive: pathlib.Path, task_id: str) -> bool:
(TZ-2 D15). Its mailbox is then only cleaned up, never read again, so owner
mail and quiz answers must not be labelled delivered into it — the routing
guard and the quiz ingress both ask this one fact. The actor's drive is read,
not the canonical row: split-root copyback can lag the settlement. A receipt for
mail that queued after the drain ended (TZ-2 B5) is not built yet: it waits for
the artifact/forwarding API TZ-1 lands in ``origin/ouroboros`` and is not to be
copied from provisional code.
not the canonical row: split-root copyback can lag the settlement. Mail that
lands after the drain ended is ``MAIL_RETAINED_UNREAD`` (``mail_write_receipt``):
the task's result keeps it as unread mail.
"""
from ouroboros.task_results import load_task_result
from ouroboros.task_status import SETTLED_STATUSES
@ -803,6 +833,32 @@ def mailbox_drain_ended(task_drive: pathlib.Path, task_id: str) -> bool:
return str((load_task_result(task_drive, task_id) or {}).get("status") or "") in SETTLED_STATUSES
# Receipt vocabulary for a message written into a task's mailbox. A write proves only
# that the row is durable; "read" is proven later by the recipient loop's acknowledgement
# (``mail_read_state``), never claimed at write time.
MAIL_QUEUED = "queued" # the recipient has not started; it reads the row when it starts
MAIL_DELIVERED = "delivered" # a live drain reads it at its next checkpoint
MAIL_RETAINED_UNREAD = "retained_unread" # its drain ended: the result keeps it unread
def mail_write_receipt(recipient_status: str, *, drain_ended: bool = False) -> Dict[str, Any]:
"""The typed receipt for one durable mailbox write to a recipient in ``recipient_status``."""
state = (MAIL_RETAINED_UNREAD if drain_ended
else MAIL_QUEUED if str(recipient_status or "") in {"requested", "scheduled"} else MAIL_DELIVERED)
return {"receipt": state, "read": False,
"read_evidence": "the recipient's task_message_acknowledged row (mail_read_state)"}
def mail_read_state(drive_root: pathlib.Path, task_id: str, msg_id: str) -> Optional[bool]:
"""True once the recipient acknowledged MSG_ID (its words entered a transcript), False
while unread, None when the acknowledgement ledger could not be read completely."""
status: Dict[str, bool] = {}
acknowledged = acknowledged_task_message_ids(drive_root, task_id, _read_status=status)
if str(msg_id) in acknowledged:
return True
return False if status.get("complete") else None
def settled_mailbox_cleanup_allowed(result: Dict[str, Any]) -> bool:
"""A settled task still owns its mailbox while post-work or input copy is owed."""
from ouroboros.post_task_checkpoint import post_task_synthesis_is_open
@ -823,14 +879,15 @@ def settled_mailbox_cleanup_allowed(result: Dict[str, Any]) -> bool:
)
def sweep_settled_owner_mailboxes(drive_root: pathlib.Path) -> Dict[str, Any]:
"""Startup sweep of mailboxes whose task died off the terminal paths (CPL4-C18).
The only regular unlink is the task_done dispatch; a task that never
reached it (crash, lost event, hard kill) leaked its mailbox forever.
A mailbox goes only after terminal file recovery, with a settled result,
settled post-task work and no pending input copy. No result keeps it.
Lock sidecars are untouched (self-healing by staleness).
def sweep_settled_owner_mailboxes(drive_root: pathlib.Path, *, stop: Any = None) -> Dict[str, Any]:
"""Off-loop sweep of canonical mailboxes whose task settled without a cleanup (CPL4-C18:
a task that never reached the task_done dispatch - crash, lost event, hard kill - leaked
its mailbox forever) or whose cleanup the loop thread declined because unread rows carry
inputs to verify or copy (carried here). Startup and the drive-custody pass call it. A
mailbox goes only with a settled result, settled post-task work, no pending input copy and
every unread row in verified canonical custody. No result keeps it. ``stop()`` (the
maintenance generation) is asked before each mailbox and fences its cleanup's copies,
row write and unlinks. Lock sidecars are untouched (self-healing by staleness).
"""
report: Dict[str, Any] = {"removed": [], "kept": 0}
mailbox_dir = pathlib.Path(drive_root) / _MAILBOX_DIR
@ -853,10 +910,10 @@ def sweep_settled_owner_mailboxes(drive_root: pathlib.Path) -> Dict[str, Any]:
settled = settled_mailbox_cleanup_allowed(result)
except Exception:
settled = False
if not settled:
if not settled or (stop is not None and stop()):
report["kept"] += 1
continue
cleanup_task_mailbox(pathlib.Path(drive_root), task_id)
cleanup_task_mailbox(pathlib.Path(drive_root), task_id, stop=stop)
if path.exists():
report["kept"] += 1 # unlink refused: still owned by the mailbox
else:

View file

@ -121,6 +121,12 @@ def project_replica_task_result_fields(
# The receiving drive's first accepted terminal transition owns provenance,
# including its absence on historical rows; replicas cannot originate it.
overlay.pop("canonical_terminal_projection_origin", None)
# Unread-mail custody is a union: a stale replica never drops a canonical row.
from ouroboros.task_custody import merge_unread_mail
custody = merge_unread_mail(canonical_fields.get("unread_mailbox"), overlay.get("unread_mailbox"))
if custody is not None:
overlay["unread_mailbox"] = custody
canonical_cost = canonical_fields.get("cost_presentation")
replica_cost = overlay.get("cost_presentation")
if (isinstance(canonical_cost, dict) and canonical_cost.get("scope") == COST_SCOPE_ROOT_TREE

View file

@ -1315,7 +1315,6 @@ def _emit_safety_mode_skip(ctx: Optional[Any], tool_name: str, mode: str, policy
P3: an advisory/off mode is legitimate ONLY while every decision it waves
through leaves a loud, durable trace at the moment it happens (review round 1)."""
log.warning("Safety mode=%s waved through LLM check for %s (policy=%s)", mode, tool_name, policy)
_emit_durable_safety_event(ctx, {
"type": "safety_mode_skip",
"tool": tool_name,

View file

@ -9,6 +9,8 @@ both outside the loop it watches.
from __future__ import annotations
import queue
import re
import sys
import threading
import time
from typing import Any, Callable, Optional
@ -28,6 +30,10 @@ from ouroboros.utils import utc_now_iso
# watches reports nothing. Older/foreign callers may pass the stamp alone.
_STAMP, _FACTS, _CPU, _LAG = 0, 1, 2, 3
# A stall row carries the loop thread's stack at onset: at most this many
# innermost frames, each one line, so the journal row stays bounded.
_STALL_STACK_FRAMES = 12
def _supervisor_loop_stalled(last_tick: float, now: float, deadline_sec: int) -> bool:
"""True when the supervisor loop has not published a liveness tick within the
@ -65,9 +71,14 @@ def loop_phase_facts(liveness: list, phase: str, *, new_tick: bool = False) -> d
``maintenance`` | ``assign``, one stamp per phase and never per sub-step — so a
stall names where the thread went silent instead of only how long it was.
``loop_thread_cpu_sec`` is the ``time.thread_time()`` delta over the interval
that ENDS with this stamp, sampled on the loop thread itself: read beside the
wall gap it separates a thread that BURNED that gap from one blocked on a lock
or starved of the GIL. ``max_event_lag_sec`` is the worst worker-stamped lag of
that ENDS with this stamp, sampled on the loop thread itself, and
``cpu_interval_sec`` is that interval's wall (monotonic) length — the number
is honest only beside the interval it covers, and at stall onset that
interval is the last HEALTHY phase, not the stall. ``loop_thread_cpu_total_sec``
is the thread's cumulative CPU at this stamp, so the watchdog can charge a
whole stall (onset stamp to recovery stamp) without seeing every stamp in
between. Read beside the wall gap they separate a thread that BURNED it from
one blocked on a lock or starved of the GIL. ``max_event_lag_sec`` is the worst worker-stamped lag of
the most recently completed drain, absent when no drained event carried a
worker stamp. ``new_tick`` opens a fresh drain maximum (the events phase opens
the tick), so a lag can never outlive the tick that observed it. No
@ -79,6 +90,8 @@ def loop_phase_facts(liveness: list, phase: str, *, new_tick: bool = False) -> d
facts = {
"phase": phase,
"loop_thread_cpu_sec": round(cpu - liveness[_CPU], 3),
"cpu_interval_sec": round(max(0.0, time.monotonic() - liveness[_STAMP]), 3),
"loop_thread_cpu_total_sec": round(cpu, 6),
"daemon_pin_matched": _daemon_pin_matched(),
}
if liveness[_LAG] is not None:
@ -165,6 +178,80 @@ def _published_loop_facts(liveness: list) -> dict:
return dict(facts) if isinstance(facts, dict) else {}
def _loop_thread_stack(ident: Optional[int], *, limit: int = _STALL_STACK_FRAMES,
facts: Optional[dict] = None) -> list[str]:
"""The loop thread's CURRENT stack, outermost first, at most ``limit`` innermost
``path:line in func`` frames (repository files relative, others absolute). Walk code
objects instead of traceback.extract_stack: linecache could read a stalled filesystem on
this watchdog thread; no locals, no message bodies. A cut walk sets
``facts["loop_stack_truncated"]`` when ``facts`` is given, so a bounded stack is never
mistaken for the whole one. No lock, disk or cooperation from the loop; [] if its
stack is unavailable."""
if not ident or limit <= 0:
return []
try:
frame = sys._current_frames().get(ident)
summaries = []
while frame is not None and len(summaries) < limit:
code = frame.f_code
summaries.append((code.co_filename, frame.f_lineno, code.co_name))
frame = frame.f_back
if frame is not None and facts is not None:
facts["loop_stack_truncated"] = True
except Exception:
log.debug("loop-thread stack capture failed", exc_info=True)
return []
finally:
frame = None # never keep a foreign frame alive past this call
try:
from ouroboros.config import REPO_DIR
root = str(REPO_DIR).rstrip("/\\") + "/"
except Exception:
root = ""
rows = []
for filename, lineno, name in reversed(summaries):
path = filename.replace("\\", "/")
if root and path.startswith(root.replace("\\", "/")):
path = path[len(root):]
rows.append(f"{path[:200]}:{lineno} in {name[:100]}")
return rows
_ABSOLUTE_PATH = re.compile(r"^([A-Za-z]:)?[/\\]")
# One ``_loop_thread_stack`` row: ``path:line in func``. The path is everything before the LAST
# ``:<line> in `` (a Windows drive ``C:`` or a colon inside the path is part of the path).
_STACK_ROW = re.compile(r"^(?P<path>.*):(?P<line>\d+) in (?P<func>.*)$")
def _innermost_repo_frame(stack: list[str]) -> str:
"""``path:function`` of the innermost frame inside the repository - a relative path, else
the innermost frame outside the Python runtime ('' when none) - without the line number,
so the samples of one stalled function fold into one row."""
runtime = tuple(prefix.replace("\\", "/") for prefix in {sys.prefix, sys.base_prefix} if prefix)
outside = ""
for row in reversed(stack):
parsed = _STACK_ROW.match(row)
path, func = (parsed.group("path"), parsed.group("func")) if parsed else ("", "")
key = f"{path}:{func}"
if path and not _ABSOLUTE_PATH.match(path):
return key
if path and not outside and not path.startswith(runtime):
outside = key
return outside
def _stall_cpu_over(onset_facts: dict, latest_facts: dict) -> Optional[float]:
"""The loop thread's CPU from the stamp it went silent on to its latest stamp
— the whole stall, however many stamps the recovery published before the
watchdog looked — from the cumulative totals; the latest stamp's own delta
when a foreign/older stamp published no total (never invented)."""
onset_total, latest_total = onset_facts.get("loop_thread_cpu_total_sec"), latest_facts.get("loop_thread_cpu_total_sec")
if isinstance(onset_total, (int, float)) and isinstance(latest_total, (int, float)):
return round(max(0.0, latest_total - onset_total), 3)
return latest_facts.get("loop_thread_cpu_sec")
def _chat_turn_wedged(busy: bool, last_activity_ts, now: float, deadline_sec: int) -> bool:
"""True when an IN-PROCESS direct-chat turn is busy but its liveness tick has been
silent past the deadline (WS3). ``last_activity_ts is None`` => the turn has not
@ -207,19 +294,29 @@ def _alert_chat_turn_wedge(task_id, gap: float) -> None:
log.debug("chat-turn wedge owner alert failed", exc_info=True)
def _start_supervisor_liveness_watchdog(liveness: list, stop_event=None) -> None:
def _start_supervisor_liveness_watchdog(
liveness: list, stop_event=None, *, loop_thread_ident: Optional[int] = None,
) -> None:
"""Dedicated daemon thread (NOT inside the supervisor loop, so it fires even when
that loop stalls). It observes two silent-wedge classes and reports them
DIFFERENTLY (owner decision 4C). A heartbeat-silent in-process direct-chat turn
ALERTS the owner, because /restart is a recovery they can perform. A supervisor
loop stall (new-message intake starvation) is JOURNAL ONLY: the ``log.error``
and one durable ``supervisor_loop_stall`` row with the phase facts the loop
published with its last stamp, closed once the loop ticks again by one
``supervisor_loop_stall_end`` — onset without an end is a generation that never
recovered. Nothing reaches the owner's chat from that half: a stall they cannot
act on is an alarm, not information, and the rows carry the diagnosis anyway.
It deliberately does NOT kill a hung thread; independent native actors keep
the chat responsive meanwhile. ``stop_event`` is
published with its last stamp plus the loop thread's stack at onset
(``stack``, bounded; ``loop_stack_truncated`` when cut), then one bounded stack
sample per watchdog interval while the stall is open, closed once the loop ticks
again by one ``supervisor_loop_stall_end`` charging the thread's CPU over the
whole stall and carrying ``samples``, up to five ``top_frames`` (innermost
repository frame -> samples) and the ``last_stack`` — where it spent the stall,
not only where it began. Onset without an end is a generation that never
recovered. Nothing reaches the
owner's chat from that half: a stall they cannot act on is an alarm, not
information, and the rows carry the diagnosis anyway. It deliberately does NOT
kill a hung thread; independent native actors keep the chat responsive
meanwhile. ``loop_thread_ident`` is the thread whose stack a stall row carries;
it defaults to the CALLER, because the loop thread starts its own watchdog
(startup phase included). ``stop_event`` is
a PER-GENERATION token: when the supervisor loop that owns ``liveness`` exits (incl.
the crash-storm death path, which never sets the global restart flag), it is set so
this watchdog stops watching a now-stale liveness list (no false post-revival alert)."""
@ -228,13 +325,23 @@ def _start_supervisor_liveness_watchdog(liveness: list, stop_event=None) -> None
deadline = get_supervisor_liveness_deadline_sec()
if deadline <= 0:
return
watched_ident = loop_thread_ident if loop_thread_ident is not None else threading.get_ident()
def _watch() -> None:
from supervisor.state import append_jsonl
interval = min(15, max(1, deadline // 3))
loop_alerted = False
stall_onset: tuple = () # (stalled stamp, phase) of the OPEN alerted stall
stall_onset: tuple = () # (stalled stamp, phase, onset facts) of the OPEN alerted stall
samples: dict = {} # innermost repository frame -> stack samples while the stall is open
last_stack: list = []
wedged_tasks: set[str] = set()
def sample(stack: list) -> None:
nonlocal last_stack
if stack:
last_stack = stack
key = _innermost_repo_frame(stack) or "(outside the repository)"
samples[key] = samples.get(key, 0) + 1
while not _restart_requested.is_set() and not (stop_event is not None and stop_event.is_set()):
time.sleep(interval)
# ONE clock: both halves measure an ELAPSED GAP against stamps taken on
@ -248,22 +355,33 @@ def _start_supervisor_liveness_watchdog(liveness: list, stop_event=None) -> None
if not loop_alerted:
gap = now - liveness[_STAMP]
facts = _published_loop_facts(liveness)
log.error(
"Supervisor loop STALLED ~%.0fs — new-message intake starved (native "
"chat still answers); investigate a blocking step.", gap,
)
flags: dict = {}
stack = _loop_thread_stack(watched_ident, facts=flags)
# A startup stall is a generation that has not finished initializing: no
# native chat answers for it, so the line says only what is true.
starved = ("startup has not finished" if facts.get("phase") == "startup"
else "new-message intake starved (native chat still answers)")
log.error("Supervisor loop STALLED ~%.0fs in phase %s — %s; loop thread is at: %s",
gap, facts.get("phase"), starved, stack[-1] if stack else "(stack unavailable)")
try:
# The facts the loop published with the stamp it went silent
# on: where it was, what its own thread burned, how far
# behind the drained worker events already were.
# on: where it was, what its own thread burned over the
# interval BEFORE the stall (cpu_interval_sec says how long
# that was), how far behind the drained worker events
# already were — and where its thread stands right now.
append_jsonl(DATA_DIR / "logs" / "supervisor.jsonl", {
"ts": utc_now_iso(), "type": "supervisor_loop_stall",
"stalled_sec": round(gap, 1), **facts,
"stalled_sec": round(gap, 1), **facts, **flags,
**({"stack": stack} if stack else {}),
})
except Exception:
log.debug("loop-stall log failed", exc_info=True)
loop_alerted = True
stall_onset = (liveness[_STAMP], facts.get("phase"))
stall_onset = (liveness[_STAMP], facts.get("phase"), facts)
samples, last_stack = {}, []
sample(stack)
else:
sample(_loop_thread_stack(watched_ident)) # one bounded sample per interval
else:
if loop_alerted:
# The loop ticked again: close the episode ONCE, and only one
@ -272,14 +390,26 @@ def _start_supervisor_liveness_watchdog(liveness: list, stop_event=None) -> None
# jump; it rounds up by at most one watchdog interval, the
# resolution at which recovery is observed at all.
try:
latest = _published_loop_facts(liveness)
stalled = round(liveness[_STAMP] - stall_onset[0], 1)
append_jsonl(DATA_DIR / "logs" / "supervisor.jsonl", {
"ts": utc_now_iso(), "type": "supervisor_loop_stall_end",
"stalled_sec": round(liveness[_STAMP] - stall_onset[0], 1),
"stalled_sec": stalled,
"phase": stall_onset[1],
# The recovery stamp's CPU delta covers the stalled interval itself:
# beside the wall gap it tells a thread that burned it from one that
# The thread's CPU from the stamp it went silent on to its
# latest stamp — the stalled interval itself, whole, however
# many phases the recovery published before this look —
# beside the wall interval it covers (the same seconds as
# stalled_sec): a thread that burned them versus one that
# waited on a lock, IO or the GIL.
"loop_thread_cpu_sec": _published_loop_facts(liveness).get("loop_thread_cpu_sec"),
"loop_thread_cpu_sec": _stall_cpu_over(stall_onset[2], latest),
"cpu_interval_sec": stalled,
# Where the thread SPENT the stall: one stack sample per watchdog
# interval, folded by innermost repository frame, plus the last one.
"samples": sum(samples.values()),
"top_frames": [{"frame": frame, "samples": count} for frame, count in
sorted(samples.items(), key=lambda item: (-item[1], item[0]))[:5]],
"last_stack": last_stack,
})
except Exception:
log.debug("loop-stall-end log failed", exc_info=True)

View file

@ -1,41 +1,36 @@
"""Upkeep a supervisor generation owes the drive.
The once-per-generation startup sweep (process custody, delegated runs, legacy
cancel latches, owed terminal deliveries, orphaned running results, pending
post-task synthesis), the throttled periodic cadences of the same surfaces, and
the delegated-snapshot GC that fails closed on an unreadable custody log.
"""
"""Upkeep a supervisor generation owes the drive: the once-per-generation startup
sweep (process custody, delegated runs, legacy cancel latches, owed terminal
deliveries, orphaned running results, pending post-task synthesis), the throttled
periodic cadences of the same surfaces — every history-sized one off the loop
thread — and the delegated-snapshot GC that fails closed on an unreadable log."""
from __future__ import annotations
import pathlib
import json
import logging
import os
import threading
import time
from typing import Any
from contextlib import contextmanager
from typing import Any, Dict
from ouroboros.server_process import DATA_DIR, log, _restart_requested, _supervisor_stop
from ouroboros.utils import utc_now_iso
def _installed_skill_names():
"""Names of skills currently installed ON DISK (disk-derived, not in-memory).
Passed to the process-custody reaper so it can tell which skill-companion
orphans are safe to reap (owner uninstalled). Disk-derived so it is correct
independent of in-memory extension-reload timing; returns None on any failure
so the reaper fails toward KEEP (never mass-kills live skills' companions).
"""Disk-derived skill owners for companion reaping; unknown means KEEP.
Reload timing and failed discovery must never mass-reap live companions.
"""
try:
from ouroboros.config import get_skills_repo_path
from ouroboros.skill_loader import discover_skills
names = {s.name for s in discover_skills(DATA_DIR, repo_path=get_skills_repo_path())}
# Coalesce an EMPTY result to None ("unknown"), NOT "everything
# uninstalled": discover_skills returns [] without raising when the skills
# dir is momentarily unavailable; treating that as an empty install set
# would let an enforced reap mass-kill live companions. None ⇒ keep-all.
# An EMPTY result is None ("unknown"), NOT "everything uninstalled":
# discover_skills returns [] when the skills dir is momentarily unavailable,
# and an enforced reap over that would mass-kill live companions.
return names or None
except Exception:
log.debug("Could not compute installed skill names for custody reaper", exc_info=True)
@ -45,58 +40,56 @@ def _installed_skill_names():
_LAST_CANCEL_INTENT_SWEEP = [0.0]
_CANCEL_INTENT_SWEEP_LOCK = threading.Lock()
_CUSTODY_SWEEP_LOCK = threading.Lock()
_RECONCILE_SWEEP_LOCK = threading.Lock()
@contextmanager
def orphan_reconcile_write_guard(task_id: str, *, stop_event: Any = None):
"""Fence the orphan reconciler's writes against assignment (queue -> row lock order) in an open generation:
the settlement probe must prove absence; a process with no supervisor (no-provider boot) dispatches nothing."""
from supervisor import queue
with queue._queue_lock:
yield not _stop_requested(stop_event) and (not queue.INITIALIZED or queue.task_settlement_liveness(task_id) is False)
def _stop_requested(stop_event: Any = None) -> bool:
"""Is this generation's mutation window closed?
True once the supervisor loop that started the pass has exited (its
per-generation ``_watchdog_stop`` token) or the process is stopping or
restarting. Off the loop thread nothing else stops the pass, so it answers two
questions with one fact: mutate nothing more, and reach the daemon ATTACH-ONLY
— an ``ensure`` between a stop request and the daemon stop starts the engine
the teardown is about to end (ARCHITECTURE §9, DEVELOPMENT Process Custody Rule).
"""Closed generation, stop or restart: no mutation or daemon start afterward.
The per-generation token also keeps an old off-loop pass from writing.
"""
return bool(_restart_requested.is_set() or _supervisor_stop.is_set()
or (stop_event is not None and stop_event.is_set()))
def _live_task_ids() -> set:
"""The ONE live-owner source both custody surfaces and the cursor refresh read.
Memory only — RUNNING and busy worker slots under ``_queue_lock``, the
direct-activity registry, in-flight post-task synthesis: the owners
``queue.task_has_live_ownership`` names, without the durable result read no
sweep may pay. Handed to its consumers as this CALLABLE so each evaluates it
after reading its own candidates (``reap_orphaned_processes`` for the rule).
"""Fresh in-memory owners for custody and cursor refresh, never result rows.
Passed as a callable so consumers recheck after collecting candidates.
"""
return _startup_live_task_ids(DATA_DIR)
def _run_cancel_delivery_ref_sweep(drive_root: pathlib.Path) -> None:
"""One existing maintenance pass; the drain retains no file/cancel work."""
"""20 s cancel/delivery/usage pass; history-sized work rides the 300 s reconcile pass."""
try:
try:
from supervisor.task_lifecycle import sweep_cancel_intents
outcomes = sweep_cancel_intents()
if outcomes:
log.info("Cancel-intent watchdog settled: %s", outcomes)
_step_recovered("cancel_intent_sweep")
except Exception:
log.debug("Cancel-intent watchdog sweep failed", exc_info=True)
_step_failed("cancel_intent_sweep")
try:
from supervisor.terminal_delivery import replay_pending_deliveries
replay_pending_deliveries(drive_root)
_step_recovered("terminal_delivery_replay")
except Exception:
log.debug("Pending terminal-delivery replay failed", exc_info=True)
try:
from ouroboros.observability import retry_pending_child_ref_promotions
retry_pending_child_ref_promotions(drive_root)
except Exception:
log.debug("Pending child-ref promotion retry failed", exc_info=True)
_step_failed("terminal_delivery_replay")
try:
_reconcile_abandoned_usage(drive_root)
_step_recovered("abandoned_usage_reconciliation")
except Exception:
log.warning("Abandoned usage reconciliation failed", exc_info=True)
_step_failed("abandoned_usage_reconciliation")
finally:
_CANCEL_INTENT_SWEEP_LOCK.release()
@ -218,44 +211,134 @@ def _reconcile_abandoned_usage(drive_root: pathlib.Path) -> None:
log.warning("Reconciled task cost refresh failed for %s", task_id, exc_info=True)
# Memory only: consecutive failures per periodic step, so a failure that recurs every cadence
# is a WARNING with its traceback on failures 1, 2, 4, 8, ... and DEBUG in between, and its
# recovery is one INFO naming the streak (never a silent DEBUG death, never a wall of WARNINGs).
_STEP_FAILURES: Dict[str, int] = {}
def _step_failed(step: str, message: str = "%s failed") -> None:
streak = _STEP_FAILURES.get(step, 0) + 1
_STEP_FAILURES[step] = streak
level = logging.WARNING if streak & (streak - 1) == 0 else logging.DEBUG
log.log(level, message + " (failure %d in a row)", step, streak, exc_info=True)
def _step_recovered(step: str) -> None:
streak = _STEP_FAILURES.pop(step, 0)
if streak:
log.info("%s recovered after %d failure(s)", step, streak)
# Memory only: the off-loop pass each latch is running (its thread, start stamp, whether its
# stall was journaled), so a cadence whose next tick finds the latch still held names the duty
# and the line its thread stands on: ``host_duty_stall`` once, ``host_duty_stall_end`` when
# the pass ends. The threshold is the duty's OWN cadence measured from the pass's start, never
# the loop deadline and never the tick alone: a marker stamped when a pass ENDS stays old while
# the next pass starts, so a due tick proves nothing about how long the current pass has run.
_DUTIES: Dict[int, Dict[str, Any]] = {}
def _duty_busy(latch: Any, cadence_sec: float) -> None:
duty = _DUTIES.get(id(latch))
if duty is None or duty["alerted"] or time.time() - duty["since"] < cadence_sec:
return
duty["alerted"] = True
from ouroboros.server_liveness import _loop_thread_stack
from supervisor.state import append_jsonl
flags: Dict[str, Any] = {}
stack = _loop_thread_stack(duty["ident"], facts=flags)
running = round(time.time() - duty["since"], 1)
log.warning("Off-loop duty %s has run %.0fs, past its %.0fs cadence; its thread is at: %s",
duty["name"], running, cadence_sec, stack[-1] if stack else "(stack unavailable)")
try:
append_jsonl(DATA_DIR / "logs" / "supervisor.jsonl", {
"ts": utc_now_iso(), "type": "host_duty_stall", "duty": duty["name"], "running_sec": running,
"cadence_sec": cadence_sec, **flags, **({"stack": stack} if stack else {})})
except Exception:
log.debug("host_duty_stall row failed", exc_info=True)
def _duty_ended(latch: Any, duty: Dict[str, Any]) -> None:
"""Close THIS pass's duty. The pass released its latch before this runs, so the next pass
may already have registered its own entry: that one is put back, never dropped (it holds
the latch, so nothing registers again meanwhile)."""
current = _DUTIES.pop(id(latch), None)
if current is not None and current is not duty:
_DUTIES.setdefault(id(latch), current)
if not duty["alerted"]:
return
from supervisor.state import append_jsonl
try:
append_jsonl(DATA_DIR / "logs" / "supervisor.jsonl", {
"ts": utc_now_iso(), "type": "host_duty_stall_end", "duty": duty["name"],
"running_sec": round(time.time() - duty["since"], 1)})
except Exception:
log.debug("host_duty_stall_end row failed", exc_info=True)
def _start_maintenance_thread(latch: Any, name: str, target: Any, args: tuple) -> bool:
"""Start one off-loop pass under a latch the CALLER already took without blocking
(busy => that tick skipped, never queued); a start that fails releases it. The pass is
watched as a duty (``_duty_busy`` / ``_duty_ended``)."""
def run() -> None:
duty["ident"] = thread.ident # the pass names its own thread: no race with its end
try:
target(*args)
finally:
_duty_ended(latch, duty)
duty: Dict[str, Any] = {"name": name, "ident": None, "since": time.time(), "alerted": False}
_DUTIES[id(latch)] = duty
try:
thread = threading.Thread(target=run, name=name, daemon=True)
thread.start()
return True
except Exception:
_DUTIES.pop(id(latch), None)
latch.release()
log.warning("%s could not start", name, exc_info=True)
return False
def _periodic_supervisor_maintenance(
last_custody_reap: list, last_review_reconcile: list, *, on_orphans_healed: Any = None,
stop_event: Any = None,
) -> None:
"""Throttled periodic upkeep extracted from the supervisor loop: cancel-intent
watchdog and pending child-ref promotion replay (every 20s), custody reap of
orphaned task-scoped processes (every 600s) + review-job zombie reconcile
(every 300s). Each cadence gates itself via its own last-run marker, updated on
the LOOP thread: the first two stamp before handing their work to a daemon
thread; the inline zombie reconcile stamps when its pass ends.
``stop_event`` is the loop's per-generation token, handed to the custody pass so
it stops mutating when that generation ends. ``on_orphans_healed(count)`` fires
when the zombie reconcile terminalized orphaned RUNNING task rows (the alarm
clock wakes early for them)."""
if time.time() - _LAST_CANCEL_INTENT_SWEEP[0] > 20 and _CANCEL_INTENT_SWEEP_LOCK.acquire(blocking=False):
_LAST_CANCEL_INTENT_SWEEP[0] = time.time()
try:
threading.Thread(target=_run_cancel_delivery_ref_sweep, args=(pathlib.Path(DATA_DIR),),
name="terminal-maintenance", daemon=True).start()
except Exception:
_CANCEL_INTENT_SWEEP_LOCK.release()
log.warning("Terminal maintenance could not start", exc_info=True)
latch = _CUSTODY_SWEEP_LOCK # the pass releases THIS object, never a later generation's
if time.time() - last_custody_reap[0] > 600 and latch.acquire(blocking=False):
last_custody_reap[0] = time.time()
try:
threading.Thread(target=_run_periodic_custody_sweep, args=(stop_event, latch),
name="custody-maintenance", daemon=True).start()
except Exception:
latch.release()
log.warning("Periodic custody sweep could not start", exc_info=True)
if time.time() - last_review_reconcile[0] > 300:
try:
_periodic_zombie_reconcile(on_orphans_healed=on_orphans_healed)
finally:
# Stamped when the pass ENDS: a pass slower than its cadence never re-arms
# on the next tick, so >=300 s of ordinary ticks separate two passes (issue #1230).
last_review_reconcile[0] = time.time()
"""Throttled upkeep on the supervisor tick: three cadences, each a non-blocking
latch plus a last-run marker, nothing history-sized inline (INV-24). Every 20 s the
cancel-intent watchdog, terminal delivery replay and abandoned usage; every 600 s
the custody block; every 300 s the reconcile block (zombie heal, then pending
child-ref promotion retry). The first two stamp here before handing off; the
reconcile pass stamps when it ENDS (issue #1230). A due cadence whose latch is still
held journals the duty's stall once, with its thread's stack (``_duty_busy``).
``stop_event`` is the loop's per-generation token: both long passes stop mutating
once it closes. ``on_orphans_healed(count)`` fires off-loop when orphaned RUNNING rows healed."""
now = time.time()
if now - _LAST_CANCEL_INTENT_SWEEP[0] > 20:
if _CANCEL_INTENT_SWEEP_LOCK.acquire(blocking=False):
_LAST_CANCEL_INTENT_SWEEP[0] = now
_start_maintenance_thread(_CANCEL_INTENT_SWEEP_LOCK, "terminal-maintenance",
_run_cancel_delivery_ref_sweep, (pathlib.Path(DATA_DIR),))
else:
_duty_busy(_CANCEL_INTENT_SWEEP_LOCK, 20)
latch = _CUSTODY_SWEEP_LOCK # a pass releases THIS object, never a later generation's
if now - last_custody_reap[0] > 600:
if latch.acquire(blocking=False):
last_custody_reap[0] = now
_start_maintenance_thread(latch, "custody-maintenance", _run_periodic_custody_sweep,
(stop_event, latch))
else:
_duty_busy(latch, 600)
latch = _RECONCILE_SWEEP_LOCK
if now - last_review_reconcile[0] > 300:
if not latch.acquire(blocking=False):
_duty_busy(latch, 300)
elif not _start_maintenance_thread(latch, "reconcile-maintenance", _run_periodic_reconcile_sweep,
(last_review_reconcile, stop_event, latch, on_orphans_healed)):
last_review_reconcile[0] = time.time() # a start that failed still waits out a cadence
def _run_periodic_custody_sweep(stop_event: Any = None, latch: Any = None) -> None:
@ -264,12 +347,10 @@ def _run_periodic_custody_sweep(stop_event: Any = None, latch: Any = None) -> No
Skill-payload hashing, the orphaned-process reaper, delegated-run reconciliation
(gateway handshake, custody replays, registration retirement) and the
settled-terminal cursor cost seconds to minutes — longer than any worker ack
wait — so they run here, on the 20 s sweep's shape: the caller took
``_CUSTODY_SWEEP_LOCK`` without blocking (busy ⇒ the tick skips, never queues)
and this pass releases it in ``finally``. Nothing serializes the pass against
assignment any more, so each step reads its CANDIDATES before the shared live
set, and the generation is re-read before every mutation.
"""
wait. The caller took ``_CUSTODY_SWEEP_LOCK`` without blocking (busy => skip,
never queue); this pass releases it in ``finally``. Nothing serializes it
against assignment, so each step reads its CANDIDATES before the shared live
set, and the generation is re-read before every mutation."""
try:
try:
if _stop_requested(stop_event):
@ -280,10 +361,9 @@ def _run_periodic_custody_sweep(stop_event: Any = None, latch: Any = None) -> No
log.warning("Terminal projection reconciliation deferred", exc_info=True)
try:
# Issue #844: release the owned-daemon start latch in ITS OWN try, ahead of
# the reap, so a raising reap can never pin it; retry once — only when THIS
# sweep released a latch — on a short-lived thread, as warm_owned_daemon()
# does (the reconcile below ensures only with orphan work). Contract, per-
# process scope and the residual: DEVELOPMENT.md "Process Custody Rule".
# the reap, so a raising reap can never pin it; retry once, only when THIS
# sweep released a latch, on a short-lived thread as warm_owned_daemon()
# does. Contract and residual: DEVELOPMENT.md "Process Custody Rule".
from ouroboros.claudexor_daemon import get_owned_daemon
if _stop_requested(stop_event):
@ -293,9 +373,8 @@ def _run_periodic_custody_sweep(stop_event: Any = None, latch: Any = None) -> No
name="owned-daemon-latch-retry", daemon=True).start()
except Exception:
log.debug("Owned daemon latch release failed", exc_info=True)
# The steps share one guard (a failure still ends the pass), but the row
# must say WHICH one died: at DEBUG, and unnamed, a block that silently
# stopped reaping for weeks looked exactly like one that had nothing to do.
# One guard for the steps (a failure still ends the pass), but the row names
# WHICH died: unnamed at DEBUG, weeks of silent non-reaping looked healthy.
step = "reap_orphaned_processes"
try:
from ouroboros.claudexor_daemon import CUSTODY_PURPOSE
@ -319,26 +398,23 @@ def _run_periodic_custody_sweep(stop_event: Any = None, latch: Any = None) -> No
if _stop_requested(stop_event):
return
_cursor_refresh_settled_terminals(_live_task_ids)
for name in ("reap_orphaned_processes", "reconcile_delegated_runs", "cursor_refresh_settled_terminals"):
_step_recovered(name)
except Exception:
log.warning("Periodic custody step %s failed", step, exc_info=True)
_step_failed(step, "Periodic custody step %s failed")
finally:
(latch or _CUSTODY_SWEEP_LOCK).release()
def _retry_latched_daemon_start() -> None:
"""The one retry of a latched owned-daemon start (#844), made by the sweep itself.
Runs on its own short-lived daemon thread, only after this sweep released
the latch: one ``ensure_owned_gateway`` with ZERO admission and ZERO
startup wait, so the supervisor loop never holds a startup wait (nor the
unbounded runtime preparation) — the spawn happens, custody keeps the child, and
``daemon_starting`` is the EXPECTED answer (the next ordinary caller joins
or settles it). Any other typed refusal (a child that died at once has
already re-latched inside the manager) is logged as a warning; nothing is
raised into the loop, nothing else is retried or scheduled, and a gateway
that did open is closed at once (the reconcile that follows attaches on
its own).
"""
"""The one retry of a latched owned-daemon start (#844), by the sweep itself, on
its own short-lived daemon thread and only after this sweep released the latch:
one ``ensure_owned_gateway`` with ZERO admission and ZERO startup wait, so no
loop thread ever holds a startup wait — the spawn happens, custody keeps the
child, and ``daemon_starting`` is the EXPECTED answer (the next ordinary caller
joins or settles it). Any other typed refusal (a child that died at once has
re-latched inside the manager) is a warning; nothing is raised into the loop or
retried again, and a gateway that did open is closed at once."""
from ouroboros.claudexor_daemon import ensure_owned_gateway
from ouroboros.gateways.claudexor import ClaudexorUnavailable
@ -367,10 +443,9 @@ def _reconcile_delegated_runs(running_task_ids: Any, *, stop_event: Any = None)
continued = {str(task["id"]) for task in pending_waits
if restore_owner_wait_allowed(DATA_DIR, task)}
# Zero admission wait: a daemon in its recovery-only window is skipped until
# the next sweep. Once a stop, restart or panic is in flight the factory is
# ATTACH-ONLY — the one ensure this surface may still make belongs to the
# latch retry above (ARCHITECTURE §9, DEVELOPMENT Process Custody Rule).
# Zero admission wait: a daemon in its recovery-only window waits for the next
# sweep. Once a stop, restart or panic is in flight the factory is ATTACH-ONLY;
# the one ensure left to this surface is the latch retry (ARCHITECTURE §9).
outcomes = reconcile_orphaned_runs(
DATA_DIR, running_task_ids=running_task_ids,
gateway_factory=lambda: (read_owned_gateway() if _stop_requested(stop_event)
@ -379,10 +454,9 @@ def _reconcile_delegated_runs(running_task_ids: Any, *, stop_event: Any = None)
)
if outcomes:
log.info("Delegated-run reconciliation handled %d orphan(s): %s", len(outcomes), outcomes)
# A run settled by this sweep may belong to a task that already wrote
# its terminal result with a non-empty unreconciled disclosure — the
# stored projection then lies forever (nanny-leaf S1). Audit-only
# refresh; never cancels.
# A run settled here may belong to a task whose stored terminal result
# carries an unreconciled disclosure that would lie forever (nanny-leaf
# S1). Audit-only refresh; never cancels.
from ouroboros.delegate_terminal import refresh_terminal_reconciliation
for tid in {str(o.get("task_id") or "") for o in outcomes
@ -529,21 +603,21 @@ def prune_agent_media_uploads(
def _startup_prune_sweeps(*, preserve_task_sources: bool = False) -> None:
"""Startup hygiene: prune stale task drives/trees and orphaned temp files."""
try:
from ouroboros.headless import prune_headless_task_drives, prune_task_drives, prune_task_trees
from ouroboros.headless import prune_task_trees
from ouroboros.utils import sweep_stale_temp_files
prune_report, task_drive_report = {}, {}
if preserve_task_sources:
log.warning("Startup task-source prune deferred: file recovery or ownership is unresolved")
else:
prune_report = prune_headless_task_drives(DATA_DIR)
task_drive_report = prune_task_drives(DATA_DIR)
# Child and direct drives are settled off the loop thread by the reconcile pass
# (``_run_drive_custody_pass``): readiness waits on no child-store copy or hash.
# Startup sweeps only the top-level tmp_scripts fallback (no script can be live
# yet); the whole-tree walk for atomic temps is owed to the first reconcile pass.
prune_task_trees(DATA_DIR)
sweep_stale_temp_files(DATA_DIR)
_prune_event("headless_task_drive_prune", ("pruned", "errors"),
report=prune_report, task_drives=task_drive_report)
sweep_stale_temp_files(DATA_DIR, atomic_temps=False)
_STARTUP_TEMP_SWEEP_OWED[0] = True
except Exception:
log.debug("Headless task drive prune failed", exc_info=True)
log.debug("Task tree prune failed", exc_info=True)
try:
# CPL4-C11 (owner batch 3A): clear owner state of tombstoned-uninstalled
# skills; grants survive as owner authority, reinstalls self-heal.
@ -590,29 +664,24 @@ def _startup_prune_sweeps(*, preserve_task_sources: bool = False) -> None:
log.debug("Agent media prune failed", exc_info=True)
if not preserve_task_sources:
try:
from ouroboros.observability import prune_observability_blobs
# Observability blobs are never deleted and never counted here: a startup census
# was 193k stat() calls that changed nothing (TZ-1 A).
from ouroboros.tools.services import prune_service_logs
_prune_event(
"runtime_artifact_prune",
("manifest_count", "blob_count", "deleted_dirs", "deleted_files", "errors"),
observability=prune_observability_blobs(DATA_DIR),
services=prune_service_logs(DATA_DIR))
_prune_event("runtime_artifact_prune", ("deleted_dirs", "deleted_files", "errors"),
services=prune_service_logs(DATA_DIR))
except Exception:
log.debug("Runtime artifact prune failed", exc_info=True)
def _cursor_refresh_settled_terminals(live_task_ids: Any = None) -> None:
"""Cursor-driven pass: runs settled OUTSIDE a generation's reconcile
outcomes (terminal-boundary settlements, earlier generations) never
reappear in the orphan sweep, so their tasks' stored evidence would stay
stale forever. Bounded to newly appended custody rows per tick, and reading
the same live-owner source as both custody surfaces: a task whose owner is
still billing is deferred rather than healed under a live writer. At BOOT
this runs AFTER the D1a backfill (see ``_startup_custody_sweep``), so a
same-generation heal keeps its pinned ``boot_backfill`` attribution and
the cursor's change-gated pass advances past it without a second write.
"""
"""Cursor-driven pass: runs settled OUTSIDE a generation's reconcile outcomes
(terminal-boundary settlements, earlier generations) never reappear in the
orphan sweep, so their stored evidence would stay stale forever. Bounded to
newly appended custody rows per tick; reads the same live-owner source as both
custody surfaces (a still-billing owner defers the heal). At BOOT it runs AFTER
the D1a backfill (``_startup_custody_sweep``), so a same-generation heal keeps
its pinned ``boot_backfill`` attribution without a second write."""
try:
from ouroboros.delegate_terminal import refresh_recently_settled_terminals
@ -682,20 +751,16 @@ def _startup_custody_sweep() -> None:
def _prune_delegated_snapshots() -> None:
"""C1 delegated execution snapshots: GC cross-checked against custody.
"""C1 delegated execution snapshots: GC cross-checked against custody. A snapshot
stays while its run is open/undisposed OR a pending invocation names it; the rest
(disposed, closed, refused) is torn down with its pinned baseline ref. Fail-soft
like every startup prune step, so the startup sequence never dies on a GC error.
A snapshot stays while its run is open/undisposed OR a pending invocation
names it; everything else (disposed, closed, refused) is torn down with its
pinned baseline ref. Fail-soft like every startup prune step — the guard
lives here so the startup sequence never dies on a GC error.
FAIL-CLOSED on an unreadable custody log (CR1-1): the keep-set comes from
replaying the custody rows, and ``_iter_rows`` swallows its own OSError —
right for the fail-soft readers, but here an unreadable log replays as
"no open runs", the keep-set goes EMPTY, and the prune destroys every
live snapshot with the child's only copy of its work. GC may delete only
over PROVEN settled && patch_disposed; an UNKNOWN custody state skips the
destructive prune entirely and says so loudly."""
FAIL-CLOSED on an unreadable custody log (CR1-1): the keep-set replays the
custody rows and ``_iter_rows`` swallows its own OSError, so an unreadable log
would replay as "no open runs", empty the keep-set and destroy every live
snapshot with the child's only copy of its work. GC deletes only over PROVEN
settled && patch_disposed; an UNKNOWN custody state skips the prune, loudly."""
try:
from ouroboros import delegate_custody as _delegate_custody
from ouroboros import subagent_worktrees as _snap_worktrees
@ -731,37 +796,121 @@ def _prune_delegated_snapshots() -> None:
log.debug("Delegated execution snapshot prune failed", exc_info=True)
def _periodic_zombie_reconcile(*, on_orphans_healed: Any = None) -> None:
"""Heal zombie 'running' records on a supervisor cadence.
# Memory only: where the last drive-custody pass stopped in each drive layout, so a bounded
# pass continues instead of re-attempting the same first drives (no ledger, no timer).
_DRIVE_PRUNE_CURSOR = {"headless": "", "direct": ""}
# Memory only: startup owes the first reconcile pass ONE whole-tree sweep of orphaned atomic
# temp files (a walk over the data root that no longer delays readiness).
_STARTUP_TEMP_SWEEP_OWED = [False]
A worker that died mid-review (crash / SIGKILL / manual stop) leaves
``review_job.json`` at status=running forever in headless/no-UI runs, where
the boot and ``GET /api/extensions`` reconciles never fire; the same death
leaves ``task_results/<id>.json`` at running. Both reconciles are
liveness-gated (pid-dead / queue-empty + worker-boot evidence), so a live
review or task is never touched.
"""
def _run_periodic_reconcile_sweep(marker: list, stop_event: Any = None, latch: Any = None,
on_orphans_healed: Any = None) -> None:
"""Off-loop 300 s history-sized heal, then the drive-custody pass (the ONE pending
child-ref promotion retry, then bounded settlement). The marker is stamped when the pass ENDS, before
its latch opens; the 20 s cancel cadence never waits on any walk and a slow pass never
restarts (issue #1230). The generation token is asked before every step and, by the
mutation owners, again before each publication commits."""
try:
_periodic_zombie_reconcile(on_orphans_healed=on_orphans_healed, stop_event=stop_event)
if _stop_requested(stop_event):
return
if _STARTUP_TEMP_SWEEP_OWED[0]:
from ouroboros.utils import sweep_stale_temp_files
_STARTUP_TEMP_SWEEP_OWED[0] = False
removed = sweep_stale_temp_files(pathlib.Path(DATA_DIR), scripts=False)
if removed:
log.info("Removed %d orphaned atomic temp file(s) left by a hard kill", removed)
if _stop_requested(stop_event):
return
_run_drive_custody_pass(stop_event)
_step_recovered("reconcile_sweep")
except Exception:
_step_failed("reconcile_sweep", "Periodic %s failed")
finally:
marker[0] = time.time()
(latch or _RECONCILE_SWEEP_LOCK).release()
def _run_drive_custody_pass(stop_event: Any = None) -> None:
"""The ONE pending child-ref promotion retry (the retry owner, before settlement asks),
the settled canonical mailboxes the loop thread's seam left (unread inputs to carry),
then settle terminal child and direct drives past retention (a cancelled subagent's at
once) through ``task_custody.settle_child_drive`` with the supervisor's probe and
ownership interlock, at most ``DRIVE_SETTLEMENTS_PER_PASS`` attempts per layout and
pass, continuing from the previous pass's cursor; crashed-settlement leftovers are
swept first. Off the loop thread: each settlement may copy and hash a child store.
A drive kept because its custody is not proven is material evidence in the event."""
from ouroboros.headless import DRIVE_SETTLEMENTS_PER_PASS, prune_headless_task_drives, prune_task_drives
from ouroboros.observability import retry_pending_child_ref_promotions
from ouroboros.owner_mailbox import sweep_settled_owner_mailboxes
from ouroboros.task_custody import sweep_custody_leftovers
from supervisor.queue import task_settlement_interlock, task_settlement_liveness
def stop() -> bool:
return _stop_requested(stop_event)
root = pathlib.Path(DATA_DIR)
report = retry_pending_child_ref_promotions(root, stop=stop) or {}
if report.get("retried") or report.get("errors"):
log.info("Child-ref promotion retry: %s", report)
if stop():
return
sweep_custody_leftovers(root)
# Canonical mailboxes the loop thread's seam left because unread rows carry inputs to
# verify or copy: carried and unlinked here, off the loop thread.
mailboxes = sweep_settled_owner_mailboxes(root, stop=stop)
reports = {}
for key, prune in (("headless", prune_headless_task_drives), ("direct", prune_task_drives)):
if stop():
return
reports[key] = prune(root, live=task_settlement_liveness, guard=lambda: task_settlement_interlock(stop=stop),
stop=stop, budget=DRIVE_SETTLEMENTS_PER_PASS, after=_DRIVE_PRUNE_CURSOR[key])
_DRIVE_PRUNE_CURSOR[key] = str(reports[key].get("cursor") or "")
_prune_event("headless_task_drive_prune", ("pruned", "errors", "custody_pending", "removed"),
report=reports["headless"], task_drives=reports["direct"], mailboxes=mailboxes)
def _periodic_zombie_reconcile(*, on_orphans_healed: Any = None, stop_event: Any = None) -> None:
"""Heal zombie 'running' records on a supervisor cadence. A worker that died
mid-review (crash / SIGKILL / manual stop) leaves ``review_job.json`` at running
forever in headless/no-UI runs, where the boot and ``GET /api/extensions``
reconciles never fire; the same death leaves ``task_results/<id>.json`` at
running. Both reconciles are liveness-gated (pid-dead / queue-empty + worker-boot
evidence), so a live review or task is never touched. Off the loop thread, so
``stop_event`` (the generation token) is re-read before every step."""
if _stop_requested(stop_event):
return
try:
from ouroboros.skill_review_runner import reconcile_stale_review_jobs
reconcile_stale_review_jobs(DATA_DIR)
except Exception:
log.debug("Periodic skill review-job reconcile failed", exc_info=True)
if _stop_requested(stop_event):
return
try:
from ouroboros.task_status import reconcile_orphaned_running_tasks
expired_quizzes: list = []
healed = reconcile_orphaned_running_tasks(DATA_DIR, expired_quizzes=expired_quizzes)
healed = reconcile_orphaned_running_tasks(
DATA_DIR, expired_quizzes=expired_quizzes,
write_guard=lambda task_id: orphan_reconcile_write_guard(task_id, stop_event=stop_event),
)
_publish_expired_quiz_frames(expired_quizzes)
if healed and callable(on_orphans_healed):
on_orphans_healed(int(healed))
except Exception:
log.debug("Periodic orphaned running-task reconcile failed", exc_info=True)
if _stop_requested(stop_event):
return
try:
from ouroboros.projects_registry import reconcile_projects
reconcile_projects(DATA_DIR)
except Exception:
log.debug("Project registry reconcile failed", exc_info=True)
_resume_interrupted_project_deletions()
if not _stop_requested(stop_event):
_resume_interrupted_project_deletions()
def _resume_interrupted_project_deletions() -> None:
@ -790,14 +939,10 @@ def _migrate_startup_cancel_latches(drive_root: pathlib.Path) -> None:
def _publish_expired_quiz_frames(expired: list) -> None:
"""Tell already-rendered cards that a healed terminal expired their question.
The same frame the task-done seam sends, from the one caller that is on the
supervisor side: the healer writes terminals off that seam, and the surfaces
Ouroboros runs on (packaged shell, mini app, phone) have no reload
affordance, so a card would keep a clickable question until navigation.
Fail-soft: the durable projection is already correct without the frame.
"""
"""Tell already-rendered cards that a healed terminal expired their question:
the same frame the task-done seam sends, from the supervisor-side healer that
writes terminals off that seam (the packaged shell, mini app and phone have no
reload affordance). Fail-soft: the durable projection is right without it."""
if not expired:
return
try:
@ -947,7 +1092,7 @@ def _run_startup_task_recovery(
drive_root: pathlib.Path, repo_dir: pathlib.Path, *, skip_live_data: bool,
prior_worker_pids: set[int] | None = None,
) -> dict:
"""File recovery precedes orphan materialization and the caller's actual prune.
"""File recovery precedes the orphan reconcile and the caller's actual prune.
Provider boot calls this from the supervisor, after process custody; the
no-provider lifespan calls it without spawning anything. There is no racing
@ -981,6 +1126,7 @@ def _run_startup_task_recovery(
expired_quizzes: list = []
reconcile_orphaned_running_tasks(
drive_root, exclude_task_ids=excluded, expired_quizzes=expired_quizzes,
write_guard=orphan_reconcile_write_guard,
)
_publish_expired_quiz_frames(expired_quizzes)
except Exception:

View file

@ -562,7 +562,8 @@ def _route_owner_message(bridge: Any, ctx: Any, incoming: Dict[str, Any]) -> Non
task_metadata = {**(task_metadata or {}), "origin_suppressed": True}
# Owner Surface Fact channel fallback: a non-web ingress (telegram/skill
# transports) carries no browser observables, but its channel IS the
# surface fact. Host-stamped here, never overwriting a real descriptor;
# surface fact, with the common ingress receipt stamp (``received_at``,
# ``enqueue_local_message``). Host-stamped here, never overwriting a real descriptor;
# source=="web" stays an honest absence (an old SPA sends no fact), and a
# synthetic A2A chat (negative id) is machine traffic — no owner sent it,
# so it must never wear an owner_client fact.
@ -574,7 +575,8 @@ def _route_owner_message(bridge: Any, ctx: Any, incoming: Dict[str, Any]) -> Non
and not _is_a2a(chat_id)
and not isinstance(task_metadata.get("client_surface"), dict)
):
task_metadata = {**task_metadata, "client_surface": {"channel": _ingress_source}}
received = {"received_at": str(incoming["received_at"])} if incoming.get("received_at") else {}
task_metadata = {**task_metadata, "client_surface": {"channel": _ingress_source, **received}}
if task_metadata.get("force_plan"):
from supervisor.worker_chat_lane import owner_conversation_admitted
from supervisor.state import budget_remaining, load_state

View file

@ -146,9 +146,11 @@ BAND_PATHS = {
"ouroboros/review_native_episode.py": "v7 follow-up F3 (owner Q6=A): a surface-declared mandatory reading became a typed FLOOR on the native episode bound (native_mandatory_read_bound / native_mandatory_read_disclosure / native_episode_transcript_bound \u2014 the one computation the advisory previews in its prompt's MANDATORY READ budget); the bound arithmetic stays with the episode that enforces it, 989->1062, no new subsystem, same seam; shrink-only direction.",
"ouroboros/reviewer_slot_config.py": "Absorbed the reviewer_slots() builder from review_substrate (altitude) and configured-subagent row resolution for the generic reviewer-actor bridge.",
"ouroboros/safety.py": "Entered the band from 954 lines with the safety-supervisor rate-limit fix: ONE shared model-call helper now serves both the primary and repair safety calls (it already deletes the duplicated call block), recognising a provider rate limit in BOTH wire shapes, taking one bounded deadline-capped backoff plus one retry, then blocking that one call with the typed non-verdict SAFETY_UNAVAILABLE outcome plus a durable audit row (a short storm latch answers further checks in the window without provider calls); the bounded newest-first conversation budget is the second half.",
"ouroboros/server_maintenance.py": "TZ-1 A: the bounded off-loop drive-custody pass (ref retry, settlement under the queue interlock) joined the reconcile block here; startup no longer copies child stores",
"ouroboros/skill_review_runner.py": None,
"ouroboros/subagent_worktrees.py": "Owner-sanctioned strict-registry delta (v7 rows 1083-1092, fork F-1=A) grew the module 1000->1082: typed refusal of a malformed registry instead of silent collapse-to-empty; shrink-only direction",
"ouroboros/subagents.py": "D07 split brought the dispatch monolith DOWN from 1593 into the band (->1370); route-health family extracted to subagent_route_health.py, shrink-only direction",
"ouroboros/task_custody.py": "Entered the band from 973 lines (TZ-1 PR-2 exact-review repair): the per-row input closure the mailbox cleanup and the drive settlement share (held projections re-verified, absent ones carried under the custody lock), the create-only placement and the generation fences at each publication boundary belong to the one custody owner; splitting them would separate the interlock from the effects it fences. Shrink-only from here.",
"ouroboros/task_pacing.py": "Cost-ceiling SSOT grew into the band while absorbing the cache-aware wrap-up reservation, the prepared/prospective wrap-up candidates, the deciding-spend basis vocabulary and the exhausted-ceiling texts; one owner for pacing decisions instead of a second pricing authority beside usage_accounting.py (at its ceiling).",
"ouroboros/task_status.py": None,
"ouroboros/tools/browser.py": None,
@ -182,6 +184,7 @@ BAND_PATHS = {
"tests/test_advisory_observability.py": None,
"tests/test_available_subagents_runtime.py": "Configured-session route and legacy custody regressions retained after removing compulsory source-request production tests.",
"tests/test_build_scripts.py": None,
"tests/test_child_drive_settlement.py": "Entered the band from 850 lines with the exact-review repair probes (current-bytes custody of mutable files, per-row input closure across mailbox cleanup, create-only placement, cancel-under-lock, generation fences, lock order): one probe per finding beside the settlement suite they constrain; split when a second custody surface lands.",
"tests/test_claudexor_runtime_delivery.py": "One suite pins managed runtime delivery end to end: closure install, exact Node selection and both POSIX tar.gz and Windows ZIP Node/npm toolchain extraction.",
"tests/test_commit_gate.py": None,
"tests/test_cybergym_dispatch.py": "CyberGym dispatch tests cover completion-order admission, transient gateway pauses and budget-refusal recovery through one existing fake campaign harness.",

1129
ouroboros/task_custody.py Normal file

File diff suppressed because it is too large Load diff

View file

@ -861,6 +861,11 @@ def write_task_result(
"""
path = task_result_path(results_drive_root, task_id)
explicit_ts = str(fields.pop("ts", "") or "")
from ouroboros.task_custody import capture_unread_mail, merge_unread_mail
# TZ-1 V10: the mailbox bytes are read BEFORE the row lock (a bounded union happens
# under it); a terminal write that the projector turns terminal captures under it.
captured = capture_unread_mail(results_drive_root, task_id) if status in _TRULY_TERMINAL_STATUSES else None
def _merge(existing: Dict[str, Any]) -> Optional[Dict[str, Any]]:
if strict_existing_dict and existing and (
@ -901,6 +906,16 @@ def write_task_result(
if resolve_task_lineage(task_id, metadata=merged.get("metadata"),
**{key: merged.get(key) for key in lineage_keys})["is_root_task"]:
projected_fields["canonical_terminal_projection_origin"] = "terminal_transition"
# TZ-1 V10: this accepted transition keeps the mail no attempt read (no ACK written);
# later late mail joins through settlement, never through a rejected write.
projected_fields["unread_mailbox"] = merge_unread_mail(
projected_fields.get("unread_mailbox"),
captured if status in _TRULY_TERMINAL_STATUSES else capture_unread_mail(results_drive_root, task_id))
# Unread-mail custody only grows: no replica or partial write can shrink it.
projected_fields["unread_mailbox"] = merge_unread_mail(
existing.get("unread_mailbox"), projected_fields.get("unread_mailbox"))
if projected_fields["unread_mailbox"] is None:
projected_fields.pop("unread_mailbox")
now = utc_now_iso()
# ABI-3 write seam: the merge BASE is normalized onto honest cost names first, so a stored alias
# neither survives nor outranks this write's fresh value; a legacy spelling IN this write still

View file

@ -7,7 +7,7 @@ import pathlib
from functools import partial
import time
from datetime import datetime, timezone
from types import SimpleNamespace
from contextlib import nullcontext
from typing import Any, Callable, Dict, Iterable, List, Optional
from ouroboros.headless import (
@ -406,16 +406,20 @@ class _EventsTailIndex:
return self._worker_boot
def _still_orphan_at_write(task_id: str, applied: List[bool], existing: Dict[str, Any],
def _still_orphan_at_write(task_id: str, observed: tuple, applied: List[bool], existing: Dict[str, Any],
fields: Dict[str, Any]) -> Optional[Dict[str, Any]]:
"""Projector for the reconciler's terminal write (bound with ``partial``): the decision was taken
outside the row lock, so a presence retry that went live meanwhile, or a row that already moved on
(requeued, settled by another writer), cancels the write. ``applied`` gets True only on a real write."""
outside the row lock, so a presence retry that went live meanwhile, or a row that moved on (requeued,
restarted as a new attempt even under the same status, settled by another writer), cancels the write:
``observed`` is the row's attempt basis plus ``updated_at`` at decision time. ``applied`` gets True
only on a real write."""
from ouroboros.presence_runner import presence_turn_is_live
from ouroboros.task_custody import attempt_basis
if presence_turn_is_live(task_id):
return None
if str(existing.get("status") or "").lower() not in {STATUS_RUNNING, STATUS_INTERRUPTED}:
if (attempt_basis(existing), existing.get("updated_at")) != observed \
or str(existing.get("status") or "").lower() not in {STATUS_RUNNING, STATUS_INTERRUPTED}:
return None
applied.append(True)
return fields
@ -614,6 +618,7 @@ def load_effective_task_result(
def reconcile_orphaned_running_tasks(
drive_root: Any, *, exclude_task_ids: frozenset[str] = frozenset(),
expired_quizzes: Optional[List[Any]] = None,
write_guard: Optional[Callable[[str], Any]] = None,
) -> int:
"""Durably finalize on-disk RUNNING task results the effective-status
projection already considers terminal.
@ -624,22 +629,23 @@ def reconcile_orphaned_running_tasks(
terminal status, but never persists it, so a headless/no-UI run that never
re-reads the result keeps the stale ``running`` on disk.
This sweep reuses ``load_effective_task_result`` so the persisted file matches
the read projection exactly and inherits ALL of its liveness gates: the grace
window, the worker-boot-after-task evidence, and the refusal to reconcile when
the queue snapshot is missing. A task that is still pending/running in the
queue, or whose worker has not booted after the task's last event, is never
reconciled. Two reads per candidate: the decision is a status-only
(``materialize_artifacts=False``) projection, and only a row it is about to
settle is read again with artifact materialization, so a live child's
scratch tree is never copied by this sweep. The monotonic guard in
``write_task_result`` additionally protects a genuinely newer terminal/cancel
write. Idempotent; safe at boot and on a periodic supervisor tick.
This sweep persists that projection and inherits ALL of its liveness gates:
the grace window, the worker-boot-after-task evidence, and the refusal to
reconcile when the queue snapshot is missing. It decides and persists on the
status-only (``materialize_artifacts=False``) projection, so no child file is
listed or copied here: the child store stays served read-only from its drive
and reaches the canonical store through ``task_custody.settle_child_drive``
before that drive can go. The write is fenced by the row's attempt basis and
``updated_at`` at decision time, and, given ``write_guard`` (the supervisor's
queue-ownership fence, queue -> row lock order), by live queue ownership; the
monotonic guard in ``write_task_result`` protects a newer terminal/cancel write.
Idempotent; safe at boot and on a periodic tick.
``expired_quizzes`` collects ``(task_id, quiz_id)`` for every question this
sweep expired, so the supervisor-side caller can send the same live frame the
task-done seam sends. This module stays free of a supervisor import.
"""
from ouroboros.task_custody import attempt_basis
from ouroboros.task_results import list_task_results, write_task_result
root = pathlib.Path(drive_root)
@ -666,15 +672,8 @@ def reconcile_orphaned_running_tasks(
except Exception:
log.debug("Orphan reconcile skipped %s: cancel authority unreadable", task_id, exc_info=True)
continue
# Decide on the status-only projection: a live row costs one projection and
# zero artifact transfers. Only a row this sweep is about to settle pays the
# materializing read, so the persisted terminal row keeps full custody
# (artifact rebase, verification receipts, artifact_bundle) (issue #1230).
try:
projected = load_effective_task_result(root, task_id, materialize_artifacts=False)
if str(projected.get("status") or "").strip().lower() not in SETTLED_STATUSES:
continue
effective = load_effective_task_result(root, task_id)
effective = load_effective_task_result(root, task_id, materialize_artifacts=False)
except Exception:
continue
eff_status = str(effective.get("status") or "").strip().lower()
@ -694,32 +693,36 @@ def reconcile_orphaned_running_tasks(
}
try:
applied: List[bool] = []
write_task_result(root, task_id, status=eff_status,
_field_projector=partial(_still_orphan_at_write, task_id, applied), **persist_fields)
if not applied:
continue # the row moved on between the effective read and this write: nothing was settled here
healed += 1
observed = (attempt_basis(row), row.get("updated_at"))
with (write_guard(task_id) if write_guard else nullcontext(True)) as allowed:
if not allowed:
continue
write_task_result(root, task_id, status=eff_status,
_field_projector=partial(_still_orphan_at_write, task_id, observed, applied),
**persist_fields)
if not applied:
continue # the row moved on between the effective read and this write: nothing was settled here
healed += 1
# This sweep is a terminal writer that never passes the task-done seam, so it closes the
# same per-task owner-control projections that seam closes (otherwise the record says
# "ended" while the card keeps an open question), before any next attempt can open new
# ones. Both legs are idempotent and fail-soft, as in the seam's own coordinator.
try:
from ouroboros.owner_hurry import reconcile_terminal as reconcile_hurry
reconcile_hurry(root, task_id)
except Exception:
log.debug("owner_hurry reconcile failed for healed %s", task_id, exc_info=True)
try:
from ouroboros.owner_quiz import reconcile_terminal as reconcile_quiz
expired = reconcile_quiz(root, task_id)
if expired_quizzes is not None:
expired_quizzes.extend((task_id, quiz_id) for quiz_id in expired)
except Exception:
log.debug("owner_quiz reconcile failed for healed %s", task_id, exc_info=True)
except Exception:
continue
# This sweep is a terminal writer that never passes the task-done seam, so
# it closes the same per-task owner-control projections that seam closes:
# otherwise the record says "ended" while the card still shows an open
# question and the paired wait never releases. Both legs are idempotent
# and fail-soft, exactly as in the seam's own coordinator.
try:
from ouroboros.owner_hurry import reconcile_terminal as reconcile_hurry
reconcile_hurry(root, task_id)
except Exception:
log.debug("owner_hurry reconcile failed for healed %s", task_id, exc_info=True)
try:
from ouroboros.owner_quiz import reconcile_terminal as reconcile_quiz
expired = reconcile_quiz(root, task_id)
if expired_quizzes is not None:
expired_quizzes.extend((task_id, quiz_id) for quiz_id in expired)
except Exception:
log.debug("owner_quiz reconcile failed for healed %s", task_id, exc_info=True)
return healed
@ -768,19 +771,19 @@ def effective_task_result(
) -> Dict[str, Any]:
"""Merge parent result, child-drive result, and active queue state.
``materialize_artifacts=False`` yields a "status/cost projection only" read
(v6.90.x P2): the entire artifact block — including the mutating child-artifact
rebase (``copy_file_to_task_artifacts``), ``collect_task_artifact_records``
scans, and the task-tree ``_project_child_result_disposition`` hash lookup — is
skipped, and the projection never carries sha-bearing/disposition claims.
``artifacts`` on a False row are the raw admission-time recorded entries,
not the merged/rebased set a materializing read would produce.
Read-only display surfaces (chat history annotation, ``api_tasks_list``, the
SSE follow loop, ``api_logs_tail`` discovery) pass ``False``; every consumer
that participates in the child-result sha economy or artifact durability
(join_ledger, wait_*/get_task_result, api_task_get/artifact, prune) keeps the
``True`` default. The orphan reconciler decides on a ``False`` read and
materializes only the row it heals.
Every read is pure (TZ-1 A): no file copy, hash, registration or manifest
write. Child files reach the canonical store only through copy-back and
``task_custody.settle_child_drive``. The default read adds the artifact VIEW
(``task_custody.store_artifact_view``: recorded canonical rows, the task's
OWN child stores' recorded rows and stat-only ``measured: False`` listings)
plus the task-tree disposition lookup. ``materialize_artifacts=False`` (the
historical name) is the "status/cost projection only" read: no view, no
disposition lookup and no sha-bearing/disposition claims; its ``artifacts``
are the raw recorded rows. Read-only display surfaces (chat history
annotation, ``api_tasks_list``, the SSE follow loop, ``api_logs_tail``
discovery) pass ``False``; child-result consumers (join_ledger,
wait_*/get_task_result, api_task_get/artifact, prune) keep the ``True``
default. The orphan reconciler decides and persists on a ``False`` read.
``_events_index`` optionally shares ONE parsed events-tail across a batch of
rows (the task-list request), so N stale-running rows cost one tail read
instead of N; ``None`` keeps the per-call read for single-row callers.
@ -966,10 +969,9 @@ def effective_task_result(
_apply_cancel_state_projection(pathlib.Path(drive_root), task_id, merged)
if not materialize_artifacts:
# Status/cost projection only: skip the whole artifact block (incl. the
# mutating child-artifact rebase and collect_task_artifact_records file
# scans) AND the disposition hash lookup. Strip the legacy mirrored
# fields so a False row never carries sha-bearing/disposition claims.
# Status/cost projection only: no artifact view (no store listing) and no
# disposition lookup. Strip the legacy mirrored fields so a False row
# never carries sha-bearing/disposition claims.
projected = dict(merged)
for field in _CHILD_DISPOSITION_FIELDS:
projected.pop(field, None)
@ -977,84 +979,24 @@ def effective_task_result(
projected["artifact_status"] = normalize_outcome_axes(projected)["artifacts"]["status"]
return projected
try:
from ouroboros.artifacts import (
collect_task_artifact_records,
copy_file_to_task_artifacts,
merge_artifact_records,
)
from ouroboros.outcomes import artifact_bundle_from_result
from ouroboros.task_custody import own_child_drives, store_artifact_view
if child_result:
parent_artifacts = [item for item in (result.get("artifacts") or []) if isinstance(item, dict)]
child_artifacts_for_merge = [item for item in (child_result.get("artifacts") or []) if isinstance(item, dict)]
if parent_artifacts or child_artifacts_for_merge:
merged["artifacts"] = merge_artifact_records(parent_artifacts, child_artifacts_for_merge)
rebased_child_artifacts: List[Dict[str, Any]] = []
if child_text:
parent_artifact_ctx = SimpleNamespace(drive_root=pathlib.Path(drive_root), task_id=task_id)
child_artifacts = merge_artifact_records(
[item for item in (child_result.get("artifacts") or []) if isinstance(item, dict)],
collect_task_artifact_records(pathlib.Path(child_text), task_id),
)
from ouroboros.outcome_receipt_store import (
is_verification_receipts_path,
publish_verification_receipt_union,
)
for child_artifact in child_artifacts:
source_text = str(child_artifact.get("path") or "").strip()
if not source_text:
continue
source = pathlib.Path(source_text).expanduser().resolve(strict=False)
if is_verification_receipts_path(child_text, task_id, source):
# Historical child results may already list the receipt
# stream as a generic artifact. Reconcile it through its
# one locked owner and never feed it to shutil.copy2.
publish_verification_receipt_union(
pathlib.Path(drive_root), task_id, pathlib.Path(child_text),
)
continue
try:
copied = copy_file_to_task_artifacts(
parent_artifact_ctx, source,
kind=str(child_artifact.get("kind") or "child_artifact"),
**({"immutable": True, "expected": child_artifact} if child_artifact.get("immutable") else {}),
)
if copied is None:
raise OSError("child artifact file is missing")
except (OSError, ValueError) as exc:
copied = {**child_artifact, "status": ARTIFACT_STATUS_FAILED,
"copy_status": "failed", "copy_error": f"{type(exc).__name__}: {exc}"}
promotion = merged.get("child_ref_promotion")
promotion = dict(promotion) if isinstance(promotion, dict) else {}
merged["child_ref_promotion"] = {
**promotion, "schema_version": 1, "status": "incomplete",
"pending_refs": [*(promotion.get("pending_refs") or []),
{"path": str(source), "kind": "task_artifact", "reason": copied["copy_error"]}],
}
rebased_child_artifacts.append(copied)
collected_artifacts = collect_task_artifact_records(drive_root, task_id)
if collected_artifacts or rebased_child_artifacts:
existing_artifacts = [item for item in (merged.get("artifacts") or []) if isinstance(item, dict)]
rebased_names = {
str(item.get("name") or pathlib.Path(str(item.get("path") or "")).name)
for item in rebased_child_artifacts
if isinstance(item, dict)
}
if rebased_names:
existing_artifacts = [
item
for item in existing_artifacts
if str(item.get("name") or pathlib.Path(str(item.get("path") or "")).name) not in rebased_names
]
collected_artifacts = collect_task_artifact_records(drive_root, task_id)
merged["artifacts"] = merge_artifact_records(existing_artifacts, rebased_child_artifacts, collected_artifacts)
# A pure view (TZ-1 A): recorded rows, the task's OWN child stores' recorded rows
# (their identity survives a suppressed cancelled child result) and stat-only
# listings; nothing is copied, hashed or registered here.
child_rows = {}
for drive in own_child_drives(drive_root, task_id):
rows = (load_task_result(drive, task_id) or {}).get("artifacts")
child_rows[drive / "task_results" / "artifacts" / task_id] = rows if isinstance(rows, list) else []
recorded = [item for item in (merged.get("artifacts") or []) if isinstance(item, dict)]
listed = store_artifact_view(drive_root, task_id, recorded, child_rows)
if listed:
merged["artifacts"] = listed
merged["artifact_bundle"] = artifact_bundle_from_result(merged)
merged["artifact_status"] = merged["artifact_bundle"].get("status")
except Exception:
pass
log.debug("Artifact view unavailable for %s", task_id, exc_info=True)
return _project_child_result_disposition(pathlib.Path(drive_root), merged)
@ -1091,12 +1033,10 @@ def wait_for_effective_tasks(
timed_out = False
early: Any = None
while True:
# Poll reads are status/cancel_state projections only: the
# materializing default performed real cross-tree writes (artifact
# copy2 + manifest rewrites) every 2s tick — a 600s wait over 5
# children was ~1500 spurious materializations. The one materializing
# read happens after the loop, on every exit path, so wait/get
# consumers keep their place in the child-result sha economy.
# Poll reads are status/cancel_state projections only (no store listing
# or disposition lookup per 2s tick). The one full read happens after the
# loop, on every exit path, so wait/get consumers keep their place in the
# child-result sha economy.
results = {
tid: load_effective_task_result(
pathlib.Path(drive_root), tid, materialize_artifacts=False
@ -1123,8 +1063,8 @@ def wait_for_effective_tasks(
timed_out = True
break
time.sleep(max(0.05, min(2.0, float(poll_interval_sec or 0.5), deadline - time.monotonic())))
# The single materializing read, on EVERY exit path (terminal, timeout,
# early beacon): the returned rows re-enter the sha/artifact economy.
# The single full read, on EVERY exit path (terminal, timeout, early
# beacon): the returned rows re-enter the sha/artifact economy.
results = {tid: load_effective_task_result(pathlib.Path(drive_root), tid) for tid in ids}
out: Dict[str, Any] = {
"mode": mode,
@ -1190,11 +1130,9 @@ def find_child_tasks(
def _raw_row_may_match(item: Dict[str, Any]) -> bool:
# Prefilter on the RAW disk row before paying for the effective
# projection: with the materializing default that projection is not a
# read — it copies files, rewrites artifact manifests and re-hashes
# every artifact of UNRELATED tasks (one finalization ran it 4-5x over
# the whole store). The lineage fields the filter needs are already on
# the raw row. Two classes must still materialize despite not matching
# projection (child-drive result, queue, store listing and disposition
# reads) for UNRELATED tasks. The lineage fields the filter needs are
# already on the raw row. Two classes must still be projected despite not matching
# raw: a row with a retry pointer (the retry chain projects the
# RETRY's lineage, which may match where the raw row does not), and a
# lineage-less row is safe to skip — a LIVE one is re-discovered by

View file

@ -378,6 +378,10 @@ def _get_task_result(
)
if isinstance(data.get("cancel_origin"), dict):
output += f"\n\n[CANCELLED_BY] {json.dumps(data['cancel_origin'], ensure_ascii=False)}"
from ouroboros.task_custody import unread_mail_notice
if unread := unread_mail_notice(data.get("unread_mailbox")):
output += f"\n\n{unread}" # TZ-1 V10: mail written to this task that its model never read
if trace and not unchanged:
output += f"\n\n[SUBTASK_TRACE]\n{trace}\n[/SUBTASK_TRACE]"
from ouroboros.task_finalization import provider_terminal_body, terminal_host_notice_text

View file

@ -1171,19 +1171,21 @@ def _code_search(ctx: ToolContext, query: str, path: str = ".",
def _forward_to_worker(
ctx: ToolContext, task_id: str, message: str, relayed_from_task_id: str = "",
) -> str:
"""Write a task-tree message into a running task's mailbox: one writer for a
"""Write a task-tree message into a running or queued task's mailbox: one writer for a
descendant (an ancestor's or relayed sibling's message), for the caller's
own parent or sibling (a peer contribution, never authority), and for any
active independent root the host lists (a message from an independent task,
owner 6C). Never owner text; WRITTEN, not read -- the recipient drains it
later, from its recorded drive or the canonical root, never the sender's."""
later, from its recorded drive or the canonical root, never the sender's.
The receipt follows ``owner_mailbox.mail_write_receipt`` (TZ-1 V10): queued (the
recipient has not started) or delivered to a live drain; never "read"."""
from ouroboros.owner_mailbox import (
PROVENANCE_INDEPENDENT_TASK, PROVENANCE_PEER_TASK, TASK_MESSAGE_MAX_CHARS, write_task_message,
)
from ouroboros.peer_roster import (
durable_descendant_of, host_listed_independent_root, peer_contribution_admission,
)
from ouroboros.task_results import STATUS_RUNNING, validate_task_id
from ouroboros.task_results import STATUS_RUNNING, STATUS_SCHEDULED, validate_task_id
from ouroboros.task_status import FINAL_STATUSES, load_effective_task_result
try:
@ -1203,8 +1205,8 @@ def _forward_to_worker(
return _publish_tool_result(ctx, ToolResult(status="unavailable", code="LEGACY_UNAVAILABLE", text=(f"⚠️ TASK_NOT_FOUND: task {tid} is not registered.")))
if status in FINAL_STATUSES:
return _publish_tool_result(ctx, ToolResult(status="blocked", code="LEGACY_BLOCKED", text=(f"⚠️ TASK_NOT_ACTIVE: task {tid} is already {status}.")))
if status != STATUS_RUNNING:
return _publish_tool_result(ctx, ToolResult(status="blocked", code="LEGACY_BLOCKED", text=(f"⚠️ TASK_NOT_ACTIVE: task {tid} is {status or 'unknown'}, not running.")))
if status not in (STATUS_RUNNING, STATUS_SCHEDULED):
return _publish_tool_result(ctx, ToolResult(status="blocked", code="LEGACY_BLOCKED", text=(f"⚠️ TASK_NOT_ACTIVE: task {tid} is {status or 'unknown'}, neither running nor queued.")))
# AR2-6: no NEW steering writes while a cancellation is pending. The
# effective status honestly stays ``running`` (cancel_state=pending rides
# beside it), so the checks above pass — consult the same predicate the
@ -1278,6 +1280,24 @@ def _forward_to_worker(
)
if not written:
return _publish_tool_result(ctx, ToolResult(status="error", code="TOOL_ERROR", text=(f"⚠️ TASK_MESSAGE_UNWRITTEN: message to task {tid} was not persisted.")))
from ouroboros.owner_mailbox import MAIL_QUEUED, MAIL_RETAINED_UNREAD, mail_write_receipt, mailbox_drain_ended
try:
drain_ended = mailbox_drain_ended(mailbox_drive, tid)
except Exception:
drain_ended = False
receipt = mail_write_receipt(status, drain_ended=drain_ended)["receipt"]
if receipt == MAIL_RETAINED_UNREAD:
return (f"Message forwarded to task {tid}: written to its mailbox ({MAIL_RETAINED_UNREAD}); task {tid}'s "
"own drain has already ended, so no checkpoint will read it: its result keeps it as unread mail. "
"Files cannot be attached to messages between tasks.")
if receipt == MAIL_QUEUED:
as_from = (f" as a message from a peer task (your {relation}; never owner text or an ancestor's steering)"
if provenance == PROVENANCE_PEER_TASK else " as a message from this task (never owner text)"
if listed_root is not None else "")
return (f"Message forwarded to task {tid}: written to its mailbox{as_from} ({MAIL_QUEUED}); task {tid} has not "
"started, so nothing has read it: it reads it when it starts, and if it ends unstarted its result keeps "
"it as unread mail. Files cannot be attached to messages between tasks.")
if provenance == PROVENANCE_PEER_TASK:
return (f"Message forwarded to task {tid}: written to its mailbox as a message from a peer task "
f"(your {relation}; never owner text or an ancestor's steering); it reads it at its next "
@ -1453,18 +1473,19 @@ def get_tools() -> List[ToolEntry]:
ToolEntry("forward_to_worker", {
"name": "forward_to_worker",
"description": (
"Write an addressed task-tree message into a running task's mailbox: a child "
"Write an addressed task-tree message into a running or queued task's mailbox: a child "
"or descendant of yours (delivered as the ancestor's message), your own parent "
"or a sibling (delivered as a message from a peer task naming the relation — "
"a contribution it weighs, never steering; relay is refused there), or any active "
"independent root the host lists (delivered as a message from an independent "
"task). It is never labelled owner dialogue, files cannot be attached, the body "
"is limited to 8000 chars (longer is refused, never truncated), and the "
"result says the message was written, not read: the task drains it at its next "
"checkpoint. To wait for a reply without spending model rounds, call await_messages."
"result says written, not read: a running task drains it at its next checkpoint, a queued "
"one when it starts, and a task that ends without reading it keeps it as unread mail in its "
"result. To wait for a reply without spending model rounds, call await_messages."
),
"parameters": {"type": "object", "properties": {
"task_id": {"type": "string", "description": "ID of the running task to forward to"},
"task_id": {"type": "string", "description": "ID of the running or queued task to forward to"},
"message": {"type": "string", "description": "Message text to forward (at most 8000 chars)"},
"relayed_from_task_id": {"type": "string", "description":
"Optional sibling/descendant task whose output this ancestor relays. "

View file

@ -367,6 +367,22 @@ def _scan_directory_output_members(
return sorted(members, key=lambda item: item.as_posix()), dir_size, "", skipped
def _record_skipped_members(ctx: ToolContext, output: str, members: List[str]) -> None:
"""The COMPLETE skip list as one durable row in the task's ``events.jsonl`` (the rendered
note is bounded); a context without a log root falls back to the process log."""
from ouroboros.utils import append_jsonl, utc_now_iso
row = {"ts": utc_now_iso(), "type": "directory_output_members_skipped",
"task_id": str(getattr(ctx, "task_id", "") or ""), "output": str(output),
"count": len(members), "members": list(members)}
try:
if not append_jsonl(ctx.drive_logs() / "events.jsonl", row):
raise OSError("events.jsonl append refused")
except Exception:
log.info("task %s: directory output %s: full export skip list (%d): %s",
row["task_id"] or "?", output, len(members), "; ".join(members))
def _register_process_outputs(
ctx: ToolContext,
outputs: List[str] | None,
@ -445,16 +461,12 @@ def _register_process_outputs(
# the task log so the omission stays resolvable (#447 P1).
shown = "; ".join(skipped_members[:5])
more = (
f" (+{len(skipped_members) - 5} more; full list in server.log,"
f" task {getattr(ctx, 'task_id', '') or '?'})"
f" (+{len(skipped_members) - 5} more; full list: directory_output_members_skipped"
f" in the task's events.jsonl)"
if len(skipped_members) > 5 else ""
)
if len(skipped_members) > 5:
log.info(
"task %s: directory output %s: full export skip list (%d): %s",
getattr(ctx, "task_id", "") or "?",
text, len(skipped_members), "; ".join(skipped_members),
)
_record_skipped_members(ctx, text, skipped_members)
notes.append(
f"skipped {len(skipped_members)} member(s) of directory output {text}: {shown}{more}"
)

View file

@ -352,7 +352,8 @@ def atomic_write_json(path: pathlib.Path, payload: Any, *, trailing_newline: boo
write_text_atomic(pathlib.Path(path), content, fsync=fsync)
def sweep_stale_temp_files(root: pathlib.Path, *, min_age_sec: float = 3600.0) -> int:
def sweep_stale_temp_files(root: pathlib.Path, *, min_age_sec: float = 3600.0,
atomic_temps: bool = True, scripts: bool = True) -> int:
"""Remove orphaned atomic-write temp files left behind by a hard kill.
``atomic_write_json`` writes to a unique ``.{name}.tmp.<pid>.<tid>.<uuid>``
@ -370,7 +371,9 @@ def sweep_stale_temp_files(root: pathlib.Path, *, min_age_sec: float = 3600.0) -
``tools/shell.py`` unlinks its ``script_<uuid>.<ext>`` files in a
``finally``, so one that survived is a hard-kill orphan. Only the
TOP-LEVEL fallback dir is swept here — task-drive copies die with their
drive's own GC prune — and only at startup, when no script can be live.
drive's own GC prune — and only at startup, when no script can be live
(``scripts``); the whole-tree walk for atomic temps (``atomic_temps``) is the
expensive half and runs off the loop thread, in the first reconcile pass.
"""
root = pathlib.Path(root)
if not root.is_dir():
@ -379,8 +382,9 @@ def sweep_stale_temp_files(root: pathlib.Path, *, min_age_sec: float = 3600.0) -
removed = 0
now = time.time()
try:
candidates = list(root.rglob(".*.tmp.*"))
candidates.extend(root.glob("tmp_scripts/script_*"))
candidates = list(root.rglob(".*.tmp.*")) if atomic_temps else []
if scripts:
candidates.extend(root.glob("tmp_scripts/script_*"))
except OSError:
return 0
fallback_scripts = root / "tmp_scripts"

View file

@ -124,7 +124,9 @@ _pytest_default_real_data_dir = (
and not os.environ.get("OUROBOROS_DATA_DIR")
and DATA_DIR == pathlib.Path.home() / "Ouroboros" / "data"
)
if _pytest_default_real_data_dir:
if _pytest_default_real_data_dir or __name__ == "__mp_main__":
# A spawn/forkserver worker re-imports this module as ``__mp_main__``: it gets a stream
# handler only, so two processes never rotate ``server.log`` against each other.
logging.basicConfig(level=logging.INFO, format=_LOG_FORMAT, handlers=[logging.StreamHandler()])
else:
_log_dir = DATA_DIR / "logs"
@ -559,7 +561,7 @@ def _handle_bridge_update_batch(bridge, updates, offset: int, ctx: Any, cursor:
"task_metadata": task_metadata,
"log_text": log_text,
"origin_message_ref": origin_message_ref,
"source": source,
"source": source, "received_at": str(msg.get("received_at") or ""),
},
)
return offset
@ -650,7 +652,13 @@ def _run_supervisor(settings: dict) -> None:
log.debug("Failed to stop previous consciousness instance", exc_info=True)
_consciousness = None
prior_worker_pids: set[int] | None = None
_watchdog_stop = threading.Event() # per-generation: set on EVERY exit of this generation
try:
# Watch startup stalls; even a failed watchdog start publishes an init outcome.
from ouroboros.server_liveness import loop_phase_facts
_loop_liveness = [time.monotonic(), {}, time.thread_time(), None] # slots: server_liveness.py
_loop_liveness[1], _loop_liveness[0] = loop_phase_facts(_loop_liveness, "startup", new_tick=True), time.monotonic()
_start_supervisor_liveness_watchdog(_loop_liveness, _watchdog_stop)
ensure_legacy_imported(pathlib.Path(DATA_DIR))
from supervisor.message_bus import LocalChatBridge, init as bus_init
@ -807,6 +815,7 @@ def _run_supervisor(settings: dict) -> None:
_supervisor_ready.clear() # never reached its loop: the API must not paint Online over the error
_supervisor_init_done.set()
_supervisor_thread = None
_watchdog_stop.set() # a generation that died in init has no loop to watch
return
_supervisor_ready.set()
@ -818,15 +827,8 @@ def _run_supervisor(settings: dict) -> None:
crash_count = 0
_last_custody_reap = [time.time()]
_last_review_job_reconcile = [time.time()]
# WS3: a dedicated watchdog thread (outside this loop, so it fires even if the
# loop stalls) surfaces a wedge as an observable signal + owner alert instead
# of silent hours; the loop publishes a liveness tick at each tick PHASE. The
# tick is MONOTONIC: it is only ever read as an elapsed gap, so a wall-clock
# jump must not turn a healthy loop into a phantom stall (nor hide a real one).
from ouroboros.server_liveness import loop_phase_facts
_loop_liveness = [time.monotonic(), {}, time.thread_time(), None] # slots: server_liveness.py
_watchdog_stop = threading.Event() # per-generation: stops the watchdog when THIS loop exits
_start_supervisor_liveness_watchdog(_loop_liveness, _watchdog_stop)
# The watchdog was started before startup recovery; never start another here.
while not _restart_requested.is_set() and not _supervisor_stop.is_set() and not _exit_signalled.is_set():
try:
_loop_liveness[1], _loop_liveness[0] = loop_phase_facts(_loop_liveness, "events", new_tick=True), time.monotonic()

View file

@ -447,12 +447,9 @@ def _publish_cancelled_task(
# the lock so the crash detector can recover the slot on a later tick.
from supervisor.task_reaper import _respawn_after_reap
_respawn_after_reap(q, workers, worker.wid, expected_worker=worker)
if str(task.get("delegation_role") or "") == "subagent":
try:
from ouroboros.headless import remove_subagent_task_drive
remove_subagent_task_drive(q.DRIVE_ROOT, str(task_id))
except Exception:
log.debug("Failed to remove cancelled subagent drive for %s", task_id, exc_info=True)
# A cancelled subagent's drive is NOT settled here: settlement copies and hashes the
# child store, which the cancel path must not carry. The off-loop reconcile pass
# settles it without waiting out retention (``headless.prune_headless_task_drives``).
try:
q.persist_queue_snapshot(reason="cancel_running")
except Exception:

View file

@ -579,7 +579,9 @@ def _finish_task_done_dispatch(
try:
from supervisor.terminal_delivery import cleanup_settled_owner_mailbox
cleanup_settled_owner_mailbox(ctx.DRIVE_ROOT, str(task_id), task)
# The loop thread copies and hashes nothing: a mailbox with unread inputs to carry
# waits for the off-loop mailbox sweep or the drive settlement.
cleanup_settled_owner_mailbox(ctx.DRIVE_ROOT, str(task_id), task, carry_inputs=False)
except Exception:
log.warning("Failed to cleanup terminal owner mailbox for %s", task_id, exc_info=True)

View file

@ -90,8 +90,9 @@ def accept_local_message(bridge, drive_root, text: str, *, retain_inputs=None, *
if retain_inputs is not None:
retain_inputs()
ref = build_owner_message_ref(chat_id=chat_id, client_message_id=message_id, ts=ts, text=logged)
# The row rides the item as its in-process witness (``record_inbound_message``).
bridge.enqueue_local_message(text, **message, accepted_source_ref=ref, accepted_source_row=row)
# The row rides the item as its in-process witness (``record_inbound_message``); its
# acceptance time is this message's receipt stamp.
bridge.enqueue_local_message(text, **message, accepted_source_ref=ref, accepted_source_row=row, received_at=ts)
return row, False
@ -357,7 +358,7 @@ class LocalChatBridge:
for key in (
"sender_label", "sender_session_id", "client_message_id", "transport",
"image_base64", "image_mime", "image_caption", "suppress_chat_log",
"task_constraint", "task_metadata", "accepted_source_ref", "accepted_source_row",
"task_constraint", "task_metadata", "accepted_source_ref", "accepted_source_row", "received_at",
):
value = msg.get(key)
if value not in (None, "", 0):
@ -508,7 +509,13 @@ class LocalChatBridge:
task_metadata: Optional[Dict[str, Any]] = None,
accepted_source_ref: Optional[Dict[str, Any]] = None,
accepted_source_row: Optional[Dict[str, Any]] = None,
received_at: str = "",
) -> None:
"""The ONE ingress every transport enqueues through, so it stamps the host receipt time
``received_at`` for all of them: an earlier host stamp of this message (its accepted row's
time, the WS acceptance's ``client_surface.received_at``) is kept, else now. The update
carries it; a surface fact without one (a host channel stamp) gets it too, so every
record that copies the fact carries it. Nothing else: a dict put, no lock, no I/O."""
clean_text = str(text or "").strip()
caption_text = str(image_caption or "").strip()
image_b64 = str(image_base64 or "").strip()
@ -516,6 +523,11 @@ class LocalChatBridge:
clean_text = caption_text
if not clean_text and not image_b64 and not (task_metadata or {}).get("chat_attachment_uploads"):
return # nothing to say and nothing attached (a file-only message carries uploads)
surface = (task_metadata or {}).get("client_surface")
surface = surface if isinstance(surface, dict) and surface else None
received_at = str(received_at or (surface or {}).get("received_at") or "") or utc_now_iso()
if surface is not None and not surface.get("received_at"):
task_metadata = {**(task_metadata or {}), "client_surface": {**surface, "received_at": received_at}}
# Invariant: the default chat/user id is the web owner (1). External
# transports (source != "web") MUST pass explicit ids — the Host Service
# injects 0 for unidentified senders so they can never bind/own the web
@ -537,6 +549,7 @@ class LocalChatBridge:
"task_metadata": dict(task_metadata or {}),
"accepted_source_ref": dict(accepted_source_ref or {}),
"accepted_source_row": dict(accepted_source_row or {}),
"received_at": received_at,
})
def send_message(

View file

@ -73,9 +73,15 @@ SCHEDULED_TASKS_FILE = pathlib.Path("state") / "scheduled_tasks.json"
OBJECTIVE_REPEAT_CAP: int = 3
# Whether THIS process owns the supervisor's live maps (``init`` ran): a process that merely
# imports the module sees empty maps, which prove nothing (``task_settlement_liveness``).
INITIALIZED = False
def init(drive_root: pathlib.Path) -> None:
global DRIVE_ROOT, FINALIZATION_GRACE_SEC, QUEUE_SNAPSHOT_PATH
global DRIVE_ROOT, FINALIZATION_GRACE_SEC, INITIALIZED, QUEUE_SNAPSHOT_PATH
DRIVE_ROOT = drive_root
INITIALIZED = True
QUEUE_SNAPSHOT_PATH = drive_root / "state" / "queue_snapshot.json"
FINALIZATION_GRACE_SEC = get_finalization_grace_sec()
BUDGET_ROOT_FENCES.clear()
@ -455,6 +461,8 @@ from supervisor.queue_transitions import ( # noqa: E402, F401 -- intentional pu
evolution_stop_report,
stop_evolution_tasks,
sweep_orphaned_budget_fences,
task_settlement_interlock,
task_settlement_liveness,
)

View file

@ -24,6 +24,7 @@ nothing from ``task_lifecycle``, which re-exports these names so
from __future__ import annotations
import contextlib
import logging
import pathlib
import threading
@ -899,6 +900,44 @@ def task_has_live_ownership(task_id: str) -> bool:
)
def task_settlement_liveness(task_id: str) -> Optional[bool]:
"""The probe a destructive custody settlement asks (``task_custody.settle_child_drive``):
True while ``task_has_live_ownership`` holds or the id waits in PENDING, None while a
reap of this task (its slot or a queued/deferred reap job) makes absence inconclusive
or the queue cannot be read, False only for proven absence. Callers without the
supervisor's live maps have no such probe, so they never delete."""
q = _queue_module()
from supervisor import task_reaper, workers
task_id = str(task_id or "").strip()
if not getattr(q, "INITIALIZED", False):
return None # empty maps outside the supervisor process prove no absence
try:
with q._queue_lock:
if task_has_live_ownership(task_id) or any(str(row.get("id") or "") == task_id for row in q.PENDING):
return True
with q._reap_queue.mutex:
jobs = [*q._reap_queue.queue, *task_reaper._deferred_reap_jobs]
if any(worker.reaping and worker.busy_task_id == task_id for worker in workers.WORKERS.values()) \
or any(isinstance(job, dict) and str(job.get("task_id") or "") == task_id for job in jobs):
return None # a reap of this task is queued or holds its slot: its process may live
return False
except Exception:
log.warning("Settlement liveness of %s is unknown", task_id, exc_info=True)
return None
@contextlib.contextmanager
def task_settlement_interlock(stop: Any = None):
"""The ownership interlock a drive settlement moves under (``task_custody.settle_child_drive``
``guard``): the queue lock admission and assignment hold, so no occupant can be admitted,
assigned or retried between the settlement's last probe and the drive's move. Yields
whether the caller's generation is still open (``stop()`` False)."""
q = _queue_module()
with q._queue_lock:
yield not (callable(stop) and stop())
def run_project_deletion(
drive_root: object,
project_id: str,

View file

@ -677,14 +677,15 @@ def _run_retry_admission_transaction(
def _discard_retry_inputs(q: Any, task: Dict[str, Any], task_id: str, retry_task_id: str) -> None:
"""Drop the inputs copied onto a retry id that will never run."""
"""Drop the inputs copied onto a retry id that will never run: by-value copies whose original id keeps
every row and input, never admitted data, so no custody settlement (``settle_child_drive``) is owed."""
if not retry_task_id or retry_task_id == task_id:
return
from ouroboros.artifacts import task_artifact_dir_path
from ouroboros.owner_mailbox import cleanup_task_mailbox
from ouroboros.owner_mailbox import discard_mailbox_copy
drive = q._task_drive_for_task(task, task_id)
cleanup_task_mailbox(drive, retry_task_id)
discard_mailbox_copy(drive, retry_task_id)
shutil.rmtree(task_artifact_dir_path(drive, retry_task_id), ignore_errors=True)

View file

@ -68,16 +68,16 @@ _HOST_SALVAGE_RECEIPT = (
def cleanup_settled_owner_mailbox(
drive_root: Any, task_id: str, task: Optional[Dict[str, Any]] = None,
drive_root: Any, task_id: str, task: Optional[Dict[str, Any]] = None, *, carry_inputs: bool = True, stop: Any = None,
) -> None:
"""Release the execution mailbox only after its canonical obligations settle."""
"""Release the execution mailbox once its canonical obligations settle and the canonical row holds its unread rows with their input closure; ``carry_inputs=False`` (the loop thread) leaves inputs to an off-loop owner, whose generation ``stop()`` fences the cleanup."""
from ouroboros.owner_mailbox import cleanup_task_mailbox, settled_mailbox_cleanup_allowed
from ouroboros.task_results import load_task_result
from supervisor.queue import _task_drive_for_task
durable = load_task_result(pathlib.Path(drive_root), str(task_id)) or {}
if settled_mailbox_cleanup_allowed(durable):
cleanup_task_mailbox(_task_drive_for_task(task or durable, str(task_id)), str(task_id))
cleanup_task_mailbox(_task_drive_for_task(task or durable, str(task_id)), str(task_id), canonical_root=drive_root, carry_inputs=carry_inputs, stop=stop)
def _registry_path(drive_root: Any) -> pathlib.Path:

View file

@ -488,8 +488,9 @@ def _stage_promoted_initial_attachments(
if attachment_manifest_all_rejected(manifest):
remove_staged_attachments(manifest)
from ouroboros.headless import remove_subagent_task_drive
remove_subagent_task_drive(DRIVE_ROOT, tid)
from supervisor.queue import task_settlement_interlock, task_settlement_liveness
remove_subagent_task_drive(DRIVE_ROOT, tid, live=task_settlement_liveness,
guard=task_settlement_interlock, admission_rollback=True)
return manifest, {
"status": "needs_manual_target",
"reason": "attachment_admission_rejected",

View file

@ -19,6 +19,16 @@ from ouroboros.task_results import (
)
def settled_off_loop(drive, task_id: str, child) -> bool:
"""TZ-1 A: the cancel path deletes no execution drive. It stays for the off-loop settlement
(``task_custody.settle_child_drive``, a cancelled subagent's at once), which removes it only
with its custody proven; True when the drive was kept and that settlement then removed it."""
from ouroboros.task_custody import settle_child_drive
return child.is_dir() and settle_child_drive(drive, task_id, child, live=lambda _task: False)["status"] == "removed" \
and not child.exists()
def _write_root_retry_pair(tmp_path, old_id: str, new_id: str, *, new_status="scheduled"):
write_task_result(
tmp_path,

View file

@ -18,7 +18,8 @@ def _managed_worker_pool_available(monkeypatch):
"""HTTP task tests model a ready server unless a case overrides the pool."""
import supervisor.workers as workers
monkeypatch.setattr(workers, "WORKERS", {0: SimpleNamespace()})
# An idle slot: the liveness readers (settlement probe, live ownership) read these facts.
monkeypatch.setattr(workers, "WORKERS", {0: SimpleNamespace(busy_task_id=None, reaping=False)})
monkeypatch.setattr(workers, "_WORKER_POOL_DISABLED_REASON", "")

View file

@ -540,6 +540,15 @@ def test_task_result_row_publishes_the_runtime_reason_alongside_the_adapter_stag
# enforces — a new runtime code with no row here fails the suite, which is the only thing
# that stops the vocabulary from being hand-copied beside the check again.
_TRUNCATION_DECISIONS: dict[str, tuple[bool, str]] = {
"artifact_archive_empty": (False, "gateway/task_archive.py: no eligible recorded directory member; HTTP refusal, not a task terminal"),
"artifact_archive_invalid": (False, "gateway/task_archive.py: invalid selector; HTTP refusal, not a task terminal"),
"artifact_archive_unavailable": (False, "gateway/task_archive.py: confined read or spool unavailable; HTTP refusal, not a task terminal"),
"artifact_archive_unverified": (False, "gateway/task_archive.py: member drift or capture verification failure; HTTP refusal, not a task terminal"),
"artifact_identity_changed": (False, "gateway/task_archive.py: a mutable file's bytes no longer match its recorded identity; HTTP 409, not a task terminal"),
"artifact_name_ambiguous": (False, "gateway/tasks.py: ambiguous nested basename; HTTP refusal, not a task terminal"),
"artifact_relpath_invalid": (False, "gateway/tasks.py: invalid exact artifact selector; HTTP refusal, not a task terminal"),
"artifact_unavailable": (False, "gateway/task_archive.py: confined single-file read unavailable; HTTP refusal, not a task terminal"),
"artifact_unverified": (False, "gateway/task_archive.py: single-file drift or capture verification failure; HTTP refusal, not a task terminal"),
"history_source_unavailable": (False, "gateway/history_paging.py: readable recent projection with explicit source gap; no task attempt was truncated"),
"late_answer_not_delivered": (False, "gateway/task_decision.py: a late quiz answer was recorded but its chat delivery failed (503, retry); no task attempt was truncated"),
"budget_pausing_no_extraction": (False, "review_verdict_extraction.py: Light verdict extraction refused while the task's exact budget pause is closing dispatch (#1196); the review row stays undispatched and the attempt is paused, not truncated"),

View file

@ -20,7 +20,7 @@ from ouroboros.task_results import (
write_task_result,
)
from tests._cancel_intents_shared import _CaptureQueue, _LiveProc, _live_split_drive_task, _seed_llm_response
from tests._cancel_intents_shared import _CaptureQueue, _LiveProc, _live_split_drive_task, _seed_llm_response, settled_off_loop
from tests._cancel_intents_shared import ( # noqa: F401 (autouse fixture applies on import)
_reap_spawned_live_procs,
)
@ -77,7 +77,7 @@ def test_e2e_tool_cancel_kills_live_worker_and_settles_with_cost(qenv, monkeypat
assert stored["parent_decision"] == "cancelled" # stamped at OUTCOME
assert stored.get("cost_accounting_status") == "available" # reconstructed
assert ci.active_intent(qenv.drive, task_id) is None
assert not child_drive.exists(), "cancelled subagent drive is cleaned up"
assert settled_off_loop(qenv.drive, task_id, child_drive), "the off-loop settlement removes the cancelled subagent's drive"
# task_done carries the reconstructed accounting — never a fabricated final $0
# (an empty ledger reconstructs to a CONFIRMED zero, which is fine).
(done,) = done_events

View file

@ -68,10 +68,10 @@ TERMINAL_WRITERS = {
# Runtime707: CURRENT-ref retry publication moved out of observability's
# locked sweep; terminal file-failure publication moved off event drain.
# Both retain CURRENT lifecycle status rather than authoring completion.
('ouroboros/headless.py::retry_child_task_refs', 'source["status"]'): 'dynamic',
('ouroboros/headless.py::_retry_child_task_refs_locked', 'source["status"]'): 'dynamic',
('ouroboros/headless.py::prepare_terminal_task_files', 'existing["status"]'): 'dynamic',
('ouroboros/headless.py::finalize_task_artifacts', 'status'): 'dynamic',
('ouroboros/headless.py::finalize_task_artifacts', 'str(existing.get("status") or status or "completed")'): 'terminal',
('ouroboros/headless.py::_finalize_task_artifacts_locked', 'status'): 'dynamic',
('ouroboros/headless.py::_finalize_task_artifacts_locked', 'str(existing.get("status") or status or "completed")'): 'terminal',
('ouroboros/mutation_attribution.py::advance_mutation_baseline', 'status'): 'dynamic',
('ouroboros/mutation_attribution.py::capture_mutation_baseline', 'status'): 'dynamic',
('ouroboros/mutation_attribution.py::record_terminal_mutation_candidates', 'status'): 'dynamic',
@ -100,6 +100,10 @@ TERMINAL_WRITERS = {
# write_task_result still preserves any terminal status under its locked reducer.
('ouroboros/server_maintenance.py::_recover_terminal_task_files', '"running"'): 'dynamic',
('ouroboros/task_status.py::reconcile_orphaned_running_tasks', 'eff_status'): 'dynamic',
# TZ-1 A/V10: the one child-drive settlement and mailbox cleanup write custody fields
# (published artifact rows, unread mail) onto CURRENT with its own status inside the
# projector; a changed attempt basis aborts, and they never originate a transition.
('ouroboros/task_custody.py::_write_custody_fields', 'str(observed.get("status") or "")'): 'dynamic',
('ouroboros/tools/control_delegation.py::record_depth_limit_refusal', 'STATUS_FAILED'): 'terminal',
('supervisor/cancel_publication.py::_finalize_cancel_intent_on_miss', 'STATUS_CANCELLED'): 'terminal',
('supervisor/events_project_routing.py::_persist_promote_rejection', 'STATUS_FAILED'): 'terminal',

View file

@ -6,7 +6,7 @@ import pytest
from ouroboros import cancel_intents, headless
from ouroboros.task_results import load_task_result, write_task_result
from tests._cancel_intents_shared import qenv # noqa: F401
from tests._cancel_intents_shared import qenv, settled_off_loop # noqa: F401
@pytest.fixture
@ -57,7 +57,7 @@ def test_cancel_prepares_early_terminal_before_cleanup_and_real_dispatch(split,
monkeypatch.setattr(headless, "finalize_task_artifacts", finalize)
cancel_intents.request_cancel(s.drive, s.task["id"], reason="owner", allow_settled_target=True)
assert s.q.cancel_task_custody(s.task["id"]) == s.q.CANCEL_ALREADY_SETTLED
assert finalized == [s.task["id"]] and not s.child.exists()
assert finalized == [s.task["id"]] and settled_off_loop(s.drive, s.task["id"], s.child)
for event in s.frames:
dispatch_event(event, s.ctx)
assert [e["status"] for e in s.pushed if e.get("type") == "task_done"] == ["completed"]
@ -83,7 +83,7 @@ def test_cancel_file_outage_retains_intent_and_retries_dead_worker(split, monkey
assert s.child.is_dir() and s.frames == [] and s.task["id"] in s.q.RUNNING
assert cancel_intents.cancel_pending(s.drive, s.task["id"])
assert s.q.cancel_task_custody(s.task["id"]) == s.q.CANCEL_ALREADY_SETTLED
assert s.state["kills"] == 1 and not s.child.exists()
assert s.state["kills"] == 1 and settled_off_loop(s.drive, s.task["id"], s.child)
assert load_task_result(s.drive, s.task["id"])["result"] == "Retained child answer"
@ -98,7 +98,7 @@ def test_genuinely_ready_current_survives_cancel_byte_identical(split, monkeypat
cancel_intents.request_cancel(s.drive, s.task["id"], reason="owner", allow_settled_target=True)
assert s.q.cancel_task_custody(s.task["id"]) == s.q.CANCEL_ALREADY_SETTLED
assert (s.drive / "task_results" / f'{s.task["id"]}.json').read_bytes() == before
assert json.loads(before)["result"] == "Accepted answer" and not s.child.exists()
assert json.loads(before)["result"] == "Accepted answer" and settled_off_loop(s.drive, s.task["id"], s.child)
def test_reconciled_terminal_without_checkpoint_still_needs_real_file_adoption(split):

File diff suppressed because it is too large Load diff

View file

@ -156,7 +156,10 @@ def test_changed_child_result_reopens_and_old_hash_is_stale(tmp_path):
def test_artifact_change_reopens_exact_hash_disposition(tmp_path):
from ouroboros.artifacts import task_artifact_dir_path
"""Reads hash no files (TZ-1 A): the recorded artifact identity is what the
exact-hash disposition covers, so a published byte change reopens it while an
unrecorded file mutation leaves the pure read unchanged."""
from ouroboros.artifacts import artifact_record, task_artifact_dir_path
from ouroboros.task_results import write_task_result
from ouroboros.task_status import load_effective_task_result
from ouroboros.tools.join_ledger import (
@ -166,6 +169,10 @@ def test_artifact_change_reopens_exact_hash_disposition(tmp_path):
from ouroboros.tools.task_tree import _tree_note
child_id = "artifact-child"
artifact_dir = task_artifact_dir_path(tmp_path, child_id)
artifact_dir.mkdir(parents=True, exist_ok=True)
artifact_path = artifact_dir / "report.md"
artifact_path.write_text("version one\n", encoding="utf-8")
write_task_result(
tmp_path,
child_id,
@ -174,11 +181,8 @@ def test_artifact_change_reopens_exact_hash_disposition(tmp_path):
root_task_id="parent1",
delegation_role="subagent",
result="artifact-backed result",
artifacts=[artifact_record(artifact_path)],
)
artifact_dir = task_artifact_dir_path(tmp_path, child_id)
artifact_dir.mkdir(parents=True, exist_ok=True)
artifact_path = artifact_dir / "report.md"
artifact_path.write_text("version one\n", encoding="utf-8")
shown_hash = _child_result_sha256(load_effective_task_result(tmp_path, child_id))
assert _tree_note(
_parent_ctx(tmp_path),
@ -191,6 +195,8 @@ def test_artifact_change_reopens_exact_hash_disposition(tmp_path):
) == "integrated"
artifact_path.write_text("version two\n", encoding="utf-8")
assert _child_result_sha256(load_effective_task_result(tmp_path, child_id)) == shown_hash
write_task_result(tmp_path, child_id, "completed", artifacts=[artifact_record(artifact_path)])
changed = load_effective_task_result(tmp_path, child_id)
assert _child_result_sha256(changed) != shown_hash
assert _current_child_result_disposition(changed) == ""
@ -651,7 +657,7 @@ def test_cancellation_wins_and_late_scratch_result_is_deleted(tmp_path):
trace_summary="late trace",
)
assert remove_subagent_task_drive(tmp_path, child_id) is True
assert remove_subagent_task_drive(tmp_path, child_id, live=lambda _task: False) is True
assert not scratch.parent.exists()
raw = load_task_result(tmp_path, child_id) or {}
assert raw["status"] == STATUS_CANCELLED

View file

@ -421,6 +421,15 @@ def test_route_owner_message_stamps_channel_for_non_web_ingress(monkeypatch):
"machine-to-machine traffic must never wear an owner_client fact"
)
# The common ingress receipt stamp rides the channel fact of a surface-less transport.
server._route_owner_message(SimpleNamespace(), make_ctx(), {
"chat_id": 1, "text": "со штампом", "client_message_id": "m5",
"task_metadata": None, "log_text": "со штампом",
"origin_message_ref": {"chat_id": 1, "client_message_id": "m5"},
"source": "skill:telegram", "received_at": "2026-09-26T00:00:00+00:00",
})
assert captured[-1]["client_surface"] == {"channel": "skill:telegram", "received_at": "2026-09-26T00:00:00+00:00"}
def test_steering_and_project_mailbox_writers_pass_client_surface():
# Tripwire complement to the behavioral tests above (the mailbox writers are
@ -527,3 +536,63 @@ def test_frontend_sends_raw_observables_without_device_taxonomy():
f"device taxonomy label {label!r} crept into {src_name} — raw "
"observables only, the model classifies"
)
def test_the_common_enqueue_stamps_one_host_receipt_for_every_transport(tmp_path, monkeypatch):
"""TZ-1 F: ``enqueue_local_message`` is the one ingress of every transport, so it stamps the
host receipt time once: an earlier host stamp (the WS acceptance's, an accepted row's time) is
kept, a host channel fact without one gets it (so a row written at dequeue measures intake lag
as ``ts - client_surface.received_at``), and a surface-less message gets no phantom fact."""
from datetime import datetime
from supervisor import message_bus
from supervisor.message_bus import LocalChatBridge
bridge = LocalChatBridge()
socket_fact = {"pywebview": True, "received_at": "2026-09-26T00:00:01+00:00"}
bridge.enqueue_local_message("from the socket", source="web", task_metadata={"client_surface": socket_fact})
bridge.enqueue_local_message("via /api/command", task_metadata={"client_surface": {"channel": "api_command"}})
bridge.enqueue_local_message("unnamed skill", chat_id=5, user_id=5, source="skill:bridge")
bridge.enqueue_local_message("accepted", chat_id=5, user_id=5, source="skill:x", received_at="2026-09-26T00:00:02+00:00")
socket, command, skill, accepted = [update["message"] for update in bridge.get_updates(offset=0, timeout=1)]
assert socket["received_at"] == socket_fact["received_at"] and socket["task_metadata"]["client_surface"] == socket_fact
assert command["task_metadata"]["client_surface"] == {"channel": "api_command", "received_at": command["received_at"]}
assert datetime.fromisoformat(command["received_at"]) and datetime.fromisoformat(skill["received_at"])
assert "client_surface" not in (skill.get("task_metadata") or {}), "no phantom surface fact"
assert accepted["received_at"] == "2026-09-26T00:00:02+00:00"
monkeypatch.setattr(message_bus, "DATA_DIR", tmp_path)
monkeypatch.setattr(message_bus, "load_state", lambda: {})
message_bus.record_inbound_message(bridge, command, chat_id=1, user_id=1, client_message_id="c1",
text="via /api/command", ts="2026-09-26T09:00:00+00:00")
row = json.loads((tmp_path / "logs" / "chat.jsonl").read_text(encoding="utf-8"))
assert row["client_surface"]["received_at"] == command["received_at"] and "logged_at" not in row
def test_a_named_acceptance_queues_its_row_time_as_the_receipt(tmp_path, monkeypatch):
from supervisor import message_bus
from supervisor.message_bus import LocalChatBridge
bridge = LocalChatBridge()
monkeypatch.setattr(message_bus, "DATA_DIR", tmp_path)
monkeypatch.setattr(message_bus, "load_state", lambda: {})
row, _rejoined = message_bus.accept_local_message(
bridge, tmp_path, "named delivery", chat_id=7, user_id=7, source="skill:telegram", client_message_id="n1",
)
[update] = bridge.get_updates(offset=0, timeout=1)
assert update["message"]["received_at"] == row["ts"]
def test_supervisor_intake_hands_the_receipt_stamp_to_routing(monkeypatch):
import server
import supervisor.message_bus as message_bus
from tests.test_transport_commands import Ctx
routed = []
monkeypatch.setattr(message_bus, "log_chat", lambda *args, **kwargs: None)
monkeypatch.setattr(server, "_route_owner_message", lambda bridge, ctx, incoming: routed.append(incoming))
bridge = message_bus.LocalChatBridge({})
bridge.enqueue_local_message("hello", chat_id=42, user_id=7, source="skill:telegram", received_at="2026-09-26T01:00:00+00:00")
server._process_bridge_updates(bridge, 0, Ctx({"owner_id": 7, "owner_chat_id": 42}))
assert [incoming["received_at"] for incoming in routed] == ["2026-09-26T01:00:00+00:00"]

View file

@ -131,10 +131,12 @@ def test_core_catalog_schema_bytes_and_handler_owners_are_stable():
# escalate description offers 0-6 alternatives (none for an open question answered in the
# human's own words), states that a shared wait ends on any incoming message and that a
# plain-text clarification ends the turn while a waited question keeps it alive; the
# `options` description says optional 0-6 and `options` leaves the required keys.
# `options` description says optional 0-6 and `options` leaves the required keys. Rolled
# again for TZ-1 V10: forward_to_worker also writes into a queued task's mailbox, so its
# description and `task_id` description say "running or queued" and when each reads it.
# Diffing the whole catalog base to head shows exactly those edits and nothing else.
assert hashlib.sha256(schema_bytes).hexdigest() == (
"8dbf49802f42ef87f0279c102103d151774db22fdfaf32668907d971599cd7cd"
"968eecad1c04b7f8a265d17bdb5dc42a8a5239373a0a5724ad8c8d5489d06ef3"
)
assert {
entry.name: (entry.handler.__module__, entry.handler.__name__)

View file

@ -419,11 +419,12 @@ def test_task_artifact_endpoint_serves_manifest_artifact_after_status_repair(tmp
assert response.text == "<h1>ok</h1>"
def test_task_artifact_endpoint_rebases_child_drive_artifact_after_status_repair(tmp_path):
def test_task_artifact_endpoint_serves_child_drive_artifact_read_only_after_status_repair(tmp_path):
from ouroboros.artifacts import collect_task_artifact_records, copy_file_to_task_artifacts
data = tmp_path / "data"
child = tmp_path / "child"
# The task's OWN headless drive (host layout); a drive merely named by the row is no authority.
child = data / "state" / "headless_tasks" / "childart" / "data"
source_dir = tmp_path / "Desktop"
source_dir.mkdir()
source = source_dir / "report.html"
@ -456,10 +457,10 @@ def test_task_artifact_endpoint_rebases_child_drive_artifact_after_status_repair
response = TestClient(app).get("/api/tasks/childart/artifacts/report.html")
parent_artifact = task_artifacts_dir(data, "childart", create=False) / "report.html"
# Two-root read: the task's own child store serves it; nothing is copied or created.
assert response.status_code == 200
assert response.text == "<h1>child</h1>"
assert parent_artifact.read_text(encoding="utf-8") == "<h1>child</h1>"
assert not task_artifacts_dir(data, "childart", create=False).exists()
def test_task_artifact_endpoint_rejects_metadata_name_path_mismatch(tmp_path):
@ -517,7 +518,7 @@ def test_startup_prune_removes_only_old_terminal_child_drives(tmp_path):
os.utime(pending_dir, (old, old))
os.utime(fresh_timestamp_dir, (old, old))
report = prune_headless_task_drives(data, retention_days=7, now=now)
report = prune_headless_task_drives(data, retention_days=7, now=now, live=lambda _task: False)
assert [item["task_id"] for item in report["pruned"]] == ["oldterminal"]
assert not terminal_dir.exists()
@ -527,7 +528,7 @@ def test_startup_prune_removes_only_old_terminal_child_drives(tmp_path):
assert any(item["task_id"] == "freshresult" and item["reason"] == "younger_than_retention" for item in report["skipped"])
def test_startup_prune_uses_effective_terminal_status(tmp_path):
def test_prune_settles_only_durably_terminal_rows_never_a_projection(tmp_path):
data = tmp_path / "data"
task_drive = data / "task_drives" / "stalerun"
child_dir = data / "state" / "headless_tasks" / "stalechild"
@ -559,8 +560,18 @@ def test_startup_prune_uses_effective_terminal_status(tmp_path):
os.utime(task_drive, (old, old))
os.utime(child_dir, (old, old))
direct_report = prune_task_drives(data, retention_days=7, now=now)
child_report = prune_headless_task_drives(data, retention_days=7, now=now)
# A projection is not custody: only the DURABLE row settles a drive (TZ-1 A).
direct_report = prune_task_drives(data, retention_days=7, now=now, live=lambda _task: False)
child_report = prune_headless_task_drives(data, retention_days=7, now=now, live=lambda _task: False)
assert [item["reason"] for item in direct_report["skipped"]] == ["task_not_terminal"]
assert [item["reason"] for item in child_report["skipped"]] == ["parent_not_terminal"]
assert task_drive.exists() and child_dir.exists()
from ouroboros.task_status import reconcile_orphaned_running_tasks
assert reconcile_orphaned_running_tasks(data) == 2 # the reconciler persists what the projection says
direct_report = prune_task_drives(data, retention_days=7, now=now, live=lambda _task: False)
child_report = prune_headless_task_drives(data, retention_days=7, now=now, live=lambda _task: False)
assert [item["task_id"] for item in direct_report["pruned"]] == ["stalerun"]
assert [item["task_id"] for item in child_report["pruned"]] == ["stalechild"]
@ -588,7 +599,7 @@ def test_startup_prune_removes_only_old_terminal_task_scratch(tmp_path):
os.utime(old_pending, (old, old))
os.utime(fresh_terminal, (old, old))
report = prune_task_drives(data, retention_days=7, now=now)
report = prune_task_drives(data, retention_days=7, now=now, live=lambda _task: False)
assert [item["task_id"] for item in report["pruned"]] == ["oldterminal"]
assert not old_terminal.exists()

View file

@ -237,13 +237,13 @@ def test_copyback_failure_retains_child_until_existing_retry_finishes(tmp_path,
copied = copy_child_task_result(parent, {"id": "custody", "drive_root": str(child)})
assert copied["artifact_bundle"]["status"] == "missing"
assert copied["child_ref_promotion"]["status"] == "incomplete"
assert not remove_subagent_task_drive(parent, "custody")
assert not remove_subagent_task_drive(parent, "custody", live=lambda _task: False)
assert Path(record["path"]).is_file()
assert retry_pending_child_ref_promotions(parent)["completed"] == ["custody"]
result = load_task_result(parent, "custody")
assert result["child_ref_promotion"]["status"] == "complete"
assert result["artifact_bundle"]["status"] == "ready"
assert remove_subagent_task_drive(parent, "custody")
assert remove_subagent_task_drive(parent, "custody", live=lambda _task: False)
assert Path(result["artifacts"][0]["path"]).read_bytes() == b"complete generated artifact"
@ -319,7 +319,7 @@ def test_full_inputs_survive_copyback_and_child_gc(tmp_path):
write_task_result(child, "inputs", "completed", result="done", task_contract=authority)
result = copy_child_task_result(parent, {"id": "inputs", "drive_root": str(child)})
assert result["child_ref_promotion"]["status"] == "complete"
assert remove_subagent_task_drive(parent, "inputs")
assert remove_subagent_task_drive(parent, "inputs", live=lambda _task: False)
rows = artifacts.resolve_attachment_manifest(parent, "inputs", result["task_contract"])
assert len(rows) == 28
for row in rows:
@ -545,17 +545,19 @@ def test_input_copy_failure_protects_each_existing_gc_root(tmp_path, monkeypatch
assert result["child_ref_promotion"]["status"] == "incomplete"
prune = headless.prune_headless_task_drives if drive_kind == "headless" else headless.prune_task_drives
assert not prune(parent, retention_days=1, now=4_000_000_000)["pruned"]
assert not headless.remove_subagent_task_drive(parent, "inputs")
assert not headless.remove_subagent_task_drive(parent, "inputs", live=lambda _task: False)
assert child.is_dir()
assert retry_pending_child_ref_promotions(parent)["completed"] == ["inputs"]
result = load_task_result(parent, "inputs")
assert result["child_ref_promotion"]["status"] == "complete"
assert headless.remove_subagent_task_drive(parent, "inputs")
assert headless.remove_subagent_task_drive(parent, "inputs", live=lambda _task: False)
assert len(artifacts.resolve_attachment_manifest(parent, "inputs", result["task_contract"])) == 28
@pytest.mark.parametrize("immutable", [False, True])
def test_materialization_preserves_capture_identity_without_freezing_mutable_outputs(tmp_path, immutable):
def test_collection_preserves_capture_identity_without_freezing_mutable_outputs(tmp_path, immutable):
"""Writer-side collection keeps an immutable capture and discloses changed bytes,
a mutable output takes its new identity; the effective read re-measures nothing."""
from ouroboros.task_results import write_task_result
from ouroboros.task_status import load_effective_task_result
@ -565,7 +567,9 @@ def test_materialization_preserves_capture_identity_without_freezing_mutable_out
captured = artifacts.copy_file_to_task_artifacts(ctx, source, immutable=immutable)
write_task_result(ctx.drive_root, ctx.task_id, "completed", artifacts=[captured])
Path(captured["path"]).write_text("later changed report")
projected = load_effective_task_result(ctx.drive_root, ctx.task_id)["artifacts"][0]
assert load_effective_task_result(ctx.drive_root, ctx.task_id)["artifacts"] == [captured]
projected = artifacts.merge_artifact_records(
[captured], artifacts.collect_task_artifact_records(ctx.drive_root, ctx.task_id))[0]
if immutable:
assert projected["immutable"] is True
assert projected["sha256"] == captured["sha256"]
@ -580,7 +584,7 @@ def test_materialization_preserves_capture_identity_without_freezing_mutable_out
assert projected["status"] == "ready"
def test_immutable_child_materialization_retains_name_bytes_and_original_identity(tmp_path):
def test_immutable_child_copy_back_retains_name_bytes_and_original_identity(tmp_path):
from ouroboros.headless import prepare_task_drive, copy_child_task_result
from ouroboros.task_results import write_task_result
from ouroboros.task_status import load_effective_task_result
@ -593,8 +597,10 @@ def test_immutable_child_materialization_retains_name_bytes_and_original_identit
SimpleNamespace(drive_root=child, task_id="capture"), source, immutable=True)
write_task_result(child, "capture", "completed", artifacts=[record], artifact_status="ready")
write_task_result(parent, "capture", "running", headless_child_drive_root=str(child))
projected = load_effective_task_result(parent, "capture")
rebased = projected["artifacts"][0]
# The read lists the child's capture where it lies; copy-back alone publishes it.
assert load_effective_task_result(parent, "capture")["artifacts"] == [record]
assert not artifacts.task_artifact_dir_path(parent, "capture").exists()
rebased = copy_child_task_result(parent, {"id": "capture", "drive_root": str(child)})["artifacts"][0]
assert rebased["name"] == record["name"] and rebased["sha256"] == record["sha256"]
assert rebased["immutable"] is True
assert Path(rebased["path"]).parent == artifacts.task_artifact_dir_path(parent, "capture")
@ -677,7 +683,7 @@ def test_file_verification_does_not_run_on_the_asgi_loop(tmp_path, monkeypatch,
@pytest.mark.parametrize("canonical_exists", [False, True])
def test_failed_child_capture_is_explicit_and_other_files_still_materialize(tmp_path, canonical_exists):
def test_failed_child_capture_is_explicit_and_other_files_still_publish(tmp_path, canonical_exists):
from ouroboros.headless import prepare_task_drive, copy_child_task_result, remove_subagent_task_drive
from ouroboros.observability import retry_pending_child_ref_promotions
from ouroboros.task_results import write_task_result
@ -695,10 +701,12 @@ def test_failed_child_capture_is_explicit_and_other_files_still_materialize(tmp_
write_task_result(child, "capture", "completed", artifacts=records, artifact_status="ready",
outcome_axes=axes, accounted_upper_bound_usd=3.5, cost_final=True)
write_task_result(parent, "capture", "running", headless_child_drive_root=str(child))
task = {"id": "capture", "drive_root": str(child)}
if canonical_exists:
assert load_effective_task_result(parent, "capture")["artifact_bundle"]["status"] == "ready"
assert copy_child_task_result(parent, task)["artifact_bundle"]["status"] == "ready"
bad = records[0]
Path(bad["path"]).write_bytes(b"changed child bytes before copy")
copied = copy_child_task_result(parent, task)
result = load_effective_task_result(parent, "capture")
row = next(item for item in result["artifacts"] if item["name"] == bad["name"])
assert result["status"] == "completed" and result["outcome_axes"] == axes
@ -715,11 +723,10 @@ def test_failed_child_capture_is_explicit_and_other_files_still_materialize(tmp_
SimpleNamespace(drive_root=parent, task_id="capture"), bad["path"], immutable=True, expected=bad)
assert reused["path"] == row["path"] and reused["sha256"] == bad["sha256"]
else:
assert row["status"] == "failed" and row["copy_status"] == "failed" and row["copy_error"]
assert row["copy_status"] == "failed" and row["copy_error"]
assert result["artifact_status"] == result["artifact_bundle"]["status"] == "missing"
copied = copy_child_task_result(parent, {"id": "capture", "drive_root": str(child)})
assert copied["child_ref_promotion"]["status"] == "incomplete"
assert not remove_subagent_task_drive(parent, "capture")
assert not remove_subagent_task_drive(parent, "capture", live=lambda _task: False)
Path(bad["path"]).write_bytes(b"a-report.txt")
assert retry_pending_child_ref_promotions(parent)["completed"] == ["capture"]
recovered = load_effective_task_result(parent, "capture")
@ -752,7 +759,7 @@ def test_failed_artifact_bundle_drives_public_and_routing_status_without_mutatin
def test_first_materialization_copy_failure_keeps_each_gc_root(tmp_path, monkeypatch, drive_kind):
import time
from ouroboros import headless
from ouroboros.task_results import write_task_result
from ouroboros.task_results import load_task_result, write_task_result
parent = tmp_path / "canonical"
child = (headless.prepare_task_drive(parent, "capture", "empty") if drive_kind == "headless"
@ -763,25 +770,33 @@ def test_first_materialization_copy_failure_keeps_each_gc_root(tmp_path, monkeyp
record = artifacts.copy_file_to_task_artifacts(
SimpleNamespace(drive_root=child, task_id="capture"), source, immutable=True)
write_task_result(child, "capture", "completed", artifacts=[record], artifact_status="ready")
# An adopted row (its promotion mark set) that still names the capture at its CHILD path:
# the settlement itself owes the canonical copy, so its copy failure is the probe.
write_task_result(parent, "capture", "completed", artifacts=[record], artifact_status="ready",
headless_child_drive_root=str(child))
headless_child_drive_root=str(child),
child_ref_promotion={"schema_version": 1, "status": "complete", "pending_refs": []})
original = artifacts.copy_artifact_file
canonical_files = artifacts.task_artifact_dir_path(parent, "capture")
def fail_copy(src, dst, **kwargs):
if Path(dst).is_relative_to(canonical_files):
# Settlement prepares every canonical copy in private staging under the canonical
# root: that copy failing is the canonical copy failing.
if Path(dst).is_relative_to(parent) and not Path(dst).is_relative_to(child):
raise OSError("controlled canonical copy failure")
return original(src, dst, **kwargs)
prune = headless.prune_headless_task_drives if drive_kind == "headless" else headless.prune_task_drives
later = time.time() + 14 * 86400
with monkeypatch.context() as patch:
patch.setattr(artifacts, "copy_artifact_file", fail_copy)
refused = prune(parent, retention_days=7, now=later)
refused = prune(parent, retention_days=7, now=later, live=lambda _task: False)
assert not refused["pruned"] and child.is_dir(), refused
assert refused["custody_pending"] == [{"task_id": "capture", "reason": "artifact_source_mismatch"}]
assert Path(record["path"]).read_bytes() == b"complete report"
copied = headless.copy_child_task_result(parent, {"id": "capture", "drive_root": str(child)})
assert copied["child_ref_promotion"]["status"] == "complete"
assert prune(parent, retention_days=7, now=later)["pruned"]
assert Path(copied["artifacts"][0]["path"]).read_bytes() == b"complete report"
assert load_task_result(parent, "capture")["artifacts"] == [record] # the row still names its source
assert not artifacts.task_artifact_dir_path(parent, "capture").exists()
settled = prune(parent, retention_days=7, now=later, live=lambda _task: False)
assert settled["pruned"] and not child.exists(), settled
published = load_task_result(parent, "capture")["artifacts"][0]
assert Path(published["path"]).read_bytes() == b"complete report" and published["sha256"] == record["sha256"]
assert Path(published["path"]).is_relative_to(artifacts.task_artifact_dir_path(parent, "capture").resolve())
@pytest.mark.parametrize("context", ["healthy", "missing", "none", "storage_failure"])
@ -910,7 +925,7 @@ def test_acknowledged_owner_inputs_survive_copyback_retry_mailbox_cleanup_and_gc
assert any(row["path"] == str(mailbox) for row in result["child_ref_promotion"]["pending_refs"])
cleanup_settled_owner_mailbox(parent, task_id, task)
assert mailbox.is_file()
assert not headless.remove_subagent_task_drive(parent, task_id)
assert not headless.remove_subagent_task_drive(parent, task_id, live=lambda _task: False)
assert observability.retry_pending_child_ref_promotions(parent)["completed"] == [task_id]
result = load_task_result(parent, task_id)
assert not mailbox.exists(), "successful retry releases retained mail through its existing owner"
@ -920,7 +935,7 @@ def test_acknowledged_owner_inputs_survive_copyback_retry_mailbox_cleanup_and_gc
assert result["child_ref_promotion"]["status"] == "complete"
assert result["child_ref_promotion"]["pending_refs"] == []
assert len(artifacts.resolve_attachment_manifest(parent, task_id, result["task_contract"])) == len(initial)
gc = headless.prune_headless_task_drives(parent, retention_days=1, now=time.time() + 90 * 86400)
gc = headless.prune_headless_task_drives(parent, retention_days=1, now=time.time() + 90 * 86400, live=lambda _task: False)
assert [row["task_id"] for row in gc["pruned"]] == [task_id]
assert not child.exists()
if ref:

View file

@ -17,6 +17,7 @@ from __future__ import annotations
import asyncio
import json
import pathlib
from types import SimpleNamespace
from ouroboros.task_results import write_task_result
@ -28,7 +29,7 @@ def _seed_child_drive_scenario(tmp_path):
from ouroboros.artifacts import collect_task_artifact_records, copy_file_to_task_artifacts
data = tmp_path / "data"
child = tmp_path / "child"
child = data / "state" / "headless_tasks" / "childart" / "data" # the task's OWN drive
source_dir = tmp_path / "Desktop"
source_dir.mkdir()
source = source_dir / "report.html"
@ -114,19 +115,24 @@ def test_child_result_sha_stable_across_false_path_reads(tmp_path):
assert sha_before == sha_after
def test_true_default_still_materializes_child_artifacts(tmp_path):
"""The default path keeps the read-repair durability: child artifacts are
rebased onto the parent drive (api_task_artifact depends on it)."""
def test_true_default_lists_child_artifacts_without_copying_them(tmp_path):
"""The default read shows the child's recorded rows at their child-drive
paths and copies nothing; copy-back alone rebases them onto the parent."""
from ouroboros.headless import copy_child_task_result
data, child = _seed_child_drive_scenario(tmp_path)
rebased = data / "task_results" / "artifacts" / "childart" / "report.html"
row = load_effective_task_result(data, "childart")
rebased = data / "task_results" / "artifacts" / "childart" / "report.html"
assert rebased.exists()
listed = next(item for item in row["artifacts"] if item.get("name") == "report.html")
assert listed["path"] == str(child / "task_results" / "artifacts" / "childart" / "report.html")
assert not rebased.exists() and not rebased.parent.exists()
copied = copy_child_task_result(data, {"id": "childart", "drive_root": str(child)})
assert rebased.read_text(encoding="utf-8") == "<h1>child</h1>"
assert any(
(item.get("name") or "") == "report.html" for item in row.get("artifacts") or []
)
assert [item["path"] for item in copied["artifacts"]] == [str(rebased)]
assert [item["path"] for item in load_effective_task_result(data, "childart")["artifacts"]] == [str(rebased)]
def test_generic_receipt_named_output_never_owns_receipt_authority(tmp_path):
@ -307,20 +313,25 @@ def test_reconcile_skips_a_live_running_child_without_materializing(tmp_path, mo
assert json.loads((data / "task_results" / f"{tid}.json").read_text(encoding="utf-8"))["status"] == "running"
def test_reconcile_heals_a_genuine_orphan_with_full_artifact_custody(tmp_path, monkeypatch):
"""The positive path: a parent row stuck at ``running`` over a finished child
drive is settled from the materializing read, so the persisted row carries
the promoted artifact, its bundle and the terminal quiz settlement."""
def test_reconcile_heals_a_genuine_orphan_and_settlement_publishes_its_files(tmp_path, monkeypatch):
"""The positive path: a parent row stuck at ``running`` over a finished child drive
is settled from the status-only read — the sweep lists, copies and hashes no file —
with its terminal quiz settlement. The child's file stays served from its OWN drive
until copy-back adopts the child and ``settle_child_drive`` publishes it before the
drive goes (TZ-1 A: publication is explicit, never a read side effect)."""
import time
from ouroboros import owner_quiz
from ouroboros.artifacts import collect_task_artifact_records, copy_file_to_task_artifacts
from ouroboros.headless import prepare_terminal_task_files
from ouroboros.task_custody import settle_child_drive
from ouroboros.task_status import SETTLED_STATUSES, reconcile_orphaned_running_tasks
from ouroboros.utils import append_jsonl
now = 1_800_000_000.0
monkeypatch.setattr(time, "time", lambda: now)
data, child, tid = tmp_path / "data", tmp_path / "child", "orphanchild"
data, tid = tmp_path / "data", "orphanchild"
child = data / "state" / "headless_tasks" / tid / "data" # the task's OWN drive (host layout)
source = tmp_path / "report.html"
source.write_text("<h1>child</h1>", encoding="utf-8")
copy_file_to_task_artifacts(SimpleNamespace(drive_root=child, task_id=tid), source, kind="user_file")
@ -341,18 +352,19 @@ def test_reconcile_heals_a_genuine_orphan_with_full_artifact_custody(tmp_path, m
assert reconcile_orphaned_running_tasks(data, expired_quizzes=expired) == 1
assert flags == [False, True]
assert copies["copy"] >= 1
# The persisted bytes and the promoted file are checked BEFORE any materializing
# loader runs again, so the oracle cannot repair what the sweep left undone.
assert flags == [False]
assert copies == {"copy": 0}
on_disk = json.loads((data / "task_results" / f"{tid}.json").read_text(encoding="utf-8"))
assert on_disk["status"] in SETTLED_STATUSES
assert on_disk["artifact_status"] and isinstance(on_disk.get("artifact_bundle"), dict)
promoted = data / "task_results" / "artifacts" / tid / "report.html"
assert promoted.read_text(encoding="utf-8") == "<h1>child</h1>"
assert not promoted.exists()
assert expired == [(tid, "q1")]
assert owner_quiz.quiz_states(data, tid)["q1"]["state"] == "expired_terminal"
direct = load_effective_task_result(data, tid)
for key in ("status", "artifact_status", "artifact_bundle"):
assert on_disk[key] == direct[key], key
assert on_disk.get("status_reconciled_from") == direct.get("status_reconciled_from")
listed = next(row for row in load_effective_task_result(data, tid)["artifacts"] if row["name"] == "report.html")
assert pathlib.Path(listed["path"]).is_relative_to(child.resolve()) # served from where it lies
assert not prepare_terminal_task_files(data, {"id": tid, "drive_root": str(child)})["error"]
assert settle_child_drive(data, tid, child, live=lambda _task: False)["status"] == "removed"
assert promoted.read_text(encoding="utf-8") == "<h1>child</h1>"
assert not child.exists()
published = next(row for row in load_effective_task_result(data, tid)["artifacts"] if row["name"] == "report.html")
assert published["path"] == str(promoted.resolve()) and published["sha256"]

View file

@ -120,11 +120,17 @@ class TestOwnerInjectPerTask(unittest.TestCase):
def test_cleanup_removes_file(self):
from ouroboros.owner_mailbox import write_owner_message, cleanup_task_mailbox, _mailbox_path
from ouroboros.task_results import load_task_result, write_task_result
write_owner_message(self.drive_root, "hello", task_id="t1", msg_id="m1")
path = _mailbox_path(self.drive_root, "t1")
self.assertTrue(path.exists())
cleanup_task_mailbox(self.drive_root, "t1")
# TZ-1 V10: an unread row leaves only into a canonical row that holds it.
self.assertFalse(cleanup_task_mailbox(self.drive_root, "t1"))
self.assertTrue(path.exists())
write_task_result(self.drive_root, "t1", "cancelled", result="Cancelled before start.")
self.assertIn('"msg_id": "m1"', load_task_result(self.drive_root, "t1")["unread_mailbox"]["rows"][0])
self.assertTrue(cleanup_task_mailbox(self.drive_root, "t1"))
self.assertFalse(path.exists())
def test_drain_nonexistent_task_returns_empty(self):
@ -226,17 +232,25 @@ class TestForwardToWorkerTool(unittest.TestCase):
with tempfile.TemporaryDirectory() as tmp:
parent_drive = pathlib.Path(tmp) / "parent"
child_drive = pathlib.Path(tmp) / "child"
queued_drive = pathlib.Path(tmp) / "queued-child"
child_drive.mkdir(parents=True)
write_task_result(parent_drive, "child1", STATUS_RUNNING, child_drive_root=str(child_drive), parent_task_id="parent1", root_task_id="parent1", result="running")
write_task_result(parent_drive, "queued1", STATUS_SCHEDULED, result="queued")
write_task_result(parent_drive, "queued1", STATUS_SCHEDULED, child_drive_root=str(queued_drive), parent_task_id="parent1", root_task_id="parent1", result="queued")
write_task_result(parent_drive, "asked1", "requested", parent_task_id="parent1", root_task_id="parent1", result="requested")
write_task_result(parent_drive, "otherchild", STATUS_RUNNING, parent_task_id="otherparent", root_task_id="otherroot", result="running")
ctx = SimpleNamespace(drive_root=parent_drive, task_id="parent1")
output = _forward_to_worker(ctx, "child1", "continue")
blocked = _forward_to_worker(ctx, "queued1", "too soon")
queued = _forward_to_worker(ctx, "queued1", "read this when you start")
blocked = _forward_to_worker(ctx, "asked1", "not admitted yet")
forbidden = _forward_to_worker(ctx, "otherchild", "wrong root")
self.assertIn("Message forwarded", output)
# TZ-1 V10: a queued task's mailbox takes the message; the receipt says nothing read it.
self.assertIn("(queued)", queued)
self.assertIn("has not started, so nothing has read it", queued)
queued_mailbox = queued_drive / "memory" / "owner_mailbox" / "queued1.jsonl"
self.assertIn("read this when you start", queued_mailbox.read_text(encoding="utf-8"))
self.assertIn("TASK_NOT_ACTIVE", blocked)
self.assertIn("TASK_FORBIDDEN", forbidden)
self.assertFalse((parent_drive / "memory" / "owner_mailbox" / "child1.jsonl").exists())

View file

@ -1,10 +1,9 @@
"""Observability census pins (CPL4-C22, owner 7A: the retention knob is GONE).
"""Observability blobs are never deleted and never counted (CPL4-C22, owner 7A; TZ-1 A).
``prune_observability_blobs`` counts manifests and blobs for startup
telemetry and deletes NOTHING — the preserve-indefinitely contract. The
former ``OUROBOROS_OBSERVABILITY_RETENTION_DAYS`` knob (parsed, clamped,
reported, deleting nothing) is retired entirely: absent from the module and
listed in ``RETIRED_SETTING_KEYS`` so stored ghosts drop on settings load.
The retention knob that deleted nothing is retired (absent from the module, listed in
``RETIRED_SETTING_KEYS`` so stored ghosts drop on settings load), and the startup census
that walked every manifest and blob to count them is gone with it: startup neither lists
nor touches the store.
"""
from __future__ import annotations
@ -12,8 +11,6 @@ from __future__ import annotations
import os
import time
from ouroboros.observability import prune_observability_blobs
def _seed_store(tmp_path):
calls = tmp_path / "observability" / "calls" / "t1"
@ -21,31 +18,33 @@ def _seed_store(tmp_path):
calls.mkdir(parents=True)
blobs.mkdir(parents=True)
aged = time.time() - 4000 * 86400
manifests = [calls / "a.json", calls / "b.json"]
blob_files = [blobs / "x.gz", blobs / "y.gz", blobs / "z.gz"]
for path in (*manifests, *blob_files):
paths = [calls / "a.json", calls / "b.json", blobs / "x.gz", blobs / "y.gz", blobs / "z.gz"]
for path in paths:
path.write_bytes(b"data")
os.utime(path, (aged, aged))
return manifests, blob_files
return paths
def test_census_counts_and_preserves_everything(tmp_path, monkeypatch):
def test_startup_neither_counts_nor_deletes_observability_blobs(tmp_path, monkeypatch):
import ouroboros.observability as observability
from ouroboros import server_maintenance as sm
monkeypatch.setattr(sm, "DATA_DIR", tmp_path)
monkeypatch.setenv("OUROBOROS_OBSERVABILITY_RETENTION_DAYS", "1") # inert: retired
manifests, blob_files = _seed_store(tmp_path)
paths = _seed_store(tmp_path)
touched = []
real_stat = os.stat
report = prune_observability_blobs(tmp_path)
def spy(path, *args, **kwargs):
if "observability" in str(path):
touched.append(str(path))
return real_stat(path, *args, **kwargs)
assert report["preserved_indefinitely"] is True
assert report["manifest_count"] == 2 and report["blob_count"] == 3
assert not report["errors"]
assert all(path.exists() for path in (*manifests, *blob_files))
def test_absent_store_reports_empty_census(tmp_path):
report = prune_observability_blobs(tmp_path)
assert report == {
"preserved_indefinitely": True, "manifest_count": 0, "blob_count": 0, "errors": [],
}
monkeypatch.setattr(os, "stat", spy)
sm._startup_prune_sweeps()
assert all(path.exists() for path in paths)
assert touched == [], touched
assert not hasattr(observability, "prune_observability_blobs")
def test_retention_knob_is_retired_everywhere():

View file

@ -858,9 +858,10 @@ def test_salvage_without_a_durable_copy_keeps_everything_in_the_note(tmp_path):
def test_cancelling_a_subagent_preserves_the_full_output_on_the_canonical_drive(
monkeypatch, tmp_path,
):
"""End to end through the REAL cancel path: publication deletes the child
drive, so the full blob must already have a copy on the canonical drive and
the terminal result must point at it (XG-7B.1, BIBLE P1)."""
"""End to end through the REAL cancel path: the child drive goes later, through the
off-loop settlement (``task_custody.settle_child_drive``), so the full blob must already
have a copy on the canonical drive and the terminal result must point at it (XG-7B.1,
BIBLE P1)."""
from ouroboros import observability
from ouroboros.headless import HEADLESS_TASKS_DIR
from ouroboros.task_results import load_task_result
@ -905,7 +906,9 @@ def test_cancelling_a_subagent_preserves_the_full_output_on_the_canonical_drive(
assert q.cancel_task_custody(task_id) == q.CANCEL_CANCELLED
assert not child_drive.exists(), "publication no longer deletes the child drive?"
from tests._cancel_intents_shared import settled_off_loop
assert settled_off_loop(tmp_path, task_id, child_drive), "the settlement no longer removes the child drive?"
result = load_task_result(tmp_path, task_id)
assert result["status"] == "cancelled"
assert "full copy preserved at " in result["result"]

View file

@ -581,7 +581,10 @@ def scan_data_paths(root: pathlib.Path = REPO) -> frozenset[str]:
# 297 -> 296 (TZ-3 PR-1): the ``knowledge_journal.jsonl`` size-telemetry writer is
# removed (its only reader was this inventory); ``knowledge_history.jsonl`` keeps the
# complete captures, now host-stamped.
EXPECTED_SCAN_PATHS = 296
# 296 -> 300 (TZ-1 child-drive custody): ``task_results/<id>.custody.lock`` (the per-task
# custody lock) and the settlement's ``state/custody_staging`` / ``state/custody_trash``
# entries; one PERSISTENCE.md row covers all three.
EXPECTED_SCAN_PATHS = 300
# Scanned paths that must always be present — guards the scanner itself
# against a silent regression that would shrink coverage while keeping counts

View file

@ -104,7 +104,7 @@ def test_copyback_promotes_trace_manifest_and_blobs_before_headless_gc(tmp_path)
"result"
] == "exact tool result"
report = prune_headless_task_drives(parent, retention_days=7, now=_future_now())
report = prune_headless_task_drives(parent, retention_days=7, now=_future_now(), live=lambda _task: False)
assert report["pruned"][0]["task_id"] == task_id
assert not child.exists()
assert read_blob_ref(parent, promoted_manifest["full_payload_ref"])["prompt"] == "exact prompt"
@ -183,8 +183,12 @@ def test_pipeline_loop_outcome_trace_refs_are_rebased_and_readable_after_gc(tmp_
nested_tool = nested_refs["tool_call_refs"][0]["manifest_ref"]
assert pathlib.Path(nested_request["path"]).is_relative_to(parent / "observability")
assert pathlib.Path(nested_tool["path"]).is_relative_to(parent / "observability")
prune_headless_task_drives(parent, retention_days=0, now=_future_now())
assert not child.exists()
# A root whose post-task synthesis is still owed keeps its drive (and mailbox) until it settles.
report = prune_headless_task_drives(parent, retention_days=0, now=_future_now(), live=lambda _task: False)
assert report["custody_pending"] == [{"task_id": task_id, "reason": "post_work_open"}] and child.exists()
write_task_result(parent, task_id, "completed", root_phase_checkpoint={"post_task_synthesis": "completed"})
report = prune_headless_task_drives(parent, retention_days=0, now=_future_now(), live=lambda _task: False)
assert not child.exists(), report
assert read_blob_ref(parent, _manifest(nested_request)["full_payload_ref"])[
"messages"
][0]["content"] == "exact pipeline prompt"
@ -258,7 +262,7 @@ def test_real_truncated_tool_source_envelope_remains_actor_readable_after_gc(tmp
request_ref = copied["loop_outcome"]["trace_refs"]["llm_call_refs"][0][
"request_ref"
]
prune_headless_task_drives(parent, retention_days=0, now=_future_now())
prune_headless_task_drives(parent, retention_days=0, now=_future_now(), live=lambda _task: False)
payload = read_blob_ref(parent, _manifest(request_ref)["full_payload_ref"])
promoted_ref = _source_ref_from_visible_result(payload["messages"][0]["content"])
assert promoted_ref == produced_ref
@ -310,7 +314,7 @@ def test_task_source_read_contract_mismatch_is_typed_unavailable(tmp_path, misma
parent / "task_results" / "artifacts" / task_id / pathlib.Path(ref["path"])
).exists()
assert prune_headless_task_drives(
parent, retention_days=0, now=_future_now()
parent, retention_days=0, now=_future_now(), live=lambda _task: False
)["pruned"]
@ -355,7 +359,7 @@ def test_copyback_promotes_service_full_log_refs_in_durable_evidence_and_tool_pa
nested_ref = json.loads(tool_payload["result"])["full_log_ref"]
assert read_blob_ref(parent, nested_ref, expected_kind="txt") == "READY\nfull service log\n"
prune_headless_task_drives(parent, retention_days=0, now=_future_now())
prune_headless_task_drives(parent, retention_days=0, now=_future_now(), live=lambda _task: False)
assert not child.exists()
assert read_blob_ref(parent, evidence_ref, expected_kind="txt").endswith("service log\n")
assert read_blob_ref(parent, nested_ref, expected_kind="txt").startswith("READY")
@ -395,10 +399,10 @@ def test_interrupted_live_ref_promotion_blocks_gc_until_idempotent_retry(
assert copied is not None
assert copied["child_ref_promotion"]["status"] == "incomplete"
assert copied["child_ref_promotion"]["pending_refs"]
assert remove_subagent_task_drive(parent, task_id) is False
report = prune_headless_task_drives(parent, retention_days=0, now=_future_now())
assert remove_subagent_task_drive(parent, task_id, live=lambda _task: False) is False
report = prune_headless_task_drives(parent, retention_days=0, now=_future_now(), live=lambda _task: False)
assert report["pruned"] == []
assert report["skipped"][0]["reason"] == "child_refs_unpromoted"
assert report["skipped"][0]["reason"] == "child_refs_pending"
assert child.exists()
monkeypatch.setattr(observability, "promote_call_manifest_ref", real)
@ -406,7 +410,7 @@ def test_interrupted_live_ref_promotion_blocks_gc_until_idempotent_retry(
assert retried is not None
assert retried["child_ref_promotion"]["status"] == "complete"
assert retried["child_ref_promotion"]["pending_refs"] == []
assert prune_headless_task_drives(parent, retention_days=0, now=_future_now())["pruned"]
assert prune_headless_task_drives(parent, retention_days=0, now=_future_now(), live=lambda _task: False)["pruned"]
def test_digest_mismatch_becomes_typed_unavailable_and_does_not_pin_drive(tmp_path):
@ -433,7 +437,7 @@ def test_digest_mismatch_becomes_typed_unavailable_and_does_not_pin_drive(tmp_pa
assert "path" not in unavailable
assert copied["child_ref_promotion"]["unavailable_refs"]
assert copied["child_ref_promotion"]["pending_refs"] == []
assert prune_headless_task_drives(parent, retention_days=0, now=_future_now())["pruned"]
assert prune_headless_task_drives(parent, retention_days=0, now=_future_now(), live=lambda _task: False)["pruned"]
def test_concurrent_copyback_is_idempotent_and_copies_only_referenced_source_handle(
@ -630,7 +634,7 @@ def test_legacy_missing_child_ref_is_typed_gap_without_permanent_retention(tmp_p
assert gap["availability"] == "unavailable"
assert gap["reason"] == "source_missing"
assert copied["child_ref_promotion"]["status"] == "complete"
assert prune_headless_task_drives(parent, retention_days=0, now=_future_now())["pruned"]
assert prune_headless_task_drives(parent, retention_days=0, now=_future_now(), live=lambda _task: False)["pruned"]
def test_startup_sweep_retries_only_pending_refs_then_prunes_without_manual_copyback(
@ -709,8 +713,11 @@ def test_startup_sweep_retries_only_pending_refs_then_prunes_without_manual_copy
# The sweep reads its drive root from its owner module (v7 server split).
monkeypatch.setattr(server_maintenance, "DATA_DIR", parent)
monkeypatch.setenv("OUROBOROS_GC_RETENTION_DAYS", "1")
# This test process owns no supervisor maps: it states the absence the probe proves in production.
monkeypatch.setattr("supervisor.queue.task_settlement_liveness", lambda _task: False)
server_maintenance._startup_prune_sweeps()
# The retry and the settlement ride the off-loop drive-custody pass, never startup.
server_maintenance._run_drive_custody_pass()
settled = load_task_result(parent, task_id) or {}
assert settled["child_ref_promotion"]["status"] == "complete"
@ -770,11 +777,12 @@ def test_startup_prune_retries_missing_pending_source_into_typed_gap(
pathlib.Path(trace["manifest_ref"]["path"]).unlink()
monkeypatch.setattr(observability, "promote_call_manifest_ref", real)
# The ONE retry owner turns the missing source into a typed gap; settlement then proceeds.
assert observability.retry_pending_child_ref_promotions(parent)["completed"] == [task_id]
report = prune_headless_task_drives(
parent, retention_days=0, now=_future_now()
parent, retention_days=0, now=_future_now(), live=lambda _task: False
)
assert report["promotion_retry"]["completed"] == [task_id]
assert report["pruned"][0]["task_id"] == task_id
settled = load_task_result(parent, task_id) or {}
gap = settled["trace_refs"]["tool_call_refs"][0]["manifest_ref"]
@ -792,24 +800,40 @@ def test_periodic_maintenance_invokes_pending_ref_promotion_sweep(
import supervisor.terminal_delivery as terminal_delivery
import threading
calls: list[pathlib.Path] = []
finished = threading.Event()
calls: list[tuple[pathlib.Path, str]] = []
threads: list = []
# The cadence state and drive root live in the maintenance owner (v7 server split).
# The retry is history-sized, so it rides the 300 s reconcile block on its own
# daemon thread and latch — never the 20 s cancel sweep, which runs beside it here.
monkeypatch.setattr(server_maintenance, "DATA_DIR", tmp_path)
monkeypatch.setattr(server_maintenance.time, "time", lambda: 10_000.0)
monkeypatch.setattr(server_maintenance, "_LAST_CANCEL_INTENT_SWEEP", [0.0])
monkeypatch.setattr(server_maintenance, "_CANCEL_INTENT_SWEEP_LOCK", threading.Lock())
monkeypatch.setattr(server_maintenance, "_RECONCILE_SWEEP_LOCK", threading.Lock())
monkeypatch.setattr(server_maintenance, "_periodic_zombie_reconcile", lambda **kwargs: None)
monkeypatch.setattr(task_lifecycle, "sweep_cancel_intents", lambda: {})
monkeypatch.setattr(terminal_delivery, "replay_pending_deliveries", lambda _root: None)
monkeypatch.setattr(
observability,
"retry_pending_child_ref_promotions",
lambda root: (calls.append(pathlib.Path(root)), finished.set(), {})[-1],
lambda root, **_kw: calls.append((pathlib.Path(root), threading.current_thread().name)) or {},
raising=False,
)
server_maintenance._periodic_supervisor_maintenance([10_000.0], [10_000.0])
def tracked(**kwargs):
thread = threading.Thread(**kwargs)
threads.append(thread)
return thread
assert finished.wait(2)
monkeypatch.setattr(server_maintenance, "threading", SimpleNamespace(Thread=tracked))
server_maintenance._periodic_supervisor_maintenance([10_000.0], [0.0])
for thread in threads:
thread.join(5)
assert [thread.name for thread in threads] == ["terminal-maintenance", "reconcile-maintenance"]
assert calls == [(tmp_path, "reconcile-maintenance")]
assert server_maintenance._RECONCILE_SWEEP_LOCK.acquire(timeout=2)
server_maintenance._RECONCILE_SWEEP_LOCK.release()
assert server_maintenance._CANCEL_INTENT_SWEEP_LOCK.acquire(timeout=2)
server_maintenance._CANCEL_INTENT_SWEEP_LOCK.release()
assert calls == [tmp_path]

View file

@ -142,7 +142,7 @@ def test_dialogue_source_survives_real_child_promotion_and_cleanup(harness):
write_task_result(child, "source", "completed")
copied = copy_child_task_result(parent, {"id": "source", "drive_root": str(child)})
assert copied["child_ref_promotion"]["status"] == "complete"
assert remove_subagent_task_drive(parent, "source") is True
assert remove_subagent_task_drive(parent, "source", live=lambda _task: False) is True
assert not child.exists()
wave = load_plan_review_state(parent, "source")["waves"][-1]
assert read_actor_source_bytes(parent, "source", wave["dialogue_source_ref"]) == raw

View file

@ -27,20 +27,29 @@ def test_remove_subagent_task_drive(tmp_path):
remove_subagent_task_drive,
)
from ouroboros.task_results import STATUS_CANCELLED, write_task_result
tid = "abcd1234"
headless_dir = tmp_path / HEADLESS_TASKS_DIR / tid / "data"
drive_dir = tmp_path / TASK_DRIVES_DIR / tid
headless_dir.mkdir(parents=True)
drive_dir.mkdir(parents=True)
# No settled row, no probe, or a live owner: custody is not proven, nothing goes.
assert remove_subagent_task_drive(tmp_path, tid, live=lambda _task: False) is False
write_task_result(tmp_path, tid, STATUS_CANCELLED, delegation_role="subagent")
assert remove_subagent_task_drive(tmp_path, tid) is False
assert remove_subagent_task_drive(tmp_path, tid, live=lambda _task: True) is False
assert remove_subagent_task_drive(tmp_path, tid, live=lambda _task: None) is False
assert headless_dir.is_dir() and drive_dir.is_dir()
assert remove_subagent_task_drive(tmp_path, tid) is True
assert remove_subagent_task_drive(tmp_path, tid, live=lambda _task: False) is True
assert not (tmp_path / HEADLESS_TASKS_DIR / tid).exists()
assert not (tmp_path / TASK_DRIVES_DIR / tid).exists()
# idempotent / no error when nothing to remove
assert remove_subagent_task_drive(tmp_path, tid) is False
assert remove_subagent_task_drive(tmp_path, tid, live=lambda _task: False) is False
# invalid task id is rejected, not raised
assert remove_subagent_task_drive(tmp_path, "../escape") is False
assert remove_subagent_task_drive(tmp_path, "../escape", live=lambda _task: False) is False
def test_remove_task_scratch_never_promotes_forged_terminal_result(tmp_path):
@ -67,7 +76,7 @@ def test_remove_task_scratch_never_promotes_forged_terminal_result(tmp_path):
artifacts=[],
)
assert remove_subagent_task_drive(tmp_path, tid) is True
assert remove_subagent_task_drive(tmp_path, tid, live=lambda _task: False) is True
stored = load_task_result(tmp_path, tid) or {}
assert "terminal_child_result_snapshot" not in stored
assert not child_drive.exists()
@ -100,7 +109,7 @@ def test_remove_subagent_drive_does_not_promote_custom_late_result(tmp_path):
scratch = tmp_path / TASK_DRIVES_DIR / tid
scratch.mkdir(parents=True)
assert remove_subagent_task_drive(tmp_path, tid) is True
assert remove_subagent_task_drive(tmp_path, tid, live=lambda _task: False) is True
stored = load_task_result(tmp_path, tid) or {}
assert stored["status"] == STATUS_CANCELLED
assert "terminal_child_result_snapshot" not in stored
@ -108,25 +117,31 @@ def test_remove_subagent_drive_does_not_promote_custom_late_result(tmp_path):
assert not scratch.exists()
def test_cancel_running_subagent_removes_drive_source():
def test_cancel_running_subagent_leaves_its_drive_to_the_off_loop_settlement(tmp_path):
# The cancellation custody family lives in task_lifecycle; its settlement
# PUBLICATION half (where the drive cleanup runs) was split into
# supervisor/cancel_publication.py at the module-size boundary.
# PUBLICATION half was split into supervisor/cancel_publication.py at the
# module-size boundary. Neither half deletes the subagent's drive any more:
# settlement copies and hashes the child store, which the cancel path must not
# carry, so the off-loop drive-custody pass settles a cancelled subagent's drive
# WITHOUT waiting out retention (the promptness the cancel path used to give).
from ouroboros import headless
from ouroboros.task_results import write_task_result
custody_src = _read("supervisor/task_lifecycle.py")
publish_src = _read("supervisor/cancel_publication.py")
assert "remove_subagent_task_drive(q.DRIVE_ROOT, str(task_id))" in publish_src
assert "delegation_role" in publish_src # gated on subagent role
# ORDER matters: the drive may only be reclaimed after the process is confirmed
# dead, or a still-running worker loses its scratch out from under it. The
# death confirmation lives in custody, which only then calls the publish
# step; inside the publish step the cleanup follows the terminal emit.
assert "remove_subagent_task_drive" not in publish_src and "remove_subagent_task_drive" not in custody_src
assert "settle_child_drive" not in publish_src and "shutil.rmtree" not in publish_src
assert "survived kill escalation" in custody_src
assert custody_src.index("survived kill escalation") < custody_src.index(
"_publish_cancelled_task(\n"
)
assert publish_src.index("_emit_cancel_task_done") < publish_src.index(
"remove_subagent_task_drive"
)
assert custody_src.index("survived kill escalation") < custody_src.index("_publish_cancelled_task(\n")
data = tmp_path / "data"
fresh = headless.prepare_task_drive(data, "young1", "empty")
write_task_result(data, "young1", "cancelled", result="killed", delegation_role="subagent", child_drive_root=str(fresh))
kept = headless.prepare_task_drive(data, "young2", "empty")
write_task_result(data, "young2", "completed", result="done", delegation_role="subagent", child_drive_root=str(kept))
report = headless.prune_headless_task_drives(data, retention_days=7, live=lambda _task: False)
assert [row["task_id"] for row in report["pruned"]] == ["young1"] and not fresh.exists()
assert kept.is_dir() and report["skipped"] == [{"task_id": "young2", "reason": "younger_than_retention"}]
# ───────────────────────── #9: orphan worker reaping ────────────────────────

View file

@ -83,7 +83,13 @@ def test_answers_survive_eighteen_quizzes_mailbox_gc_and_rotation(runtime):
state.rotate_jsonl_log_if_needed(runtime.root, "chat.jsonl", "chat", max_bytes=1)
assert len(quiz_states(runtime.root, runtime.task["id"])) == 16
assert "q00" not in quiz_states(runtime.root, runtime.task["id"])
cleanup_task_mailbox(runtime.root, runtime.task["id"])
# TZ-1 V10: an unread answer leaves the mailbox only into the settled row that holds it.
assert not cleanup_task_mailbox(runtime.root, runtime.task["id"])
from ouroboros.task_results import write_task_result
held = write_task_result(runtime.root, runtime.task["id"], "completed", result="done")["unread_mailbox"]
assert held["total"] == 18 and held["read_complete"] is True
assert cleanup_task_mailbox(runtime.root, runtime.task["id"])
state.rotate_jsonl_log_if_needed(runtime.root, "chat.jsonl", "chat", max_bytes=1)
assert not drain_owner_entries(runtime.root, runtime.task["id"], include_acknowledged=True)
facts = _facts(runtime)

View file

@ -35,9 +35,12 @@ CHAPTER_BYTE_BUDGETS: dict[str, int] = {
# 7 bytes on the official line); re-based here, no text of this chapter was touched.
# 165550 -> 165900 (PR #1300): the net_transport row and the data-layout row for the merged
# extra-CA bundle; the base sat 174 bytes under the previous budget.
# 165900 -> 166100 (steer sprint 2026-09-26, measured 165937 on the merged tree: delegate_message,
# 165900 -> 166450 (TZ-1 PR-2 custody, measured 166432): module-map rows for the two new
# owners (task_custody.py, gateway/task_archive.py) and the data-layout row for the
# settlement's staging/trash; the artifact-route sentence was replaced, not appended to.
# 166450 -> 166700 (steer sprint 2026-09-26, measured 166595 on the merged tree: delegate_message,
# truthful waiting A-E, low-water reclaim; see the sprint ledger).
"docs/architecture/01-high-level-architecture.md": 166100,
"docs/architecture/01-high-level-architecture.md": 166700,
# 15517 -> 16200 (#1195): the session-custodied startup historical audit is a
# new node of the startup flow (readiness no longer waits for the historical
# seal diagnostic); the chapter had no older description of that pass to replace.
@ -75,7 +78,9 @@ CHAPTER_BYTE_BUDGETS: dict[str, int] = {
# vocabulary (#931/#1061); the touched descriptions were REPLACED and
# compressed (net chapter growth is under the added owner's paragraph size),
# and the merged #1236 base already sat 5 bytes under the previous budget.
"docs/architecture/03-web-ui-pages-and-buttons.md": 107000,
# 107000 -> 107200 (TZ-1 PR-2, measured 107174): one module row for result_files.js, the
# card's Files row; no older text described task result files on the card.
"docs/architecture/03-web-ui-pages-and-buttons.md": 107200,
"docs/architecture/04-server-api-endpoints.md": 26833,
# 27137 -> 30400: the schedule table gains a documented write contract the
# chapter had no text for — one transaction owning the lock ORDER, the strict
@ -112,7 +117,11 @@ CHAPTER_BYTE_BUDGETS: dict[str, int] = {
# typed timeout-cause sentence join TZ-1's bridge-intake paragraph; TZ-2 had compressed the
# owner-wait and heartbeat paragraphs it touched in place (+103 bytes alone), TZ-1's
# paragraph is new, so the union displaces nothing.
"docs/architecture/05-supervisor-loop.md": 33700,
# 33700 -> 34100 (TZ-1 PR-2, measured 34059): the reconcile pass moved off the loop thread
# (own latch, the attempt-basis fence, the drive-custody pass: child-ref retry then bounded
# drive settlements under the queue interlock, generation re-asked per item and commit) and
# the watchdog watches startup with the stall stack; the sentences they change were replaced.
"docs/architecture/05-supervisor-loop.md": 34100,
# 286850 -> 287600: "an answer that has not arrived is a gap" is a new invariant of
# plan review and task acceptance (the slot census vocabulary, the `awaiting`
# projection, the only-awaited task outcome); the in-flight sentence it grew from is
@ -196,9 +205,14 @@ CHAPTER_BYTE_BUDGETS: dict[str, int] = {
# free host_task_facts, stat-only files_rescued and post-work settlement clauses
# replace their prior paragraphs (+242 bytes) independently of the memory writer;
# both contracts survive the merge, with no duplicated prose to displace.
# 316800 -> 322400 (steer sprint 2026-09-26, measured 322273 on the merged tree: delegate_message,
# 316800 -> 317750 (TZ-1 PR-2 custody, measured 317713): the startup-prune sentence is
# replaced by the one deletion owner (settle_child_drive): its obligations (recorded rows
# first, unrecorded files, the input closure, exact unread lines), the shared custody lock,
# the queue interlock the move needs and the identity-ranked view; the forward_to_worker
# clause gains the queued receipt and the terminal unread-mail custody.
# 317750 -> 323400 (steer sprint 2026-09-26, measured 323255 on the merged tree: delegate_message,
# truthful waiting A-E, low-water reclaim; see the sprint ledger).
"docs/architecture/06-agent-core.md": 322400,
"docs/architecture/06-agent-core.md": 323400,
# 36991 -> 37300: the facade paragraph names the three loop constants runtime_limits.py
# gained (events batch bound, budget-projection retry interval); no older text to displace.
# 37300 -> 38400 (PR #1207): the Z.ai (`zai::`) direct provider gets its own route
@ -315,7 +329,7 @@ CHAPTER_BYTE_BUDGETS: dict[str, int] = {
# each side fit alone (TZ-2 96666, TZ-1 96682); TZ-2's reflection-custody and stop-freshness
# clauses and TZ-1's off-loop ingress-lock clause rewrite different bullets in place, so
# the union displaces nothing.
# 96800 -> 97500 (steer sprint 2026-09-26, measured 97396 on the merged tree: delegate_message,
# 96800 -> 97500 (steer sprint 2026-09-26, measured 97380 on the merged tree: delegate_message,
# truthful waiting A-E, low-water reclaim; see the sprint ledger).
"docs/development/06-rules-by-change-class.md": 97500,
"docs/development/07-managed-update-rule.md": 4166,

View file

@ -31,7 +31,7 @@ def test_child_source_closure_survives_real_cleanup(tmp_path, source):
write_task_result(child, "source", "completed", **_field(ref, source))
copied = copy_child_task_result(parent, {"id": "source", "drive_root": str(child)})
assert copied["child_ref_promotion"]["promoted_source_handle_count"] == 1
assert remove_subagent_task_drive(parent, "source") is True
assert remove_subagent_task_drive(parent, "source", live=lambda _task: False) is True
assert not child.exists()
assert artifacts.read_actor_source_bytes(parent, "source", ref) == raw
assert artifacts.collect_task_artifact_records(parent, "source") == []
@ -48,11 +48,11 @@ def test_failed_completion_promotion_retains_child_until_retry(tmp_path, monkeyp
patch.setattr(artifacts, "store_actor_source_bytes", lambda *_a, **_k: (_ for _ in ()).throw(OSError("copy failed")))
copied = copy_child_task_result(parent, {"id": "source", "drive_root": str(child)})
assert copied["child_ref_promotion"]["status"] == "incomplete"
assert remove_subagent_task_drive(parent, "source") is False
assert remove_subagent_task_drive(parent, "source", live=lambda _task: False) is False
assert child.exists()
copied = copy_child_task_result(parent, {"id": "source", "drive_root": str(child)})
assert copied["child_ref_promotion"]["status"] == "complete"
assert remove_subagent_task_drive(parent, "source") is True
assert remove_subagent_task_drive(parent, "source", live=lambda _task: False) is True
assert artifacts.read_actor_source_bytes(parent, "source", ref) == raw
@ -136,7 +136,7 @@ def test_nested_acceptance_sources_survive_copy_back_and_cleanup(tmp_path, outer
assert copied["child_ref_promotion"]["status"] == "complete"
assert copied["child_ref_promotion"]["promoted_source_handle_count"] == 3
assert copied["review_projection"]["panels"][0]["applied_source_ref"] == refs[0]
assert remove_subagent_task_drive(parent, "source") is True
assert remove_subagent_task_drive(parent, "source", live=lambda _task: False) is True
assert not child.exists()
assert [artifacts.read_actor_source_bytes(parent, "source", ref) for ref in refs] == before
assert artifacts.collect_task_artifact_records(parent, "source") == []
@ -166,11 +166,11 @@ def test_nested_copy_failure_holds_child_and_rechecks_existing_checkpoint(tmp_pa
artifacts.task_artifact_dir_path(child, "source") / result["path"])
assert artifacts.read_actor_source_bytes(parent, "source", checkpoint)
assert artifacts.read_actor_source_bytes(parent, "source", trajectory)
assert remove_subagent_task_drive(parent, "source") is False
assert remove_subagent_task_drive(parent, "source", live=lambda _task: False) is False
copied = copy_child_task_result(parent, {"id": "source", "drive_root": str(child)})
assert copied["child_ref_promotion"]["status"] == "complete"
assert copied["child_ref_promotion"]["pending_refs"] == []
assert remove_subagent_task_drive(parent, "source") is True
assert remove_subagent_task_drive(parent, "source", live=lambda _task: False) is True
assert artifacts.read_actor_source_bytes(parent, "source", result) == b"complete output beyond the preview"
@ -227,7 +227,7 @@ def test_nested_trajectory_promotes_existing_call_blobs_without_crawling_prose(t
copied = copy_child_task_result(parent, task)
assert copied["child_ref_promotion"]["status"] == "incomplete"
assert copied["child_ref_promotion"]["pending_refs"]
assert remove_subagent_task_drive(parent, "source") is False
assert remove_subagent_task_drive(parent, "source", live=lambda _task: False) is False
assert retry_pending_child_ref_promotions(parent)["completed"] == ["source"]
first_ref = None
for _ in range(2):
@ -238,7 +238,7 @@ def test_nested_trajectory_promotes_existing_call_blobs_without_crawling_prose(t
assert ref == first_ref
assert "corpus_sha256" not in ref # Checkpoints keep their own byte identity.
assert artifacts.read_actor_source_bytes(checkpoint_root, "source", checkpoint) == original_checkpoint
assert remove_subagent_task_drive(parent, "source") is True
assert remove_subagent_task_drive(parent, "source", live=lambda _task: False) is True
assert not child.exists()
# Rebase the already promoted source again, then repeat after the child is
# gone. Neither a second physical digest nor idempotent copying renames rows.
@ -339,5 +339,5 @@ def test_copyback_prepares_bulk_artifacts_and_selected_review_refs_outside_lock(
copied = copy_child_task_result(parent, {"id": "source", "drive_root": str(child)})
assert copied["child_ref_promotion"]["status"] == "complete"
assert observed[0] == "bulk" and "review" in observed
assert remove_subagent_task_drive(parent, "source") is True
assert remove_subagent_task_drive(parent, "source", live=lambda _task: False) is True
assert (artifacts.task_artifact_dir_path(parent, "source") / "report.txt").read_bytes() == b"actual deliverable"

View file

@ -788,3 +788,20 @@ def test_owner_safety_mode_response_in_frozen_contract():
assert "OwnerSafetyModeResponse" in contracts.__all__
assert set(contracts.OwnerSafetyModeResponse.__annotations__) == {"ok", "safety_mode"}
def test_safety_mode_skip_keeps_its_durable_row_without_a_log_line(tmp_path, caplog):
"""Owner decision В6 (TZ-1): the waved-through check leaves ONLY its durable audit row
(and the counter/Logs it feeds), never a process-log WARNING beside it."""
import json as _json
import logging
ctx = ToolContext(repo_dir=tmp_path / "system", drive_root=tmp_path / "data", task_id="t-quiet", task_metadata={})
(tmp_path / "data" / "logs").mkdir(parents=True, exist_ok=True)
with caplog.at_level(logging.DEBUG, logger=safety_mod.log.name):
safety_mod._emit_safety_mode_skip(ctx, "run_command", "light", "check_conditional")
assert not [r for r in caplog.records if "waved through" in r.getMessage()]
rows = [_json.loads(line) for line in (tmp_path / "data" / "logs" / "events.jsonl").read_text().splitlines()]
assert [r for r in rows if r.get("type") == "safety_mode_skip"] == [
{**row, "ts": row["ts"]} for row in rows if row.get("type") == "safety_mode_skip"]
assert rows[-1]["tool"] == "run_command" and rows[-1]["policy"] == "check_conditional"

View file

@ -198,7 +198,11 @@ def test_server_extraction_size_bounds_have_meaningful_headroom():
for module in _LEAVES
}
counts["server"] = len((REPO / "server.py").read_text(encoding="utf-8").splitlines())
assert all(count <= 1000 for name, count in counts.items() if name != "server")
# server_maintenance.py entered the size band with TZ-1 A (its rationale in
# size_ratchet_manifest.BAND_PATHS): the bounded off-loop drive-custody pass joined the
# reconcile block it runs in; it stays under the band's ceiling, shrink-only from here.
assert all(count <= 1000 for name, count in counts.items() if name not in {"server", "ouroboros.server_maintenance"})
assert counts["ouroboros.server_maintenance"] <= 1150
# server.py keeps the lifespan, the supervisor loop, the owner-command
# dispatch, the process state those three need, AND (on this tree) the
# deferred restart transaction plus post-cutoff upstream drift, so the

View file

@ -0,0 +1,34 @@
"""``server.py`` imported by a spawn/forkserver worker (``__mp_main__``) attaches a stream
handler only: two processes must never rotate ``logs/server.log`` against each other. A
module-level ``multiprocessing.parent_process()`` check would be None in such a child, so the
proof is a REAL child re-running the module under that name."""
from __future__ import annotations
import os
import pathlib
import subprocess
import sys
import pytest
REPO = pathlib.Path(__file__).resolve().parents[1]
@pytest.mark.serial
def test_a_spawn_child_importing_server_gets_a_stream_handler_only(tmp_path):
data = tmp_path / "data"
data.mkdir()
code = (
"import logging, runpy, sys\n"
"runpy.run_path('server.py', run_name='__mp_main__')\n"
"print('HANDLERS', sorted(type(h).__name__ for h in logging.getLogger().handlers))\n"
)
env = {**os.environ, "OUROBOROS_DATA_DIR": str(data), "PYTHONPATH": str(REPO)}
env.pop("PYTEST_CURRENT_TEST", None)
completed = subprocess.run([sys.executable, "-c", code], cwd=REPO, env=env, capture_output=True,
text=True, timeout=180)
assert completed.returncode == 0, completed.stderr[-2000:]
handlers = next(line for line in completed.stdout.splitlines() if line.startswith("HANDLERS"))
assert "RotatingFileHandler" not in handlers and "StreamHandler" in handlers, handlers
assert not (data / "logs" / "server.log").exists()

View file

@ -705,7 +705,6 @@ def _supervisor_harness(monkeypatch, tmp_path, steps):
monkeypatch.setattr(workers_mod, name, noop)
monkeypatch.setattr(workers_mod, "get_event_q", lambda: queue_mod.Queue())
monkeypatch.setattr("ouroboros.delegate_recovery.pre_adopt_planned_handoffs", noop)
monkeypatch.setattr("ouroboros.observability.prune_observability_blobs", lambda _root: {})
monkeypatch.setattr("ouroboros.tools.services.prune_service_logs", lambda _root: {})
monkeypatch.setattr("ouroboros.consciousness.BackgroundConsciousness", _Consciousness)
return rec

View file

@ -0,0 +1,33 @@
"""A directory output's complete skip list is a durable row in the task's ``events.jsonl``
(``directory_output_members_skipped``); the rendered note stays bounded and names that row. A
context without a log root falls back to the process log rather than losing the list."""
from __future__ import annotations
import json
import logging
from types import SimpleNamespace
from ouroboros.tools import shell_outputs
def test_the_complete_skip_list_is_one_durable_task_event(tmp_path, caplog):
logs = tmp_path / "logs"
ctx = SimpleNamespace(task_id="t-1", drive_logs=lambda: logs)
members = [f"site/.env.{index}: dotenv secret" for index in range(7)]
with caplog.at_level(logging.INFO, logger=shell_outputs.log.name):
shell_outputs._record_skipped_members(ctx, "site", members)
rows = [json.loads(line) for line in (logs / "events.jsonl").read_text(encoding="utf-8").splitlines()]
assert len(rows) == 1 and rows[0]["type"] == "directory_output_members_skipped"
assert rows[0]["task_id"] == "t-1" and rows[0]["output"] == "site" and rows[0]["count"] == 7
assert rows[0]["members"] == members and rows[0]["ts"]
assert not [r for r in caplog.records if "skip list" in r.getMessage()]
def no_root():
raise RuntimeError("no drive")
with caplog.at_level(logging.INFO, logger=shell_outputs.log.name):
shell_outputs._record_skipped_members(SimpleNamespace(task_id="t-2", drive_logs=no_root), "site", members)
assert [r for r in caplog.records if "full export skip list (7)" in r.getMessage()]

View file

@ -66,7 +66,7 @@ def test_saved_body_and_sources_precede_orphan_reader_and_actual_prune(roots, mo
root, repo = roots
child = _terminal(root, family=family)
seen = []
def orphan_reader(_root, *, exclude_task_ids, expired_quizzes=None):
def orphan_reader(_root, *, exclude_task_ids, expired_quizzes=None, write_guard=None):
row = load_task_result(root, "saved", strict=True)
assert row["result"] == "full retained answer"
manifest = observability.read_call_manifest_ref(root, row["trace_refs"]["response"], task_id="saved")
@ -83,7 +83,11 @@ def test_saved_body_and_sources_precede_orphan_reader_and_actual_prune(roots, mo
assert _recovery(root, repo)["recovered"] == []
row = load_task_result(root, "saved")
monkeypatch.setattr("ouroboros.retention.age_cutoff", lambda *a, **k: 4_000_000_000)
# The supervisor probe proves no owner of this dead task (unknown would keep the drive).
monkeypatch.setattr("supervisor.queue.task_settlement_liveness", lambda _task: False)
maintenance._startup_prune_sweeps()
assert child.exists(), "startup copies and hashes no child store: settlement is the off-loop pass's"
maintenance._run_drive_custody_pass()
assert not child.exists()
manifest = observability.read_call_manifest_ref(root, row["trace_refs"]["response"], task_id="saved")
assert observability.read_blob_ref(root, manifest["full_payload_ref"])["answer"] == "full retained answer"
@ -126,7 +130,7 @@ def test_orphan_exclusion_filters_before_effective_materialization(roots, monkey
return {"task_id": tid, "status": "failed", "result": "proven orphan"}
monkeypatch.setattr("ouroboros.task_status.load_effective_task_result", effective)
assert reconcile_orphaned_running_tasks(root, exclude_task_ids={"live"}) == 1
assert read == [("dead", False), ("dead", True)] # decide on a projection, then heal with custody
assert read == [("dead", False)] # decide and persist on the status-only projection: no file work
assert load_task_result(root, "live")["status"] == "running"
assert load_task_result(root, "dead")["status"] == "failed"
@ -180,8 +184,8 @@ def test_host_terminal_without_child_terminal_does_not_disable_retention(
assert report["unresolved"] == report["errors"] == report["protected"] == []
assert result_path.read_bytes() == settled_bytes
monkeypatch.setattr("ouroboros.retention.age_cutoff", lambda *a, **k: 4_000_000_000)
maintenance._startup_prune_sweeps(preserve_task_sources=bool(
report["unresolved"] or report["protected"] or report["errors"]))
monkeypatch.setattr("supervisor.queue.task_settlement_liveness", lambda _task: False)
maintenance._run_drive_custody_pass()
assert not child.exists()
assert result_path.read_bytes() == settled_bytes
@ -382,38 +386,72 @@ def test_late_split_root_start_cannot_replace_canonical_terminal(roots):
@pytest.mark.parametrize("failure", [False, True])
def test_periodic_bulk_work_does_not_block_drain_or_duplicate_sweep(roots, monkeypatch, failure):
"""The child-ref promotion walk is history-sized (every child drive, a result load
each). It rides the 300 s reconcile block, never the 20 s cancel sweep: while the
walk is held the tick returns, the cancel sweep keeps its OWN cadence (a second one
starts and finishes 25 s later), and no second walk starts. Whether the walk ends or
raises, its latch opens and its marker is stamped at the END (issue #1230)."""
root, _ = roots
entered, release, done = threading.Event(), threading.Event(), threading.Event()
lock = threading.Lock()
clock = [100.0]
calls = []
monkeypatch.setattr(maintenance, "_CANCEL_INTENT_SWEEP_LOCK", lock)
cancel_lock, reconcile_lock = threading.Lock(), threading.Lock()
clock = [1000.0]
calls, threads = [], []
monkeypatch.setattr(maintenance, "_CANCEL_INTENT_SWEEP_LOCK", cancel_lock)
monkeypatch.setattr(maintenance, "_RECONCILE_SWEEP_LOCK", reconcile_lock)
monkeypatch.setattr(maintenance, "_LAST_CANCEL_INTENT_SWEEP", [0.0])
monkeypatch.setattr(maintenance, "time", SimpleNamespace(time=lambda: clock[0]))
def tracked(**kwargs):
thread = threading.Thread(**kwargs)
threads.append(thread)
return thread
monkeypatch.setattr(maintenance, "threading", SimpleNamespace(Thread=tracked))
monkeypatch.setattr("supervisor.task_lifecycle.sweep_cancel_intents", lambda: calls.append("cancel"))
monkeypatch.setattr("supervisor.terminal_delivery.replay_pending_deliveries", lambda root: calls.append("delivery"))
def bulk(root):
monkeypatch.setattr(maintenance, "_reconcile_abandoned_usage", lambda root: calls.append("usage"))
monkeypatch.setattr(maintenance, "_periodic_zombie_reconcile", lambda **kw: calls.append("heal"))
def bulk(root, **_kwargs):
calls.append("refs")
entered.set()
assert release.wait(3)
done.set()
if failure:
raise OSError("copy failed")
monkeypatch.setattr(observability, "retry_pending_child_ref_promotions", bulk)
maintenance._periodic_supervisor_maintenance([100.0], [100.0])
marker = [0.0]
def cancel_sweeps():
return [thread for thread in threads if thread.name == "terminal-maintenance"]
maintenance._periodic_supervisor_maintenance([clock[0]], marker) # both cadences due
assert entered.wait(2)
try:
clock[0] = 125
maintenance._periodic_supervisor_maintenance([125.0], [125.0])
assert calls == ["cancel", "delivery", "refs"] # drain returned while I/O is still held
for thread in cancel_sweeps():
thread.join(3)
assert calls.count("cancel") == calls.count("delivery") == calls.count("usage") == 1
assert calls.index("heal") < calls.index("refs"), "heal first, then promote"
assert not cancel_lock.locked(), "the cancel sweep finished while the walk is still held"
clock[0] = 1025.0
maintenance._periodic_supervisor_maintenance([clock[0]], marker)
for thread in cancel_sweeps():
thread.join(3)
assert [t.name for t in threads] == ["terminal-maintenance", "reconcile-maintenance", "terminal-maintenance"]
assert calls.count("cancel") == 2 and calls.count("refs") == 1, "cancel cadence held; no duplicate walk"
assert marker[0] == 0.0 and reconcile_lock.locked(), "stamped only when the walk ENDS"
finally:
release.set()
assert done.wait(2) and lock.acquire(timeout=2)
lock.release()
maintenance._periodic_supervisor_maintenance([125.0], [125.0])
assert lock.acquire(timeout=2)
lock.release()
assert calls == ["cancel", "delivery", "refs"] * 2
for thread in threads:
thread.join(3)
assert done.wait(2) and all(not thread.is_alive() for thread in threads)
assert marker[0] == 1025.0 and not reconcile_lock.locked() and not cancel_lock.locked()
clock[0] = 1050.0
maintenance._periodic_supervisor_maintenance([clock[0]], marker) # 25 s after the END
for thread in threads:
thread.join(3)
assert calls.count("cancel") == 3 and calls.count("refs") == 1, "the walk waits its 300 s, the cancel sweep does not"
def test_thread_start_failure_releases_maintenance_latch(roots, monkeypatch):

View file

@ -60,6 +60,15 @@ def subscription_ui():
self.end_headers()
self.wfile.write(body.encode())
return
if self.path.startswith("/api/tasks/") and backend.get("task_gateway"):
response = backend["task_gateway"].get(self.path)
self.send_response(response.status_code)
for key in ("content-type", "content-disposition", "content-length"):
if key in response.headers:
self.send_header(key, response.headers[key])
self.end_headers()
self.wfile.write(response.content)
return
self.path = self.path.removeprefix("/static")
super().do_GET()

View file

@ -144,9 +144,11 @@ def test_stall_end_is_written_once_per_alerted_stall(monkeypatch, journal):
end = next(row for _p, row in journal if row["type"] == "supervisor_loop_stall_end")
assert end["stalled_sec"] == pytest.approx(100.0, abs=1.0)
assert end["phase"] == "maintenance" # where it was stuck, not where it resumed
# The recovery stamp publishes the loop thread's CPU over the stalled interval itself.
assert end["loop_thread_cpu_sec"] == liveness[1]["loop_thread_cpu_sec"]
# One recovery stamp: the CPU charged to the stall is that stamp's delta, over
# the same wall interval the stall lasted.
assert end["loop_thread_cpu_sec"] == pytest.approx(liveness[1]["loop_thread_cpu_sec"], abs=1e-3)
assert isinstance(end["loop_thread_cpu_sec"], float)
assert end["cpu_interval_sec"] == end["stalled_sec"]
def test_a_healthy_loop_journals_neither_row(monkeypatch, journal):
@ -223,7 +225,7 @@ def test_the_loop_publishes_one_monotonic_stamp_per_tick_phase():
r'[^)]*\), time\.(\w+)\(\)',
source,
)
assert [phase for phase, _clock in stamps] == ["events", "maintenance", "assign"], stamps
assert [phase for phase, _clock in stamps] == ["startup", "events", "maintenance", "assign"], stamps
assert {clock for _phase, clock in stamps} == {"monotonic"}, stamps
# The bounded drain receives the loop's own liveness list and observes the lag itself.
from ouroboros import server_liveness
@ -361,3 +363,403 @@ def test_failed_init_is_not_ready_on_the_state_api_while_boot_waiters_still_sett
payload = json.loads(asyncio.run(api_state(request)).body)
assert payload["supervisor_ready"] is False
assert payload["supervisor_error"] == "Supervisor init failed: boot dependency refused"
def test_stall_end_charges_the_whole_stall_not_the_last_phase(monkeypatch, journal):
"""Several phases can pass between the loop's recovery and the watchdog's next
look. The end row charges the thread's CPU from the stamp it went silent on to
its LATEST stamp (the cumulative totals), never only the last phase's delta."""
import server
from ouroboros import server_liveness
monkeypatch.setenv("OUROBOROS_SUPERVISOR_LIVENESS_DEADLINE_SEC", "1")
liveness = _live_liveness("maintenance")
onset_total = liveness[1]["loop_thread_cpu_total_sec"]
clock = _Clock(mono=liveness[0] + 100.0)
monkeypatch.setattr(server_liveness, "time", clock)
stop = threading.Event()
try:
server._start_supervisor_liveness_watchdog(liveness, stop)
_wait_until(lambda: journal)
# The loop recovers and publishes THREE stamps, burning CPU before each,
# before it moves the trigger stamp the watchdog reads.
for phase in ("assign", "events", "maintenance"):
burn_until = time.thread_time() + 0.02
while time.thread_time() < burn_until:
pass
liveness[1] = server_liveness.loop_phase_facts(liveness, phase)
liveness[0] = clock.mono
_wait_until(lambda: any(row["type"] == "supervisor_loop_stall_end" for _p, row in journal))
finally:
_stop_watchdog(stop)
end = next(row for _p, row in journal if row["type"] == "supervisor_loop_stall_end")
whole = liveness[1]["loop_thread_cpu_total_sec"] - onset_total
assert end["loop_thread_cpu_sec"] == pytest.approx(whole, abs=2e-3)
assert whole >= 0.05 and end["loop_thread_cpu_sec"] > liveness[1]["loop_thread_cpu_sec"]
assert end["cpu_interval_sec"] == end["stalled_sec"] == pytest.approx(100.0, abs=1.0)
def test_stall_row_carries_the_loop_threads_stack(monkeypatch, journal):
"""The onset row names the LINE the loop thread is on, bounded to a few frames."""
import server
from ouroboros import server_liveness
monkeypatch.setenv("OUROBOROS_SUPERVISOR_LIVENESS_DEADLINE_SEC", "1")
# The watchdog must not fetch source through linecache: disk may be stalled.
import linecache
monkeypatch.setattr(linecache, "getline", lambda *_a, **_k: (_ for _ in ()).throw(AssertionError("disk read")))
liveness = _live_liveness("maintenance")
clock = _Clock(mono=liveness[0] + 100.0)
monkeypatch.setattr(server_liveness, "time", clock)
release = threading.Event()
def _pretend_stalled_maintenance_step():
release.wait(10)
stalled = threading.Thread(target=_pretend_stalled_maintenance_step, name="fake-supervisor-loop")
stalled.start()
stop = threading.Event()
try:
server._start_supervisor_liveness_watchdog(liveness, stop, loop_thread_ident=stalled.ident)
_wait_until(lambda: journal)
finally:
release.set()
stalled.join(5)
_stop_watchdog(stop)
row = next(row for _p, row in journal if row["type"] == "supervisor_loop_stall")
stack = row["stack"]
assert isinstance(stack, list) and 1 <= len(stack) <= server_liveness._STALL_STACK_FRAMES, stack
assert any("_pretend_stalled_maintenance_step" in frame for frame in stack), stack
assert all(isinstance(frame, str) and " in " in frame for frame in stack), stack
assert row["phase"] == "maintenance" # the coarse phase still rides beside the exact line
def test_the_watched_thread_defaults_to_the_one_that_started_the_watchdog(monkeypatch, journal):
"""The loop thread starts its own watchdog (startup phase included), so the
default identity is the CALLER: this test thread, caught in its own wait."""
import server
from ouroboros import server_liveness
monkeypatch.setenv("OUROBOROS_SUPERVISOR_LIVENESS_DEADLINE_SEC", "1")
liveness = _live_liveness("startup")
clock = _Clock(mono=liveness[0] + 100.0)
monkeypatch.setattr(server_liveness, "time", clock)
stop = threading.Event()
try:
server._start_supervisor_liveness_watchdog(liveness, stop)
_wait_until(lambda: journal)
finally:
_stop_watchdog(stop)
row = next(row for _p, row in journal if row["type"] == "supervisor_loop_stall")
assert row["phase"] == "startup"
assert any("_wait_until" in frame for frame in row["stack"]), row["stack"]
def test_a_gone_thread_yields_no_stack_rather_than_an_invented_one():
from ouroboros import server_liveness
assert server_liveness._loop_thread_stack(None) == []
assert server_liveness._loop_thread_stack(-1) == []
def test_a_stamp_names_the_interval_its_cpu_covers():
"""``loop_thread_cpu_sec`` is honest only beside ``cpu_interval_sec``, and the
cumulative total lets the watchdog charge a whole stall it did not watch stamp by stamp."""
from ouroboros import server_liveness
liveness = [time.monotonic(), {}, time.thread_time(), None]
time.sleep(0.05)
first = server_liveness.loop_phase_facts(liveness, "events")
assert 0.045 <= first["cpu_interval_sec"] < 5.0, first
assert first["loop_thread_cpu_total_sec"] == pytest.approx(time.thread_time(), abs=0.05)
liveness[0] = time.monotonic()
second = server_liveness.loop_phase_facts(liveness, "maintenance")
assert second["cpu_interval_sec"] < 0.05, second
assert second["loop_thread_cpu_total_sec"] >= first["loop_thread_cpu_total_sec"]
assert server_liveness._stall_cpu_over(first, second) == pytest.approx(
second["loop_thread_cpu_total_sec"] - first["loop_thread_cpu_total_sec"], abs=1e-3)
# A foreign stamp without a total falls back to the latest delta; nothing is invented.
assert server_liveness._stall_cpu_over({}, second) == second["loop_thread_cpu_sec"]
assert server_liveness._stall_cpu_over({}, {}) is None
def test_the_watchdog_watches_startup_and_every_generation_exit_stops_it():
"""Source pin: the watchdog starts BEFORE the init block, so a hung recovery,
worktree prune or worker spawn is a journaled "startup" stall with its stack
instead of a silent wedge behind ``_supervisor_ready``; and BOTH generation
exits — the init-failure return and the loop exit — set the per-generation
stop token, so no watchdog outlives the liveness list it reads."""
import server
source = inspect.getsource(server._run_supervisor)
start = source.index("_start_supervisor_liveness_watchdog(_loop_liveness, _watchdog_stop)")
assert start < source.index("ensure_legacy_imported("), "the watchdog must start before init"
assert source.index("\n try:\n", source.index("prior_worker_pids:")) < start, "watchdog setup must use the init failure rail"
assert source.index('loop_phase_facts(_loop_liveness, "startup", new_tick=True)') < start
assert source.count("_watchdog_stop.set()") == 2, "init-failure exit and loop exit"
failure = source.index('_supervisor_error = f"Supervisor init failed: {exc}"')
assert source.index("_watchdog_stop.set()", failure) < source.index("\n return\n", failure)
def test_watchdog_start_failure_publishes_init_failure_without_running_recovery_on_live_data(monkeypatch):
"""The earlier watchdog start must not bypass the supervisor init-outcome rail."""
import server
ready = threading.Event()
init_done = threading.Event()
recovery = []
monkeypatch.setattr(server, "_supervisor_ready", ready)
monkeypatch.setattr(server, "_supervisor_init_done", init_done)
monkeypatch.setattr(server, "_supervisor_thread", threading.current_thread())
monkeypatch.setattr(server, "_supervisor_error", None)
monkeypatch.setattr(server, "_consciousness", None)
monkeypatch.setattr(server, "_apply_settings_to_env", lambda _settings: None)
monkeypatch.setattr(server, "_startup_worker_pids", lambda _root: set())
monkeypatch.setattr(server, "_run_startup_task_recovery", lambda *_a, **_k: recovery.append(True))
monkeypatch.setattr(server, "_start_supervisor_liveness_watchdog", lambda *_a: (_ for _ in ()).throw(RuntimeError("watchdog start refused")))
server._run_supervisor({})
assert init_done.is_set(), "the outcome is published for boot waiters"
assert not ready.is_set(), "a failed init is an outcome, never readiness"
assert "watchdog start refused" in server._supervisor_error
assert server._supervisor_thread is None
assert recovery == [True]
def test_a_cut_stack_is_flagged_and_a_startup_stall_does_not_claim_a_chat_that_answers(monkeypatch, journal, caplog):
"""The onset row says when its bounded stack is not the whole one (``loop_stack_truncated``),
and a stall in the startup phase is worded as an unfinished start: no native chat answers
for a generation that has not initialized."""
import logging
import server
from ouroboros import server_liveness
monkeypatch.setenv("OUROBOROS_SUPERVISOR_LIVENESS_DEADLINE_SEC", "1")
liveness = _live_liveness("startup")
clock = _Clock(mono=liveness[0] + 100.0)
monkeypatch.setattr(server_liveness, "time", clock)
release = threading.Event()
def _deep(depth):
if depth:
return _deep(depth - 1)
release.wait(10)
stalled = threading.Thread(target=lambda: _deep(server_liveness._STALL_STACK_FRAMES + 5), name="fake-loop")
stalled.start()
stop = threading.Event()
try:
with caplog.at_level(logging.ERROR):
server._start_supervisor_liveness_watchdog(liveness, stop, loop_thread_ident=stalled.ident)
_wait_until(lambda: journal)
finally:
release.set()
stalled.join(5)
_stop_watchdog(stop)
row = next(row for _p, row in journal if row["type"] == "supervisor_loop_stall")
assert row["loop_stack_truncated"] is True and len(row["stack"]) == server_liveness._STALL_STACK_FRAMES
assert all(" in " in frame and "locals" not in frame for frame in row["stack"])
message = next(r.getMessage() for r in caplog.records if "STALLED" in r.getMessage())
assert "startup has not finished" in message and "native chat still answers" not in message
shallow: dict = {}
assert server_liveness._loop_thread_stack(threading.get_ident(), limit=500, facts=shallow)
assert "loop_stack_truncated" not in shallow # a whole stack carries no flag
def test_stall_end_carries_samples_top_frames_and_the_last_stack(monkeypatch, journal):
"""While a stall is open the watchdog samples the thread's stack once per interval; the
closing row says how many samples it took, which repository frames they fold into (at
most five) and the last stack - where the thread SPENT the stall, not only where it began."""
import server
from ouroboros import server_liveness
monkeypatch.setenv("OUROBOROS_SUPERVISOR_LIVENESS_DEADLINE_SEC", "1")
liveness = _live_liveness("maintenance")
clock = _Clock(mono=liveness[0] + 100.0)
monkeypatch.setattr(server_liveness, "time", clock)
release = threading.Event()
def _pretend_stalled_maintenance_step():
release.wait(10)
stalled = threading.Thread(target=_pretend_stalled_maintenance_step, name="fake-supervisor-loop")
stalled.start()
stop = threading.Event()
try:
server._start_supervisor_liveness_watchdog(liveness, stop, loop_thread_ident=stalled.ident)
_wait_until(lambda: journal)
ticks = clock.ticks
_wait_until(lambda: clock.ticks > ticks + 3) # several intervals inside the open stall
liveness[1] = server_liveness.loop_phase_facts(liveness, "assign")
liveness[0] = clock.mono
_wait_until(lambda: any(row["type"] == "supervisor_loop_stall_end" for _p, row in journal))
finally:
release.set()
stalled.join(5)
_stop_watchdog(stop)
end = next(row for _p, row in journal if row["type"] == "supervisor_loop_stall_end")
assert end["samples"] >= 3 and 1 <= len(end["top_frames"]) <= 5
assert sum(item["samples"] for item in end["top_frames"]) == end["samples"]
hot = end["top_frames"][0]["frame"]
assert "test_supervisor_loop_measurements.py:_pretend_stalled_maintenance_step" in hot, hot
assert any("_pretend_stalled_maintenance_step" in frame for frame in end["last_stack"])
import sys
runtime_only = [f"{sys.prefix}/lib/python3/threading.py:1 in wait".replace("\\", "/")]
assert server_liveness._innermost_repo_frame(runtime_only) == "" # a stall entirely inside the runtime names nothing
assert server_liveness._innermost_repo_frame(["/srv/app/x.py:3 in f", *runtime_only]) == "/srv/app/x.py:f"
def test_stack_rows_parse_windows_drives_and_colons_inside_paths():
"""A row is ``path:line in func`` with the path before the LAST ``:<line> in ``: a Windows
drive letter (``C:``) is part of an absolute path, never a relative path named ``C``."""
from ouroboros import server_liveness
windows = ["C:/Users/owner/app/ouroboros/tool.py:7 in helper", "ouroboros/loop.py:12 in run_step"]
assert server_liveness._innermost_repo_frame(windows[::-1]) == "ouroboros/loop.py:run_step"
assert server_liveness._innermost_repo_frame(windows[:1]) == "C:/Users/owner/app/ouroboros/tool.py:helper"
assert server_liveness._innermost_repo_frame(["C:\\Tools\\x.py:3 in f"]) == "C:\\Tools\\x.py:f"
assert server_liveness._innermost_repo_frame(["pkg/a:b.py:9 in <lambda>"]) == "pkg/a:b.py:<lambda>"
assert server_liveness._innermost_repo_frame(["not a stack row"]) == ""
def test_a_recurring_step_failure_backs_off_to_powers_of_two_and_reports_recovery(monkeypatch, caplog):
"""Failures 1, 2, 4, 8 of one periodic step are WARNING with the traceback; the ones between
are DEBUG; the first success afterwards is one INFO naming the streak."""
import logging
import ouroboros.server_maintenance as sm
monkeypatch.setattr(sm, "_STEP_FAILURES", {})
with caplog.at_level(logging.DEBUG, logger=sm.log.name):
for _ in range(5):
try:
raise RuntimeError("boom")
except RuntimeError:
sm._step_failed("demo_step")
sm._step_recovered("demo_step")
sm._step_recovered("demo_step") # no streak: nothing said
rows = [(r.levelname, r.getMessage()) for r in caplog.records if "demo_step" in r.getMessage()]
assert [level for level, _ in rows] == ["WARNING", "WARNING", "DEBUG", "WARNING", "DEBUG", "INFO"], rows
assert rows[0][1].endswith("(failure 1 in a row)") and rows[-1][1] == "demo_step recovered after 5 failure(s)"
assert all(r.exc_info for r in caplog.records if "failure" in r.getMessage() and "recovered" not in r.getMessage())
def test_a_due_cadence_finding_its_latch_held_journals_the_duty_stall_with_its_stack(monkeypatch, journal):
"""An off-loop pass that outlives its own cadence is a host duty stall: the next due tick
journals it once with the pass thread's stack (its threshold is the pass's cadence, never
the loop deadline), and the pass's end journals the closing row."""
import ouroboros.server_maintenance as sm
monkeypatch.setattr(sm, "_DUTIES", {})
monkeypatch.setattr(sm, "_LAST_CANCEL_INTENT_SWEEP", [time.time()])
release = threading.Event()
def _slow_custody_pass(stop_event, latch):
try:
release.wait(10)
finally:
latch.release()
monkeypatch.setattr(sm, "_run_periodic_custody_sweep", _slow_custody_pass)
last_custody, last_reconcile = [0.0], [time.time()]
try:
sm._periodic_supervisor_maintenance(last_custody, last_reconcile)
assert sm._CUSTODY_SWEEP_LOCK.locked() and not journal
last_custody[0] = 0.0 # the next cadence is due while the pass still runs...
sm._periodic_supervisor_maintenance(last_custody, last_reconcile)
assert not journal, "...but the pass has not run a cadence yet: a due tick alone is no stall"
sm._DUTIES[id(sm._CUSTODY_SWEEP_LOCK)]["since"] -= 601 # now it has outlived its cadence
sm._periodic_supervisor_maintenance(last_custody, last_reconcile)
sm._periodic_supervisor_maintenance(last_custody, last_reconcile) # journaled once, not per tick
finally:
release.set()
_wait_until(lambda: not sm._CUSTODY_SWEEP_LOCK.locked())
_wait_until(lambda: any(row["type"] == "host_duty_stall_end" for _p, row in journal))
kinds = [row["type"] for _p, row in journal]
assert kinds == ["host_duty_stall", "host_duty_stall_end"], kinds
stall = journal[0][1]
assert stall["duty"] == "custody-maintenance" and stall["cadence_sec"] == 600
assert any("_slow_custody_pass" in frame for frame in stall["stack"]), stall
assert journal[1][1]["duty"] == "custody-maintenance" and journal[1][1]["running_sec"] >= 0
def test_a_reconcile_pass_just_started_after_an_old_end_stamp_is_not_a_stall(monkeypatch, journal):
"""The reconcile marker is stamped when a pass ENDS, so it is still old while the next pass
starts: every following tick is "due" at once. The duty threshold is the pass's own running
time, so a pass that has run a moment is never journaled, and one past its cadence is."""
import ouroboros.server_maintenance as sm
monkeypatch.setattr(sm, "_DUTIES", {})
monkeypatch.setattr(sm, "_LAST_CANCEL_INTENT_SWEEP", [time.time()])
release = threading.Event()
def _slow_reconcile_pass(marker, stop_event=None, latch=None, on_orphans_healed=None):
try:
release.wait(10)
finally:
marker[0] = time.time()
latch.release()
monkeypatch.setattr(sm, "_run_periodic_reconcile_sweep", _slow_reconcile_pass)
last_custody, last_reconcile = [time.time()], [time.time() - 3600] # the previous pass ended long ago
try:
sm._periodic_supervisor_maintenance(last_custody, last_reconcile) # starts the pass
assert sm._RECONCILE_SWEEP_LOCK.locked()
for _ in range(3):
sm._periodic_supervisor_maintenance(last_custody, last_reconcile) # due every tick, latch held
assert not journal, journal
sm._DUTIES[id(sm._RECONCILE_SWEEP_LOCK)]["since"] -= 301
sm._periodic_supervisor_maintenance(last_custody, last_reconcile)
assert [row["type"] for _p, row in journal] == ["host_duty_stall"]
assert journal[0][1]["duty"] == "reconcile-maintenance" and journal[0][1]["running_sec"] >= 300
finally:
release.set()
_wait_until(lambda: not sm._RECONCILE_SWEEP_LOCK.locked())
_wait_until(lambda: any(row["type"] == "host_duty_stall_end" for _p, row in journal))
def test_a_finished_pass_never_drops_the_duty_its_successor_registered(monkeypatch, journal):
"""A pass releases its latch inside its target and its wrapper closes the duty afterwards;
a successor started in between registers its own duty, which the old wrapper must keep, so
the successor's own overrun is still journaled."""
import ouroboros.server_maintenance as sm
monkeypatch.setattr(sm, "_DUTIES", {})
monkeypatch.setattr(sm, "_LAST_CANCEL_INTENT_SWEEP", [time.time()])
released, first_may_return, second_release = threading.Event(), threading.Event(), threading.Event()
passes = []
def _custody_pass(stop_event, latch):
passes.append(threading.current_thread())
if len(passes) == 1:
latch.release() # the first pass frees its latch ...
released.set()
first_may_return.wait(10) # ... and its wrapper closes the duty only later
return
try:
second_release.wait(10)
finally:
latch.release()
monkeypatch.setattr(sm, "_run_periodic_custody_sweep", _custody_pass)
last_custody, last_reconcile = [0.0], [time.time()]
try:
sm._periodic_supervisor_maintenance(last_custody, last_reconcile)
_wait_until(released.is_set)
last_custody[0] = 0.0
sm._periodic_supervisor_maintenance(last_custody, last_reconcile) # the successor starts
successor = sm._DUTIES[id(sm._CUSTODY_SWEEP_LOCK)]
first_may_return.set()
_wait_until(lambda: not passes[0].is_alive())
assert sm._DUTIES.get(id(sm._CUSTODY_SWEEP_LOCK)) is successor, "the old wrapper dropped its successor"
successor["since"] -= 601
last_custody[0] = 0.0
sm._periodic_supervisor_maintenance(last_custody, last_reconcile)
assert [row["type"] for _p, row in journal] == ["host_duty_stall"]
finally:
first_may_return.set()
second_release.set()
_wait_until(lambda: not sm._CUSTODY_SWEEP_LOCK.locked())
_wait_until(lambda: any(row["type"] == "host_duty_stall_end" for _p, row in journal))
assert not sm._DUTIES, "the successor closed its own duty"

View file

@ -10,11 +10,19 @@ It now runs on a daemon thread under a non-blocking module lock (busy => skip,
never queue), reads its CANDIDATES before it reads LIVENESS through one shared
live-owner source, and stops mutating the moment its loop generation ends —
reaching the daemon attach-only while a stop, restart or panic is in flight.
Every guard below is pinned in both directions.
The ~300 s reconcile block (zombie heal over every stored task result, artifact
materialization of a healed row, the child-ref promotion walk over every child
drive) was the residual the invariant tolerated INLINE on the tick, and the
child-ref walk also sat inside the 20 s cancel sweep, holding that latch for its
whole length. Both ride the same off-loop shape now: own latch, own daemon
thread, marker stamped when the pass ENDS (issue #1230), generation token
re-read before every step. Every guard below is pinned in both directions.
"""
from __future__ import annotations
import os
import threading
import time
from types import SimpleNamespace
@ -23,14 +31,16 @@ import pytest
@pytest.fixture(autouse=True)
def _fresh_custody_sweep_latch(monkeypatch):
"""The custody latch is process-global and a pass may outlive the test that started it
(the first tick of any real loop starts one): every test here gets its own latch, so a
busy one left behind by another test can neither skip this sweep nor be released by it."""
"""The custody and reconcile latches are process-global and a pass may outlive the test
that started it (the first tick of any real loop starts one): every test here gets its
own latches, so a busy one left behind by another test can neither skip this sweep nor
be released by it."""
import threading
from ouroboros import server_maintenance
monkeypatch.setattr(server_maintenance, "_CUSTODY_SWEEP_LOCK", threading.Lock())
monkeypatch.setattr(server_maintenance, "_RECONCILE_SWEEP_LOCK", threading.Lock())
def _track_threads(monkeypatch) -> list:
@ -287,13 +297,19 @@ def test_a_stop_in_flight_makes_the_sweep_gateway_attach_only(tmp_path, monkeypa
assert used[-1] == ("attach", {}), "a restart in flight never ensures either"
def test_reconcile_cadence_is_stamped_when_the_pass_ends(quiet_tick, monkeypatch):
"""The 300-s zombie reconcile stamps its marker when the pass ENDS, so a pass
slower than its cadence never re-arms on the very next tick (issue #1230); a
pass that raises still stamps, and the next eligible run still happens."""
def test_reconcile_cadence_is_stamped_when_the_pass_ends(quiet_tick, monkeypatch, caplog):
"""Off the loop thread the rule of issue #1230 still holds: the marker is stamped
when the pass ENDS (before its latch opens), so a pass slower than its cadence never
re-arms on the very next tick; a pass that raises still stamps and still releases,
and its failure is a WARNING on the maintenance thread, never a crash of the loop."""
import logging
sm = quiet_tick
clock = [1_000_000.0]
monkeypatch.setattr(sm.time, "time", lambda: clock[0])
monkeypatch.setattr("ouroboros.observability.retry_pending_child_ref_promotions", lambda root, **kwargs: {})
monkeypatch.setattr(sm, "_STEP_FAILURES", {})
threads = _track_threads(monkeypatch)
calls = []
def slow_pass(**kwargs):
@ -303,23 +319,297 @@ def test_reconcile_cadence_is_stamped_when_the_pass_ends(quiet_tick, monkeypatch
raise RuntimeError("the pass itself failed")
monkeypatch.setattr(sm, "_periodic_zombie_reconcile", slow_pass)
busy = threading.Lock()
busy.acquire()
monkeypatch.setattr(sm, "_CANCEL_INTENT_SWEEP_LOCK", busy) # the 20 s sweep is skipped
last_custody_reap = [clock[0] + 10_000] # the 600 s sweep is not due
marker = [clock[0] - 301]
sm._periodic_supervisor_maintenance(last_custody_reap, marker)
def tick(marker):
sm._periodic_supervisor_maintenance(last_custody_reap, marker)
for thread in threads:
thread.join(5)
marker = [clock[0] - 301]
tick(marker)
assert len(calls) == 1 and marker[0] == clock[0] # stamped at the END of the 400 s pass
sm._periodic_supervisor_maintenance(last_custody_reap, marker)
tick(marker)
assert len(calls) == 1, "a pass slower than its cadence must not re-arm on the next tick"
clock[0] += 301.0
with pytest.raises(RuntimeError):
sm._periodic_supervisor_maintenance(last_custody_reap, marker)
with caplog.at_level(logging.WARNING):
tick(marker)
assert len(calls) == 2 and marker[0] == clock[0], "a failing pass still stamps when it ends"
sm._periodic_supervisor_maintenance(last_custody_reap, marker)
assert not sm._RECONCILE_SWEEP_LOCK.locked(), "and still releases its latch"
assert any("Periodic reconcile_sweep failed (failure 1 in a row)" in r.getMessage() for r in caplog.records)
tick(marker)
assert len(calls) == 2
clock[0] += 301.0
sm._periodic_supervisor_maintenance(last_custody_reap, marker)
tick(marker)
assert len(calls) == 3, "the next eligible run still happens after the cadence"
assert all(not thread.is_alive() for thread in threads)
def _quiet_reconcile_steps(monkeypatch, sm, done: list, *, heal=None):
"""Stub every step of the reconcile block; each records its name (and the thread it
ran on) so a test can pin order, placement and the generation cut."""
def step(name, value=0):
def run(*_a, **_k):
done.append((name, threading.current_thread().name))
return value
return run
monkeypatch.setattr("ouroboros.skill_review_runner.reconcile_stale_review_jobs", step("review_jobs"))
monkeypatch.setattr("ouroboros.task_status.reconcile_orphaned_running_tasks", heal or step("orphans"))
monkeypatch.setattr("ouroboros.projects_registry.reconcile_projects", step("projects"))
monkeypatch.setattr(sm, "_resume_interrupted_project_deletions", step("deletions"))
monkeypatch.setattr("ouroboros.observability.retry_pending_child_ref_promotions", step("child_refs", {}))
def test_a_slow_reconcile_block_never_holds_the_loop_tick(quiet_tick, monkeypatch):
"""The 300 s zombie/artifact reconcile ran INLINE on the tick: a history-sized heal
held the drain, the fence acks and assignment for its whole walk. The tick that
STARTS it now returns at once; a tick while it is busy SKIPS (one thread, no queue);
the marker is stamped only when the pass ENDS; the latch opens in ``finally``; and
the orphan-heal notification reaches the alarm clock from the maintenance thread."""
sm = quiet_tick
threads = _track_threads(monkeypatch)
entered, release = threading.Event(), threading.Event()
done, healed = [], []
def slow_heal(root, **kwargs):
entered.set()
assert release.wait(5), "the test must release the reconcile block"
return 2
_quiet_reconcile_steps(monkeypatch, sm, done, heal=slow_heal)
marker = [0.0]
try:
started = time.monotonic()
sm._periodic_supervisor_maintenance(
[time.time()], marker,
on_orphans_healed=lambda count: healed.append((count, threading.current_thread().name)))
tick = time.monotonic() - started
assert entered.wait(5), "the reconcile block really ran"
assert tick < 2.0, f"the loop tick waited {tick:.1f}s for the reconcile block"
assert marker[0] == 0.0, "not stamped until the pass ENDS (issue #1230)"
sm._periodic_supervisor_maintenance([time.time()], marker)
assert [thread.name for thread in threads] == ["reconcile-maintenance"], "busy => skipped"
assert threads[0].daemon and threads[0].is_alive()
finally:
release.set()
for thread in threads:
thread.join(5)
assert all(not thread.is_alive() for thread in threads)
assert healed == [(2, "reconcile-maintenance")]
assert [name for name, _ in done] == ["review_jobs", "projects", "deletions", "child_refs"]
assert {thread for _, thread in done} == {"reconcile-maintenance"}
assert marker[0] > 0.0 and not sm._RECONCILE_SWEEP_LOCK.locked(), "stamped, then released"
def test_child_ref_promotion_retries_left_the_cancel_sweep(quiet_tick, monkeypatch):
"""The child-ref promotion walk (every child drive, a result load each) rode the
20 s cancel sweep under ITS latch, so the cancel-intent watchdog could not run again
until the walk ended. It now runs AFTER the heal in the reconcile block; a 20 s
sweep that is due runs its three steps and nothing else."""
sm = quiet_tick
done: list = []
_quiet_reconcile_steps(monkeypatch, sm, done)
monkeypatch.setattr("supervisor.task_lifecycle.sweep_cancel_intents", lambda: done.append(("cancel", "")) or {})
monkeypatch.setattr("supervisor.terminal_delivery.replay_pending_deliveries", lambda root: done.append(("delivery", "")))
monkeypatch.setattr(sm, "_reconcile_abandoned_usage", lambda root: done.append(("usage", "")))
monkeypatch.setattr(sm, "_LAST_CANCEL_INTENT_SWEEP", [0.0])
monkeypatch.setattr(sm, "_CANCEL_INTENT_SWEEP_LOCK", threading.Lock())
threads = _track_threads(monkeypatch)
sm._periodic_supervisor_maintenance([time.time()], [time.time()]) # the 20 s sweep alone
for thread in threads:
thread.join(5)
assert [name for name, _ in done] == ["cancel", "delivery", "usage"], "no history-sized step"
assert not sm._CANCEL_INTENT_SWEEP_LOCK.locked()
done.clear()
sm._periodic_supervisor_maintenance([time.time()], [0.0]) # the 300 s block alone
for thread in threads:
thread.join(5)
assert [name for name, _ in done] == ["review_jobs", "orphans", "projects", "deletions", "child_refs"]
assert [thread.name for thread in threads] == ["terminal-maintenance", "reconcile-maintenance"]
def test_a_closed_generation_stops_the_reconcile_block_before_its_next_mutation(quiet_tick, monkeypatch):
"""Like the custody block, the reconcile block outlives the loop that started it,
so it re-reads the per-generation token before EVERY step and stops; an OPEN
generation runs every step, heal before promote; a restart in flight closes it the
same way. The marker is stamped and the latch released whichever way it ends."""
sm = quiet_tick
done: list = []
stop = threading.Event()
def heal_then_close(root, **kwargs):
done.append(("orphans", threading.current_thread().name))
stop.set()
return 0
_quiet_reconcile_steps(monkeypatch, sm, done, heal=heal_then_close)
marker = [0.0]
assert sm._RECONCILE_SWEEP_LOCK.acquire(blocking=False)
sm._run_periodic_reconcile_sweep(marker, stop)
assert [n for n, _ in done] == ["review_jobs", "orphans"], "generation ended mid-pass: nothing further"
assert marker[0] > 0.0 and not sm._RECONCILE_SWEEP_LOCK.locked()
done.clear()
_quiet_reconcile_steps(monkeypatch, sm, done)
assert sm._RECONCILE_SWEEP_LOCK.acquire(blocking=False)
sm._run_periodic_reconcile_sweep(marker, threading.Event())
assert [n for n, _ in done] == ["review_jobs", "orphans", "projects", "deletions", "child_refs"]
done.clear()
closed = threading.Event()
closed.set()
assert sm._RECONCILE_SWEEP_LOCK.acquire(blocking=False)
sm._run_periodic_reconcile_sweep(marker, closed)
assert done == [] and not sm._RECONCILE_SWEEP_LOCK.locked(), "closed at thread start: mutates nothing"
sm._restart_requested.set()
try:
assert sm._RECONCILE_SWEEP_LOCK.acquire(blocking=False)
sm._run_periodic_reconcile_sweep(marker, threading.Event())
finally:
sm._restart_requested.clear()
assert done == [] and not sm._RECONCILE_SWEEP_LOCK.locked(), "a restart in flight closes it too"
def test_a_reconcile_thread_that_cannot_start_releases_its_latch_and_waits_a_cadence(
quiet_tick, monkeypatch, caplog,
):
"""A start refusal is the one failure the tick sees itself: the latch it took opens
again, the marker is stamped (one warning per cadence, not one per 0.5 s tick)."""
import logging
sm = quiet_tick
clock = [5_000.0]
monkeypatch.setattr(sm, "time", SimpleNamespace(time=lambda: clock[0]))
def refuse(**kwargs):
raise RuntimeError("thread unavailable")
monkeypatch.setattr(sm, "threading", SimpleNamespace(Thread=refuse))
marker = [0.0]
with caplog.at_level(logging.WARNING):
sm._periodic_supervisor_maintenance([clock[0]], marker)
assert marker[0] == clock[0]
assert sm._RECONCILE_SWEEP_LOCK.acquire(blocking=False)
sm._RECONCILE_SWEEP_LOCK.release()
assert any("reconcile-maintenance could not start" in r.getMessage() for r in caplog.records)
clock[0] += 10.0
sm._periodic_supervisor_maintenance([clock[0]], marker)
assert marker[0] == clock[0] - 10.0, "not due again until a full cadence has passed"
def test_startup_custody_still_runs_inline_and_starts_no_maintenance_thread(tmp_path, monkeypatch):
"""Startup custody is once-per-generation and synchronous by contract: the loop's
readiness follows it. The off-loop move touched only the tick; the startup sweep
still runs every step on the caller's thread and starts nothing."""
from ouroboros import process_custody as pc
from ouroboros import server_maintenance as sm
order: list = []
monkeypatch.setattr(sm, "DATA_DIR", tmp_path)
monkeypatch.setattr(sm, "_installed_skill_names", lambda: None)
monkeypatch.setattr(pc, "reap_orphaned_processes",
lambda root, **kw: order.append(("reap", threading.current_thread().name)) or [])
monkeypatch.setattr(sm, "_reconcile_delegated_runs",
lambda live, **kw: order.append(("reconcile", threading.current_thread().name)))
monkeypatch.setattr("ouroboros.delegate_terminal.backfill_terminal_reconciliations",
lambda root: order.append(("backfill", "")) or [])
monkeypatch.setattr(sm, "_cursor_refresh_settled_terminals", lambda live=None: order.append(("cursor", "")))
monkeypatch.setattr("supervisor.terminal_delivery.replay_pending_deliveries",
lambda root: order.append(("replay", "")))
monkeypatch.setattr("ouroboros.delegate_state_sweep.sweep_settled_delegate_state",
lambda root: order.append(("delegate_state", "")) or {})
threads = _track_threads(monkeypatch)
sm._startup_custody_sweep()
assert [name for name, _ in order] == ["reap", "reconcile", "backfill", "cursor", "replay", "delegate_state"]
assert {thread for _, thread in order[:2]} == {threading.current_thread().name}
assert threads == [], "startup custody never hands its work to a maintenance thread"
def test_drive_custody_rides_the_reconcile_pass_bounded_and_never_startup(tmp_path, monkeypatch):
"""Child and direct drives are settled by the off-loop reconcile pass through the one
settlement owner, with the supervisor's probe and ownership interlock, at most
DRIVE_SETTLEMENTS_PER_PASS attempts per layout and pass from a memory-only cursor;
the startup sweep copies and hashes no child store (readiness waits on nothing)."""
from ouroboros import headless, server_maintenance as sm
from ouroboros.task_results import load_task_result, write_task_result
monkeypatch.setattr(sm, "DATA_DIR", tmp_path)
monkeypatch.setattr(sm, "_DRIVE_PRUNE_CURSOR", {"headless": "", "direct": ""})
monkeypatch.setenv("OUROBOROS_GC_RETENTION_DAYS", "1")
monkeypatch.setattr("ouroboros.retention.age_cutoff", lambda *a, **k: 4_000_000_000)
monkeypatch.setattr(headless, "DRIVE_SETTLEMENTS_PER_PASS", 2)
monkeypatch.setattr("supervisor.queue.task_settlement_liveness", lambda _task: False)
interlocks = []
import supervisor.queue as queue_mod
real = queue_mod.task_settlement_interlock
def counted(stop=None):
interlocks.append(threading.current_thread().name)
return real(stop=stop)
monkeypatch.setattr(queue_mod, "task_settlement_interlock", counted)
for name in ("d1", "d2", "d3"):
drive = headless.prepare_task_drive(tmp_path, name, "empty")
write_task_result(tmp_path, name, "cancelled", result="x", delegation_role="subagent", child_drive_root=str(drive))
base = tmp_path / "state" / "headless_tasks"
sm._startup_prune_sweeps()
assert sorted(p.name for p in base.iterdir()) == ["d1", "d2", "d3"], "startup settles no drive"
sm._run_drive_custody_pass()
assert sorted(p.name for p in base.iterdir()) == ["d3"] and sm._DRIVE_PRUNE_CURSOR["headless"] == "d2"
assert len(interlocks) == 2
sm._run_drive_custody_pass()
assert list(base.iterdir()) == [] and sm._DRIVE_PRUNE_CURSOR["headless"] == "d3"
assert all(load_task_result(tmp_path, name)["status"] == "cancelled" for name in ("d1", "d2", "d3"))
closed = threading.Event()
closed.set()
drive = headless.prepare_task_drive(tmp_path, "d4", "empty")
write_task_result(tmp_path, "d4", "cancelled", result="x", delegation_role="subagent", child_drive_root=str(drive))
sm._run_drive_custody_pass(closed)
assert drive.is_dir(), "a closed generation settles nothing"
def test_startup_sweeps_only_the_script_fallback_and_owes_the_tree_walk_to_the_first_pass(tmp_path, monkeypatch):
"""The whole-tree walk for orphaned atomic temp files left the startup path (it delayed
readiness by hundreds of thousands of stat calls): startup sweeps only the top-level
tmp_scripts fallback, and the first off-loop reconcile pass of the generation sweeps the
tree once; a deferred startup owes nothing."""
from ouroboros import server_maintenance as sm
monkeypatch.setattr(sm, "DATA_DIR", tmp_path)
monkeypatch.setattr(sm, "_STARTUP_TEMP_SWEEP_OWED", [False])
monkeypatch.setattr(sm, "_periodic_zombie_reconcile", lambda **kwargs: None)
monkeypatch.setattr(sm, "_run_drive_custody_pass", lambda stop_event=None: None)
aged = time.time() - 7200
script = tmp_path / "tmp_scripts" / "script_dead.py"
orphan = tmp_path / "state" / "deep" / ".state.json.tmp.1.2.abc"
for path in (script, orphan):
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text("x", encoding="utf-8")
os.utime(path, (aged, aged))
sm._startup_prune_sweeps()
assert not script.exists() and orphan.exists() and sm._STARTUP_TEMP_SWEEP_OWED == [True]
assert sm._RECONCILE_SWEEP_LOCK.acquire(blocking=False)
sm._run_periodic_reconcile_sweep([0.0], threading.Event())
assert not orphan.exists() and sm._STARTUP_TEMP_SWEEP_OWED == [False]
orphan.write_text("x", encoding="utf-8")
os.utime(orphan, (aged, aged))
assert sm._RECONCILE_SWEEP_LOCK.acquire(blocking=False)
sm._run_periodic_reconcile_sweep([0.0], threading.Event())
assert orphan.exists(), "the walk runs once per generation, not every pass"
monkeypatch.setattr(sm, "_STARTUP_TEMP_SWEEP_OWED", [False])
sm._startup_prune_sweeps(preserve_task_sources=True)
assert sm._STARTUP_TEMP_SWEEP_OWED == [False]

View file

@ -289,6 +289,13 @@ def test_reaper_terminal_cleans_split_drive_not_canonical_mailbox(
task_done_event={"type": "task_done", "task_id": task_id, "status": "failed"},
)
# The loop-thread seam copies nothing: a split drive's mailbox (its acknowledged history
# may carry inputs to promote) waits for the off-loop owner, which releases it; the
# canonical mailbox is never the split task's to clean.
assert _mailbox_path(child_drive, task_id).exists()
from supervisor.terminal_delivery import cleanup_settled_owner_mailbox
cleanup_settled_owner_mailbox(tmp_path, task_id, {}, carry_inputs=True)
assert not _mailbox_path(child_drive, task_id).exists()
assert not _ack_path(child_drive, task_id).exists()
assert _mailbox_path(tmp_path, task_id).exists()

View file

@ -0,0 +1,535 @@
"""TZ-1 V12: a task's recorded files leave through one confined descriptor - the exact nested
file (``?relpath=``), or one recorded directory as a ZIP (``?archive=``) - read from the task's
OWN stores only (canonical, then its own child drive's), with the detail's
``artifact_archives`` projection. A bare name never picks a nested file; forged rows, escaping
symlinks and components swapped after attribution are refused or excluded, never followed; a
captured file serves only the bytes it verified; a platform without directory-relative
no-follow opens answers a typed 503 (owner decision, issue #1297)."""
from __future__ import annotations
import errno
import io
import os
import stat
import threading
import zipfile
from hashlib import sha256
from pathlib import Path
from types import SimpleNamespace
import pytest
from starlette.applications import Starlette
from starlette.routing import Route
from starlette.testclient import TestClient
from ouroboros import artifacts, headless
from ouroboros.gateway import task_archive
from ouroboros.task_custody import task_artifact_stores
from ouroboros.task_results import write_task_result
TASK = "childpub"
URL = f"/api/tasks/{TASK}/artifacts/reports.zip"
SECRET = "outside secret"
confined = pytest.mark.skipif(not task_archive.CONFINED, reason="needs directory-relative no-follow opens")
def _split_child(tmp_path, *, status="completed"):
"""A canonical subagent row plus its own headless child drive holding two nested
deliverables that share one basename."""
data = tmp_path / "data"
child = headless.prepare_task_drive(data, TASK, "empty")
store = artifacts.task_artifact_dir_path(child, TASK, create=True)
for sub, text in (("a", "alpha"), ("b", "beta")):
(store / "reports" / sub).mkdir(parents=True)
(store / "reports" / sub / "summary.txt").write_text(text, encoding="utf-8")
write_task_result(child, TASK, status, result="child done", artifacts=artifacts.collect_task_artifact_records(child, TASK),
artifact_status="ready")
write_task_result(data, TASK, "running", child_drive_root=str(child), delegation_role="subagent",
parent_task_id="parent1", root_task_id="parent1")
(data / "state" / "queue_snapshot.json").write_text('{"pending": [], "running": []}', encoding="utf-8")
return data, child, store
def _client(data):
from ouroboros.gateway.tasks import api_task_artifact, api_task_get
app = Starlette(routes=[
Route("/api/tasks/{task_id}", endpoint=api_task_get, methods=["GET"]),
Route("/api/tasks/{task_id}/artifacts/{name}", endpoint=api_task_artifact, methods=["GET", "HEAD"]),
])
app.state.drive_root = data
return TestClient(app)
def _zip(response):
assert response.status_code == 200, response.text
assert response.headers["content-type"] == "application/zip"
assert int(response.headers["content-length"]) == len(response.content)
return zipfile.ZipFile(io.BytesIO(response.content))
def _link_or_skip(link: Path, target: Path, *, directory: bool = False) -> None:
try:
link.symlink_to(target, target_is_directory=directory)
except (OSError, NotImplementedError) as exc:
pytest.skip(f"symlinks unavailable: {exc}")
def _spy_spools(monkeypatch):
spools, real = [], task_archive.tempfile.TemporaryFile
def spy(*args, **kwargs):
spool = real(*args, **kwargs)
spools.append(spool)
return spool
monkeypatch.setattr(task_archive.tempfile, "TemporaryFile", spy)
return spools
def _swap_after_members(monkeypatch, swap):
"""The attacker's move right after the members were chosen from their stats and before the
first open: the window a path-based open would lose."""
real = task_archive._archive_members
def raced(*args, **kwargs):
members = real(*args, **kwargs)
swap()
return members
monkeypatch.setattr(task_archive, "_archive_members", raced)
@confined
def test_exact_nested_selection_reads_both_own_stores_read_only(tmp_path):
data, child, store = _split_child(tmp_path)
client = _client(data)
url = f"/api/tasks/{TASK}/artifacts/summary.txt"
ambiguous = client.get(url)
assert ambiguous.status_code == 409
assert ambiguous.json() == {
"error": "artifact name matches several nested files; select one with ?relpath=",
"reason_code": "artifact_name_ambiguous", "task_id": TASK, "artifact": "summary.txt",
"relpaths": ["reports/a/summary.txt", "reports/b/summary.txt"]}
exact = client.get(url, params={"relpath": "reports/b/summary.txt"})
assert (exact.status_code, exact.text) == (200, "beta")
(store / "reports/b/summary.txt").unlink()
write_task_result(child, TASK, "completed", artifacts=artifacts.collect_task_artifact_records(child, TASK))
assert client.get(url).status_code == 404 # a sole nested basename is no bare-name route
for bad in ("../a/summary.txt", "reports//summary.txt", "/reports/a/summary.txt", "reports/./summary.txt",
"reports/a/other.txt", "reports\\a\\summary.txt", ""):
refused = client.get(url, params={"relpath": bad})
assert (refused.status_code, refused.json()["reason_code"]) == (400, "artifact_relpath_invalid"), bad
assert client.get(url, params={"relpath": "reports/a/summary.txt", "source": "x"}).status_code == 400
assert client.get(url, params={"relpath": "reports/c/summary.txt"}).status_code == 404
(store / "summary.txt").write_text("top", encoding="utf-8")
write_task_result(child, TASK, "completed", artifacts=artifacts.collect_task_artifact_records(child, TASK))
assert (client.get(url).status_code, client.get(url).text) == (200, "top")
detail = client.get(f"/api/tasks/{TASK}").json()
assert sorted(row.get("relpath", row["name"]) for row in detail["artifacts"]) == [
"reports/a/summary.txt", "summary.txt"]
assert not artifacts.task_artifact_dir_path(data, TASK).exists() # nothing created or copied
headless.copy_child_task_result(data, {"id": TASK, "drive_root": str(child)})
(store / "reports/a/summary.txt").unlink()
after = client.get(url, params={"relpath": "reports/a/summary.txt"})
assert (after.status_code, after.text) == (200, "alpha") # the canonical copy at its relpath
@confined
def test_forged_rows_and_escaping_links_authorize_no_read(tmp_path):
data = tmp_path / "data"
sibling = headless.prepare_task_drive(data, "sibling1", "empty")
planted = artifacts.task_artifact_dir_path(sibling, TASK, create=True) / "secret.txt"
planted.write_text("sibling secret", encoding="utf-8")
write_task_result(data, TASK, "completed", child_drive_root=str(sibling), drive_root=str(sibling),
artifacts=[{"name": "secret.txt", "path": str(planted)}],
metadata={"drive_root": str(sibling), "child_drive_root": str(sibling)})
client = _client(data)
assert task_artifact_stores(data, TASK) == [artifacts.task_artifact_dir_path(data.resolve(), TASK)]
refused = client.get(f"/api/tasks/{TASK}/artifacts/secret.txt")
assert refused.status_code == 500 and "sibling" not in refused.text
assert client.get(f"/api/tasks/{TASK}/artifacts/secret.txt", params={"relpath": "secret.txt"}).status_code == 404
data2, child, store = _split_child(tmp_path / "two")
outside = tmp_path / "outside"
outside.mkdir()
(outside / "secret.txt").write_text(SECRET, encoding="utf-8")
_link_or_skip(store / "link.txt", outside / "secret.txt")
write_task_result(child, TASK, "completed", artifacts=[{"name": "link.txt", "path": str(store / "link.txt")}])
client2 = _client(data2)
escaped = client2.get(f"/api/tasks/{TASK}/artifacts/link.txt") # a child row escaping its store is not listed
assert escaped.status_code == 404 and SECRET not in escaped.text
assert client2.get(f"/api/tasks/{TASK}/artifacts/link.txt", params={"relpath": "link.txt"}).status_code == 404
@confined
def test_an_immutable_download_serves_only_its_verified_bytes(tmp_path, monkeypatch):
data = tmp_path / "data"
source = tmp_path / "report.txt"
source.write_text("captured", encoding="utf-8")
record = artifacts.copy_file_to_task_artifacts(SimpleNamespace(drive_root=data, task_id=TASK), source, immutable=True)
write_task_result(data, TASK, "completed", artifacts=[record])
client = _client(data)
url = f"/api/tasks/{TASK}/artifacts/{record['name']}"
real = artifacts.stream_artifact_file
def swap_after_verify(*args, **kwargs):
measured = real(*args, **kwargs)
Path(record["path"]).write_text("swapped!", encoding="utf-8") # after verification
return measured
monkeypatch.setattr(task_archive.artifact_store, "stream_artifact_file", swap_after_verify)
assert client.get(url).text == "captured"
monkeypatch.undo()
refused = client.get(url) # the pathname now holds other bytes: nothing unverified leaves
assert (refused.status_code, refused.json()["reason_code"]) == (404, "artifact_unverified")
assert "swapped" not in refused.text
@confined
def test_the_descriptor_response_answers_head_and_ranges_from_one_open(tmp_path, monkeypatch):
data, child, store = _split_child(tmp_path)
(store / "blob.bin").write_bytes(b"0123456789")
write_task_result(child, TASK, "completed", artifacts=artifacts.collect_task_artifact_records(child, TASK))
client = _client(data)
url = f"/api/tasks/{TASK}/artifacts/blob.bin"
full = client.get(url)
assert full.content == b"0123456789" and full.headers["accept-ranges"] == "bytes"
assert full.headers["content-length"] == "10" and full.headers["etag"] and full.headers["last-modified"]
head = client.head(url)
assert head.status_code == 200 and head.content == b"" and head.headers["content-length"] == "10"
part = client.get(url, headers={"range": "bytes=2-4"})
assert (part.status_code, part.content, part.headers["content-range"]) == (206, b"234", "bytes 2-4/10")
suffix = client.get(url, headers={"range": "bytes=-3"})
assert (suffix.status_code, suffix.content) == (206, b"789")
assert client.get(url, headers={"range": "bytes=20-30"}).status_code == 416
assert client.get(url, headers={"range": "bytes=0-1,4-5"}).content == b"0123456789" # several: served whole
@confined
def test_chat_media_serves_only_bytes_that_hash_to_its_name(tmp_path):
data = tmp_path / "data"
payload = b"\x89PNG fake image"
name = f"chat-media-{sha256(payload).hexdigest()}.png"
media = artifacts.task_artifact_dir_path(data, TASK, create=True) / "chat_media" / name
media.parent.mkdir(parents=True)
media.write_bytes(payload)
client = _client(data)
assert client.get(f"/api/tasks/{TASK}/artifacts/{name}").content == payload
media.write_bytes(b"tampered")
assert client.get(f"/api/tasks/{TASK}/artifacts/{name}").status_code == 404
def test_without_confined_opens_files_and_archives_fail_closed(tmp_path, monkeypatch):
data, child, _store = _split_child(tmp_path)
client = _client(data)
monkeypatch.setattr(task_archive, "CONFINED", False)
assert client.get(f"/api/tasks/{TASK}").json()["artifact_archives"] == {
"reports": {"name": "reports.zip", "files": 0, "size": 0, "excluded": 2, "available": False}}
refused = client.get(URL, params={"archive": "reports"})
assert (refused.status_code, refused.json()["reason_code"]) == (503, "artifact_archive_unavailable")
plain = client.get(f"/api/tasks/{TASK}/artifacts/summary.txt", params={"relpath": "reports/a/summary.txt"})
assert (plain.status_code, plain.json()["reason_code"]) == (503, "artifact_unavailable")
assert client.get(URL, params={"archive": "../reports"}).status_code == 400 # refusals come first
@confined
def test_directory_archive_keeps_relative_paths_one_member_per_relpath_canonical_first(tmp_path, monkeypatch):
data, child, store = _split_child(tmp_path)
client = _client(data)
assert client.get(f"/api/tasks/{TASK}").json()["artifact_archives"] == {
"reports": {"name": "reports.zip", "files": 2, "size": 9, "excluded": 0, "available": True}}
response = client.get(URL, params={"archive": "reports"})
archive = _zip(response)
assert response.headers["content-disposition"] == 'attachment; filename="reports.zip"'
assert archive.namelist() == ["reports/a/summary.txt", "reports/b/summary.txt"]
assert archive.read("reports/b/summary.txt") == b"beta"
assert _zip(client.get(f"/api/tasks/{TASK}/artifacts/a.zip", params={"archive": "reports/a"})).namelist() == [
"a/summary.txt"]
canonical = artifacts.task_artifact_dir_path(data, TASK, create=True)
(canonical / "reports/a").mkdir(parents=True)
(canonical / "reports/a/summary.txt").write_text("alpha", encoding="utf-8") # a relocated copy: same bytes
opened = []
real_open = task_archive._open_member
monkeypatch.setattr(task_archive, "_open_member", lambda parents, route: opened.append(route) or real_open(parents, route))
merged = _zip(client.get(URL, params={"archive": "reports"}))
assert merged.namelist() == ["reports/a/summary.txt", "reports/b/summary.txt"]
assert merged.read("reports/a/summary.txt") == b"alpha" and opened[0][1][0] == "task_results" # canonical first
(canonical / "reports/a/summary.txt").write_text("alpha canonical", encoding="utf-8") # other bytes than the record
refused = client.get(URL, params={"archive": "reports"})
assert (refused.status_code, refused.json()["reason_code"], refused.json()["member"]) == (
404, "artifact_archive_unverified", "reports/a/summary.txt") # never new bytes under the recorded identity
(canonical / ".github").mkdir()
(canonical / ".github/ci.yml").write_text("name: CI", encoding="utf-8")
dotted = _zip(client.get(f"/api/tasks/{TASK}/artifacts/.github.zip", params={"archive": ".github"}))
assert dotted.namelist() == [".github/ci.yml"]
@confined
def test_directory_archive_excludes_and_counts_rows_it_cannot_serve(tmp_path):
data, child, store = _split_child(tmp_path)
(store / "reports/b/summary.txt").unlink()
outside = tmp_path / "outside"
outside.mkdir()
(outside / "secret.txt").write_text(SECRET, encoding="utf-8")
_link_or_skip(store / "reports/link.txt", outside / "secret.txt")
for name in ("failed.txt", "errored.txt"):
(store / "reports" / name).write_text(name, encoding="utf-8")
def ready(path, **extra):
return {"name": path.name, "path": str(path), "status": "ready", "errors": [], **extra}
rows = [ready(store / "reports/a/summary.txt"), ready(store / "reports/link.txt", relpath="reports/link.txt"),
ready(store / "reports/gone.txt"), {**ready(store / "reports/failed.txt"), "status": "failed"},
{**ready(store / "reports/errored.txt"), "errors": ["copy failed"]}]
write_task_result(child, TASK, "completed", artifacts=rows)
client = _client(data)
# The escaping link row never enters the view; gone, failed and errored rows are counted out.
assert client.get(f"/api/tasks/{TASK}").json()["artifact_archives"] == {
"reports": {"name": "reports.zip", "files": 1, "size": 5, "excluded": 3, "available": True}}
archive = _zip(client.get(URL, params={"archive": "reports"}))
assert archive.namelist() == ["reports/a/summary.txt"]
@confined
def test_directory_archive_verifies_captures_and_types_every_refusal(tmp_path, monkeypatch):
data, child, store = _split_child(tmp_path)
rows = [artifacts.artifact_record(store / "reports/a/summary.txt"),
{**artifacts.artifact_record(store / "reports/b/summary.txt"), "immutable": True}]
write_task_result(child, TASK, "completed", artifacts=rows)
client = _client(data)
spools = _spy_spools(monkeypatch)
(store / "reports/a/summary.txt").write_text("alpha v2", encoding="utf-8") # a mutable row, not re-recorded
stale = client.get(URL, params={"archive": "reports"})
assert (stale.status_code, stale.json()["member"]) == (404, "reports/a/summary.txt")
rows[0] = artifacts.artifact_record(store / "reports/a/summary.txt") # re-recorded: the new identity serves
write_task_result(child, TASK, "completed", artifacts=rows)
assert _zip(client.get(URL, params={"archive": "reports"})).read("reports/a/summary.txt") == b"alpha v2"
(store / "reports/b/summary.txt").write_text("beta v2", encoding="utf-8")
refused = client.get(URL, params={"archive": "reports"})
assert refused.json() == {"error": "archive member is missing, changed while read, or failed its capture verification",
"reason_code": "artifact_archive_unverified", "task_id": TASK,
"artifact": "reports.zip", "directory": "reports", "member": "reports/b/summary.txt"}
assert len(spools) == 3 and all(spool.closed for spool in spools)
for name, params in (("reports.zip", {"archive": "reports", "relpath": "reports.zip"}),
("reports.zip", {"archive": "../reports"}), ("reports.zip", {"archive": ""}),
("other.zip", {"archive": "reports"}), ("reports", {"archive": "reports"})):
bad = client.get(f"/api/tasks/{TASK}/artifacts/{name}", params=params)
assert (bad.status_code, bad.json()["reason_code"]) == (400, "artifact_archive_invalid"), params
def no_spool(*_args, **_kwargs):
raise OSError(errno.EMFILE, "too many open files")
monkeypatch.setattr(task_archive.tempfile, "TemporaryFile", no_spool)
failed = client.get(URL, params={"archive": "reports"})
assert (failed.status_code, failed.json()["reason_code"]) == (503, "artifact_archive_unavailable")
@confined
def test_a_directory_swapped_for_a_symlink_after_the_stat_is_refused_never_followed(tmp_path, monkeypatch):
data, child, store = _split_child(tmp_path)
client = _client(data)
outside = tmp_path / "outside" / "a"
outside.mkdir(parents=True)
(outside / "summary.txt").write_text(SECRET, encoding="utf-8")
def swap():
(store / "reports/a").rename(tmp_path / "moved-a")
_link_or_skip(store / "reports/a", outside, directory=True)
_swap_after_members(monkeypatch, swap)
refused = client.get(URL, params={"archive": "reports"})
assert (refused.status_code, refused.json()["reason_code"]) == (404, "artifact_archive_unverified")
assert SECRET not in refused.text
@pytest.mark.skipif(not task_archive.CONFINED or not hasattr(os, "mkfifo"), reason="confined opens and FIFOs required")
def test_a_member_swapped_for_a_fifo_is_refused_without_blocking(tmp_path, monkeypatch):
data, child, store = _split_child(tmp_path)
client = _client(data)
target = store / "reports/a/summary.txt"
def swap():
target.unlink()
os.mkfifo(target)
_swap_after_members(monkeypatch, swap)
answer = []
worker = threading.Thread(target=lambda: answer.append(client.get(URL, params={"archive": "reports"})), daemon=True)
worker.start()
worker.join(timeout=15)
assert not worker.is_alive() and answer, "opening the FIFO blocked the archive build"
assert (answer[0].status_code, answer[0].json()["reason_code"]) == (404, "artifact_archive_unverified")
assert stat.S_ISFIFO(os.lstat(target).st_mode)
@pytest.mark.serial
@confined
def test_real_http_consumer_downloads_a_nested_file_and_a_directory_zip_off_the_loop(tmp_path):
"""The real server path (uvicorn + the HTTP client) for the two V12 addresses the UI builds."""
import asyncio
import socket
import time
import httpx
import uvicorn
data, child, _store = _split_child(tmp_path)
app = _client(data).app
sock = socket.socket()
sock.bind(("127.0.0.1", 0))
server = uvicorn.Server(uvicorn.Config(app, log_level="warning"))
thread = threading.Thread(target=server.run, kwargs={"sockets": [sock]}, daemon=True)
thread.start()
try:
end = time.monotonic() + 10
while not server.started and thread.is_alive() and time.monotonic() < end:
time.sleep(0.01)
assert server.started
base = f"http://127.0.0.1:{sock.getsockname()[1]}/api/tasks/{TASK}"
async def fetch():
async with httpx.AsyncClient(timeout=10) as client:
nested = await client.get(f"{base}/artifacts/summary.txt", params={"relpath": "reports/a/summary.txt"})
archive = await client.get(f"{base}/artifacts/reports.zip", params={"archive": "reports"})
return nested, archive
nested, archive = asyncio.run(fetch())
assert (nested.status_code, nested.text) == (200, "alpha")
assert zipfile.ZipFile(io.BytesIO(archive.content)).namelist() == ["reports/a/summary.txt",
"reports/b/summary.txt"]
finally:
server.should_exit = True
thread.join(10)
sock.close()
@confined
@pytest.mark.parametrize("immutable", [False, True])
def test_an_empty_file_still_completes_the_response(tmp_path, immutable):
"""F8: a zero-byte file (mutable, or a verified capture through the spool) answers GET
with a complete empty body, HEAD with its length, and every range as unsatisfiable."""
data, child, store = _split_child(tmp_path)
(store / "empty.bin").write_bytes(b"")
row = artifacts.artifact_record(store / "empty.bin")
write_task_result(child, TASK, "completed", artifacts=[{**row, "immutable": True} if immutable else row])
client = _client(data)
url = f"/api/tasks/{TASK}/artifacts/empty.bin"
full = client.get(url)
assert (full.status_code, full.content, full.headers["content-length"]) == (200, b"", "0")
head = client.head(url)
assert (head.status_code, head.content, head.headers["content-length"]) == (200, b"", "0")
assert client.get(url, headers={"range": "bytes=0-0"}).status_code == 416
assert client.get(url, headers={"range": "bytes=-1"}).status_code == 416
@confined
def test_a_failed_spool_allocation_closes_the_member_it_opened(tmp_path, monkeypatch):
"""A verified download opens the member first; when the private spool cannot be
allocated the open descriptor is closed with the typed refusal, never leaked."""
data, child, store = _split_child(tmp_path)
write_task_result(child, TASK, "completed",
artifacts=[{**artifacts.artifact_record(store / "reports/a/summary.txt"), "immutable": True}])
client = _client(data)
handles = []
real_open = task_archive._open_member
def spy_open(*args, **kwargs):
handle, observed = real_open(*args, **kwargs)
handles.append(handle)
return handle, observed
def no_spool(*_args, **_kwargs):
raise OSError(errno.EMFILE, "too many open files")
monkeypatch.setattr(task_archive, "_open_member", spy_open)
monkeypatch.setattr(task_archive.tempfile, "TemporaryFile", no_spool)
refused = client.get(f"/api/tasks/{TASK}/artifacts/summary.txt", params={"relpath": "reports/a/summary.txt"})
assert (refused.status_code, refused.json()["reason_code"]) == (503, "artifact_unavailable")
assert len(handles) == 1 and handles[0].closed
@confined
def test_a_store_reached_through_a_link_serves_and_lists_nothing(tmp_path):
"""Path safety: a store or a parent component swapped for a symlink belongs to no
store: the view lists nothing through it and no byte leaves through it."""
data, child, store = _split_child(tmp_path)
outside = tmp_path / "outside"
outside.mkdir()
(outside / "summary.txt").write_text(SECRET, encoding="utf-8")
(store / "reports/a").rename(tmp_path / "moved-a")
_link_or_skip(store / "reports/a", outside, directory=True)
client = _client(data)
detail = client.get(f"/api/tasks/{TASK}").json()
assert "reports/a/summary.txt" not in {row.get("relpath") for row in detail["artifacts"]}
refused = client.get(f"/api/tasks/{TASK}/artifacts/summary.txt", params={"relpath": "reports/a/summary.txt"})
assert refused.status_code == 404 and SECRET not in refused.text
linked_store = data / "task_results" / "artifacts"
linked_store.mkdir(parents=True, exist_ok=True)
_link_or_skip(linked_store / TASK, outside, directory=True)
refused = client.get(f"/api/tasks/{TASK}/artifacts/summary.txt")
assert refused.status_code in {404, 409} and SECRET not in refused.text
@confined
def test_a_measured_files_changed_bytes_are_refused_with_the_recorded_digest_and_unmeasured_bytes_are_labelled(tmp_path):
"""A row that records a digest (immutable or not) is served only when its bytes still match
it: a same-length replacement answers 409 ``artifact_identity_changed`` naming the recorded
digest, never the new bytes under the old identity. A listing without a digest streams its
current bytes and says so (``x-ouroboros-artifact-identity: unmeasured``). A directory ZIP
applies the same rule to each member."""
data, child, store = _split_child(tmp_path)
client = _client(data)
url = f"/api/tasks/{TASK}/artifacts/summary.txt"
(store / "reports/a/summary.txt").write_text("ALPHA", encoding="utf-8") # same length as "alpha"
refused = client.get(url, params={"relpath": "reports/a/summary.txt"})
assert refused.status_code == 409, refused.text
body = refused.json()
assert body["reason_code"] == "artifact_identity_changed" and body["task_id"] == TASK
assert body["recorded_sha256"] == sha256(b"alpha").hexdigest() and body["recorded_size"] == 5
assert "ALPHA" not in refused.text
served = client.get(url, params={"relpath": "reports/b/summary.txt"})
assert served.text == "beta" and served.headers["x-ouroboros-artifact-identity"] == "verified"
assert served.headers["x-ouroboros-artifact-sha256"] == sha256(b"beta").hexdigest()
(store / "reports" / "c").mkdir()
(store / "reports/c/summary.txt").write_text("gamma", encoding="utf-8") # nobody recorded it
current = client.get(url, params={"relpath": "reports/c/summary.txt"})
assert current.text == "gamma" and current.headers["x-ouroboros-artifact-identity"] == "unmeasured"
assert "x-ouroboros-artifact-sha256" not in current.headers
archive = client.get(URL, params={"archive": "reports"})
assert (archive.status_code, archive.json()["reason_code"]) == (404, "artifact_archive_unverified")
assert archive.json()["member"] == "reports/a/summary.txt"
@confined
def test_a_stale_accepted_disposition_cannot_certify_bytes_the_store_no_longer_serves(tmp_path):
"""The parent's disposition hash covers recorded identities and a read stays pure (no hash
on read), so an unrecorded byte change leaves it accepted; the store makes that honest by
refusing to serve the changed bytes under the recorded identity, so the disposition never
certifies bytes a consumer can obtain."""
from ouroboros.task_status import load_effective_task_result
from ouroboros.tools.join_ledger import _child_result_sha256, _current_child_result_disposition
from ouroboros.tools.task_tree import _tree_note
data = tmp_path / "data"
artifact_dir = artifacts.task_artifact_dir_path(data, TASK, create=True)
report = artifact_dir / "report.md"
report.write_text("version one\n", encoding="utf-8")
write_task_result(data, TASK, "completed", parent_task_id="parent1", root_task_id="parent1",
delegation_role="subagent", result="artifact-backed", artifacts=[artifacts.artifact_record(report)])
parent = SimpleNamespace(drive_root=str(data), budget_drive_root=str(data), task_id="parent1", role="orchestrator",
task_metadata={"budget_drive_root": str(data), "root_task_id": "parent1"})
shown = _child_result_sha256(load_effective_task_result(data, TASK))
payload = {"type": "child_result_disposition", "child_task_id": TASK, "disposition": "integrated",
"child_result_sha256": shown}
assert _tree_note(parent, "decision", "integrated the report", payload=payload).startswith("OK:")
report.write_text("version two\n", encoding="utf-8") # same length, no re-record
view = load_effective_task_result(data, TASK)
assert _child_result_sha256(view) == shown and _current_child_result_disposition(view) == "integrated"
refused = _client(data).get(f"/api/tasks/{TASK}/artifacts/report.md")
assert refused.status_code == 409 and refused.json()["reason_code"] == "artifact_identity_changed"
assert refused.json()["recorded_sha256"] == sha256(b"version one\n").hexdigest()
assert "version two" not in refused.text

View file

@ -780,13 +780,14 @@ def test_materializing_child_read_cannot_overwrite_canonical_zero_run_receipt(tm
"ts": "2026-01-01T00:00:02+00:00",
})
# Repeated polling must preserve the canonical-only row. Final copy-back
# then unions the ordinary child check into that same authority file.
# Repeated polling must preserve the canonical-only row, and a read writes no
# receipt (TZ-1 A: reads are pure). Final copy-back then unions the ordinary
# child check into that same authority file.
effective_task_result(tmp_path, load_task_result(tmp_path, tid) or {})
assert [
row.get("contract_kind")
for row in read_verification_receipts(tmp_path, tid)
] == [None, "delegation_zero_run"]
] == ["delegation_zero_run"]
copied = copy_child_task_result(
tmp_path, {"id": tid, "drive_root": str(child_drive)},

View file

@ -54,7 +54,7 @@ def test_clean_source_in_model_request_survives_child_copyback_and_pruning(tmp_p
promoted = read_blob_ref(parent, manifest["full_payload_ref"])["messages"][0]["content"]
assert warning in promoted
assert _marker_ref(promoted, "PRODUCER_RESULT_SOURCE_JSON=") == source
prune_headless_task_drives(parent, retention_days=0, now=4_000_000_000.0)
prune_headless_task_drives(parent, retention_days=0, now=4_000_000_000.0, live=lambda _task: False)
assert not child.exists()
assert read_actor_source_bytes(parent, task_id, source) == payload.encode("utf-8")
canonical_ctx = ToolContext(repo_dir=repo, drive_root=parent, task_id=task_id)

View file

@ -62,3 +62,62 @@ def test_failed_supervisor_init_never_paints_online(subscription_ui):
assert page.get_by_text("Online", exact=True).count() == 0
capture(page, "tz1-failed-init-starting")
assert not ui["errors"], ui["errors"]
def test_recorded_folder_download_from_real_gateway_in_chat(subscription_ui, tmp_path):
"""The served SPA reads real task detail and ZIP bytes from the gateway,
rather than an authored HTML fragment or a fake detail response."""
import io
import zipfile
from tests.test_task_file_serving import TASK, _client, _split_child
from ouroboros import headless
data, child, _store = _split_child(tmp_path)
headless.copy_child_task_result(data, {"id": TASK, "drive_root": str(child)})
client = _client(data)
subscription_ui["backend"]["task_gateway"] = client
page = subscription_ui["page"]
url_prefix = f"/api/tasks/{TASK}"
seen = []
def gateway(route):
from urllib.parse import urlparse
path = urlparse(route.request.url).path
suffix = route.request.url.split(path, 1)[1]
response = client.get(path + suffix)
seen.append((path, response.status_code))
route.fulfill(status=response.status_code,
headers={"content-type": response.headers.get("content-type", "application/json"),
"content-disposition": response.headers.get("content-disposition", "")},
body=response.content)
page.route(f"**{url_prefix}*", gateway)
page.goto(subscription_ui["url"])
# The event is only an activation fixture. The browser's task-detail read,
# archive availability and downloaded bytes all come from the real gateway.
page.evaluate("row => window.__ouroWs.emit('chat', row)", {
"chat_id": 1, "task_id": TASK, "role": "assistant", "is_progress": True,
"content": "Building reports", "ts": "2026-09-25T07:00:00Z",
})
page.evaluate("row => window.__ouroWs.emit('chat', row)", {
"chat_id": 1, "task_id": TASK, "role": "system", "system_type": "task_summary",
"task_terminal_status": "completed", "content": "Done", "ts": "2026-09-25T07:01:00Z",
})
card = page.locator(f'#chat-messages .chat-live-card[data-task-id="{TASK}"]')
card.wait_for()
assert card.locator('[data-live-phase]').inner_text() == "Done"
card.locator('[data-live-summary-button]').click()
page.wait_for_selector(f'#chat-messages .chat-live-card[data-task-id="{TASK}"] [data-result-files]')
folder = card.locator('[data-result-files] a[href*="?archive=reports"]')
assert folder.count() == 1 and "reports/" in folder.inner_text()
with page.expect_download() as item:
folder.click()
assert item.value.failure() is None, (item.value.failure(), seen, subscription_ui["errors"])
payload = item.value.path().read_bytes()
with zipfile.ZipFile(io.BytesIO(payload)) as archive:
assert archive.namelist() == ["reports/a/summary.txt", "reports/b/summary.txt"]
assert archive.read("reports/a/summary.txt") == b"alpha"
assert (url_prefix, 200) in seen, seen # real detail, not a fixture JSON object
capture(page, "tz1-real-folder-download")
assert not subscription_ui["errors"], subscription_ui["errors"]

View file

@ -0,0 +1,437 @@
"""TZ-1 V10 and A: mail written to a task that its model never read is kept by the
accepted terminal write itself (no ACK, no second writer per terminal path), custody only
grows across both canonical/replica seams, forward_to_worker reaches a queued task with an
honest receipt, and every effective read stays pure."""
from __future__ import annotations
import json
import threading
from pathlib import Path
from types import SimpleNamespace
import pytest
from ouroboros import artifacts, headless, owner_mailbox, task_custody
from ouroboros.task_results import load_task_result, write_task_result
TASK = "mailtask"
def _msg_ids(value) -> list:
return [json.loads(row)["msg_id"] for row in (value or {}).get("rows") or []]
@pytest.mark.parametrize("terminal", ["cancelled", "failed", "completed"])
def test_an_accepted_terminal_transition_keeps_every_unread_row_without_acknowledging(tmp_path, terminal):
data = tmp_path / "data"
drive = headless.prepare_task_drive(data, TASK, "forked")
write_task_result(data, TASK, "scheduled", child_drive_root=str(drive))
owner_mailbox.write_owner_message(drive, "owner words", TASK, msg_id="o1")
owner_mailbox.write_task_message(drive, "x" * 7000, TASK, source_task_id="parent1", msg_id="t1")
owner_mailbox.write_owner_message(drive, "hurry", TASK, msg_id="h1", kind=owner_mailbox.KIND_HURRY)
owner_mailbox.write_owner_message(data, "canonical words", TASK, msg_id="c1")
owner_mailbox.acknowledge_task_messages(drive, TASK, ["o1"], wake_id="test") # o1 was read
stored = write_task_result(data, TASK, terminal, result="ended")
custody = stored["unread_mailbox"]
assert sorted(_msg_ids(custody)) == ["c1", "t1"] and custody["read_complete"] is True
assert json.loads(next(row for row in custody["rows"] if '"t1"' in row))["text"] == "x" * 7000 # full bytes
assert owner_mailbox.acknowledged_task_message_ids(drive, TASK) == {"o1"} # capture never ACKs
# A rejected regression and a same-status enrichment fabricate no new transition.
owner_mailbox.write_task_message(drive, "too late", TASK, source_task_id="parent1", msg_id="t2")
write_task_result(data, TASK, "running", result="stale mirror")
write_task_result(data, TASK, terminal, cost_note="enrichment")
assert sorted(_msg_ids(load_task_result(data, TASK)["unread_mailbox"])) == ["c1", "t1"]
def test_get_task_result_shows_unread_mail_and_the_exact_rows_ride_the_authority(tmp_path):
from ouroboros.tools.control_task_results import _get_task_result
data = tmp_path / "data"
owner_mailbox.write_task_message(data, "please also check the logs", TASK, source_task_id="parent1", msg_id="t1")
write_task_result(data, TASK, "cancelled", result="Cancelled before start.", parent_task_id="parent1",
root_task_id="parent1", delegation_role="subagent")
ctx = SimpleNamespace(drive_root=data, task_id="parent1", task_metadata={})
text = _get_task_result(ctx, TASK)
assert "[UNREAD_MAILBOX] 1 message(s)" in text and "please also check the logs" in text
authority = json.loads(_get_task_result(ctx, TASK, include_authority=True))["authority"]
assert _msg_ids(authority["unread_mailbox"]) == ["t1"]
def test_a_torn_mailbox_read_is_disclosed_and_never_proof_of_emptiness(tmp_path):
data = tmp_path / "data"
mailbox = owner_mailbox._mailbox_path(data, TASK)
mailbox.parent.mkdir(parents=True)
mailbox.write_text('{"msg_id": "a", "kind": "owner_text", "text": "whole"}\n{"msg_id": "b", "te', encoding="utf-8")
custody = write_task_result(data, TASK, "failed", result="x")["unread_mailbox"]
assert custody["read_complete"] is False
assert not owner_mailbox.cleanup_task_mailbox(data, TASK) and mailbox.is_file()
def test_exact_rows_keep_unicode_line_separators(tmp_path):
data = tmp_path / "data"
text = "first
second
third"
owner_mailbox.write_owner_message(data, text, TASK, msg_id="u1")
custody = write_task_result(data, TASK, "cancelled", result="x")["unread_mailbox"]
assert custody["read_complete"] is True and json.loads(custody["rows"][0])["text"] == text
def test_unread_custody_is_a_union_at_both_replica_seams(tmp_path):
from ouroboros.post_task_checkpoint import project_replica_task_result_fields
from ouroboros.task_status import load_effective_task_result
data = tmp_path / "data"
drive = headless.prepare_task_drive(data, TASK, "empty")
owner_mailbox.write_owner_message(drive, "one", TASK, msg_id="m1")
write_task_result(drive, TASK, "completed", result="child answer") # the child's own capture
write_task_result(data, TASK, "running", child_drive_root=str(drive))
owner_mailbox.write_owner_message(data, "two", TASK, msg_id="m2")
canonical = write_task_result(data, TASK, "failed", result="orphaned") # canonical capture: m1 + m2
assert sorted(_msg_ids(canonical["unread_mailbox"])) == ["m1", "m2"]
stale = {"status": "completed", "unread_mailbox": {"rows": [], "read_complete": True}}
# Effective-read seam and copy-back seam: a stale replica never drops a held row.
assert sorted(_msg_ids(project_replica_task_result_fields(canonical, stale)["unread_mailbox"])) == ["m1", "m2"]
assert sorted(_msg_ids(load_effective_task_result(data, TASK)["unread_mailbox"])) == ["m1", "m2"]
copied = headless.copy_child_task_result(data, {"id": TASK, "drive_root": str(drive)})
assert sorted(_msg_ids(copied["unread_mailbox"])) == ["m1", "m2"]
write_task_result(data, TASK, "failed", unread_mailbox={"rows": [], "read_complete": True})
assert sorted(_msg_ids(load_task_result(data, TASK)["unread_mailbox"])) == ["m1", "m2"]
def test_forward_to_a_queued_task_is_read_when_it_starts_and_kept_if_it_never_does(tmp_path):
import queue as queue_mod
from ouroboros.loop_round_limits import _drain_incoming_messages
from ouroboros.tools.core import _forward_to_worker
data = tmp_path / "data"
drive = headless.prepare_task_drive(data, TASK, "forked")
write_task_result(data, TASK, "scheduled", child_drive_root=str(drive), parent_task_id="parent1",
root_task_id="parent1", delegation_role="subagent")
ctx = SimpleNamespace(drive_root=data, task_id="parent1", task_metadata={})
receipt = _forward_to_worker(ctx, TASK, "start with the logs")
assert "(queued)" in receipt and "has not started, so nothing has read it" in receipt
rows, complete = task_custody.unread_mail_rows(drive, TASK)
assert complete and json.loads(rows[0])["text"] == "start with the logs"
msg_id = json.loads(rows[0])["msg_id"]
assert owner_mailbox.mail_read_state(drive, TASK, msg_id) is False
# Its first round reads (and acknowledges) it like any running delivery.
messages: list = []
_drain_incoming_messages(messages, queue_mod.Queue(), drive, TASK, None, set())
assert "start with the logs" in json.dumps(messages) and owner_mailbox.mail_read_state(drive, TASK, msg_id) is True
def test_mail_write_receipt_vocabulary():
assert owner_mailbox.mail_write_receipt("scheduled")["receipt"] == owner_mailbox.MAIL_QUEUED
assert owner_mailbox.mail_write_receipt("running")["receipt"] == owner_mailbox.MAIL_DELIVERED
assert owner_mailbox.mail_write_receipt("running", drain_ended=True)["receipt"] == owner_mailbox.MAIL_RETAINED_UNREAD
assert all(owner_mailbox.mail_write_receipt(status)["read"] is False for status in ("scheduled", "running"))
def test_canonical_only_late_mail_after_cleanup_is_held_by_the_next_sweep(tmp_path):
data = tmp_path / "data"
write_task_result(data, TASK, "completed", result="done",
child_ref_promotion={"schema_version": 1, "status": "complete", "pending_refs": []},
root_phase_checkpoint={"post_task_synthesis": "completed"})
assert owner_mailbox.cleanup_task_mailbox(data, TASK) # nothing there
owner_mailbox.write_owner_message(data, "after the old settlement", TASK, msg_id="late")
report = owner_mailbox.sweep_settled_owner_mailboxes(data)
assert report["removed"] == [TASK]
assert _msg_ids(load_task_result(data, TASK)["unread_mailbox"]) == ["late"]
def _tree_snapshot(root: Path) -> dict:
return {str(path.relative_to(root)): (path.stat().st_size, path.stat().st_mtime_ns)
for path in sorted(root.rglob("*")) if path.is_file()}
def test_every_observation_surface_is_pure(tmp_path, monkeypatch):
"""No read copies, hashes or registers a file, or writes any byte (the fail-soft
quarantine of an inadmissible row stays the one owner-accepted exception)."""
from starlette.applications import Starlette
from starlette.routing import Route
from starlette.testclient import TestClient
from ouroboros.gateway.tasks import api_task_get, api_tasks_list
from ouroboros.task_status import find_child_tasks, load_effective_task_result, wait_for_effective_tasks
from ouroboros.tools.control_task_results import _get_task_result
data = tmp_path / "data"
drive = headless.prepare_task_drive(data, TASK, "empty")
store = artifacts.task_artifact_dir_path(drive, TASK, create=True)
(store / "reports").mkdir()
(store / "reports" / "summary.txt").write_text("nested", encoding="utf-8")
(store / "top.txt").write_text("top", encoding="utf-8")
write_task_result(drive, TASK, "completed", result="child done",
artifacts=artifacts.collect_task_artifact_records(drive, TASK))
write_task_result(data, TASK, "running", child_drive_root=str(drive), parent_task_id="parent1",
root_task_id="parent1", delegation_role="subagent")
(data / "state" / "queue_snapshot.json").write_text('{"pending": [], "running": []}', encoding="utf-8")
before = _tree_snapshot(data)
def forbidden(*_args, **_kwargs):
raise AssertionError("an observation surface touched artifact bytes or registrations")
for name in ("stream_artifact_file", "copy_artifact_file", "copy_file_to_task_artifacts",
"_register_task_artifact_records", "store_actor_source_bytes"):
monkeypatch.setattr(artifacts, name, forbidden)
app = Starlette(routes=[Route("/api/tasks", endpoint=api_tasks_list, methods=["GET"]),
Route("/api/tasks/{task_id}", endpoint=api_task_get, methods=["GET"])])
app.state.drive_root = data
client = TestClient(app)
ctx = SimpleNamespace(drive_root=data, task_id="parent1", task_metadata={})
view = load_effective_task_result(data, TASK)
assert sorted(row.get("relpath") or row["name"] for row in view["artifacts"]) == ["reports/summary.txt", "top.txt"]
assert all(row.get("measured") is False or row.get("sha256") for row in view["artifacts"])
load_effective_task_result(data, TASK, materialize_artifacts=False)
find_child_tasks(data, parent_task_id="parent1")
wait_for_effective_tasks(data, [TASK], timeout_sec=0)
assert client.get(f"/api/tasks/{TASK}").status_code == 200
assert client.get("/api/tasks").status_code == 200
_get_task_result(ctx, TASK)
assert _tree_snapshot(data) == before
assert not artifacts.task_artifact_dir_path(data, TASK).exists()
def test_the_terminal_capture_reads_the_mailbox_before_the_row_lock(tmp_path, monkeypatch):
"""V10: the mailbox bytes are read outside the task-result row lock; only the bounded
union runs under it (no file read inside a 4 s lock hold)."""
from ouroboros import platform_layer, task_results
data = tmp_path / "data"
owner_mailbox.write_owner_message(data, "unread words", TASK, msg_id="u1")
events = []
real_capture, real_acquire = task_custody.capture_unread_mail, platform_layer.acquire_exclusive_file_lock
def capture(*args, **kwargs):
events.append("capture")
return real_capture(*args, **kwargs)
def acquire(path, **kwargs):
if str(path).endswith(f"{TASK}.json.lock"):
events.append("row_lock")
return real_acquire(path, **kwargs)
monkeypatch.setattr(task_results, "capture_unread_mail", capture, raising=False)
monkeypatch.setattr(task_custody, "capture_unread_mail", capture)
monkeypatch.setattr(platform_layer, "acquire_exclusive_file_lock", acquire)
stored = write_task_result(data, TASK, "cancelled", result="x")
assert _msg_ids(stored["unread_mailbox"]) == ["u1"]
assert events == ["capture", "row_lock"], events
def test_forward_receipt_says_retained_unread_once_the_recipient_drain_ended(tmp_path, monkeypatch):
"""A message written after the recipient's own drain ended (TZ-2's ``mailbox_drain_ended``
fact, read at the recipient's drive) is never promised to a next checkpoint: the receipt
names the unread retention honestly; the row stays for the settlement to keep."""
from ouroboros.tools.core import _forward_to_worker
data = tmp_path / "data"
drive = headless.prepare_task_drive(data, TASK, "forked")
write_task_result(data, TASK, "running", child_drive_root=str(drive), parent_task_id="parent1",
root_task_id="parent1", delegation_role="subagent")
ctx = SimpleNamespace(drive_root=data, task_id="parent1", task_metadata={})
asked = []
monkeypatch.setattr(owner_mailbox, "mailbox_drain_ended", lambda root, tid: asked.append((root, tid)) or True)
receipt = _forward_to_worker(ctx, TASK, "too late for a checkpoint")
assert asked == [(drive, TASK)] # the recipient's own drive, never the sender's
assert f"({owner_mailbox.MAIL_RETAINED_UNREAD})" in receipt and "next checkpoint" not in receipt
rows, complete = task_custody.unread_mail_rows(drive, TASK)
assert complete and json.loads(rows[0])["text"] == "too late for a checkpoint"
def test_the_seam_keeps_a_canonical_mailbox_with_uncarried_inputs_and_the_off_loop_sweep_carries_them(tmp_path):
"""The task-done seam runs on the loop thread and hashes nothing: a settled task's mailbox
whose unread rows carry attachments waits for the off-loop sweep, which verifies and records
the canonical closure per exact row (inline and >25-row manifest alike) before unlinking."""
from ouroboros import artifacts
data = tmp_path / "data"
write_task_result(data, TASK, "completed", result="done", root_phase_checkpoint={"post_task_synthesis": "completed"})
sources = []
for index in range(30):
source = tmp_path / f"input-{index}.txt"
source.write_text(f"input {index}", encoding="utf-8")
sources.append(str(source))
manifest = artifacts.stage_task_attachments(data, TASK, sources)
assert owner_mailbox.write_owner_message(data, "see the files", TASK, msg_id="late", attachment_manifest=manifest)
row = task_custody.unread_mail_rows(data, TASK)[0][0]
assert "attachment_manifest_ref" in json.loads(row)
assert owner_mailbox.cleanup_task_mailbox(data, TASK, carry_inputs=False) is False
assert owner_mailbox._mailbox_path(data, TASK).is_file()
assert owner_mailbox.sweep_settled_owner_mailboxes(data)["removed"] == [TASK]
custody = load_task_result(data, TASK)["unread_mailbox"]
assert custody["rows"] == [row]
resolved = artifacts.resolve_attachment_manifest(data, TASK, custody["inputs"][task_custody._row_key(row)])
assert len(resolved) == 30
for item in resolved:
artifacts.stream_artifact_file(Path(item["abs_path"]), expected=item)
def test_the_drive_custody_pass_sweeps_the_canonical_mailboxes_the_seam_left(tmp_path, monkeypatch):
"""The off-loop drive-custody pass owns the mailbox sweep too, so a canonical mailbox the
loop thread kept (unread inputs to carry) is carried and unlinked within a cadence."""
from ouroboros import artifacts
from ouroboros import server_maintenance as sm
data = tmp_path / "data"
monkeypatch.setattr(sm, "DATA_DIR", data)
monkeypatch.setattr(sm, "_DRIVE_PRUNE_CURSOR", {"headless": "", "direct": ""})
write_task_result(data, TASK, "completed", result="done", root_phase_checkpoint={"post_task_synthesis": "completed"})
source = tmp_path / "input.txt"
source.write_text("input", encoding="utf-8")
manifest = artifacts.stage_task_attachments(data, TASK, [str(source)])
assert owner_mailbox.write_owner_message(data, "see the file", TASK, msg_id="late", attachment_manifest=manifest)
assert owner_mailbox.cleanup_task_mailbox(data, TASK, carry_inputs=False) is False
sm._run_drive_custody_pass()
assert not owner_mailbox._mailbox_path(data, TASK).exists()
custody = load_task_result(data, TASK)["unread_mailbox"]
assert _msg_ids(custody) == ["late"] and len(custody["inputs"]) == 1
closed = threading.Event()
closed.set()
assert owner_mailbox.write_owner_message(data, "again", TASK, msg_id="again", attachment_manifest=manifest)
sm._run_drive_custody_pass(closed)
assert owner_mailbox._mailbox_path(data, TASK).exists(), "a closed generation unlinks nothing"
def _staged(tmp_path: Path, drive: Path, count: int, prefix: str) -> list:
sources = []
for index in range(count):
source = tmp_path / f"{prefix}-{index}.txt"
source.write_text(f"{prefix} input {index}", encoding="utf-8")
sources.append(str(source))
manifest = artifacts.stage_task_attachments(drive, TASK, sources)
assert len(manifest) == count and all(row["status"] == "staged" for row in manifest)
return manifest
@pytest.mark.parametrize("count", [2, 30])
def test_an_acknowledged_owner_row_with_line_separators_keeps_its_inputs_through_drive_gc(tmp_path, count):
"""R1: every mailbox reader splits rows on "\\n" only (``owner_mailbox.mailbox_lines``). Owner
text holding a literal U+2028/U+2029 that was delivered and acknowledged is no unread capture,
so the copy-back promotion alone carries its follow-up inputs: it reads that row whole (inline
and >25-row manifest alike) and the canonical store keeps the files after the drive goes."""
data = tmp_path / "data"
drive = headless.prepare_task_drive(data, TASK, "empty")
write_task_result(data, TASK, "running", headless_child_drive_root=str(drive))
manifest = _staged(tmp_path, drive, count, "follow-up")
text = "first
second
third"
assert owner_mailbox.write_owner_message(drive, text, TASK, msg_id="u1", attachment_manifest=manifest)
assert [entry["text"] for entry in owner_mailbox.drain_owner_entries(drive, TASK)] == [text]
assert owner_mailbox.acknowledge_task_messages(drive, TASK, ["u1"], wake_id="test")
assert task_custody.unread_mail_rows(drive, TASK) == ([], True) # read: no unread custody holds it
write_task_result(drive, TASK, "completed", result="done")
copied = headless.copy_child_task_result(data, {"id": TASK, "drive_root": str(drive)})
assert copied["child_ref_promotion"]["status"] == "complete"
report = headless.prune_headless_task_drives(data, retention_days=1, now=4_000_000_000, live=lambda _task: False)
assert [row["task_id"] for row in report["pruned"]] == [TASK] and not drive.exists()
store = artifacts.task_artifact_dir_path(data, TASK)
for row in manifest:
artifacts.stream_artifact_file(store / row["relpath"], expected=row)
@pytest.mark.parametrize("count", [2, 30])
def test_an_unreadable_owner_history_row_fails_input_promotion_closed_until_repaired(tmp_path, count):
"""R1: a torn append followed by the next row leaves ONE unreadable line that swallowed an owner
row with inputs. Its kind cannot be proven, so the input history read fails closed: the
promotion keeps a pending ref (retry evidence), the drive stays, and a repair converges."""
from ouroboros import observability
data = tmp_path / "data"
drive = headless.prepare_task_drive(data, TASK, "empty")
write_task_result(data, TASK, "running", headless_child_drive_root=str(drive))
mailbox = owner_mailbox._mailbox_path(drive, TASK)
mailbox.parent.mkdir(parents=True, exist_ok=True)
mailbox.write_text('{"msg_id": "torn", "kind": "owner_text", "te', encoding="utf-8") # a crashed append
manifest = _staged(tmp_path, drive, count, "swallowed")
assert owner_mailbox.write_owner_message(drive, "see the files", TASK, msg_id="m1", attachment_manifest=manifest)
assert len(mailbox.read_text(encoding="utf-8").split("\n")) == 2 # one unreadable line, one terminator
write_task_result(drive, TASK, "completed", result="done")
copied = headless.copy_child_task_result(data, {"id": TASK, "drive_root": str(drive)})
assert copied["child_ref_promotion"]["status"] == "incomplete"
assert any(ref.get("path") == str(mailbox) for ref in copied["child_ref_promotion"]["pending_refs"])
later = 4_000_000_000
assert not headless.prune_headless_task_drives(data, retention_days=1, now=later, live=lambda _t: False)["pruned"]
assert drive.is_dir()
content = mailbox.read_text(encoding="utf-8")
mailbox.write_text(content[content.index("{", 1):], encoding="utf-8") # the torn prefix repaired away
assert observability.retry_pending_child_ref_promotions(data)["completed"] == [TASK]
report = headless.prune_headless_task_drives(data, retention_days=1, now=later, live=lambda _task: False)
assert [row["task_id"] for row in report["pruned"]] == [TASK]
store = artifacts.task_artifact_dir_path(data, TASK)
for row in manifest:
artifacts.stream_artifact_file(store / row["relpath"], expected=row)
custody = load_task_result(data, TASK)["unread_mailbox"]
assert _msg_ids(custody) == ["m1"] and len(custody["inputs"]) == 1
@pytest.mark.parametrize("boundary", ["inside_carry", "mail_lock_wait"])
def test_a_generation_closed_inside_a_mailbox_cleanup_writes_and_unlinks_nothing_more(tmp_path, monkeypatch, boundary):
"""R3: the off-loop cleanup (the drive-custody sweep and the ref retry) threads the generation
into its mutation owners: a close observed inside the input carry places no further copy, and
one observed while the mail lock was awaited writes no row and unlinks nothing. The mailbox
and the row stay exactly as retry evidence; the next generation converges."""
data = tmp_path / "data"
if boundary == "inside_carry":
drive = headless.prepare_task_drive(data, TASK, "empty")
write_task_result(data, TASK, "scheduled", child_drive_root=str(drive))
manifest = _staged(tmp_path, drive, 30, "carried")
write_task_result(data, TASK, "cancelled", result="Cancelled before start.")
else:
drive = data
write_task_result(data, TASK, "completed", result="done", root_phase_checkpoint={"post_task_synthesis": "completed"})
manifest = _staged(tmp_path, data, 2, "canonical")
assert owner_mailbox.write_owner_message(drive, "see the files", TASK, msg_id="m1", attachment_manifest=manifest)
mailbox = owner_mailbox._mailbox_path(drive, TASK)
before = load_task_result(data, TASK)
closed, written_after = [], []
if boundary == "inside_carry":
store = artifacts.task_artifact_dir_path(data, TASK).resolve()
real_copy = artifacts.copy_artifact_file
def copy_then_close(source, destination, **kwargs):
placed = Path(destination).resolve().is_relative_to(store) and Path(source).resolve() != Path(destination).resolve()
was_closed = bool(closed)
measured = real_copy(source, destination, **kwargs)
if placed:
written_after.extend([str(destination)] if was_closed else [])
closed.append(1) # the generation closes right after the first placed copy
return measured
monkeypatch.setattr(artifacts, "copy_artifact_file", copy_then_close)
assert owner_mailbox.cleanup_task_mailbox(drive, TASK, canonical_root=data, stop=lambda: bool(closed)) is False
assert closed and written_after == [], "a copy was placed after the close"
else:
real_lock = task_custody.task_mail_lock
def lock_then_close(*args, **kwargs):
closed.append(1) # the generation closes while this lock is awaited
return real_lock(*args, **kwargs)
monkeypatch.setattr(task_custody, "task_mail_lock", lock_then_close)
assert owner_mailbox.sweep_settled_owner_mailboxes(data, stop=lambda: bool(closed)) == {"removed": [], "kept": 1}
assert closed
assert mailbox.is_file() and load_task_result(data, TASK) == before
monkeypatch.undo()
assert owner_mailbox.cleanup_task_mailbox(drive, TASK, canonical_root=data) is True
custody = load_task_result(data, TASK)["unread_mailbox"]
resolved = artifacts.resolve_attachment_manifest(data, TASK, next(iter(custody["inputs"].values())))
assert len(resolved) == len(manifest)
for item in resolved:
artifacts.stream_artifact_file(Path(item["abs_path"]), expected=item)

View file

@ -102,12 +102,35 @@ export function cancelTask(taskId, { cascade = false, stopPolicy = '' } = {}) {
return Object.keys(body).length ? jsonPost(url, body) : fetchJson(url, { method: 'POST' });
}
/** Canonical task-file address shared by live delivery, replay and source links. */
export function taskArtifactDownloadUrl(taskId, name) {
if (!/^[A-Za-z0-9][A-Za-z0-9_.-]{0,127}$/.test(String(taskId || ''))
const TASK_ID_RE = /^[A-Za-z0-9][A-Za-z0-9_.-]{0,127}$/;
const plainSegments = (text) => typeof text === 'string' && !!text && !text.includes('\\') && !text.includes('\0')
&& text.split('/').every((part) => part && part !== '.' && part !== '..');
// Python quote(safe='') spelling of one path segment.
const encodeSegment = (text) => encodeURIComponent(text).replace(/[!'()*]/g, char => `%${char.charCodeAt(0).toString(16).toUpperCase()}`);
/**
* Canonical task-file address shared by live delivery, replay and source links. A nested
* result file (`relpath` = its store-relative path, ending in `name`) is addressed exactly
* with `?relpath=`; the bare name only ever selects a top-level file.
*/
export function taskArtifactDownloadUrl(taskId, name, relpath = '') {
if (!TASK_ID_RE.test(String(taskId || ''))
|| typeof name !== 'string' || !name || name.startsWith('.') || /[/\\]/.test(name)) return '';
const encodedName = encodeURIComponent(name).replace(/[!'()*]/g, char => `%${char.charCodeAt(0).toString(16).toUpperCase()}`);
return `/api/tasks/${encodeURIComponent(taskId)}/artifacts/${encodedName}`;
const url = `/api/tasks/${encodeURIComponent(taskId)}/artifacts/${encodeSegment(name)}`;
if (!relpath || relpath === name) return url;
if (!plainSegments(relpath) || relpath.split('/').at(-1) !== name) return '';
return `${url}?relpath=${encodeURIComponent(relpath)}`;
}
/**
* Address of the on-demand ZIP of one recorded result directory (store-relative, plain
* segments): `{basename}.zip?archive={directory}` on the task-file route; '' for a
* directory the route would refuse.
*/
export function taskArtifactArchiveUrl(taskId, directory) {
if (!TASK_ID_RE.test(String(taskId || '')) || !plainSegments(directory)) return '';
const name = `${directory.split('/').at(-1)}.zip`;
return `/api/tasks/${encodeURIComponent(taskId)}/artifacts/${encodeSegment(name)}?archive=${encodeURIComponent(directory)}`;
}
/** URL for one published immutable source handle. */

View file

@ -1159,6 +1159,12 @@
* intent is the SOFT stop ("finalize_then_cancel") — the UI shows
* "Finalizing…" and offers the hard escalation; absent on immediate intents.
* @typedef {Object} TaskDetailResponse
* @property {Array<{name:string, path?:string, relpath?:string, size?:number, measured?:boolean, status?:string, errors?:string[]}>=} artifacts
* Recorded result rows; a nested file keeps its store-relative `relpath`, and `measured: false`
* marks a stat-only listing, not a verified capture.
* @property {Object.<string, {name:string, files:number, size:number, excluded:number, available:boolean}>=} artifact_archives
* Per top-level result directory, what `?archive=<dir>` would stream now (members from one
* confined stat each, rows left out counted); a folder offers its `.zip` only when `available`.
* @property {Object.<string,Object>=} model_waits
* @property {TaskCostBreakdown=} cost_breakdown
* @property {string=} cancel_state

View file

@ -10,6 +10,7 @@ import { createChatDecision } from './chat_decision.js';
import { bindProjectWorkPointer } from './project_work_pointer.js';
import { createModelWaitController, isModelWaitReference } from './model_wait.js';
import { clientSurfaceField } from './client_surface.js';
import { syncResultFilesItem } from './result_files.js';
import { createChatHistoryPager } from './chat_history.js';
import { mergeHistoricalTimelineItem, historyNodeIsProtected, historyRowIds, stampHistoryNode, compareHistoryPosition } from './chat_history_replay.js';
import { apiClient, apiFetch, fetchTaskDetail, fetchTaskDetailStrict } from './api_client.js';
@ -1167,18 +1168,24 @@ export function createChatInstance({
return withStableViewport(() => {
const id = taskKey(taskId);
modelWaits.observe(id, detail);
const filed = noteResultFiles(liveCardRecords.get(id), detail);
const groups = reviewGroupsFromTaskDetail(detail, id);
if (!id || groups.length === 0) return false;
if (!id || groups.length === 0) return filed;
const fresh = !liveCardRecords.has(id);
const record = getLiveCardRecord(id);
if (fresh) reanchorTaskCard(record, detail?.ts || detail?.timestamp || '');
const changed = record.reviewController.updateMany(groups);
ensureLiveCardVisible(record);
const reconciled = reconcileCancelCardFromDetail(record, id, detail);
return Boolean(changed || reconciled);
return Boolean(changed || reconciled || filed);
});
}
// V12: a settled detail keeps the card's one Files row current.
function noteResultFiles(record, detail) {
return Boolean(syncResultFilesItem(record, detail) && (updateLiveCardCount(record), renderLiveCardTimeline(record), true));
}
function hydrateCardReviews(taskId, revision = null) {
return destroyed ? Promise.resolve(false) : reviewHydrator.hydrate(taskId, revision, {
onDomWrite: _remoteActivityDepth > 0 ? withRemoteActivity : withStableViewport,

View file

@ -2,6 +2,7 @@
// live-card presentation projections (moved verbatim from chat.js) plus the
// in-flight direct/ephemeral turn status reducer and snapshot hydration.
import { executorIdentityMarkup, joinMetaParts } from './harness_presentation.js';
import { resultFilesItemHtml } from './result_files.js';
import { compactModel, formatLogDuration, modelExecutionLabel } from './log_events.js';
import { createSystemMessageActions } from './ui_helpers.js';
import { projectReference } from './project_reference.js';
@ -71,6 +72,7 @@ export function isLiveLineExpandable(item) {
}
export function buildTimelineItemHtml(item, record) {
if (item.resultArtifacts) return resultFilesItemHtml(item);
const expandable = isLiveLineExpandable(item);
const expanded = expandable && record.expandedLineKeys.has(item.lineKey);
const displayHeadline = expanded && item.fullHeadline ? item.fullHeadline : item.headline;

162
web/modules/result_files.js Normal file
View file

@ -0,0 +1,162 @@
// V12 (TZ-1): a task card's one "Files" row. The host's recorded result rows are
// projected into root files and ONE folder record per nested tree (name, file count,
// size, member paths as its tooltip), with real download / directory-ZIP links built
// only from host-written fields (`artifacts`, `artifact_archives`); no backend is faked.
import { taskArtifactArchiveUrl, taskArtifactDownloadUrl } from './api_client.js';
import { isTerminalTaskDetail } from './log_events.js';
import { escapeHtmlAttr, escapeHtmlText as escapeHtml } from './utils.js';
const RESULT_FILE_ROWS = 8;
const FOLDER_TOOLTIP_PATHS = 20;
function artifactBytes(value) {
const bytes = Number(value);
if (value === null || value === undefined || value === '' || !Number.isFinite(bytes) || bytes < 0) return '';
if (bytes < 1024) return `${bytes} B`;
const units = ['KB', 'MB', 'GB', 'TB'];
let size = bytes / 1024;
let unit = 0;
while (size >= 1024 && unit < units.length - 1) { size /= 1024; unit += 1; }
return `${size >= 10 ? size.toFixed(0) : size.toFixed(1)} ${units[unit]}`;
}
// The host's word on one folder's archive (`artifact_archives[name]`): null when it says
// nothing, `{available: false}` when it refuses or no address can be built, else the address
// with the HOST's member count/size and the rows it leaves out.
function folderArchive(taskId, name, archives) {
const fact = archives && typeof archives === 'object' && Object.hasOwn(archives, name) ? archives[name] : null;
if (!fact || typeof fact !== 'object') return null;
const fileCount = Number(fact.files);
const url = fact.available === true && Number.isInteger(fileCount) && fileCount > 0
? taskArtifactArchiveUrl(taskId, name) : '';
if (!url) return { available: false };
const size = Number(fact.size);
const excluded = Number(fact.excluded);
return {
available: true, url, fileCount,
name: typeof fact.name === 'string' && fact.name ? fact.name : `${name}.zip`,
bytes: fact.size !== null && fact.size !== '' && Number.isFinite(size) && size >= 0 ? size : null,
excluded: Number.isInteger(excluded) && excluded > 0 ? excluded : 0,
};
}
/**
* V12: a task's result records as its card lists them. A root file stays one row; every
* nested tree is ONE folder record (name, file count, size, member paths in its tooltip),
* so a cloned repo or venv never floods the card and a bare name never stands for a
* nested file. A root file offers its download when its capture still serves bytes
* (status ready or unstated, no errors); a stat-only listing says `unverified`. A folder
* offers its `.zip` only where the host's `artifact_archives` calls it available, with the
* host's member count/size and `excluded` rows; a folder the host refuses says `no
* archive`. Reads only host-written fields. null when the list names nothing.
*/
export function projectResultArtifacts(taskId, records, archives = null) {
const files = [];
const folders = new Map();
for (const row of Array.isArray(records) ? records : []) {
if (!row || typeof row !== 'object') continue;
const parts = String(row.relpath || '').split('/').filter((part) => part && part !== '.');
const size = Number(row.size);
const bytes = row.size !== null && row.size !== '' && Number.isFinite(size) && size >= 0 ? size : null;
const status = String(row.status || '').trim().toLowerCase();
const serving = (status === '' || status === 'ready') && !(Array.isArray(row.errors) && row.errors.length)
&& !row.copy_status;
if (parts.length > 1) {
const folder = folders.get(parts[0])
|| { name: parts[0], fileCount: 0, bytes: 0, unavailableCount: 0, relpaths: [], archive: null };
folder.fileCount += 1;
folder.bytes = folder.bytes === null || bytes === null ? null : folder.bytes + bytes;
if (!serving) folder.unavailableCount += 1;
folder.relpaths.push(parts.join('/'));
folders.set(parts[0], folder);
continue;
}
const name = String(row.name || parts[0] || '').trim();
if (!name) continue;
files.push({
name, bytes,
note: serving ? (row.measured === false ? 'unverified' : '') : (status || 'unavailable'),
url: serving ? taskArtifactDownloadUrl(taskId, name) : '',
});
}
const dirs = [...folders.values()].sort((a, b) => a.name.localeCompare(b.name));
for (const folder of dirs) folder.archive = folderArchive(taskId, folder.name, archives);
const fileCount = files.length + dirs.reduce((sum, folder) => sum + folder.fileCount, 0);
return fileCount ? { files, folders: dirs, fileCount } : null;
}
/**
* Keep a card's one Files row in step with a SETTLED detail's record list: a detail without
* the list leaves the row as it was, an empty list removes it. true when the items changed.
*/
export function syncResultFilesItem(record, detail) {
if (!record || !isTerminalTaskDetail(detail) || !Array.isArray(detail?.artifacts)) return false;
const view = projectResultArtifacts(record.groupId, detail.artifacts, detail.artifact_archives);
const key = `files|${record.groupId}`;
const index = record.items.findIndex((item) => item.dedupeKey === key);
const current = record.items[index];
if (JSON.stringify(current?.resultArtifacts ?? null) === JSON.stringify(view)) return false;
if (!view) record.items.splice(index, 1);
else if (current) current.resultArtifacts = view;
else {
record.items.push({
phase: 'result', headline: 'Files', fullHeadline: 'Files', body: '', fullBody: '', fullRef: '',
truncated: false, receipt: false, ts: '', sourceTs: String(detail.ts || ''), count: 1,
dedupeKey: key, lineKey: `files-${String(record.groupId).replace(/[^A-Za-z0-9_-]/g, '-')}`,
resultArtifacts: view,
});
}
return true;
}
const fileCountText = (count) => `${count.toLocaleString('en-US')} ${count === 1 ? 'file' : 'files'}`;
const metaTail = (parts) => parts.filter(Boolean).map((part) => ` · ${escapeHtml(part)}`).join('');
function folderTooltip(folder) {
const shown = folder.relpaths.slice(0, FOLDER_TOOLTIP_PATHS);
const more = folder.relpaths.length - shown.length;
return shown.join('\n') + (more > 0 ? `\n… ${more.toLocaleString('en-US')} more` : '');
}
// Rows use the markdown list markup (`md-li`, `md-link`) this body already styles.
function resultArtifactsHtml(view) {
const rows = [
...view.files.map((file) => ({
files: 1,
html: (file.url
? `<a class="md-link" href="${escapeHtmlAttr(file.url)}" download="${escapeHtmlAttr(file.name)}">${escapeHtml(file.name)}</a>`
: escapeHtml(file.name))
+ metaTail([artifactBytes(file.bytes), file.note]),
})),
...view.folders.map((folder) => ({
files: folder.fileCount,
html: `<span title="${escapeHtmlAttr(folderTooltip(folder))}">`
+ (folder.archive?.available
? `<a class="md-link" href="${escapeHtmlAttr(folder.archive.url)}" download="${escapeHtmlAttr(folder.archive.name)}">${escapeHtml(folder.name)}/</a>`
: `${escapeHtml(folder.name)}/`)
+ metaTail(folder.archive?.available
? ['folder', fileCountText(folder.archive.fileCount), artifactBytes(folder.archive.bytes),
folder.archive.excluded ? `${folder.archive.excluded.toLocaleString('en-US')} not archived` : '']
: ['folder', fileCountText(folder.fileCount), artifactBytes(folder.bytes),
folder.unavailableCount ? `${folder.unavailableCount.toLocaleString('en-US')} unavailable` : '',
folder.archive ? 'no archive' : ''])
+ '</span>',
})),
];
const hidden = rows.slice(RESULT_FILE_ROWS).reduce((sum, row) => sum + row.files, 0);
return rows.slice(0, RESULT_FILE_ROWS).map((row) => `<span class="md-li">• ${row.html}</span>`).join('')
+ (hidden ? `<span class="md-li">+${fileCountText(hidden).replace(/ (files?)$/, ' more $1')}</span>` : '');
}
/** The timeline markup of a Files item (`item.resultArtifacts`, see `syncResultFilesItem`). */
export function resultFilesItemHtml(item) {
return `
<div class="chat-live-line ${item.phase || 'result'}" data-live-line-key="${escapeHtmlAttr(item.lineKey || '')}" data-result-files data-expanded="0">
<div class="chat-live-line-head">
<span class="chat-live-line-title">Files</span>
<span class="chat-live-line-time">${fileCountText(item.resultArtifacts.fileCount)}</span>
</div>
<div class="chat-live-line-body">${resultArtifactsHtml(item.resultArtifacts)}</div>
</div>
`;
}

View file

@ -0,0 +1,87 @@
// V12: a task's nested result tree is ONE folder record on its card (name, file count,
// size, member paths in its tooltip); a root file stays one row. A root file offers
// bytes only while its record still serves them, a nested file is never addressed by
// its bare name, a folder offers its `.zip` only where the host's `artifact_archives`
// confirms it (with the host's count/size), and a stat-only listing says `unverified`.
import assert from 'node:assert/strict';
import test from 'node:test';
import { projectResultArtifacts, resultFilesItemHtml, syncResultFilesItem } from '../modules/result_files.js';
import { installDom, restoreDom } from './chat_dom_fixture.js';
const TASK = 'task-files';
const RECORDS = [
{ kind: 'task_artifact', name: 'report.md', size: 2150, status: 'ready', errors: [] },
{ kind: 'task_artifact', name: 'README.md', relpath: 'repo/README.md', size: 500, status: 'ready', errors: [] },
{ kind: 'task_artifact', name: 'app.py', relpath: 'repo/src/app.py', size: 1548, status: 'missing', errors: [] },
{ kind: 'task_artifact', name: 'json.py', relpath: 'venv/lib/json.py', size: 1000, measured: false },
{ kind: 'task_artifact', name: 'notes.md', size: 10, status: 'missing', errors: [] },
{ kind: 'task_artifact', name: 'listed.txt', size: 5, measured: false },
];
test('root files stay rows; each nested tree is one folder record keeping its relative paths', () => {
const view = projectResultArtifacts(TASK, RECORDS);
assert.deepEqual(view.files, [
{ name: 'report.md', bytes: 2150, note: '', url: `/api/tasks/${TASK}/artifacts/report.md` },
{ name: 'notes.md', bytes: 10, note: 'missing', url: '' },
{ name: 'listed.txt', bytes: 5, note: 'unverified', url: `/api/tasks/${TASK}/artifacts/listed.txt` },
]);
assert.deepEqual(view.folders, [
{ name: 'repo', fileCount: 2, bytes: 2048, unavailableCount: 1, relpaths: ['repo/README.md', 'repo/src/app.py'], archive: null },
{ name: 'venv', fileCount: 1, bytes: 1000, unavailableCount: 0, relpaths: ['venv/lib/json.py'], archive: null },
]);
assert.equal(view.fileCount, 6);
assert.ok(!view.files.some((file) => file.name === 'json.py'), 'a nested file never stands in for a root name');
assert.equal(projectResultArtifacts(TASK, []), null);
assert.equal(projectResultArtifacts(TASK, [null, 'x', { size: 3 }]), null);
});
function render(view) {
const { prior } = installDom();
try { return resultFilesItemHtml({ phase: 'result', lineKey: 'line-1', resultArtifacts: view }); }
finally { restoreDom(prior); }
}
const ARCHIVES = {
repo: { name: 'repo.zip', files: 1, size: 500, excluded: 1, available: true },
venv: { name: 'venv.zip', files: 0, size: 0, excluded: 1, available: false },
};
test('the Files row renders real links and a folder tooltip, never a nested file address', () => {
const html = render(projectResultArtifacts(TASK, RECORDS, ARCHIVES));
assert.match(html, /data-result-files/);
assert.match(html, /<span class="chat-live-line-time">6 files<\/span>/);
assert.match(html, /<a class="md-link" href="\/api\/tasks\/task-files\/artifacts\/report\.md" download="report\.md">report\.md<\/a> · 2\.1 KB/);
assert.match(html, /notes\.md · 10 B · missing/);
assert.match(html, /listed\.txt<\/a> · 5 B · unverified/);
assert.match(html, /<span title="repo\/README\.md\nrepo\/src\/app\.py"><a class="md-link" href="\/api\/tasks\/task-files\/artifacts\/repo\.zip\?archive=repo" download="repo\.zip">repo\/<\/a> · folder · 1 file · 500 B · 1 not archived<\/span>/);
assert.match(html, /<span title="venv\/lib\/json\.py">venv\/ · folder · 1 file · 1000 B · no archive<\/span>/);
assert.equal((html.match(/href=/g) || []).length, 3);
assert.doesNotMatch(html, /artifacts\/json\.py|artifacts\/README|relpath=/, 'no bare or nested member address');
});
test('the row bounds its length, escapes names and lists at most twenty tooltip paths', () => {
const many = Array.from({ length: 12 }, (_, index) => ({ name: `f${index}.txt`, size: 1, status: 'ready' }));
const tree = Array.from({ length: 25 }, (_, index) => ({ name: `m${index}`, relpath: `tree/m${index}`, size: 1 }));
const html = render(projectResultArtifacts(TASK, [...many, ...tree]));
assert.equal((html.match(/class="md-li">•/g) || []).length, 8);
assert.match(html, /\+29 more files/, 'four root files and the 25-file folder are counted, not dropped');
assert.match(render(projectResultArtifacts(TASK, tree)), /tree\/m19\n… 5 more"/);
const hostile = render(projectResultArtifacts(TASK, [
{ name: '<img src=x onerror=1>.md', size: 1, status: 'missing' },
{ name: 'a', relpath: '"><b>/a', size: 1, status: 'ready' },
]));
assert.doesNotMatch(hostile, /<img|<b>/);
assert.match(hostile, /&lt;img src=x onerror=1&gt;\.md/);
});
test('a card keeps one Files row in step with a settled detail only', () => {
const record = { groupId: TASK, items: [] };
assert.equal(syncResultFilesItem(record, { status: 'running', artifacts: RECORDS }), false, 'not settled');
assert.equal(syncResultFilesItem(record, { status: 'completed' }), false, 'no record list: unchanged');
assert.equal(syncResultFilesItem(record, { status: 'completed', artifacts: RECORDS, artifact_archives: ARCHIVES }), true);
assert.equal(record.items.length, 1);
assert.equal(record.items[0].dedupeKey, `files|${TASK}`);
assert.equal(syncResultFilesItem(record, { status: 'completed', artifacts: RECORDS, artifact_archives: ARCHIVES }), false);
assert.equal(syncResultFilesItem(record, { status: 'completed', artifacts: [] }), true);
assert.deepEqual(record.items, []);
});

View file

@ -1,13 +1,31 @@
import assert from 'node:assert/strict';
import test from 'node:test';
import { taskArtifactDownloadUrl } from '../modules/api_client.js';
import { taskArtifactArchiveUrl, taskArtifactDownloadUrl } from '../modules/api_client.js';
test('captured task files use the backend canonical URL encoding', () => {
assert.equal(taskArtifactDownloadUrl('task', "résumé's (complete).zip"),
'/api/tasks/task/artifacts/r%C3%A9sum%C3%A9%27s%20%28complete%29.zip');
assert.equal(taskArtifactDownloadUrl('task', 'ordinary.bin'), '/api/tasks/task/artifacts/ordinary.bin');
assert.equal(taskArtifactDownloadUrl('task', 'report.pdf', 'nested/a/report.pdf'),
'/api/tasks/task/artifacts/report.pdf?relpath=nested%2Fa%2Freport.pdf');
for (const relpath of ['../report.pdf', '/report.pdf', 'nested/../report.pdf',
'nested//report.pdf', 'nested\\report.pdf', 'nested/other.pdf']) {
assert.equal(taskArtifactDownloadUrl('task', 'report.pdf', relpath), '');
}
for (const name of ['../other', 'folder/file', 'folder\\file', '.artifact_manifest.json']) {
assert.equal(taskArtifactDownloadUrl('task', name), '');
}
assert.equal(taskArtifactDownloadUrl('other/task', 'file.bin'), '');
});
test('a recorded directory archive is addressed by its basename plus .zip and the exact directory', () => {
assert.equal(taskArtifactArchiveUrl('task', 'repo'), '/api/tasks/task/artifacts/repo.zip?archive=repo');
assert.equal(taskArtifactArchiveUrl('task', "nested/src dir's"),
"/api/tasks/task/artifacts/src%20dir%27s.zip?archive=nested%2Fsrc%20dir's");
for (const directory of ['', '../repo', '/repo', 'repo/', 'repo/../x', 'repo\\src', '.', 'a/./b', 'a\0b',
42, null, undefined]) {
assert.equal(taskArtifactArchiveUrl('task', directory), '', JSON.stringify(directory));
}
assert.equal(taskArtifactArchiveUrl('task', '.github'), '/api/tasks/task/artifacts/.github.zip?archive=.github');
assert.equal(taskArtifactArchiveUrl('other/task', 'repo'), '');
});