Merge phase P5+P7 (terminal presentation: coherent depth tuple, settled result overrides early finality)

This commit is contained in:
Ouroboros 2026-09-04 04:14:11 +03:00
commit 33d4124c36
46 changed files with 1720 additions and 177 deletions

View file

@ -39,7 +39,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de
│ ├── task_lifecycle.py ← Cancellation custody — the ONE settle owner of durable cancel intents: claim → capture → confirmed death → natural-completion re-check → owed delivery registration → settle → delivery/cleanup, plus the `sweep_cancel_intents` watchdog and the queue-owned root-budget admission fence; every custody rule it enforces is stated once in §10 (cancellation custody)
│ ├── cancel_publication.py ← Cancellation settlement publication, re-imported by `task_lifecycle.py`: typed CANCEL_* outcome vocabulary, artifact-honest cancelled result fields, physical-ledger cost reconstruction, salvage adapter, owed-before-settle outbox registration, publication of the STORED terminal truth, capture-miss terminalization/delivery adapter
│ ├── queue_transitions.py ← Queue-owned lifecycle transitions that are not cancellation custody: acceptance-fence open/inspect/seal, explicit budget resume, typed `stop_evolution_tasks` (per-task typed outcomes through the durable-intent ingress, never an in-place prune; an incomplete stop leaves the campaign OPEN via the durable `evolution_owner_stopped` flag, the settle-time backstop in events.py closes it when the last live evolution task settles, and both start ingresses clear the flag BEFORE minting a fresh campaign), and fenced Project deletion (cascade only lineage ROOTS — descendants fall with their trees, one cascade and one summary per tree; tombstone only after provable quiescence; a settled-but-LIVE root still mints the coordination intent, and wind-down defers and RE-CHECKS bounded instead of re-running the cancel pass over a settled-lingering set, which would deliver duplicate owner summaries); imports nothing from task_lifecycle; `supervisor.queue` re-exports these names
│ ├── terminal_delivery.py ← Durable terminal-answer delivery seam: restart-surviving `delivery_id` dedupe + bounded PENDING outbox `state/terminal_deliveries.json` (owed before enqueue, cleared in the delivering write, replayed on boot and on the supervisor tick), shared by natural final answers (every non-ephemeral root registers at durable-result persistence), cancel salvage, cascade digest, and non-retry reap; the cascade digest enumerates descendants by ANCESTRY (parent-chain walk, never `root_task_id` equality); eviction past outbox capacity is disclosed via typed `terminal_delivery_exhausted`, never a silent pop; salvage messages carry a bounded preview plus a full-copy receipt (path, size, full 64-hex sha256 or an explicit marker) and route by lineage chat — no resolvable chat records a typed `terminal_delivery_handoff` row; reads/mutations are row-strict per §10; the delivery id digests only the STABLE part (task id + status framing + core answer) so a replay whose rebuilt note shrank dedups instead of double-sending
│ ├── terminal_delivery.py ← Durable terminal-answer delivery seam: restart-surviving `delivery_id` dedupe + bounded PENDING outbox `state/terminal_deliveries.json` (owed before enqueue, cleared in the delivering write, replayed on boot and on the supervisor tick), shared by natural final answers (every non-ephemeral root registers at durable-result persistence), cancel salvage, cascade digest, and non-retry reap; the cascade digest enumerates descendants by ANCESTRY (parent-chain walk, never `root_task_id` equality); eviction past outbox capacity is disclosed via typed `terminal_delivery_exhausted`, never a silent pop; salvage messages carry a bounded preview plus a full-copy receipt (path, size, full 64-hex sha256 or an explicit marker) and route by lineage chat — no resolvable chat records a typed `terminal_delivery_handoff` row; reads/mutations are row-strict per §10; the delivery id digests only the STABLE part (task id + status framing + core answer) so a replay whose rebuilt note shrank dedups instead of double-sending; the per-origin projection is `host_salvage` receipt / `host_notice` own text kept as a System row with its markdown / `model_final` assistant projection
│ ├── task_reaper.py ← Off-loop single-owner reaper thread: kill/join/archive/respawn a timed-out worker; STRICT fail-closed — an unconfirmed death holds the slot `reaping`, leaves the task RUNNING, emits `task_reaper_wedged` + an owner /restart hint (no terminal/retry/respawn while the worker may be alive); after confirmed death it reconciles the task's open delegated runs through the custody seam before the retry/respawn decision and discloses still-open runs; mints no cancel intents
│ ├── owner_stop.py ← Owner graceful stop: `finalize_then_cancel` policy as an axis on the SAME durable cancel intent (monotonic — immediate HARDENS a pending graceful, never softens back; hardening revokes an unread control via mailbox revocation, and the loop revalidates durable policy at drain); one deterministic typed `finalize_now` control whose first line is the `owner_requested_finalization` literal, routed by the loop to its own rail (zero or one tool-less turn); descendants settle first with a bounded child projection fed to the root's final turn; the grace budget starts at the durable control DRAIN (first drain wins), bounded by `request + OWNER_STOP_OUTER_CAP_SEC`, and neither anchor is ever progress-extended; `running_owner_stop_tasks` bypasses only the generic idle/finalization-grace rails; a COMPLETED finalize root suppresses the redundant cascade summary
│ ├── schedule_time.py ← Cron/timezone schedule time parsing helpers
@ -72,7 +72,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de
├── agent.py ← Task orchestrator; the dispatch-note pair lives in `subagent_dispatch_notes.py` (same-name re-exports)
├── agent_startup_checks.py ← Worker-boot verification: dirty repo, version sync, budget, memory files, health checks
├── agent_task_pipeline.py ← Task execution pipeline orchestration; freezes one shared non-final subtree-cost snapshot for summary/reflection before the terminal checkpoint records final spend; hands the summary and reflection prompts the commit/advisory review lens PLUS the task's own acceptance-panel projection, and an absence statement names the lens it describes; calls the swarm-efficiency rollup owned by task_finalization.py at pipeline end
├── task_finalization.py ← Terminal delivery + sealed final ground truth: live final-answer delivery before blocking post-task (final event selected by the finalizing task's id; buffered copy retained under one `delivery_id`), the sealed final package (delivered text + the durable result's own artifact manifest) fed to summary/reflection as a prompt input, never a validator; owns the per-task `swarm_efficiency` rollup (subagent_count / wave_count / inter-wave latency / `lanes_requested` — a rollup built from pre-dispatch fanout events cannot truthfully report effective lanes, which are per-child dispatch facts; `planned` stays null, never inferred as 0 from absent events; host-attested Swarm intent is the typed metadata `force_plan_source == "swarm"`, never prompt inspection, and a Swarm task that fanned out nothing records a minimal `no_fanout_observed` block instead of silence)
├── task_finalization.py ← Terminal delivery + sealed final ground truth: live final-answer delivery before blocking post-task (final event selected by the finalizing task's id; buffered copy retained under one `delivery_id`), the sealed final package (delivered text + the durable result's own artifact manifest) fed to summary/reflection as a prompt input, never a validator; owns the per-task `swarm_efficiency` rollup (subagent_count / wave_count / inter-wave latency / `lanes_requested` — a rollup built from pre-dispatch fanout events cannot truthfully report effective lanes, which are per-child dispatch facts; `planned` stays null, never inferred as 0 from absent events; host-attested Swarm intent is the typed metadata `force_plan_source == "swarm"`, never prompt inspection, and a Swarm task that fanned out nothing records a minimal `no_fanout_observed` block instead of silence); a fanned-out root additionally carries a `depth` block (`requested_depth`/`permitted_depth`/`attempted_depth`/`achieved_depth` plus a typed status, `host_visible_only`) built from the root contract and its subtree's depth provenance, so a root that never carried its own request recovers it from the children it scheduled, and the terminal task_summary row reports it only when a request exists
├── mutation_attribution.py ← Root-task baseline capture in the existing task result; clean-at-baseline Git candidate projection; terminal projection includes the committed interval delta
├── process_interpreters.py ← Interpreter resolvers for the user process launch surfaces: one-time pre-guard unversioned-Python resolver + post-gates Node ladder (PATH-first health probe, bundled fallback, attested child-env PATH prepend; the probe EXECUTES a candidate, so it runs only after the dispatch gates approve the call)
├── post_task_checkpoint.py ← Durable root post-task phase/final-cost checkpoint shared by task finalization and Project naming recovery
@ -108,7 +108,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de
├── owner_hurry.py ← Owner "hurry": a typed TASK-LOCAL acceleration latch, never a chat message; the durable `owner_hurry` projection is written by `update_json_locked` touching only its own keys — never `write_task_result`, whose status-regression guard could drop concurrent terminal fields — keyed by the real attempt identity `task["_attempt"]`; while latched, the next acceptance panel is skipped with zero reviewer calls (`acceptance_skip_applied`), remaining improvement passes overlay to 0 through `effective_budget_profile` (the immutable task_contract is never rewritten), and force-plan becomes task-locally advisory; the effect DIES WITH THE ATTEMPT (`retry_reset` on every same-id requeue producer), a never-applied request is marked `not_applied_before_terminal`, and the non-chat `owner_hurry` events are hidden from chat by `log_events.js`
├── owner_quiz.py ← Owner-quiz lifecycle projection: worker-side `record_asked`, request-id-idempotent first-answer-wins `record_answered` (option index validated against the STORED labels), structural-only `reconcile_terminal` (open → expired_terminal at task done; no host TTL), `quiz_states` replay; same locked-writer idiom as owner_hurry, touching only the `owner_quiz` key
├── routing_wait.py ← Root-parameterized SSOT of the durable routing-receipt waits (`wait_for_promotion_admission`, `wait_for_routing_annotation`); tools/control.py keeps thin wrappers, so the gateway picker dispatcher confirms clicks through the SAME receipts the LLM routing tools poll
├── outcomes.py ← Typed task-outcome and acceptance-decision authority keeping the lifecycle/execution/objective/review/artifact/verification/child-absorption axes separate; policy denials, cosmetic exits, and ignored outcomes never masquerade as genuine tool failures; receipt reconciliation lives in `_outcome_receipts.py`, trace classification in `_outcome_tool_errors.py`
├── outcomes.py ← Typed task-outcome and acceptance-decision authority keeping the lifecycle/execution/objective/review/artifact/verification/child-absorption axes separate; policy denials, cosmetic exits, and ignored outcomes never masquerade as genuine tool failures; receipt reconciliation lives in `_outcome_receipts.py`, trace classification in `_outcome_tool_errors.py`; a verification ledger above the inline threshold rides as a stub whose `summary` is re-projected from the refreshed artifact file at finalization, and the stub is never a source for entries or outcome axes (the ledger embeds the task contract, which its entry count excludes, so a stub is the normal shape for a swarm root)
├── outcome_receipt_store.py ← Durable verification-receipt path/append/read authority + exact-row union of forked-child and canonical replicas; owns the zero-run WRITE enum (`incomplete`/`unknown` only — a zero-run "complete" is unverifiable self-report); outcomes.py re-exports the compatibility names
├── depth_evidence.py ← Pure requested/permitted/attempted/achieved depth projection for root acceptance; missing admitted permission stays evidence-unknown rather than reconstructed from mutable live config
├── _outcome_receipts.py ← Receipt parsing and the ONE canonical receipt identity (`receipt_canonical_identity` → `ReceiptIdentity`; invariant in §10): three independent components — `criterion_id`; structurally canonical `check` text PAIRED with its `check_rendering` stamp (quoted shell punctuation is data, not syntax, and receipts from different renderings are never the same verification — the stored string alone cannot say which renderer wrote it); and the raw-sorted `canonical_path_set` (whitespace untouched — a leading space is a legal filename byte); `ReceiptIdentity.key` selects ONE typed (kind, value) and sameness is that key's equality, never a match across kinds — the parts are disclosures, never the comparison; the per-kind normalization answer lives in the closed `IDENTITY_KINDS`/`KIND_NORMALIZES_COMMAND_TEXT` table, so a fourth kind must state its own answer in its own row rather than inherit a default; the outstanding sets `unreconciled_failed`/`unreconciled_masked` scan every candidate against ALL later reconcilers and collapse repeated failures of one check onto the freshest receipt; the shared disclosed projections (`receipt_identity_projection`, `disclosed_list_projection`) make every bound explicit — exact omitted counts plus a hash over the injective serialization, string bounding via the SSOT `utils.truncate_review_artifact`, never a hand-rolled slice; `verification_receipt_ledger_row` splats that projection, so a new receipt key is dropped unless added there
@ -445,7 +445,7 @@ Workspace binding changes the contextual repo, never the system repo for BIBLE/p
Workspace preflight snapshots git state (bounded porcelain rows), manifests/scripts, and tool availability into the full `workspace_preflight.json` artifact with a bounded summary in metadata; `tools_on_path`/`tools_missing_from_path` are named that way because `shutil.which` measures PATH presence, not executability, and the structured keys stay frozen because they ride durable, replaying task metadata; a collection failure is a disclosed error summary, never a fictitious full artifact.
Completion compares against the captured preflight base (task-local commits stay in the delta, not `git diff HEAD`); the patch is bound to `task_constraint.base_sha`; a moved HEAD fails closed only for `self_worktree` (a shared tree relies on reverse-patch verification); an unborn repo diffs against the canonical empty tree. Patch capture streams the tracked binary diff plus admitted untracked files, excluding scratch/cache/junk/oversized/binary-untracked/incidental-lockfile entries with per-file reasons; a sensitive-looking untracked credential is excluded per-file and disclosed as `sensitive_blocked`. `workspace_patch.json` is written for EVERY workspace finalization (including no-change and failed) and is the truth source for CLI strict-patch — it distinguishes omitted vs no-op vs failed; `workspace.patch` exists only for `ready_with_changes`.
Completion compares against the captured preflight base (task-local commits stay in the delta, not `git diff HEAD`); the patch is bound to `task_constraint.base_sha`; a moved HEAD fails closed only for `self_worktree` (a shared tree relies on reverse-patch verification); an unborn repo diffs against the canonical empty tree. Patch capture streams the tracked binary diff plus admitted untracked files, excluding scratch/cache/junk/oversized/binary-untracked/incidental-lockfile entries with per-file reasons; generated output (`dist/`, `build/`) is governed by the project's own `.gitignore`, honoured through `--exclude-standard`, not by a host name rule — git-ignored files are outside the capture universe and are not listed as exclusions; a sensitive-looking untracked credential is excluded per-file and disclosed as `sensitive_blocked`. `workspace_patch.json` is written for EVERY workspace finalization (including no-change and failed) and is the truth source for CLI strict-patch — it distinguishes omitted vs no-op vs failed; `workspace.patch` exists only for `ready_with_changes`.
Forked/empty task state lives under `data/state/headless_tasks/<task_id>/data`: a forked drive copies `identity.md`, `WORLD.md`, `registry.md` (a project fork carries `memory/knowledge/patterns.md` and omits global knowledge; an empty drive starts blank); dialogue, scratchpad, mailbox, and history never cross. The child drive is execution state: the result copies back to the canonical root, declared artifacts rebase to `data/task_results/artifacts/<task_id>/` (missing source = copy failure; collisions get a deterministic suffix), and verification-receipt replicas union with exact-row de-dup. Once the canonical result is terminal, late copy-back and effective reads pass through the same pure field-custody projection — the parent-owned terminal marker and cost/round/token fields cannot be overwritten. `memory_export.json` is an explicit artifact, never merged automatically.
@ -601,7 +601,7 @@ One shared WebSocket serves the whole application; Projects open no independent
### Navigation and shared UI contracts
Primary navigation exposes Chat (Main), a collapsible Projects group, Files, Skills, Widgets, Dashboard, and Settings; About is a Settings sub-tab. `syncNavigationState()` is the single presentation state machine for active page, active Project, Projects expansion, mobile drawer, and panel backdrop — independent toggles must not leave multiple rows active or a hidden surface looking selected. Sidebar and Project panel widths are owner-local UI preferences, not runtime settings.
Primary navigation exposes Chat (Main), a collapsible Projects group, Files, Skills, Widgets, Dashboard, and Settings; About is a Settings sub-tab. `syncNavigationState()` is the single presentation state machine for active page, active Project, Projects expansion, mobile drawer, and panel backdrop — independent toggles must not leave multiple rows active or a hidden surface looking selected. Sidebar and Project panel widths are owner-local UI preferences, not runtime settings. The application has exactly one client-side route: a `#<page>` fragment for the injected page ids, honoured once on load and validated against the existing `#page-<name>` section, so an unknown fragment is ignored rather than painting a blank surface. It is never written back on navigation — the packaged desktop shell and the Linux browser fallback have no address bar to read it from, and the Telegram mini app always loads at `/` — so a browser gains a shareable `/#widgets` link while no other surface changes. Projects are panels, not routes, and the sidebar is the document's one `<nav>` landmark.
Each active or deleting Project has a sidebar row with pointer- and keyboard-operable open/rename/delete; the backend owns the 80-character name limit and lifecycle truth. Unread Projects sort ahead of read, then by durable activity; a deleting Project becomes non-openable and stays visibly transitional until the server publishes authoritative registry state. On narrow screens navigation is an explicit drawer and the Project chat a full-width overlay; no gesture-only navigation layer competes with message scroll, text selection, or the software keyboard.
@ -628,7 +628,7 @@ Messages, media bubbles, and task-card roots order by raw numeric timestamps (ti
History reconciliation is two-pass: progress and system records first rebuild timestamped task-card state, messages and cards insert chronologically, and only then may terminal state seal a card — so terminal replay cannot discard earlier progress, and cards whose final summary was missed offline recover. Live echoes and history rows deduplicate on stable message identities. That rebuild is also the memory bound: cards are minted from progress rows, `task_summary` rows and subagent lineage, so the cap is relative to the population the last rebuild produced — once more than 200 live cards exist beyond it the next history sync replays durable history instead of folding into the existing cards, and that rebuild sets the new floor (`web/modules/chat.js`).
The Main chat receives ordinary main-thread dialogue plus exactly the two host-stamped Project lifecycle rows — `project_started` and the terminal `project_completion_summary`, each with the shared "Open Project" action; all other Project traffic stays in the Project thread. Lifecycle rows are plain dashboard text: the producer strips markdown once before durable write and live send, history normalizes older rows on read (`chat.jsonl` is never rewritten), and the renderer escapes any system row without `markdown: true` (except `skill_review`'s dedicated renderer). Project panels accept only their own registered `chat_id`; the complete Project chat-id set is separate from the sidebar's bounded summary, and a `projects_changed` frame adds a new chat id synchronously before the asynchronous state refresh, so an early frame cannot be misclassified as Main.
The Main chat receives ordinary main-thread dialogue plus exactly the two host-stamped Project lifecycle rows — `project_started` and the terminal `project_completion_summary`, each with the shared "Open Project" action; all other Project traffic stays in the Project thread. Lifecycle rows are plain dashboard text: the producer strips markdown once before durable write and live send, history normalizes older rows on read (`chat.jsonl` is never rewritten), and the renderer escapes any system row without `markdown: true` (except `skill_review`'s dedicated renderer). Finality metadata is row-specific. Durable `task_summary` rows carry the SAME phase the browser paints: `project_dialogue.outcome_phase` mirrors `log_events.js` — the terminality gate plus the severity fold over normalized axes — and stamps it as `outcome_phase` beside the status word taken from `OUTCOME_PHASE_HEADLINE`, so there is no second status-word family. The pre-finalization authored row carries the phase with `outcome_final=false` but no host verdict clause; only terminal `task_summary` rows append the host verdict clause (`_completion_verdict`, the host twin of `taskReasonDetail`'s acceptance branch) when one exists. The Main `project_completion_summary` instead puts the shared headline and any verdict directly in its plain text; it does not carry `outcome_phase`. The start row carries neither outcome finality nor a verdict. A verdict is the execution reason on ordinary terminal summaries, or — when the host's acceptance decision is anything but accepted — the decision status with its stored rationale, markdown-stripped and flattened like every other lifecycle-row text; the latter leads the Main completion row so a host-authored row never presents an unaccepted claim as the whole story. History keeps older status words unrewritten, and the Telegram skill reads the stamped task-summary phase instead of deriving one of its own. Project panels accept only their own registered `chat_id`; the complete Project chat-id set is separate from the sidebar's bounded summary, and a `projects_changed` frame adds a new chat id synchronously before the asynchronous state refresh, so an early frame cannot be misclassified as Main.
The Main composer exposes one-shot Swarm planning, the owner Low/Max context choice, file attachment, and Send. Swarm places a structural `force_plan` fact on the next ordinary message and disarms after that send — never inferred from keywords. Low/Max uses the dedicated owner endpoint; a derived Low fit never masquerades as an owner-selected posture, and selecting Max may invoke the exact-route capability acknowledgement when provider metadata cannot prove the window. Project panels omit global Restart, Panic, evolution, consciousness, review, and budget controls — those belong to the one Ouroboros process, not a Project thread.
@ -640,7 +640,7 @@ Direct in-process chat turns and ephemeral decision turns create no supervisor q
Owner-message continuity is journal-backed: each locally sent owner row is kept (bounded, with its routing annotation) until a fetched history response returns the same `client_message_id`, and a full rebuild re-renders unconfirmed journal rows after the server response, so a stale history snapshot cannot erase a message the owner just sent. Task finalization is honest: a root's early final answer carries `task_phase="finalizing"` (the same fact replay derives from the open post-task checkpoint), the card holds a sticky `Finalizing…` phase, and `task_cost_finalized` is bookkeeping that never resolves a card. If an observed managed root disappears from a queue-authoritative snapshot whose request began after the observation, the state-refresh fan-out removes activity/cancel authority and starts one single-flight durable task-detail read; request generations prevent an older response from undoing a newer projection. Two liveness invariants close the stuck-`Working...` class: the history window is lineage-closed — subagent lineage older than the window's progress recency floor is kept only while the child still runs or finalizes, or its parent is REPRESENTED by the response (a row that proves the task's card: telemetry, its own message or summary, a folded review group or plan reference it owns — a typed photo/document/quiz delivery does NOT count), or the parent is alive, and an anchored child represents its own children in turn, so a nested swarm is kept or dropped whole; a FINAL row failing that test is emitted without lineage fields (`ouroboros/gateway/history.py`, one anchor set computed in the quota pass), so replay cannot mint an unfinishable parent card while a visible or live parent keeps its completed children and their executor receipts — and liveness reconciles over the CARD SET, not only the activity registry: an unvouched connected root card is finished ONLY by proven durable terminal detail, and a card with no durable result honestly keeps its state. The header badge has exactly one writer, the status reducer, with the panel-boot "Online" seed as the sole exception.
Task activity collapses into a live task card per root instead of flooding the transcript; `log_events.js` keeps the live task card and grouped task cards on one reducer across Chat and Dashboard Logs. Non-terminal LLM/tool/checkpoint failures stay inspectable timeline facts but do not promote the card; only authoritative terminal task truth changes terminal status, and unknown Chat event names do not acquire severity from keyword substrings. A failed child keeps a local `Failed` chip while its root continues. A card keeps a concise plain-text latest-activity line and an expandable timeline; server-truncated rows fetch the complete typed task record on demand into a bounded viewer. The compact card is a navigation projection, not the result authority; urgent toast/unread behavior stays confined to explicit `task_incident` facts, never inferred from severity.
Task activity collapses into a live task card per root instead of flooding the transcript; `log_events.js` keeps the live task card and grouped task cards on one reducer across Chat and Dashboard Logs. Non-terminal LLM/tool/checkpoint failures stay inspectable timeline facts but do not promote the card; only authoritative terminal task truth changes terminal status, and unknown Chat event names do not acquire severity from keyword substrings. A failed child keeps a local `Failed` chip while its root continues. A card keeps a concise plain-text latest-activity line and an expandable timeline; server-truncated rows fetch the complete typed task record on demand into a bounded viewer. The compact card is a navigation projection, not the result authority; urgent toast/unread behavior stays confined to explicit `task_incident` facts, never inferred from severity. The ONE explanatory line under a terminal headline is `taskReasonDetail`: an owner-requested soft stop has none, a hard failure or a cancellation names its execution reason, a non-accepted host acceptance decision names the decision status plus its stored rationale, and everything else keeps the typed reason phrase (an unknown code stays raw); Logs meta additionally names `review <status>` and `acceptance <status>`.
Subagents render as distinct child cards keyed by their actual child task ids; parents keep lineage references without duplicating the child's final answer. Nested children collapse by default: headline `role · model` (short task id only when a sibling shares both; Logs keep the full form), status carried by the chip alone. Reviews are not task lineage: a reviewer run's execution receipt belongs inside the real owning task card, and its harness or neutral API mark identifies the delivery channel without minting a child card or proving execution. Selected `subagent_id`/configured snapshot, requested route, effective engine route/model/account, and terminal execution evidence stay separate facts — intent must not be redrawn as proof of where the run settled. Legacy lane/executor fields stay readable on historical cards only.
@ -936,7 +936,7 @@ The delivery-control protocol is resolved here and only here (DEVELOPMENT keeps
The child-absorption gate is an action gate: while undispositioned direct children remain, the loop HOLDS the candidate (`child_absorption_or_revision_required`) instead of arming the JSON-only control instruction — the hold-vs-arm split exists so the model never receives two contradictory instructions in one round. A typed keep cannot close the gate, and after the one bounded reminder the gate forces the best-effort `children_unabsorbed` rail with a current `id [status] sha256` listing. The absorption digest's `## child` header carries the child's typed custody debt (`delegated_runs_unreconciled`, the first ten ids plus an omitted count and a `get_task_result` pointer) — visibility only: the parent's authority over that patch is exactly the orphan rule, and a child's debt never relabels the root card (DESIGN §4, a failed child keeps a compact factual marker inside its parent while the root continues under its own authoritative status). The acceptance-subtree copy of that debt is a disclosed gap, deferred with the pre-finalization reminder. Finalizing over an UNDISPOSED OWN delegated patch is deliberately NOT gated: the consequence is disclosed where the decision is made — in the `integrate_delegated_patch` schema and in the apply receipts — and lands as the additive Done-with-warnings custody overlay rather than a hold. A pre-finalization reminder for that case is deferred.
Provider death is the one forced rail that is NOT a best-effort completion: `_handle_provider_unavailable` salvages the best available text but stamps `infra_failed`, so the task terminalizes `failed` with the typed `provider_unavailable` reason and an immediate "provider outage — NOT completed" owner notification; a waited-out transport outage reaches the same rail through its deterministic no-resend branch (`transport_unavailable_no_resend`). The rail makes its one forced model call only while a call can still land: a transport that spent its same-model retry wall stamps `_llm_retry_wall_exhausted`, which the rail reads as a no-call gate beside `context_overflow` and `provider_outcome_unknown`, shipping the salvage with no request — so the forced rail does not re-pay a second retry window over a proven-dead provider; the marker is a last-invocation bool on the shared usage dict, a disclosed residual. Terminal delivery preserves producer authorship: complete model answers appear as Ouroboros, host-authored incident receipts as System.
Provider death is the one forced rail that is NOT a best-effort completion: `_handle_provider_unavailable` salvages the best available text but stamps `infra_failed`, so the task terminalizes `failed` with the typed `provider_unavailable` reason and an immediate "provider outage — NOT completed" owner notification; a waited-out transport outage reaches the same rail through its deterministic no-resend branch (`transport_unavailable_no_resend`). The rail makes its one forced model call only while a call can still land: a transport that spent its same-model retry wall stamps `_llm_retry_wall_exhausted`, which the rail reads as a no-call gate beside `context_overflow` and `provider_outcome_unknown`, shipping the salvage with no request — so the forced rail does not re-pay a second retry window over a proven-dead provider; the marker is a last-invocation bool on the shared usage dict, a disclosed residual. Terminal delivery preserves producer authorship. Every forced rail stamps one closed-vocabulary producer word at the single forced-finalization sink: `model_final` for complete model text (including a host disclosure glued inside it — plan review left open, deferred children — exactly as the ordinary no-tool final does), `host_notice` for a terminal text the host wrote alone (kept verbatim on every transport, with its markdown, and projected as a System row without a system_type), and `host_salvage` only on the provider-death rail, where the text is replaced by the outage receipt and the full bytes stay in task details; the provider-death wrapper upgrades a bare notice to salvage so a no-call/no-resend arm keeps its receipt. A missing origin identifies only rows written before that stamp existed.
Host-enforced task acceptance is a root-owned completion coach, not the P3 commit gate; its mechanism is stated once, here. `off` disables it. In `auto` and `required`, substantive queued, headless, and scheduled roots are eligible; direct chat becomes eligible only after an observable reviewable effect or a typed deliverable/criterion — pure conversation, read-only exploration, routing turns, and cognitive-memory updates create no eligibility. Child reviews are advisory evidence superseded by the root decision.
@ -1065,7 +1065,7 @@ Children coordinate through `tree_note` and `tree_read`; only the parent may use
An undisclosed spend contributes `0.0` to `accounted_usd` — inventing a conservative bound would fabricate a number the harness never gave (BIBLE P1) — so a `TOTAL_BUDGET` fence cannot stop spend it was never told about; the honest consequence is loss of finality, not a guessed charge.
**The nanny model.** An `agent_session` subagent is an ordinary recursive task-tree child acting as a **nanny** supervising at most one active bounded external leaf. The task node keeps lineage, authority, deadline, budget, acceptance, cancellation, and descendants; the harness process remains a non-recursive tool leaf — session rows never flatten the task tree into harness processes. The nanny is the host: verification receipts stay host-authored, and harness output is a claim to check, never proof. Harness-agnostic by construction: the row holds an opaque Claudexor target, Ouroboros asks for an access profile derived from task authority and lets Claudexor choose the mechanism; no harness-name branch selects a capability or fallback in core dispatch, and login-wire asymmetries stay presentation adapters in `gateway/claudexor_accounts.py`.
**The nanny model.** An `agent_session` subagent is an ordinary recursive task-tree child acting as a **nanny** supervising at most one active bounded external leaf. The task node keeps lineage, authority, deadline, budget, acceptance, cancellation, and descendants; the harness process remains a non-recursive tool leaf — session rows never flatten the task tree into harness processes. The nanny is the host: verification receipts stay host-authored, and harness output is a claim to check, never proof. A nanny-to-nanny chain through `schedule_subagent` is the host-attested realization of a nested subscription swarm, each level one metered supervisor task plus one free harness run; `schedule_subagent.requested_depth` is the typed ABSOLUTE request counted from the root, recorded as telemetry that never narrows the configured caps, while the legacy `depth_remaining` envelope keeps its narrowing semantics — two semantics, both disclosed on the contract, and the root's `swarm_efficiency.depth` block plus its terminal summary row report requested, permitted and achieved with the typed status. Harness-agnostic by construction: the row holds an opaque Claudexor target, Ouroboros asks for an access profile derived from task authority and lets Claudexor choose the mechanism; no harness-name branch selects a capability or fallback in core dispatch, and login-wire asymmetries stay presentation adapters in `gateway/claudexor_accounts.py`.
**Transport.** `gateways/claudexor.py` is pure transport (descriptor read, `/v2` handshake, `config.CLAUDEXOR_MIN_VERSION` 3.2.0 floor). The daemon bearer token grants the entire `/v2` surface, so it never leaves this module, and the HTTP client runs `trust_env=False` so an ambient proxy cannot intercept the loopback control plane. Production starts obtain a handshaken owned gateway from `claudexor_daemon.ensure_owned_gateway` (exact reviewed engine/Node pins; never a PATH install); keeping lifecycle above transport keeps account status side-effect-free and harness mechanisms out of Ouroboros (`claudexor_daemon.py`/`claudexor_runtime.py` docstrings).
@ -1540,7 +1540,7 @@ otherwise the view is partial and the consumer remains non-final or abstains.
| Surface | Canonical source and owner | Bounded projection | Actor-readable source/ref | Decision and retention rule |
|---|---|---|---|---|
| Owner authority and biography | Canonical `logs/chat.jsonl`, archive generations, and `memory/dialogue_blocks.json` owned by the canonical drive | Main/Project context sections and archive-aware history windows | Existing `chat_history`/archive readers with generation and gap metadata | A known gap is disclosed; summaries/blocks never replace exact current owner directives. Raw generations and durable blocks follow their existing retention owner. |
| Execution evidence | Task results, observability call manifests/blobs, service logs, and process-custody records | Status cards, terminal rows, bounded tails, and compact child summaries | Exact artifact/blob/service-log refs carried by the task result or canonical promotion | A projection cannot certify a missing child/source. Referenced canonical artifacts are promoted before child-drive GC; disposable execution scratch follows unified GC. |
| Execution evidence | Task results, observability call manifests/blobs, service logs, and process-custody records | Status cards, terminal rows, bounded tails, and compact child summaries | Exact artifact/blob/service-log refs carried by the task result or canonical promotion | A projection cannot certify a missing child/source. Referenced canonical artifacts are promoted before child-drive GC; disposable execution scratch follows unified GC. An omitted-to-artifact verification ledger stub carries only its re-projected `summary`; entries and axes are read from the artifact file it points at. |
| Terminal task/project memory | Root terminal result plus existing task/project summary producers | Cognitive Main terminal summaries and the two Project-root UI lifecycle rows (started + terminal completion) | Task-result ID, project binding, and summary/source refs | Summary is a biography projection, not raw evidence. Terminal outcomes, including failed/cancelled/degraded, remain retained through their canonical result owner. |
| Background Consciousness observations | `data/state/consciousness_observations.jsonl`, append-only enqueue/ACK rows owned by `BackgroundConsciousness` | Pending count/oldest metadata and a bounded recent observation rendering | `read_file(root='runtime_data', path='state/consciousness_observations.jsonl')` | Unacknowledged rows survive restart/overflow/error. Gaps block ACK and the existing direct identity rewrite; only a settled successful cycle appends ACK. |
| Plan/review authority | Exact task-artifact/observability wave bodies, evidence selectors, reviewer route/thread receipts, and the bounded review hot index | Review status, latest wave, obligations, and compact findings | Exact artifact/source handle plus SHA/range/thread selectors | Missing or partial evidence is `DEGRADED`/`NOT_RUN`, never PASS. Exact artifacts remain bound to the reviewed candidate SHA; hot indexes may rotate only after the source is retained. |

View file

@ -176,7 +176,9 @@ Status, owner action, and urgent notification are separate product concepts:
- **Status** states a fact about the affected object. It does not imply that the
owner can or must act. Task status uses one factual word family: `Working`,
`Done`, `Done with warnings`, `Failed`, `Cancelled`.
`Done`, `Done with warnings`, `Failed`, `Cancelled`. The same five words are
the host's durable label vocabulary for its own task rows in Main and the
Project thread (`OUTCOME_PHASE_HEADLINE`), not only the browser's.
- **Owner action** exists only when the responsible domain exposes a current
concrete continuation, such as Resume, Retry, Connect, Repair, Grant access,
or Restart. The action is a real adjacent control; severity alone never

View file

@ -1320,7 +1320,11 @@ both critical. The imperatives:
`integrate_subagent_patch` and runs its own `commit_reviewed`. The shared
`external_workspace` surface verifies and records without re-applying; a
genesis project is durable because the project directory IS the
deliverable. The canonical/replica terminal field-custody projection is
deliverable. A genesis project starts without a `.gitignore`, so its small
text build output (`dist/`, `build/`) rides the `workspace.patch` record
until the project declares one — a disclosed residual, bounded only by the
per-file size cap and git's binary verdict, since there is no total-patch
cap. The canonical/replica terminal field-custody projection is
ONE pure reducer reused by copy-back and effective reads — every change
adds a stale-replica regression at BOTH seams
(`tests/test_available_subagents_runtime_review_fixes.py`). Do not broaden
@ -1719,7 +1723,9 @@ by "Provider Independence" above. Call-site imperatives:
Binding the complete-result SHA-256 means a parent cannot claim it
integrated a result that later changed. `deferred` suppresses only the
reminder and forces an honest degraded/best-effort terminal answer until
resolved. A child wedged in the legacy `cancel_requested` latch is intent,
resolved. That per-value consequence is carried by the
`tree_note` payload schema itself, which is the SSOT for when to choose each
value, and its enum reads the validator's own set. A child wedged in the legacy `cancel_requested` latch is intent,
not outcome — it stays visible as cancel-pending until custody settles it.
- Host task acceptance is root-only; eligibility uses structured facts
(`outcomes.turn_has_reviewable_effects` plus a typed
@ -1753,9 +1759,11 @@ by "Provider Independence" above. Call-site imperatives:
resulting pair stays outside `_ACCEPTANCE_BLOCKED_TERMINAL_REASONS`.
The reviewer verdict vocabulary `PASS|FAIL|DEGRADED` is NOT narrowable;
`adaptive_quorum` applies, any contributing FAIL fails, DEGRADED abstains,
and no quorum is a terminal HOST decision. Chat and Logs use the same
severity reducer; degraded review or a best-effort objective must never
render as green solved. Do not add task scope review or reuse the commit
and no quorum is a terminal HOST decision. Chat, Logs, the
durable Project lifecycle rows and the Telegram notifier use the same phase
(`taskOutcomeSeverity`/`taskTerminalPhase` mirrored by
`project_dialogue.outcome_phase`); degraded review or a best-effort
objective must never render as green solved on any of them. Do not add task scope review or reuse the commit
gate.
- The acceptance improvement loop is a reviewer-authored DIALOGUE: obligation
identity comes from the reviewer's typed
@ -2009,7 +2017,14 @@ rules, not a copied color/radius/dimension inventory.
- Task outcome truth stays in `log_events.js::taskOutcomeSeverity` and
`taskTerminalPhase`; `taskPresentation` is the one compact factual
projection consumed by chips, live completion, history replay, and child
terminal presentation. A non-terminal diagnostic may add a timeline fact
terminal presentation. Its host mirror is
`project_dialogue.outcome_phase`, pinned to the browser by one shared
fixture (`web/tests/fixtures/outcome_phase_parity.json`): a new axis, reason
or acceptance status is added to both sides in the same commit, with a row
in that fixture. The detail line under the headline comes from
`taskReasonDetail` in the order soft stop, hard failure or cancellation
reason, host acceptance decision (status plus stored rationale), typed
reason phrase or raw code — never from a second producer. A non-terminal diagnostic may add a timeline fact
but must not promote the whole task; unknown event names never acquire
Chat severity from `error`/`crash`/`fail` keyword matching. The Chat
header reports connection and server-authoritative activity only; failed

View file

@ -1207,7 +1207,7 @@ def _run_task_summary(env, llm, task, usage, llm_trace, drive_logs, review_evide
sealed_final=None):
"""Generate a detailed task summary and inject it into chat.jsonl."""
try:
from ouroboros.project_dialogue import append_authored_task_summary, completion_status_label
from ouroboros.project_dialogue import append_authored_task_summary, completion_status_label, outcome_phase
from ouroboros.projects_registry import project_thread_note_for_task
from ouroboros.consolidator import CONSOLIDATION_REASONING_EFFORT, _consolidation_route
task_id = str(task.get("id") or "unknown")
@ -1230,7 +1230,7 @@ def _run_task_summary(env, llm, task, usage, llm_trace, drive_logs, review_evide
"summary_kind": "authored_root_summary", "summary_id": summary_id,
"task_id": task_id, "parent_task_id": str(task.get("parent_task_id") or ""), "root_task_id": str(task.get("root_task_id") or task_id),
"project_id": str(task.get("project_id") or ""), "chat_id": int(task.get("chat_id") or 0), "delegation_role": str(task.get("delegation_role") or ""), "role": str(task.get("role") or ""),
"status": str(stored_result.get("status") or "completed"), "outcome": completion_status_label(stored_result, usage),
"status": str(stored_result.get("status") or "completed"), "outcome": completion_status_label(stored_result, usage), "outcome_phase": outcome_phase(stored_result, usage),
"outcome_final": False, "outcome_authority": "pre_finalization_narrative_context",
"text": value, "tool_calls": n_tool_calls, "rounds": rounds, "outcome_axes": outcome_axes, "reason_code": reason_code,
"result_ref": result_ref, "source_coverage": {"task_result": result_ref}, **_summary_row_cost_fields(usage), **presence_fields,

View file

@ -82,11 +82,11 @@ def build_depth_summary(
statuses = [row for row in subtree_statuses if isinstance(row, dict)]
provenances = [task_depth_provenance(row) for row in statuses]
provenances = [row for row in provenances if row]
if requested is None:
requested = next(
(row.get("requested_depth") for row in provenances if row.get("requested_depth") is not None),
None,
)
requested_values = [
value for value in [requested, *(row.get("requested_depth") for row in provenances)]
if value is not None
]
requested = max(requested_values) if requested_values else None
permitted_values = [
value
for value in [
@ -97,15 +97,57 @@ def build_depth_summary(
]
permitted = min(permitted_values) if permitted_values else None
def _maximum(key: str) -> Any:
values = [row.get(key) for row in provenances if row.get(key) is not None]
def _maximum(key: str, rows: list = provenances) -> Any:
values = [row.get(key) for row in rows if row.get(key) is not None]
if values:
return max(values)
return 0 if not statuses else None
attempted = _maximum("attempted_depth")
achieved = _maximum("achieved_depth")
if requested is None:
# Decision: rows are per-task depth facts of one tree. Rows sharing one
# (request, permitted) pair form a chain whose achievement is its deepest
# row; a tree with several chains reports the most-reduced chain first
# (capability_reduced > evidence_unknown > chosen_shallower > achieved), so
# the verdict and its coherent chain tuple never depend on child order.
def _chain_status(ask: Any, cap: Any, rows: list) -> str:
if ask is None:
return "request_unknown"
reached = [row.get("achieved_depth") for row in rows]
tried = [row.get("attempted_depth") for row in rows]
if cap is not None and cap < ask:
return "capability_reduced"
known = [value for value in reached if value is not None]
if known and max(known) >= ask:
return "achieved"
if cap is None or None in reached or None in tried:
return "evidence_unknown"
return "chosen_shallower"
chains: Dict[Any, list] = {}
for row in provenances:
chains.setdefault((row.get("requested_depth"), row.get("permitted_depth")), []).append(row)
if chains:
order = ["request_unknown", "capability_reduced", "evidence_unknown", "chosen_shallower", "achieved"]
decisions = [
(
_chain_status(ask, cap, rows), ask, cap,
_maximum("attempted_depth", rows), _maximum("achieved_depth", rows),
)
for (ask, cap), rows in chains.items()
]
status, requested, permitted, attempted, achieved = min(
decisions,
key=lambda item: (
order.index(item[0]),
-(item[1] if item[1] is not None else -1),
item[4] if item[4] is not None else -1,
item[2] if item[2] is not None else -1,
item[3] if item[3] is not None else -1,
),
)
elif requested is None:
status = "request_unknown"
elif permitted is None or attempted is None or achieved is None:
status = "evidence_unknown"

View file

@ -484,11 +484,9 @@ def _copy_task_summary_metadata(rec: Dict[str, Any], entry: Dict[str, Any]) -> N
rec["reason_code"] = str(entry.get("reason_code") or "")
if isinstance(entry.get("review_projection"), dict):
rec["review_projection"] = dict(entry.get("review_projection") or {})
# v6.82 P1: the summary row now carries the flat task-scope cost snapshot
# written by agent_task_pipeline; replay it so a reload still shows cost.
# _annotate_terminal_task_truth later OVERRIDES these with the persisted
# task_results values when the result file survives (row = fallback only).
for key in _TASK_COST_META_FIELDS:
# The summary row is the pruned-result fallback for cost and finality.
# Persisted task_results values still override cost fields below.
for key in ("outcome_phase", "outcome_final", *_TASK_COST_META_FIELDS):
if key in entry:
rec[key] = entry[key]
@ -546,6 +544,7 @@ def _annotate_terminal_task_truth(
the pre-computed set — zero extra reads."""
try:
from ouroboros.project_dialogue import outcome_phase
from ouroboros.task_status import FINAL_STATUSES
cache = result_cache if result_cache is not None else {}
@ -599,6 +598,7 @@ def _annotate_terminal_task_truth(
terminal_status_by_task[task_id] = status
terminal_truth: Dict[str, Any] = {
"outcome_axes": normalize_outcome_axes(result),
"outcome_phase": outcome_phase(result, {}), "outcome_final": True,
}
if result.get("reason_code"):
terminal_truth["reason_code"] = str(result.get("reason_code") or "")

View file

@ -832,10 +832,16 @@ def finalize_task_artifacts(parent_drive_root: pathlib.Path, task: Dict[str, Any
item["size"] = len(data)
item["sha256"] = sha256(data).hexdigest()
item["status"] = ARTIFACT_STATUS_READY
stub = fields.get("verification_ledger")
if isinstance(stub, dict) and stub.get("omitted_to_artifact") and isinstance(refreshed_artifact_ledger.get("summary"), dict):
fields["verification_ledger"] = {**stub, "summary": dict(refreshed_artifact_ledger["summary"])}
except Exception:
log.debug("Failed to refresh verification ledger artifact for task %s", task_id, exc_info=True)
# The loop rewrote the ledger file in place, so the bundle computed
# before it no longer describes the bytes on disk.
fields["artifact_bundle"] = artifact_bundle_from_result(provisional)
except Exception:
pass
log.warning("Artifact bundle/ledger refresh failed for task %s", task_id, exc_info=True)
write_task_result(
parent_drive_root,
task_id,

View file

@ -68,6 +68,7 @@ from ouroboros.usage_accounting import (
last_physical_attempt_capture,
)
from ouroboros.task_finalization import (
TERMINAL_ORIGIN_HOST_NOTICE,
TERMINAL_ORIGIN_HOST_SALVAGE,
TERMINAL_ORIGIN_MODEL_FINAL,
)
@ -2701,17 +2702,16 @@ def _handle_provider_unavailable(
ctx: _RoundLimitContext, *, error_kind: str = "provider_unavailable",
wait_cause: str = "", waited: bool = False, wait_eligible: bool = True,
) -> Tuple[str, Dict[str, Any], Dict[str, Any]]:
"""Provider-death rail wrapper: every arm carries a terminal provenance;
``setdefault`` keeps explicit stamps authoritative. Not every exit is a
provider death: the deadline grace arm can return a MODEL-AUTHORED final
(``deadline_local``) and a scheduled swarm handoff pops its reason code —
both keep their legacy shape."""
"""Provider-death rail wrapper: every arm carries terminal provenance.
The forced-finalization sink stamps ``host_notice``; retained/generated
model candidates stamp ``model_final``."""
text, usage, llm_trace = _provider_unavailable_result(
ctx, error_kind=error_kind, wait_cause=wait_cause, waited=waited,
wait_eligible=wait_eligible,
)
if str(usage.get("reason_code") or "") not in ("", "deadline_local"):
usage.setdefault("terminal_origin", TERMINAL_ORIGIN_HOST_SALVAGE)
if str(usage.get("reason_code") or "") not in ("", "deadline_local") and usage.get(
"terminal_origin") in (None, TERMINAL_ORIGIN_HOST_NOTICE):
usage["terminal_origin"] = TERMINAL_ORIGIN_HOST_SALVAGE
return text, usage, llm_trace
@ -2759,9 +2759,7 @@ def _provider_unavailable_result(
)
return text, usage, llm_trace
if is_transport_wait:
# No-resend terminal: salvage, no forced-final call over a dead
# egress. Stamp BEFORE the composer (owner-stop pattern): a SCHEDULED
# swarm handoff clears it; guard mirrors the sibling below.
# No-resend terminal over dead egress.
live_trace = getattr(ctx, "llm_trace", None)
llm_trace = live_trace if isinstance(live_trace, dict) else {}
ctx.accumulated_usage["execution_status"] = RESULT_INFRA_FAILED
@ -2773,7 +2771,6 @@ def _provider_unavailable_result(
if str(usage.get("reason_code") or "") == "provider_unavailable":
usage["execution_status"] = RESULT_INFRA_FAILED
return text, usage, llm_trace
# No-call shapes; see provider_no_call_source
no_call, wall = provider_no_call_source(ctx.accumulated_usage, is_deadline_exhausted)
if no_call:
if wall:
@ -3404,6 +3401,7 @@ def _record_forced_finalization(
# answer (`_forced_final_answer`) and the no-spend host-fallback fence
# path (`_handle_budget_exceeded` -> `_forced_fallback_result`).
_record_forced_acceptance_bypass(ctx, llm_trace, reason_code)
ctx.accumulated_usage.setdefault("terminal_origin", TERMINAL_ORIGIN_HOST_NOTICE)
binding = dict(candidate.acceptance_binding or {}) if candidate is not None else {}
tools = getattr(ctx, "tools", None)
current_fingerprint = str(
@ -4429,8 +4427,7 @@ def _forced_fallback_result(
candidate_reason: str = "",
provider_terminal: bool = False,
) -> Tuple[str, Dict[str, Any], Dict[str, Any]]:
"""Return a current or fallback candidate."""
"""Compose fallback."""
router_result = _forced_swarm_router_result(ctx, llm_trace, reason_code)
if router_result is not None:
return router_result
@ -4451,11 +4448,10 @@ def _forced_fallback_result(
candidate.model_text or candidate.full_text if provider_terminal else
sanitize_tool_result_for_log(_compose_delivery_suffix(candidate.full_text, suffix))
)
if provider_terminal:
ctx.accumulated_usage.update(
terminal_origin=TERMINAL_ORIGIN_MODEL_FINAL,
terminal_plan_review_open=bool(plan_suffix),
)
ctx.accumulated_usage.update(
terminal_origin=TERMINAL_ORIGIN_MODEL_FINAL,
terminal_plan_review_open=bool(plan_suffix),
)
if composed != candidate.full_text:
candidate = _publish_model_forced_candidate(
ctx, llm_trace, composed, reason_code,
@ -4757,13 +4753,11 @@ def _forced_final_answer(
_force_plan_disclosure(tools_ctx, llm_trace, forced_reason=reason_code)
if tools_ctx is not None else ""
)
if provider_terminal:
ctx.accumulated_usage["terminal_plan_review_open"] = bool(plan_suffix)
ctx.accumulated_usage["terminal_plan_review_open"] = bool(plan_suffix)
full_text = extracted if provider_terminal else _compose_delivery_suffix(
extracted, plan_suffix + _forced_orphan_note(ctx),
)
if provider_terminal:
ctx.accumulated_usage["terminal_origin"] = TERMINAL_ORIGIN_MODEL_FINAL
ctx.accumulated_usage["terminal_origin"] = TERMINAL_ORIGIN_MODEL_FINAL
candidate = _publish_model_forced_candidate(
ctx, llm_trace, full_text, reason_code,
degraded_reason=degraded,

View file

@ -1376,6 +1376,11 @@ def refresh_verification_ledger_artifacts(
if not isinstance(ledger, dict):
return ledger
# An omitted-to-artifact stub is a PROJECTION of the artifact file, not a
# source: it carries no entries, so rebuilding from it would mint "0
# entries / no failures / execution ok" over the real ledger's summary.
if ledger.get("omitted_to_artifact"):
return ledger
entries = [
item for item in (ledger.get("entries") or [])
if not (isinstance(item, dict) and item.get("kind") == "artifact_bundle")

View file

@ -419,42 +419,51 @@ def routing_option_label(option: Any) -> str:
return "Project" if option.get("project_id") and not option.get("task_id") else "Task"
def completion_status_label(result: Dict[str, Any], event: Dict[str, Any]) -> str:
from ouroboros.task_results import (
STATUS_CANCELLED, STATUS_COMPLETED, STATUS_FAILED, STATUS_REJECTED_DUPLICATE,
)
OUTCOME_PHASE_HEADLINE = {"working": "Working", "done": "Done", "warn": "Done with warnings",
"error": "Failed", "cancelled": "Cancelled"}
status = str(result.get("status") or event.get("status") or "").strip().lower()
axes = {}
for source in (event, result):
value = source.get("outcome_axes")
if isinstance(value, dict):
axes.update({key: axis for key, axis in value.items() if isinstance(axis, dict)})
axis_status = {key: str(axis.get("status") or "").lower() for key, axis in axes.items()}
failed = (
status == STATUS_FAILED
or axis_status.get("lifecycle") == STATUS_FAILED
or axis_status.get("execution") in {"failed", "infra_failed"}
or axis_status.get("objective") == "fail"
or axis_status.get("review") == "fail"
or axis_status.get("artifacts") in {"failed", "missing"}
or str(result.get("artifact_status") or event.get("artifact_status") or "").lower()
in {"failed", "missing"}
)
degraded = any(value in {"degraded", "partial", "best_effort"}
for value in axis_status.values())
checkpoint = result.get("root_phase_checkpoint")
degraded |= bool(isinstance(checkpoint, dict)
and str(checkpoint.get("post_task_synthesis") or "").lower() == "degraded")
if status == STATUS_CANCELLED:
return "Cancelled"
if failed:
return "Failed"
if status == STATUS_COMPLETED:
return "Completed with limitations" if degraded else "Completed"
if status == STATUS_REJECTED_DUPLICATE:
return "Not started"
return status.replace("_", " ").title() or "Finished"
def outcome_phase(result: Dict[str, Any], event: Dict[str, Any]) -> str:
"""The host mirror of the browser's terminality gate and severity fold.
Durable host rows read exactly what ``log_events.js`` paints, over
NORMALIZED axes; web/tests/fixtures/outcome_phase_parity.json pins both.
"""
from ouroboros.outcomes import REASON_OWNER_REQUESTED_FINALIZATION, normalize_outcome_axes
from ouroboros.post_task_checkpoint import post_task_synthesis_is_open
from ouroboros.task_status import FINAL_STATUSES
record = {**event, **{key: value for key, value in result.items() if value not in (None, "")}}
sources = [s.get("outcome_axes") for s in (event, result) if isinstance(s.get("outcome_axes"), dict)]
record["outcome_axes"] = {k: v for source in sources for k, v in source.items() if isinstance(v, dict)}
axes = normalize_outcome_axes(record)
axis = {k: str(v.get("status") or "").lower() for k, v in axes.items() if isinstance(v, dict)}
status = str(record.get("task_terminal_status") or record.get("status") or "").strip().lower()
checkpoint = record.get("root_phase_checkpoint")
synthesis = checkpoint.get("post_task_synthesis") if isinstance(checkpoint, dict) else ""
if not (status in {"done", "cancel_requested"} or (status in FINAL_STATUSES and not (
status == "completed" and post_task_synthesis_is_open(synthesis)))):
return "working"
lifecycle = axis.get("lifecycle") or status
if lifecycle in {"cancelled", "cancel_requested"}:
return "cancelled"
if (lifecycle == "failed" or axis.get("execution") in {"failed", "infra_failed"}
or axis.get("objective") == "fail" or axis.get("review") == "fail"
or {axis.get("artifacts"), str(record.get("artifact_status") or "").lower()} & {"failed", "missing"}):
return "error"
if str(record.get("reason_code") or "") == REASON_OWNER_REQUESTED_FINALIZATION:
return "done"
if (lifecycle == "rejected_duplicate" or bool((axes.get("objective") or {}).get("warning"))
or axis.get("execution") in {"degraded", "best_effort"}
or axis.get("objective") in {"degraded", "best_effort"}
or axis.get("review") == "degraded"):
return "warn"
return "done"
def completion_status_label(result: Dict[str, Any], event: Dict[str, Any]) -> str:
"""The one owner-visible status word for a host-authored task row."""
return OUTCOME_PHASE_HEADLINE[outcome_phase(result, event)]
def append_canonical_task_summary(drive_root: Any, row: Dict[str, Any]) -> bool:
@ -711,7 +720,8 @@ def _append_terminal_task_projection(
project_id = resolve_project_id({**task, **effective})
role = str(effective.get("role") or task.get("role") or ("root" if is_root else "child"))
reason = str(effective.get("reason_code") or event.get("reason_code") or "")
outcome = completion_status_label(effective, event)
phase = outcome_phase(effective, event)
outcome = OUTCOME_PHASE_HEADLINE[phase]
excerpt = _completion_excerpt(effective)
details = f'Details: get_task_result(task_id="{tid}")'
text = (
@ -720,8 +730,12 @@ def _append_terminal_task_projection(
)
if excerpt:
text += f" {excerpt}"
if reason:
text += f" Reason: {reason}."
verdict = _completion_verdict(effective, event)
if verdict:
text += f" {verdict}"
depth = (effective.get("swarm_efficiency") or {}).get("depth") if isinstance(effective.get("swarm_efficiency"), dict) else None
if isinstance(depth, dict) and depth.get("requested_depth") is not None:
text += f" Depth requested={depth['requested_depth']}, permitted={depth.get('permitted_depth')}, achieved={depth.get('achieved_depth')} ({depth.get('status')})."
result_ref = {"kind": "task_result", "task_id": tid, "reader": "get_task_result"}
row = {
"ts": str(event.get("ts") or effective.get("ts") or utc_now_iso()),
@ -732,7 +746,7 @@ def _append_terminal_task_projection(
"chat_id": int(event.get("chat_id") or task.get("chat_id") or 0),
"delegation_role": str(effective.get("delegation_role") or task.get("delegation_role") or ""),
"role": role, "status": str(effective.get("status") or status),
"outcome": outcome, "outcome_final": True,
"outcome": outcome, "outcome_phase": phase, "outcome_final": True,
"outcome_authority": "canonical_task_result_after_finalization",
"outcome_axes": effective.get("outcome_axes") or event.get("outcome_axes") or {},
"reason_code": reason, "result_ref": result_ref,
@ -787,6 +801,37 @@ def _completion_excerpt(result: Dict[str, Any]) -> str:
return ""
def _completion_verdict(result: Dict[str, Any], event: Dict[str, Any]) -> str:
"""One TERMINATED host clause for BOTH lifecycle rows.
A host row must not present an unaccepted claim as the whole story: a
non-accepted decision leads with its full upstream-bounded rationale;
otherwise the execution reason stands. The Python twin of
``taskReasonDetail``'s acceptance branch; callers add no punctuation.
"""
from ouroboros.outcomes import ACCEPTANCE_ACCEPTED, REASON_OWNER_REQUESTED_FINALIZATION
decision: Dict[str, Any] = {}
for source in (event, result):
axes = source.get("outcome_axes") if isinstance(source.get("outcome_axes"), dict) else {}
for holder in (source.get("review_status"), axes.get("review")):
if isinstance(holder, dict) and isinstance(holder.get("acceptance_decision"), dict):
decision = holder["acceptance_decision"]
status = str(decision.get("status") or "").strip()
reason = str(result.get("reason_code") or event.get("reason_code") or "")
if (reason != REASON_OWNER_REQUESTED_FINALIZATION and status != ACCEPTANCE_ACCEPTED
and status and outcome_phase(result, event) in {"done", "warn"}):
clause = f"Acceptance: {status}"
rationale = " ".join(strip_markdown(str(decision.get("rationale") or "")).split())
if rationale:
clause += " — " + rationale
elif reason and reason != REASON_OWNER_REQUESTED_FINALIZATION:
clause = f"Reason: {reason}"
else:
return ""
return clause if clause.endswith((".", "!", "?", "…")) else clause + "."
def _run_lives_in_its_project(
drive_root: Any, task_id: str, project_id: str, task: Dict[str, Any], result: Dict[str, Any],
) -> bool:
@ -867,11 +912,13 @@ def enqueue_project_completion_summary(
# a Main row leading into an empty room.
return False
excerpt = _completion_excerpt(result)
verdict = _completion_verdict(result, task_done_event)
lead = f"{verdict} " if verdict else ""
event = {
"type": "send_message", "chat_id": 1, "task_id": tid,
"text": (f"{snapshot['target_label']} · "
f"{completion_status_label(result, task_done_event)}\n"
f"{excerpt or 'Open the Project for details.'}"),
f"{lead}{excerpt or 'Open the Project for details.'}"),
"role": "system", "system_type": "project_completion_summary",
"delivery_id": f"project-completion:{tid}",
"progress_meta": {
@ -942,6 +989,7 @@ __all__ = [
"latest_chat_annotations",
"enqueue_project_completion_summary",
"completion_status_label",
"outcome_phase",
"owner_message_ref_is_valid",
"project_origin_rows",
"project_recent_dialogue",

View file

@ -41,6 +41,17 @@ _SEALED_FINAL_TEXT_PROMPT_CHARS = 4000
# never be inferred from result text or lifecycle status.
TERMINAL_ORIGIN_MODEL_FINAL = "model_final"
TERMINAL_ORIGIN_HOST_SALVAGE = "host_salvage"
# A terminal text the HOST wrote alone (a budget rejection, a round-limit rail
# with nothing to deliver, a scheduled swarm handoff). It is not salvage: its
# own words ARE the answer, so they are published verbatim on every transport
# instead of being replaced by the outage receipt.
TERMINAL_ORIGIN_HOST_NOTICE = "host_notice"
HOST_AUTHORED_TERMINAL_ORIGINS = frozenset({
TERMINAL_ORIGIN_HOST_SALVAGE, TERMINAL_ORIGIN_HOST_NOTICE,
})
_STAMPED_TERMINAL_ORIGINS = frozenset({
TERMINAL_ORIGIN_MODEL_FINAL, *HOST_AUTHORED_TERMINAL_ORIGINS,
})
TERMINAL_PLAN_REVIEW_NOTE = (
"Plan review was still open when the outage forced finalization; "
"its details remain in the task."
@ -98,9 +109,7 @@ def prepare_terminal_send_event(
# branch (supervisor/workers.py stamps task_terminal_status="failed")
# and lets the live concludesTurn gate settle the activity.
send_event.setdefault("progress_meta", {})["task_terminal_status"] = "completed"
if ephemeral or presence or origin not in {
TERMINAL_ORIGIN_MODEL_FINAL, TERMINAL_ORIGIN_HOST_SALVAGE,
}:
if ephemeral or presence or origin not in _STAMPED_TERMINAL_ORIGINS:
return send_event
canonical_root = pathlib.Path(task.get("budget_drive_root") or env_drive_root)
preserved_path = ""
@ -126,7 +135,7 @@ def terminal_result_fields(usage: Dict[str, Any]) -> Dict[str, Any]:
"""Additive durable origin/full-copy fields; unknown producers stay legacy."""
fields: Dict[str, Any] = {}
origin = str(usage.get("terminal_origin") or "")
if origin in {TERMINAL_ORIGIN_MODEL_FINAL, TERMINAL_ORIGIN_HOST_SALVAGE}:
if origin in _STAMPED_TERMINAL_ORIGINS:
fields["terminal_origin"] = origin
path = str(usage.get("terminal_salvage_path") or "")
if path:
@ -388,6 +397,25 @@ def build_swarm_efficiency(env: Any, task: Dict[str, Any]) -> Dict[str, Any] | N
"inter_wave_latency_sec_total": round(inter_wave_latency_total, 3),
"lanes_requested": lanes,
}
try:
# Depth is the one swarm fact the root could not see: its own
# contract carries the request, the subtree carries what was
# actually reached. Its own try/except — the enclosing one returns
# None for the WHOLE rollup, and a subtree read failure must not
# erase the fan-out numbers.
from ouroboros.depth_evidence import build_depth_summary
from ouroboros.task_status import find_child_tasks
canonical = pathlib.Path(task.get("budget_drive_root") or drive_root)
rollup["depth"] = build_depth_summary(
task.get("task_contract"),
find_child_tasks(
canonical, parent_task_id=task_id, root_task_id=task_id,
scope="subtree", materialize_artifacts=False,
),
)
except Exception:
log.debug("swarm depth summary failed", exc_info=True)
if swarm_intent:
rollup["intent_source"] = "swarm"
# The planned figure under its existing event name — the waves'

View file

@ -426,10 +426,13 @@ def _subtask_outcome_summary(data: Dict[str, Any], receipts: list | None = None)
if isinstance(data.get("artifact_bundle"), dict):
summary["artifact_bundle"] = data.get("artifact_bundle")
if ledger:
# An omitted-to-artifact stub carries no entries; its summary is the
# count authority, and for a full ledger the two always agree.
ledger_summary = ledger.get("summary") if isinstance(ledger.get("summary"), dict) else {}
summary["verification_ledger"] = {
"schema_version": ledger.get("schema_version"),
"summary": ledger.get("summary") if isinstance(ledger.get("summary"), dict) else {},
"entry_count": len(ledger.get("entries") or []) if isinstance(ledger.get("entries"), list) else 0,
"summary": ledger_summary,
"entry_count": ledger_summary.get("entry_count", len(ledger.get("entries") or []) if isinstance(ledger.get("entries"), list) else 0),
}
if receipts:
# W2: bounded per-receipt rows for the FULL single-child handoff ONLY
@ -1704,6 +1707,10 @@ def schedule_subagent_properties() -> Dict[str, Any]:
"may_mutate": {"type": "boolean", "default": False, "description": "Optional: grant this child the intent to spawn MUTATIVE (acting) descendants of its own. Still bounded by the usual mutative-subagent gating and depth/active caps."},
"may_fan_out": {"type": "boolean", "default": True, "description": "Optional: whether this child may spawn MULTIPLE children (a wave). Bounded by the per-root active cap."},
"max_children": {"type": "integer", "default": 0, "description": "Optional soft cap on this child's own direct children (0 = inherit / configured cap)."},
"requested_depth": {
"type": "integer", "default": 0,
"description": "Optional: how deep, counted ABSOLUTELY FROM THE ROOT, you intend this branch to nest (root=0, direct children=1; asking for children, grandchildren and great-grandchildren is 3). Recorded as your attested request and reported back as requested/permitted/achieved on the root result; it never widens or narrows the configured caps. 0 or omitted = no request.",
},
"required_capabilities": {
"type": "array",
"items": {"type": "string", "enum": list(SUBAGENT_CAPABILITIES)},
@ -2014,6 +2021,7 @@ def _schedule_task(ctx: ToolContext, internal: Dict[str, Any] | None = None, /,
may_mutate=may_mutate, may_fan_out=params.get("may_fan_out", True),
max_children=params.get("max_children", 0),
intent_note=params.get("delegation_intent", ""),
requested_depth=params.get("requested_depth", 0),
)
child_contract = _build_child_subagent_contract({

View file

@ -207,8 +207,14 @@ def admitted_depth_cap(parent_contract: Any, live_max_depth: Any) -> int:
def depth_provenance_for_schedule(
parent_budget: Dict[str, Any], *, new_depth: int, max_depth: int,
achieved_depth: Any = None, use_remaining_envelope: bool = False,
requested_depth: Any = None,
) -> Dict[str, Any]:
"""Carry requested/permitted/attempted/achieved depth as additive facts."""
"""Carry requested/permitted/attempted/achieved depth as additive facts.
A request is telemetry, never a cap: only the configured cap and the legacy
remaining envelope narrow what a branch may do, so asking for less than the
cap records the intent without silently binding the descendants to it.
"""
budget = parent_budget if isinstance(parent_budget, dict) else {}
inherited = normalize_depth_provenance(budget.get("depth_provenance"))
requested = inherited.get("requested_depth")
@ -226,11 +232,22 @@ def depth_provenance_for_schedule(
requested = max(0, int(budget.get("depth_remaining")))
except (TypeError, ValueError):
requested = None
if requested is None:
# An explicit request of record on THIS call stays unchanged. A
# non-integer fails soft to "no request recorded": the
# field is telemetry, and refusing the whole schedule over it would
# trade a real capability for a bookkeeping nicety.
try:
explicit = max(0, int(requested_depth or 0))
except (TypeError, ValueError):
explicit = 0
if explicit > 0:
requested = explicit
try:
cap = min(MAX_SUBAGENT_DEPTH_HARD_CAP, max(0, int(max_depth)))
except (TypeError, ValueError):
cap = 0
current_permitted = cap if requested is None else min(cap, max(0, int(requested)))
current_permitted = cap
remaining = budget.get("depth_remaining")
if (
use_remaining_envelope
@ -337,12 +354,7 @@ def stamp_task_assignment_depth(
requested = admitted.get("requested_depth")
permitted = _bounded_permitted_depth(admitted.get("permitted_depth"))
if permitted is None:
permitted = min(
min(MAX_SUBAGENT_DEPTH_HARD_CAP, max(0, int(max_depth))),
min(MAX_SUBAGENT_DEPTH_HARD_CAP, max(0, int(requested)))
if requested is not None
else min(MAX_SUBAGENT_DEPTH_HARD_CAP, max(0, int(max_depth))),
)
permitted = min(MAX_SUBAGENT_DEPTH_HARD_CAP, max(0, int(max_depth)))
provenance = {
"requested_depth": requested,
"permitted_depth": permitted,
@ -418,6 +430,7 @@ def record_depth_limit_refusal(
may_fan_out=params.get("may_fan_out", True),
max_children=params.get("max_children", 0),
intent_note=params.get("delegation_intent", ""),
requested_depth=params.get("requested_depth", 0),
)
from ouroboros.contracts.task_contract import build_task_contract
@ -793,6 +806,7 @@ def child_budget_for_schedule(
may_fan_out: bool,
max_children: int,
intent_note: str,
requested_depth: Any = None,
) -> Dict[str, Any]:
"""Resolve a child's delegation_budget at schedule time (C3.1): decrement
depth_remaining one generation (falling back to the configured max_depth/new_depth
@ -803,7 +817,7 @@ def child_budget_for_schedule(
parent_budget = parent_budget if isinstance(parent_budget, dict) else {}
provenance = depth_provenance_for_schedule(
parent_budget, new_depth=new_depth, max_depth=max_depth,
use_remaining_envelope=True,
use_remaining_envelope=True, requested_depth=requested_depth,
)
permitted_remaining = max(
0, int(provenance.get("permitted_depth") or 0) - int(new_depth),

View file

@ -77,10 +77,13 @@ def _task_record(
record["artifact_bundle"] = data.get("artifact_bundle")
ledger = data.get("verification_ledger") if isinstance(data.get("verification_ledger"), dict) else {}
if ledger:
# An omitted-to-artifact stub carries no entries; its summary is the
# count authority, and for a full ledger the two always agree.
ledger_summary = ledger.get("summary") if isinstance(ledger.get("summary"), dict) else {}
record["verification_ledger"] = {
"schema_version": ledger.get("schema_version"),
"summary": ledger.get("summary") if isinstance(ledger.get("summary"), dict) else {},
"entry_count": len(ledger.get("entries") or []) if isinstance(ledger.get("entries"), list) else 0,
"summary": ledger_summary,
"entry_count": ledger_summary.get("entry_count", len(ledger.get("entries") or []) if isinstance(ledger.get("entries"), list) else 0),
}
if include_results:
record["result"] = result

View file

@ -84,7 +84,24 @@ def _tree_read(
def get_tools() -> List[ToolEntry]:
from ouroboros.task_tree_ledger import DELEGATION_CONSTRAINT_DIRECTIVES, LEDGER_KINDS
from ouroboros.task_tree_ledger import (
CHILD_RESULT_DISPOSITIONS, DELEGATION_CONSTRAINT_DIRECTIVES, LEDGER_KINDS,
)
# The schema is where the model learns WHEN to choose each value, so the
# per-value consequence belongs here rather than in prose elsewhere. Read
# from the validator's own frozenset, sorted so the enum order is stable
# across worker processes (a frozenset's iteration order is not).
disposition_schema = {
"type": "string",
"enum": sorted(CHILD_RESULT_DISPOSITIONS),
"description": (
"integrated = consumed by this answer (item closes); irrelevant = not needed "
"here (the child result stays as evidence; item closes); deferred = STILL OWED: "
"your terminal answer reads degraded/best_effort until you re-disposition this "
"exact child_result_sha256."
),
}
return [
ToolEntry("tree_note", {
@ -125,10 +142,7 @@ def get_tools() -> List[ToolEntry]:
"properties": {
"type": {"type": "string", "enum": ["child_result_disposition"]},
"child_task_id": {"type": "string"},
"disposition": {
"type": "string",
"enum": ["integrated", "irrelevant", "deferred"],
},
"disposition": disposition_schema,
"child_result_sha256": {"type": "string"},
"children": {
"type": "array",
@ -140,10 +154,7 @@ def get_tools() -> List[ToolEntry]:
"type": "object",
"properties": {
"child_task_id": {"type": "string"},
"disposition": {
"type": "string",
"enum": ["integrated", "irrelevant", "deferred"],
},
"disposition": disposition_schema,
"child_result_sha256": {"type": "string"},
},
"required": [

View file

@ -16,13 +16,15 @@ import pathlib
import re
from typing import List
# v6.35.0 (T7): bumped to 2 with binary + size + junk-artifact hygiene so the
# real-usage workspace.patch (consumed by subagents / PR integration) never
# carries a compiled `go build` binary, a Redis dump, or other untracked build
# junk. Kept consistent with the bench capture_patch.sh JUNK_RE + numstat
# binary detection. This is patch-transport hygiene only (artifact path/extension
# + git's own binary verdict), never code/content inference (Bible P5).
_PATCH_EXCLUDE_RULES_VERSION = 2
# Version 3: generated output directories are no longer excluded by NAME. The
# project's own `.gitignore` (honoured through `--exclude-standard`), the 5 MiB
# per-file cap below, git's own binary verdict and the credential-name check
# decide what a workspace.patch may carry; a project whose deliverable IS its
# build output must not have it silently dropped. The benchmark capture script
# owns its own any-depth rule and does not import this module. This is
# patch-transport hygiene only (artifact path/extension + git's binary
# verdict), never code/content inference (Bible P5).
_PATCH_EXCLUDE_RULES_VERSION = 3
_PATCH_MAX_UNTRACKED_FILE_BYTES = 5 * 1024 * 1024 # 5 MiB per untracked file
_TOP_LEVEL_EXCLUDE_DIRS = {".ouroboros", ".venv", "venv", "env"}
_ANY_SEGMENT_EXCLUDE_DIRS = {
@ -37,11 +39,12 @@ _ANY_SEGMENT_EXCLUDE_DIRS = {
"__pycache__",
"node_modules",
}
# Junk file tails / build dirs the dir-sets above don't already cover; the same
# JUNK_RE the bench capture_patch.sh uses (devtools/benchmarks/swe_bench_pro/).
# Junk file tails and caches the dir-sets above don't already cover. Runtime
# dumps, compiled bytecode and coverage output only; generated-output
# directories are the project's own .gitignore decision, not a name rule here.
_PATCH_JUNK_RE = re.compile(
r"appendonlydir|\.rdb$|\.aof$|\.manifest$|\.log$|\.tmp$|\.pid$|\.sock$"
r"|\.pyc$|\.pyo$|^(dist|build)/|\.DS_Store|(^|/)\.coverage$"
r"|\.pyc$|\.pyo$|\.DS_Store|(^|/)\.coverage$"
r"|coverage\.xml$|(^|/)htmlcov/"
)
_LOCKFILE_MANIFESTS = {

View file

@ -137,11 +137,13 @@ def _summary_ids_in_tail(api, limit: int = 200) -> list:
max_entries=limit,
tail_bytes=256 * 1024,
)
ids = []
summaries = {}
for e in rows:
if str(e.get("type") or "") == "task_summary" and e.get("task_id"):
ids.append((str(e.get("task_id")), e))
return ids
if (str(e.get("type") or "") == "task_summary" and e.get("task_id")
and e.get("outcome_final") is not False
and str(e.get("outcome_phase") or "") != "working"):
summaries[str(e.get("task_id"))] = e
return list(summaries.items())
async def _check_tasks_notify(
@ -180,7 +182,14 @@ async def _check_tasks_notify(
if outcome and outcome not in ("completed", "done"):
parts.append(outcome)
tail = (" · " + " · ".join(parts)) if parts else ""
healthy = outcome in ("", "completed", "done") and not degraded
# The host stamps one status phase on its own task rows; consume it
# instead of re-deriving a third status ladder from the axes. Legacy
# rows and pre-finalization "working" rows keep the axes rule.
phase = str(e.get("outcome_phase") or "")
if phase and phase != "working":
healthy = phase == "done"
else:
healthy = outcome in ("", "completed", "done") and not degraded
icon = "✅" if healthy else "⚠️"
msg = (f"{icon} Задача {tid[:8]} готова{tail}" if lang == "ru" else f"{icon} Task {tid[:8]} done{tail}")
send_outcome, exc = await _push_notification(api, chat_id, msg, trust_env=trust_env)

View file

@ -34,6 +34,7 @@ from typing import Any, Dict, List, Optional
from ouroboros.utils import update_json_locked, utc_now_iso
from ouroboros.task_finalization import (
HOST_AUTHORED_TERMINAL_ORIGINS,
TERMINAL_ORIGIN_HOST_SALVAGE,
TERMINAL_ORIGIN_MODEL_FINAL,
)
@ -719,10 +720,13 @@ def project_terminal_result_event(
"""Project one terminal event from producer-stamped origin.
``host_salvage`` becomes one short keyed plain System receipt (inherited
``format``/``log_text`` dropped; the full bytes stay in task details);
``model_final`` and a missing legacy origin keep the assistant projection
untouched — no text/status/length inference. The delivery id always
digests the stable core result rather than a mutable disclosure suffix.
``format``/``log_text`` dropped; the full bytes stay in task details).
``host_notice`` is a text the host wrote alone, so it keeps its OWN words
and inherited markdown and becomes a System row WITHOUT a system_type,
which is what lets a replayed card conclude on it. ``model_final`` and a
missing legacy origin keep the assistant projection untouched — no
text/status/length inference. The delivery id always digests the stable
core result rather than a mutable disclosure suffix.
"""
tid = str(task_id or "")
core_text = str(result_text or "")
@ -732,19 +736,22 @@ def project_terminal_result_event(
event.setdefault("chat_id", lineage_chat_id(drive_root, task or {}, tid))
event["delivery_id"] = delivery_id_for(tid, core_text)
origin = str(terminal_origin or "")
if origin == TERMINAL_ORIGIN_HOST_SALVAGE:
if origin in HOST_AUTHORED_TERMINAL_ORIGINS:
salvage = origin == TERMINAL_ORIGIN_HOST_SALVAGE
event.update({
"text": _HOST_SALVAGE_RECEIPT,
"text": _HOST_SALVAGE_RECEIPT if salvage else (event.get("text") or core_text),
"role": "system",
"system_type": "terminal_incident",
"terminal_origin": TERMINAL_ORIGIN_HOST_SALVAGE,
"terminal_origin": origin,
**({"system_type": "terminal_incident"} if salvage else {}),
})
event.pop("log_text", None)
event.pop("format", None)
if salvage:
event.pop("log_text", None)
event.pop("format", None)
return event
if origin == TERMINAL_ORIGIN_MODEL_FINAL:
event["terminal_origin"] = TERMINAL_ORIGIN_MODEL_FINAL
# Missing origin remains absent so legacy rows replay byte-compatibly.
# A missing origin now identifies only a row written before every forced
# rail stamped one; it stays absent so legacy rows replay byte-compatibly.
return event

View file

@ -1526,3 +1526,22 @@ def test_select_subagent_constraint_read_only_token_is_readonly():
constraint = _select_subagent_constraint(surface, "", False, [], "")
assert isinstance(constraint, dict), surface
assert constraint == baseline, surface
def test_schedule_subagent_publishes_the_depth_request_and_the_handler_accepts_it():
"""A nanny chain could describe its intended nesting only in prose: the tool
refused ``requested_depth`` by name, so the root's own depth summary could
never say what had been asked for."""
from ouroboros.tools import control
props = control.schedule_subagent_properties()
row = props["requested_depth"]
assert row["type"] == "integer"
assert row["default"] == 0
# Absolute from the root, and telemetry rather than a cap: both semantics
# have to be readable by the model that fills the field in.
assert "ABSOLUTELY FROM THE ROOT" in row["description"]
assert "never widens or narrows" in row["description"]
# The handler's closed keyword set DERIVES from the same schema.
assert "requested_depth" in control.schedule_subagent_param_names()
assert "requested_depth" not in control.HIDDEN_LEGACY_SCHEDULE_PARAMS

View file

@ -914,3 +914,25 @@ def test_full_blackboard_never_locks_out_validated_dispositions(tmp_path, monkey
)
assert "ledger is full" not in accepted
assert not accepted.startswith("⚠️")
def test_the_disposition_enum_and_its_cost_come_from_one_place():
"""The enum was hand-written twice with no per-value meaning, although the
consequence of ``deferred`` is real and host-enforced. Both payload sites
now read the validator's own set, in a stable order, and the schema states
what each value costs — the schema is where the model actually reads it."""
from ouroboros.task_tree_ledger import CHILD_RESULT_DISPOSITIONS
from ouroboros.tools import task_tree
entry = next(tool for tool in task_tree.get_tools() if tool.name == "tree_note")
payload = entry.schema["parameters"]["properties"]["payload"]["properties"]
single = payload["disposition"]
batch = payload["children"]["items"]["properties"]["disposition"]
assert single is batch or single == batch
for schema in (single, batch):
assert schema["enum"] == sorted(CHILD_RESULT_DISPOSITIONS)
assert schema["enum"] == ["deferred", "integrated", "irrelevant"]
for value in CHILD_RESULT_DISPOSITIONS:
assert value in schema["description"], value
assert "degraded/best_effort" in schema["description"]

View file

@ -4,6 +4,8 @@ task contract and be surfaced in the child's prompt — instead of being lost in
freeform objective prose (the cyber-racing 'maximum subagents' request that
collapsed into 3 flat research leaves)."""
from types import SimpleNamespace
def test_delegation_budget_defaults_and_normalization():
from ouroboros.contracts.task_contract import build_task_contract, normalize_delegation_budget
@ -216,3 +218,117 @@ def test_child_budget_strict_boolean_parsing():
)
assert child2["may_mutate"] is True
assert child2["may_fan_out"] is True
def test_requested_depth_is_recorded_as_telemetry_and_never_narrows_the_cap():
"""A nanny chain had no way to SAY how deep it meant to nest, so the root's
depth summary read "request unknown" and a shallow outcome could not be told
apart from a reduced capability. The request is now recorded — and recording
it must not become a second cap: asking for less than the configured depth
used to narrow what the descendants were permitted."""
from ouroboros.depth_evidence import build_depth_summary
from ouroboros.tools.control_delegation import child_budget_for_schedule
root = {"delegation_budget": {"may_delegate": True}}
def _schedule(**kwargs):
return child_budget_for_schedule(
root, current_depth=0, new_depth=1, max_depth=4, may_mutate=False,
may_fan_out=True, max_children=0, intent_note="", **kwargs,
)["depth_provenance"]
asked = _schedule(requested_depth=2)
assert asked["requested_depth"] == 2
assert asked["permitted_depth"] == 4, "a request is telemetry, not a cap"
assert asked["attempted_depth"] == 1
assert asked["achieved_depth"] is None
# Absent or zero means no request of record — never a recorded zero.
assert _schedule()["requested_depth"] is None
assert _schedule(requested_depth=0)["requested_depth"] is None
# The immutable host ceiling bounds authority, not the attested request.
over_cap = _schedule(requested_depth=99)
assert over_cap["requested_depth"] == 99
assert over_cap["permitted_depth"] == 4
assert build_depth_summary(
{"delegation_budget": {"depth_provenance": over_cap}}, [],
)["status"] == "capability_reduced"
# A non-integer fails SOFT: the handler validates argument names, not value
# types, and refusing a whole schedule over a telemetry field would trade a
# real capability for bookkeeping.
assert _schedule(requested_depth="x")["requested_depth"] is None
def test_an_inherited_request_of_record_survives_a_nannys_own_number():
from ouroboros.tools.control_delegation import child_budget_for_schedule
parent = {"delegation_budget": {
"may_delegate": True, "depth_provenance": {"requested_depth": 3, "permitted_depth": 4},
}}
provenance = child_budget_for_schedule(
parent, current_depth=1, new_depth=2, max_depth=4, may_mutate=False,
may_fan_out=True, max_children=0, intent_note="", requested_depth=1,
)["depth_provenance"]
assert provenance["requested_depth"] == 3
assert provenance["permitted_depth"] == 4
def test_assignment_stamp_without_a_permitted_fact_uses_the_configured_cap():
"""The stamp used to fold the REQUEST into the permitted number, so a branch
that asked for two lost the depth the configuration actually allowed."""
from ouroboros.tools.control_delegation import stamp_task_assignment_depth
task = {
"depth": 1,
"task_contract": {"delegation_budget": {
"depth_provenance": {"requested_depth": 2, "permitted_depth": "not-a-number"},
}},
}
stamped = stamp_task_assignment_depth(task, max_depth=4)
provenance = stamped["task_contract"]["delegation_budget"]["depth_provenance"]
assert provenance["requested_depth"] == 2
assert provenance["permitted_depth"] == 4
def test_the_swarm_rollup_reports_requested_permitted_and_achieved_depth(tmp_path):
from ouroboros.task_finalization import build_swarm_efficiency
from ouroboros.task_results import STATUS_COMPLETED, write_task_result
from ouroboros.utils import append_jsonl
root_id = "depth-root"
(tmp_path / "logs").mkdir(parents=True, exist_ok=True)
append_jsonl(tmp_path / "logs" / "events.jsonl", {
"type": "swarm_fanout", "parent_task_id": root_id, "task_ids": ["c1"],
"requested_count": 1, "ts": "2026-09-03T10:00:00Z",
})
write_task_result(
tmp_path, "c1", STATUS_COMPLETED, result="done",
parent_task_id=root_id, root_task_id=root_id, delegation_role="subagent",
depth_provenance={
"requested_depth": 2, "permitted_depth": 4,
"attempted_depth": 1, "achieved_depth": 1,
},
)
rollup = build_swarm_efficiency(
SimpleNamespace(drive_root=tmp_path),
{
"id": root_id, "budget_drive_root": str(tmp_path),
"task_contract": {"delegation_budget": {
"depth_provenance": {"requested_depth": 2, "permitted_depth": 4},
}},
},
)
assert rollup["subagent_count"] == 1
assert rollup["depth"]["requested_depth"] == 2
assert rollup["depth"]["permitted_depth"] == 4
assert rollup["depth"]["achieved_depth"] == 1
assert rollup["depth"]["status"] == "chosen_shallower"
assert rollup["depth"]["host_visible_only"] is True
def test_a_plain_task_gets_no_depth_block(tmp_path):
from ouroboros.task_finalization import build_swarm_efficiency
assert build_swarm_efficiency(
SimpleNamespace(drive_root=tmp_path), {"id": "lonely"},
) is None

View file

@ -48,6 +48,15 @@ def test_architecture_mentions_shared_log_grouping_and_direct_provider_review_fa
assert "claudeRuntimeHasError" not in arch
def test_architecture_limits_finality_and_verdict_claims_to_actual_rows():
arch = _read("docs/ARCHITECTURE.md")
assert "The start row carries neither outcome finality nor a verdict" in arch
assert "The pre-finalization authored row carries the phase with `outcome_final=false`" in arch
assert "only terminal `task_summary` rows append the host verdict clause" in arch
assert "Both Main rows, the Project thread rows" not in arch
def test_architecture_documents_skill_schedule_lifecycle_and_evolution_light_block():
arch = _read("docs/ARCHITECTURE.md")
@ -246,6 +255,8 @@ def test_architecture_mirror_matches_the_split_axes_contracts():
# swarm_efficiency reports the REQUEST: lanes_requested, never lanes_used.
assert "lanes_requested" in arch
# A fanned-out root also reports the depth REQUEST beside those lanes.
assert "`requested_depth`" in arch
assert "lanes_used" not in arch
# Task-group compaction left with the degenerate lane fan-out (v6.87.28).
assert "task-group compaction" not in arch

View file

@ -662,6 +662,67 @@ def test_chat_history_task_summary_row_passes_flat_cost_fields_through(tmp_path)
assert "cost_usd" not in rec
def test_chat_history_replays_task_summary_finality_without_task_result(tmp_path):
logs = tmp_path / "logs"
logs.mkdir()
(logs / "chat.jsonl").write_text(
json.dumps({
"ts": "2026-09-03T00:00:00Z",
"direction": "system",
"type": "task_summary",
"task_id": "open-summary",
"chat_id": 1,
"text": "Narrative written before finalization.",
"outcome_phase": "warn",
"outcome_final": False,
}) + "\n",
encoding="utf-8",
)
(logs / "progress.jsonl").write_text("", encoding="utf-8")
endpoint = make_chat_history_endpoint(tmp_path)
response = asyncio.run(endpoint(SimpleNamespace(query_params={"limit": "10"})))
payload = json.loads(response.body.decode("utf-8"))["messages"]
rec = next(item for item in payload if item.get("task_id") == "open-summary")
assert rec["outcome_phase"] == "warn"
assert rec["outcome_final"] is False
assert "task_terminal_status" not in rec
def test_chat_history_settled_result_overrides_pre_final_summary_finality(tmp_path):
logs = tmp_path / "logs"
logs.mkdir()
(logs / "chat.jsonl").write_text(
json.dumps({
"ts": "2026-09-03T00:00:00Z",
"direction": "system",
"type": "task_summary",
"task_id": "settled-summary",
"chat_id": 1,
"text": "Narrative written before finalization.",
"outcome_phase": "working",
"outcome_final": False,
}) + "\n",
encoding="utf-8",
)
(logs / "progress.jsonl").write_text("", encoding="utf-8")
results = tmp_path / "task_results"
results.mkdir()
(results / "settled-summary.json").write_text(
json.dumps({"task_id": "settled-summary", "status": "completed"}),
encoding="utf-8",
)
endpoint = make_chat_history_endpoint(tmp_path)
response = asyncio.run(endpoint(SimpleNamespace(query_params={"limit": "10"})))
payload = json.loads(response.body.decode("utf-8"))["messages"]
rec = next(item for item in payload if item.get("task_id") == "settled-summary")
assert rec["outcome_phase"] == "done"
assert rec["outcome_final"] is True
def test_chat_history_attaches_terminal_cost_truth_from_task_result(tmp_path):
"""v6.82 P1: a terminal task_results/<id>.json carries the final cost truth;
it is attached to the surviving progress anchor on replay."""

View file

@ -1472,7 +1472,12 @@ def test_workspace_patch_preserves_lockfile_when_other_changes_are_junk(tmp_path
repo.mkdir()
subprocess.run(["git", "init"], cwd=repo, check=True, capture_output=True)
(repo / "README.md").write_text("base\n", encoding="utf-8")
subprocess.run(["git", "add", "README.md"], cwd=repo, check=True, capture_output=True)
# Version 3: generated output is excluded because the PROJECT declares it,
# not because a host name rule guesses. Without the .gitignore the dist file
# would ride the patch and, being a real change beside the lockfile, would
# also stop the lockfile from reading as incidental.
(repo / ".gitignore").write_text("dist/\n", encoding="utf-8")
subprocess.run(["git", "add", "README.md", ".gitignore"], cwd=repo, check=True, capture_output=True)
subprocess.run(
["git", "-c", "user.email=t@example.com", "-c", "user.name=T", "commit", "-m", "init"],
cwd=repo,
@ -1489,7 +1494,9 @@ def test_workspace_patch_preserves_lockfile_when_other_changes_are_junk(tmp_path
assert "package-lock.json" in patch
assert "dist/out.txt" not in patch
assert manifest["counts"]["untracked_included"] == 1
assert manifest["counts"]["untracked_excluded"] == 1
# A git-ignored file is outside the capture universe, so it is not listed
# as an exclusion either.
assert manifest["counts"]["untracked_excluded"] == 0
def test_workspace_patch_excludes_binary_junk_and_oversize(tmp_path, monkeypatch):
@ -1520,7 +1527,7 @@ def test_workspace_patch_excludes_binary_junk_and_oversize(tmp_path, monkeypatch
artifacts, manifest = write_workspace_patch_artifacts(repo, tmp_path / "artifacts", task={})
assert manifest["exclude_rules_version"] == 2
assert manifest["exclude_rules_version"] == 3
excluded = {item["path"]: item["reason"] for item in manifest["untracked_excluded"]}
assert "binary file" in excluded.get("app", "")
assert "binary file" in excluded.get("dump.rdb", "") or "junk artifact" in excluded.get("dump.rdb", "")

View file

@ -18,6 +18,10 @@ def test_navigation_shell_dom_has_sidebar_drawer_and_project_slots():
html = _read("web/index.html")
chat_js = _read("web/modules/chat.js")
assert 'id="primary-sidebar"' in html
# The sidebar is the document's one navigation landmark.
assert '<nav id="primary-sidebar"' in html
assert '<aside id="primary-sidebar"' not in html
assert '<aside id="project-panel"' in html
assert 'data-mobile-nav-toggle' not in html
assert 'id="nav-drawer-backdrop"' in html
assert 'data-nav-page="chat"' in html
@ -58,6 +62,13 @@ def test_navigation_state_and_mobile_drawer_are_first_class():
assert ".nav-drawer-backdrop" in css
assert ".mobile-nav-toggle" in css
assert "grid-template-columns: var(--sidebar-width) minmax(0, 1fr) auto;" in css
# One client-side route, read once on load and validated against the DOM
# rather than a duplicated page list; never written back on navigation
# (the desktop shell and the Telegram mini app have no address bar).
assert "function pageFromHash()" in app_js
assert "document.getElementById(`page-${name}`)" in app_js
assert "hashchange" not in app_js
assert "replaceState" not in app_js
assert ".mobile-nav-toggle {\n position: fixed" not in css
assert "flex: 0 0 44px;" in css

View file

@ -113,7 +113,7 @@ def test_depth_summary_reports_lower_cap_as_typed_reduction(monkeypatch):
assert build_depth_summary(root_contract, statuses) == {
"requested_depth": 3, "permitted_depth": 2,
"attempted_depth": 2, "achieved_depth": 2,
"status": "capability_reduced", "host_visible_only": True,
"status": "capability_reduced", "host_visible_only": True
}
@ -140,24 +140,64 @@ def test_depth_summary_is_order_independent_and_allows_chosen_shallower():
expected = {
"requested_depth": 3, "permitted_depth": 2,
"attempted_depth": 2, "achieved_depth": 2,
"status": "capability_reduced", "host_visible_only": True,
"status": "capability_reduced", "host_visible_only": True
}
assert build_depth_summary(root_contract, mixed) == expected
assert build_depth_summary(root_contract, reversed(mixed)) == expected
assert build_depth_summary(root_contract, [mixed[0]]) == {
"requested_depth": 3, "permitted_depth": 3,
"attempted_depth": 1, "achieved_depth": 1,
"status": "chosen_shallower", "host_visible_only": True,
"status": "chosen_shallower", "host_visible_only": True
}
def test_depth_summary_mixed_requests_uses_strongest_ask_and_branch_status():
mixed = [
{"depth_provenance": {
"requested_depth": 2, "permitted_depth": 2,
"attempted_depth": 2, "achieved_depth": 2,
}},
{"depth_provenance": {
"requested_depth": 4, "permitted_depth": 4,
"attempted_depth": 3, "achieved_depth": 3,
}},
]
expected = {
"requested_depth": 4, "permitted_depth": 4,
"attempted_depth": 3, "achieved_depth": 3,
"status": "chosen_shallower", "host_visible_only": True
}
assert build_depth_summary({}, mixed) == expected
assert build_depth_summary({}, reversed(mixed)) == expected
def test_depth_summary_reduced_chain_decides_over_deeper_achieved_chain():
mixed = [
{"depth_provenance": {
"requested_depth": 5, "permitted_depth": 5,
"attempted_depth": 5, "achieved_depth": 5,
}},
{"depth_provenance": {
"requested_depth": 3, "permitted_depth": 2,
"attempted_depth": 2, "achieved_depth": 2,
}},
]
expected = {
"requested_depth": 3, "permitted_depth": 2,
"attempted_depth": 2, "achieved_depth": 2,
"status": "capability_reduced", "host_visible_only": True,
}
assert build_depth_summary({}, mixed) == expected
assert build_depth_summary({}, reversed(mixed)) == expected
def test_depth_summary_never_recomputes_missing_history_from_live_settings(monkeypatch):
root_contract = build_task_contract({"delegation_budget": {"depth_remaining": 3}})
monkeypatch.setenv("OUROBOROS_MAX_SUBAGENT_DEPTH", "7")
assert build_depth_summary(root_contract, []) == {
"requested_depth": 3, "permitted_depth": None,
"attempted_depth": 0, "achieved_depth": 0,
"status": "evidence_unknown", "host_visible_only": True,
"status": "evidence_unknown", "host_visible_only": True
}

View file

@ -1,6 +1,7 @@
import gzip
import json
import os
import pathlib
from concurrent.futures import ThreadPoolExecutor
from types import SimpleNamespace
@ -1238,3 +1239,76 @@ def test_has_failures_true_for_failed_receipt():
{"status": "ready", "artifacts": [], "errors": []},
)
assert refreshed["summary"]["has_failures"] is True
def test_omitted_ledger_stub_keeps_the_real_summary_across_artifact_finalization(tmp_path):
"""A ledger above the inline threshold rides as a stub with no entries.
Artifact finalization rebuilt the ledger FROM that stub, so a task whose
real ledger held nine entries and a failing verification receipt was
rewritten to "0 entries / no failures / execution ok" on the record the
owner and the parent read — the stub is a projection of the artifact file,
never a source. The stub's summary is now re-projected from the refreshed
file, and the recomputed bundle describes the bytes actually on disk.
"""
from hashlib import sha256
from tests.test_headless_cli import _init_repo_with_file
from ouroboros import headless
from ouroboros.task_results import STATUS_COMPLETED, load_task_result, write_task_result
parent = tmp_path / "data"
parent.mkdir()
repo = tmp_path / "repo"
_init_repo_with_file(repo)
task_id = "stubledger"
ledger = {
"schema_version": 2, "created_at": "now", "task_id": task_id,
"entries": [{"kind": "verification_receipt", "status": "fail", "detail": "x" * 1600}]
+ [{"kind": "objective_outcome", "status": "not_evaluated", "detail": "y" * 1600} for _ in range(8)],
"summary": {"entry_count": 9, "has_failures": True},
}
refs = maybe_write_verification_artifact(parent, task_id, ledger)
assert refs["artifact"] is not None, "the fixture must exceed the inline threshold"
assert refs["inline"]["omitted_to_artifact"] is True
assert "entries" not in refs["inline"]
write_task_result(
parent, task_id, STATUS_COMPLETED, result="done",
verification_ledger=refs["inline"], artifacts=[refs["artifact"]],
artifact_status="ready",
outcome_axes={"lifecycle": {"status": STATUS_COMPLETED}, "execution": {"status": EXECUTION_OK}},
)
for run in ("first", "second"):
headless.finalize_task_artifacts(parent, {"id": task_id, "workspace_root": str(repo)})
stored = load_task_result(parent, task_id)
stub = stored["verification_ledger"]
assert stub["omitted_to_artifact"] is True, run
assert "entries" not in stub, run
assert "outcome_axes" not in stub, run
assert stub["summary"]["entry_count"] == 9, run
assert stub["summary"]["has_failures"] is True, run
ledger_item = next(
item for item in stored["artifact_bundle"]["artifacts"]
if item.get("kind") == "verification_ledger"
)
data = pathlib.Path(ledger_item["path"]).read_bytes()
assert ledger_item["size"] == len(data), run
assert ledger_item["sha256"] == sha256(data).hexdigest(), run
on_disk = json.loads(data.decode("utf-8"))
assert on_disk["summary"]["entry_count"] == 9, run
assert on_disk["summary"]["has_failures"] is True, run
def test_refreshing_an_omitted_ledger_stub_is_an_identity():
stub = {
"schema_version": 1, "created_at": "now", "task_id": "t",
"summary": {"entry_count": 9, "has_failures": True}, "omitted_to_artifact": True,
}
for status in ("ready", "ready_with_changes", "failed", "pending", "missing"):
assert refresh_verification_ledger_artifacts(
dict(stub), {"status": status, "artifacts": [], "errors": []},
) == stub

View file

@ -161,13 +161,13 @@ def test_missing_role_terminal_root_is_never_labeled_child(tmp_path):
def test_terminal_child_projection_is_idempotent_and_honest_for_all_outcomes(tmp_path):
from ouroboros.project_dialogue import append_terminal_task_projection
from ouroboros.project_dialogue import OUTCOME_PHASE_HEADLINE, append_terminal_task_projection
cases = [
("ok", "completed", {"execution": {"status": "ok"}}, "Completed"),
("ok", "completed", {"execution": {"status": "ok"}}, "Done"),
("failed", "failed", {"execution": {"status": "failed"}}, "Failed"),
("cancelled", "cancelled", {"execution": {"status": "ok"}}, "Cancelled"),
("degraded", "completed", {"execution": {"status": "best_effort"}}, "Completed with limitations"),
("degraded", "completed", {"execution": {"status": "best_effort"}}, "Done with warnings"),
]
for suffix, status, axes, label in cases:
task_id = f"child-{suffix}"
@ -196,6 +196,7 @@ def test_terminal_child_projection_is_idempotent_and_honest_for_all_outcomes(tmp
assert row["role"] == "reviewer"
assert row["status"] == status
assert row["outcome"] == label
assert OUTCOME_PHASE_HEADLINE[row["outcome_phase"]] == label
assert row["reason_code"] == f"reason-{suffix}"
assert row["result_ref"] == {
"kind": "task_result", "task_id": task_id, "reader": "get_task_result",
@ -285,7 +286,7 @@ def test_terminal_root_fallback_covers_cancel_without_preempting_open_synthesis(
)
normal = next(row for row in _chat_rows(tmp_path) if row.get("task_id") == "normal-root")
assert normal["summary_kind"] == "terminal_root_projection"
assert normal["outcome"] == "Completed"
assert normal["outcome"] == "Done"
assert normal["outcome_final"] is True
@ -313,7 +314,7 @@ def test_running_async_root_truth_survives_restart_degradation_once(tmp_path):
assert stored["root_phase_checkpoint"]["post_task_synthesis"] == "degraded"
rows = [row for row in _chat_rows(tmp_path) if row.get("task_id") == "restart-root"]
assert len(rows) == 1
assert rows[0]["outcome"] == "Completed with limitations"
assert rows[0]["outcome"] == "Done"
assert rows[0]["outcome"] == completion_status_label(stored, {})
assert rows[0]["outcome_final"] is True
assert recover_pending_root_post_task_synthesis(tmp_path, repo_dir=tmp_path / "repo") == 0

View file

@ -93,7 +93,7 @@ def test_completion_summary_event_text_is_plain_and_fully_normalized(
ctx.DRIVE_ROOT, {"status": "completed"}, "root-project", root, result, done,
) is True
assert queued[0]["text"] == (
f"Launch 🚀 › Ship release · Completed\n{PLAIN_EXCERPT}"
f"Launch 🚀 › Ship release · Done\n{PLAIN_EXCERPT}"
)
for marker in ("#", "**", "`"):
assert marker not in queued[0]["text"]
@ -136,6 +136,17 @@ def test_host_salvage_terminal_incident_drops_inherited_markdown_format(tmp_path
)
assert legacy["format"] == "markdown"
# A host NOTICE is not salvage: its own words are the answer, and its
# markdown must survive or the host's own code spans render escaped.
notice = project_terminal_result_event(
tmp_path, {"chat_id": 7}, "terminal-a",
result_text=raw, terminal_origin="host_notice", base_event=dict(base),
)
assert notice["format"] == "markdown"
assert notice["role"] == "system"
assert "system_type" not in notice
assert notice["text"] == raw
def test_history_normalizes_old_project_rows_on_read_without_rewriting_log(tmp_path):
"""Bug report #10: rows persisted BEFORE the producer stripped markdown are
@ -285,3 +296,259 @@ def test_cancel_receipt_rides_verbatim_live_durable_and_on_replay(
)
assert replayed["text"] == verbatim
assert replayed["markdown"] is False
def _parity_cases():
import pathlib
fixture = pathlib.Path(__file__).resolve().parents[1] / "web" / "tests" / "fixtures" / "outcome_phase_parity.json"
return json.loads(fixture.read_text(encoding="utf-8"))["cases"]
def test_host_status_phase_mirrors_the_browser_over_the_shared_fixture():
"""S5-03: one status-word family. The same fixture is read by
``web/tests/reason_detail.test.js``, so a divergence between the browser's
severity fold and the host's durable label fails on both sides."""
from ouroboros.project_dialogue import completion_status_label, outcome_phase
cases = _parity_cases()
assert len(cases) >= 10
for case in cases:
record = case["record"]
assert outcome_phase(record, {}) == case["phase"], case["name"]
assert completion_status_label(record, {}) == case["headline"], case["name"]
# The event frame is the other half of the same merge: a record read
# from the task_done event alone must resolve identically.
assert outcome_phase({}, record) == case["phase"], case["name"]
def test_host_status_phase_folds_the_legacy_partial_result_status_to_a_warning():
"""The browser already shows a warning here (the gateway normalizes axes on
read); the host label used to agree only by accident, through a 'partial'
member of its own degraded set."""
from ouroboros.project_dialogue import completion_status_label, outcome_phase
legacy = {"status": "completed", "result_status": "partial"}
assert outcome_phase(legacy, {}) == "warn"
assert completion_status_label(legacy, {}) == "Done with warnings"
def test_owner_requested_stop_is_done_on_the_host_row_too():
from ouroboros.project_dialogue import _completion_verdict, completion_status_label
stopped = {
"status": "completed", "reason_code": "owner_requested_finalization",
"outcome_axes": {"execution": {"status": "best_effort"}},
}
assert completion_status_label(stopped, {}) == "Done"
assert _completion_verdict(stopped, {}) == ""
A4_DECISION = {
"status": "finalized_unaccepted",
"rationale": "Acceptance reviewers did not reach a valid quorum.",
}
A4_CLAUSE = "Acceptance: finalized_unaccepted — Acceptance reviewers did not reach a valid quorum."
def _a4_result(**overrides):
result = {
"task_id": "root-project", "status": "completed", "reason_code": "final_message",
"project_id": "launch", "title": "Ship release", "result": "Release shipped.",
"outcome_axes": {
"execution": {"status": "ok"},
"review": {"status": "degraded", "acceptance_decision": dict(A4_DECISION)},
},
}
result.update(overrides)
return result
def test_host_verdict_states_an_unaccepted_acceptance_decision_in_its_own_words():
"""S5-04: a warning caused by REVIEW used to be explained by the execution
reason that happened to sit beside it (``Reason: final_message``), which
named the delivery step rather than the cause."""
from ouroboros.project_dialogue import _completion_verdict
assert _completion_verdict(_a4_result(), {}) == A4_CLAUSE
# The stored rationale already ends in a period; the clause must not double it.
assert not _completion_verdict(_a4_result(), {}).endswith("..")
def test_host_verdict_keeps_the_execution_reason_when_acceptance_was_reached():
from ouroboros.project_dialogue import _completion_verdict
accepted = _a4_result(outcome_axes={
"execution": {"status": "ok"},
"review": {"status": "pass", "acceptance_decision": {"status": "accepted"}},
})
assert _completion_verdict(accepted, {}) == "Reason: final_message."
assert _completion_verdict({"status": "completed"}, {}) == ""
# A hard failure explains itself by its execution reason, not by a decision.
assert _completion_verdict(
_a4_result(status="failed", reason_code="delegated_custody_unreconciled",
outcome_axes={"execution": {"status": "failed"},
"review": {"acceptance_decision": dict(A4_DECISION)}}),
{},
) == "Reason: delegated_custody_unreconciled."
def test_host_verdict_flattens_and_strips_a_markdown_rationale():
"""These are durable plain-text rows: the rationale is free owner-visible
text up to 500 characters and may carry newlines and markdown markers."""
from ouroboros.project_dialogue import _completion_verdict
noisy = _a4_result(outcome_axes={
"execution": {"status": "ok"},
"review": {"status": "degraded", "acceptance_decision": {
"status": "revision_requested",
"rationale": "## Verdict\n\nThe **tests** never ran with `pytest`",
}},
})
verdict = _completion_verdict(noisy, {})
assert verdict == "Acceptance: revision_requested — Verdict The tests never ran with pytest."
for marker in ("#", "**", "`", "\n"):
assert marker not in verdict
# A rationale that already terminates itself keeps its own punctuation: a
# question mark is as terminal as a period, and appending one would render
# "…did the tests run?." to the owner.
asking = _a4_result(outcome_axes={
"execution": {"status": "ok"},
"review": {"status": "degraded", "acceptance_decision": {
"status": "revision_requested", "rationale": "Did the tests ever run?",
}},
})
assert _completion_verdict(asking, {}) == "Acceptance: revision_requested — Did the tests ever run?"
def test_host_verdict_keeps_the_full_bounded_acceptance_rationale():
from ouroboros.project_dialogue import _completion_verdict
rationale = ("Review evidence " + ("remains material and owner-visible. " * 9)).strip()
result = _a4_result(outcome_axes={
"execution": {"status": "ok"},
"review": {"status": "degraded", "acceptance_decision": {
"status": "revision_requested", "rationale": rationale,
}},
})
assert len(rationale) > 240
assert _completion_verdict(result, {}) == f"Acceptance: revision_requested — {rationale}"
def test_host_verdict_states_a_decision_without_a_rationale_alone():
from ouroboros.project_dialogue import _completion_verdict
bare = _a4_result(outcome_axes={
"execution": {"status": "degraded"},
"review": {"status": "degraded", "acceptance_decision": {"status": "revision_requested"}},
})
assert _completion_verdict(bare, {}) == "Acceptance: revision_requested."
def test_host_verdict_leads_both_lifecycle_rows(tmp_path, monkeypatch):
"""The verdict must reach the owner where the owner looks: the Main row's
second line and the Project thread's terminal row."""
from ouroboros.project_dialogue import (
append_terminal_task_projection, enqueue_project_completion_summary,
)
from ouroboros.projects_registry import bind_task_to_project, create_project
project = create_project(tmp_path, "launch", name="Launch 🚀")
bind_task_to_project(
tmp_path, "root-project", project["id"], project["chat_id"],
origin={"absent": "system"},
)
queued = []
monkeypatch.setattr(
"supervisor.terminal_delivery.enqueue_terminal_delivery",
lambda _root, event, **_kwargs: queued.append(dict(event)) or True,
)
root = {
"id": "root-project", "project_id": "launch",
"title": "Ship release", "chat_id": project["chat_id"],
}
result = _a4_result()
done = {"status": "completed", "outcome_axes": result["outcome_axes"]}
assert enqueue_project_completion_summary(
tmp_path, {"status": "completed"}, "root-project", root, result, done,
) is True
assert queued[0]["text"] == (
f"Launch 🚀 › Ship release · Done with warnings\n{A4_CLAUSE} Release shipped."
)
assert "final_message" not in queued[0]["text"]
ordinary = _a4_result(
reason_code="budget_exhausted",
outcome_axes={"execution": {"status": "degraded"}},
)
assert enqueue_project_completion_summary(
tmp_path, {"status": "completed"}, "root-project", root, ordinary,
{"status": "completed", "outcome_axes": ordinary["outcome_axes"]},
) is True
assert queued[1]["text"] == (
"Launch 🚀 › Ship release · Done with warnings\n"
"Reason: budget_exhausted. Release shipped."
)
assert append_terminal_task_projection(tmp_path, "root-project", root, result, done)
rows = [
json.loads(line)
for line in (tmp_path / "logs" / "chat.jsonl").read_text(encoding="utf-8").splitlines()
if line.strip()
]
projection = next(row for row in rows if row.get("summary_kind") == "terminal_root_projection")
assert A4_CLAUSE in projection["text"]
assert "Reason: final_message" not in projection["text"]
assert projection["reason_code"] == "final_message"
assert projection["outcome"] == "Done with warnings"
assert projection["text"].endswith('Details: get_task_result(task_id="root-project")')
def test_host_verdict_and_the_card_line_compose_the_same_sentence():
"""The shared fixture is the only place the two languages agree; a clause
present there must be produced by the host verdict as well."""
from ouroboros.project_dialogue import _completion_verdict
for case in _parity_cases():
clause = case.get("acceptance_clause") or ""
if clause:
assert _completion_verdict(case["record"], {}) == clause, case["name"]
def test_terminal_row_reports_the_depth_request_only_when_one_exists(tmp_path):
"""S3-05: the owner-visible row is where a nested swarm's depth becomes
checkable — numbers first, and no line at all on swarms nobody asked to
nest, so an ordinary task's row stays byte-unchanged."""
from ouroboros.project_dialogue import append_terminal_task_projection
task = {"id": "swarm-root", "chat_id": 3, "role": "root"}
result = {
"task_id": "swarm-root", "status": "completed", "result": "Shipped.",
"outcome_axes": {"execution": {"status": "ok"}},
"swarm_efficiency": {
"subagent_count": 3,
"depth": {
"requested_depth": 2, "permitted_depth": 4, "attempted_depth": 2,
"achieved_depth": 2, "status": "achieved", "host_visible_only": True,
},
},
}
done = {"chat_id": 3, "status": "completed", "outcome_axes": result["outcome_axes"]}
assert append_terminal_task_projection(tmp_path, "swarm-root", task, result, done)
flat = {"id": "flat-root", "chat_id": 3, "role": "root"}
flat_result = {**result, "task_id": "flat-root", "swarm_efficiency": {"subagent_count": 3}}
assert append_terminal_task_projection(tmp_path, "flat-root", flat, flat_result, done)
rows = {
json.loads(line)["task_id"]: json.loads(line)
for line in (tmp_path / "logs" / "chat.jsonl").read_text(encoding="utf-8").splitlines()
if line.strip()
}
assert "Depth requested=2, permitted=4, achieved=2 (achieved)." in rows["swarm-root"]["text"]
assert "Depth" not in rows["flat-root"]["text"]
for marker in ("#", "**", "`"):
assert marker not in rows["swarm-root"]["text"]

View file

@ -856,7 +856,7 @@ def test_project_completion_enqueues_once_for_root_and_never_for_child_or_epheme
"type": "send_message",
"chat_id": 1,
"task_id": "root-project",
"text": "Launch 🚀 › Ship release · Completed\nRelease shipped.",
"text": "Launch 🚀 › Ship release · Done\nRelease shipped.",
"role": "system",
"system_type": "project_completion_summary",
"delivery_id": "project-completion:root-project",
@ -893,15 +893,15 @@ def test_project_completion_enqueues_once_for_root_and_never_for_child_or_epheme
(
{"status": "completed"},
{"outcome_axes": {"execution": {"status": "degraded"}}},
"Completed with limitations",
"Done with warnings",
),
(
{"status": "completed", "outcome_axes": {"execution": {"status": "degraded"}}},
{},
"Completed with limitations",
"Done with warnings",
),
(
{"status": "completed", "outcome_axes": {"objective": {"status": "fail"}}},
{"status": "completed", "outcome_axes": {"objective": {"status": "fail", "source": "task_acceptance_review"}}},
{},
"Failed",
),

View file

@ -1521,6 +1521,39 @@ def test_recent_tasks_includes_outcome_contract_and_ledger(tmp_path):
assert record["artifact_bundle"]["status"] == "ready_no_changes"
assert record["verification_ledger"]["entry_count"] == 1
# A ledger above the inline threshold rides as a stub with NO entries: its
# own summary is the count authority, so the row must not report zero.
write_task_result(
tmp_path,
"recent2",
STATUS_COMPLETED,
result="done",
verification_ledger={
"schema_version": 1, "omitted_to_artifact": True,
"summary": {"entry_count": 9, "has_failures": True},
},
)
payload = json.loads(_handle_recent_tasks(SimpleNamespace(drive_root=tmp_path), limit=2))
stub_record = next(row for row in payload["tasks"] if row["task_id"] == "recent2")
assert stub_record["verification_ledger"]["entry_count"] == 9
assert stub_record["verification_ledger"]["summary"]["has_failures"] is True
def test_subtask_outcome_summary_reports_a_stub_ledger_count(tmp_path):
"""The same count authority on the parent-visible child summary: a stub
ledger used to report ``0 entries / no failures`` to its parent."""
from ouroboros.tools.control import _subtask_outcome_summary
summary = json.loads(_subtask_outcome_summary({
"status": "completed", "result": "done",
"verification_ledger": {
"schema_version": 1, "omitted_to_artifact": True,
"summary": {"entry_count": 9, "has_failures": True},
},
}))
assert summary["verification_ledger"]["entry_count"] == 9
assert summary["verification_ledger"]["summary"]["has_failures"] is True
def test_effective_status_keeps_workspace_finalization_nonterminal_without_child_drive(tmp_path):
from ouroboros.headless import ARTIFACT_STATUS_FINALIZING

View file

@ -126,6 +126,41 @@ def test_task_summary_scan_uses_bounded_live_tail(tmp_path, monkeypatch):
assert seen == {"path": path, "max_entries": 123, "tail_bytes": 256 * 1024}
def test_tasks_notify_waits_for_latest_final_summary_for_each_task(tmp_path, monkeypatch):
nt = _load(); api, data = _api(tmp_path); _Rec.sent = []
monkeypatch.setattr(nt, "TelegramClient", _Rec)
chat = data / "logs" / "chat.jsonl"
working = {
"type": "task_summary", "task_id": "same1", "outcome_final": False,
"outcome_phase": "working",
"outcome_axes": {"lifecycle": {"status": "completed"},
"execution": {"status": "ok"}},
}
chat.write_text(json.dumps(working) + "\n", encoding="utf-8")
state = {"notified_task_ids": []}
asyncio.run(nt._check_tasks_notify(
api, {"TELEGRAM_NOTIFY_TASKS": "on"}, 42, state, "en",
))
assert _Rec.sent == []
assert state["notified_task_ids"] == []
terminal = {
**working, "outcome_final": True, "outcome_phase": "warn",
"outcome_axes": {"lifecycle": {"status": "completed"},
"execution": {"status": "ok"},
"review": {"status": "degraded"}},
}
with open(chat, "a", encoding="utf-8") as f:
f.write(json.dumps(terminal) + "\n")
asyncio.run(nt._check_tasks_notify(
api, {"TELEGRAM_NOTIFY_TASKS": "on"}, 42, state, "en",
))
assert _Rec.sent == [(42, "⚠️ Task same1 done")]
assert state["notified_task_ids"] == ["same1"]
def test_notify_disabled_is_silent(tmp_path, monkeypatch):
nt = _load(); api, data = _api(tmp_path); _Rec.sent = []
monkeypatch.setattr(nt, "TelegramClient", _Rec)
@ -540,3 +575,42 @@ def test_tasks_notify_reads_lifecycle_status_and_severity_icon(tmp_path, monkeyp
]
# The container is never stringified into an owner-visible push.
assert all("{'status'" not in text for _chat, text in _Rec.sent)
def test_tasks_notify_consumes_the_host_status_phase_when_the_row_carries_one(
tmp_path, monkeypatch,
):
"""S5-03: the host stamps ONE phase on its own task rows, so the notifier
consumes it instead of re-deriving a third status ladder.
Two owner-visible flips follow from the card's rule: a task whose execution
was clean but whose acceptance review degraded now warns (it used to read
✅), and an owner-requested stop is a success (it used to warn because the
stop leaves a best_effort execution axis). Legacy rows without the field
keep the axes rule; a pre-finalization ``working`` row is not a candidate.
"""
nt = _load(); api, data = _api(tmp_path); _Rec.sent = []
monkeypatch.setattr(nt, "TelegramClient", _Rec)
rows = [
{"task_id": "review1", "outcome_phase": "warn",
"outcome_axes": {"lifecycle": {"status": "completed"},
"execution": {"status": "ok"},
"review": {"status": "degraded"}}},
{"task_id": "stop1", "outcome_phase": "done",
"outcome_axes": {"lifecycle": {"status": "completed"},
"execution": {"status": "best_effort"}}},
{"task_id": "open1", "outcome_phase": "working",
"outcome_axes": {"lifecycle": {"status": "completed"},
"execution": {"status": "degraded"}}},
]
with open(data / "logs" / "chat.jsonl", "w", encoding="utf-8") as f:
for row in rows:
f.write(json.dumps({"type": "task_summary", **row}) + "\n")
state = {"notified_task_ids": []}
asyncio.run(nt._check_tasks_notify(api, {"TELEGRAM_NOTIFY_TASKS": "on"}, 42, state, "en"))
assert [text for _chat, text in _Rec.sent] == [
"⚠️ Task review1 done",
"✅ Task stop1 done",
]
assert state["notified_task_ids"] == ["review1", "stop1"]

View file

@ -304,10 +304,12 @@ def test_deadline_grace_final_is_never_stamped_host_salvage(monkeypatch):
assert usage.get("terminal_origin") != L.TERMINAL_ORIGIN_HOST_SALVAGE
def test_budget_and_round_limit_rails_keep_legacy_missing_origin(monkeypatch):
"""The wrapper deviation's whole point (fable lane B gap #2): budget and
round-limit rails never enter the wrapper, so their terminals still carry
NO terminal_origin (missing = legacy shape)."""
def test_budget_and_round_limit_rails_stamp_host_notice(monkeypatch):
"""INTENTIONAL behaviour change, not a bug fix of the test: these rails used
to leave terminal_origin absent because they never enter the provider-death
wrapper, so "missing" meant both "written before the stamp existed" and "the
host wrote this text alone". The forced-finalization sink now stamps
host_notice, and a missing origin identifies only legacy rows."""
import pathlib
from types import SimpleNamespace
@ -324,4 +326,131 @@ def test_budget_and_round_limit_rails_keep_legacy_missing_origin(monkeypatch):
)
monkeypatch.setattr(L, "call_llm_with_retry", lambda *a, **k: (None, 0.0))
_t, usage, _tr = L._handle_round_limit(ctx)
assert "terminal_origin" not in usage
assert usage.get("terminal_origin") == L.TERMINAL_ORIGIN_HOST_NOTICE
def _rail_ctx(task_id="t-rail", round_idx=11):
import pathlib
from types import SimpleNamespace
return SimpleNamespace(
messages=[{"role": "user", "content": "do"}],
llm=None, active_model="m", active_effort="low", max_retries=1,
drive_logs=pathlib.Path("/tmp"), task_id=task_id, round_idx=round_idx,
event_queue=None, accumulated_usage={},
task_type="", active_use_local=False, max_rounds=10, deadline_ts=None,
drive_root=None, budget_drive_root=None, root_task_id="", tools=None,
llm_trace={},
)
def test_budget_rejection_before_any_work_is_a_host_notice(monkeypatch):
"""The host wrote this text alone, so it must be attributable — and it must
be published verbatim rather than replaced by the outage receipt."""
import ouroboros.loop as L
ctx = _rail_ctx(task_id="t-budget", round_idx=1)
result = L._check_budget_limits(ctx, 0.0)
assert result is not None
text, usage, _trace = result
assert text.startswith("🚫 Task rejected")
assert usage["terminal_origin"] == L.TERMINAL_ORIGIN_HOST_NOTICE
assert usage["reason_code"] == "budget_exhausted"
def test_a_host_notice_publishes_its_own_words_with_its_markdown(tmp_path):
"""A notice is NOT salvage: replacing its text with the outage receipt would
name the wrong cause, and dropping its markdown would render the host's own
code spans as escaped plain text. It also carries no system_type, which is
what lets a replayed task card conclude on it."""
from supervisor.terminal_delivery import project_terminal_result_event
text = "🚫 Task rejected. Total budget exhausted.\n\nPlan review left open: `plan_review_advisory`."
base = {
"type": "send_message", "chat_id": 7, "task_id": "terminal-n",
"text": text, "format": "markdown", "log_text": "kept",
}
notice = project_terminal_result_event(
tmp_path, {"chat_id": 7}, "terminal-n",
result_text=text, terminal_origin="host_notice", base_event=dict(base),
)
assert notice["text"] == text
assert "model-provider outage" not in notice["text"]
assert notice["role"] == "system"
assert notice["terminal_origin"] == "host_notice"
assert "system_type" not in notice
assert notice["format"] == "markdown"
assert notice["log_text"] == "kept"
def test_the_durable_result_carries_the_third_producer_word():
from ouroboros.task_finalization import terminal_result_fields
assert terminal_result_fields({"terminal_origin": "host_notice"})["terminal_origin"] == "host_notice"
assert terminal_result_fields({"terminal_origin": "model_final"})["terminal_origin"] == "model_final"
# An unknown producer stays legacy rather than acquiring a word.
assert "terminal_origin" not in terminal_result_fields({"terminal_origin": "something_new"})
def test_provider_death_arms_are_not_downgraded_by_the_forced_sink(monkeypatch):
"""Regression for the sink's interaction with the provider rail: the sink
stamps host_notice FIRST on the three no-call/no-resend provider-death arms,
so a plain setdefault in the wrapper would have become a no-op and a
provider-death salvage would have published verbatim."""
import ouroboros.loop as L
monkeypatch.setattr(L, "call_llm_with_retry", lambda *a, **k: (None, 0.0))
for kind, wait_cause in (
("provider_outcome_unknown", ""),
("provider_unavailable", "transport_unavailable"),
("context_overflow", ""),
):
ctx = _rail_ctx(task_id=f"t-{kind}")
ctx.accumulated_usage = {"terminal_origin": L.TERMINAL_ORIGIN_HOST_NOTICE}
_text, usage, _trace = L._handle_provider_unavailable(
ctx, error_kind=kind, wait_cause=wait_cause,
)
assert usage.get("terminal_origin") == L.TERMINAL_ORIGIN_HOST_SALVAGE, kind
def test_a_notice_keeps_its_completion_excerpt_unlike_a_salvage():
"""Only salvage hides its text behind the neutral details copy."""
from ouroboros.project_dialogue import _completion_excerpt
assert _completion_excerpt({"result": "x", "terminal_origin": "host_notice"}) == "x"
assert _completion_excerpt({"result": "x", "terminal_origin": "host_salvage"}) == ""
def test_a_non_provider_rail_with_a_complete_candidate_stays_model_final(tmp_path, monkeypatch):
"""The candidate branch only stamped provenance on the provider rail, so a
round-limit stop that delivered the model's own complete answer went out
unattributed. The ordinary no-tool final composes the same disclosure glue
and stays model_final; this rail now says the same thing, and the glue is
still appended to the model's text."""
from tests.test_delivery_forced_finalization import _forced_test_context
loop, registry, limit_ctx, trace = _forced_test_context(tmp_path)
answer = "Complete current model answer."
loop._replace_delivery_candidate(registry, limit_ctx, trace, answer, control="replace")
monkeypatch.setattr(
loop, "_force_plan_disclosure",
lambda *_a, **_k: "\n\nPlan review was left open.",
)
text, usage, _returned_trace = loop._handle_round_limit(limit_ctx)
assert text.startswith(answer)
assert text.endswith("Plan review was left open.")
assert usage["terminal_origin"] == loop.TERMINAL_ORIGIN_MODEL_FINAL
assert usage["terminal_plan_review_open"] is True
def test_a_non_provider_rail_without_a_candidate_is_a_host_notice(tmp_path, monkeypatch):
from tests.test_delivery_forced_finalization import _forced_test_context
loop, _registry, limit_ctx, _trace = _forced_test_context(tmp_path)
monkeypatch.setattr(loop, "call_llm_with_retry", lambda *_a, **_k: (None, 0.0))
_text, usage, _returned_trace = loop._handle_round_limit(limit_ctx)
assert usage["terminal_origin"] == loop.TERMINAL_ORIGIN_HOST_NOTICE

View file

@ -1042,7 +1042,9 @@ def test_expired_local_deadline_does_not_dispatch_a_paid_final_call(monkeypatch,
assert result is not None
assert result[0].startswith("⚠️ Task reached its deadline")
assert ctx.accumulated_usage == {
usage = dict(ctx.accumulated_usage)
assert usage.pop("terminal_origin") == "host_notice"
assert usage == {
"execution_status": "failed",
"reason_code": "deadline_local",
}

View file

@ -914,3 +914,37 @@ def test_ui_smoke_module_widget_temporal_convergence(direct_server_with_data, br
if "Executable doesn't exist" in str(exc) or "playwright install" in str(exc).lower():
pytest.skip(str(exc))
raise
@pytest.mark.ui_browser
def test_ui_smoke_widgets_page_has_a_typed_address_and_one_nav_landmark(
direct_server_with_data,
):
"""A page could not be linked to and the sidebar was not a landmark.
Opening the application with a `#widgets` fragment must land on Widgets
(browsers gain a shareable link; the desktop shell and the Linux fallback
get the same load-time route with no address bar to break), and an
automation or assistive client must be able to resolve the navigation by
its landmark instead of by an id.
"""
pytest.importorskip("playwright.sync_api", reason="Playwright is not installed")
from playwright.sync_api import Error as PlaywrightError
from playwright.sync_api import sync_playwright
url = direct_server_with_data["url"]
try:
with sync_playwright() as pw:
browser = pw.chromium.launch(headless=True)
page = browser.new_page(viewport={"width": 1280, "height": 800})
try:
page.goto(f"{url}/#widgets", wait_until="domcontentloaded", timeout=30_000)
page.wait_for_selector("nav#primary-sidebar", timeout=15_000)
page.wait_for_selector("#page-widgets.active", timeout=15_000)
assert page.locator('[data-nav-page="widgets"].active').count() == 1
finally:
browser.close()
except PlaywrightError as exc:
if "Executable doesn't exist" in str(exc) or "playwright install" in str(exc).lower():
pytest.skip(str(exc))
raise

View file

@ -0,0 +1,117 @@
"""Who decides that a workspace change is junk.
A skill whose deliverable IS its build output (a bundled widget under
``dist/``) had that file silently dropped from ``workspace.patch``, because the
capture rules excluded generated-output directories by NAME — a rule inherited
from a benchmark capture script that owns its own copy and never imported this
module. The project's own ``.gitignore`` is the authority instead, and the
remaining host rules stay: the per-file size cap, git's binary verdict, and the
credential-shaped name check.
"""
from __future__ import annotations
import pathlib
import subprocess
def _genesis_repo(root: pathlib.Path, gitignore: str = "") -> pathlib.Path:
"""A freshly created project: one commit, optionally a .gitignore."""
root.mkdir()
subprocess.run(["git", "init"], cwd=root, check=True, capture_output=True)
(root / "README.md").write_text("base\n", encoding="utf-8")
tracked = ["README.md"]
if gitignore:
(root / ".gitignore").write_text(gitignore, encoding="utf-8")
tracked.append(".gitignore")
subprocess.run(["git", "add", *tracked], cwd=root, check=True, capture_output=True)
subprocess.run(
["git", "-c", "user.email=t@example.com", "-c", "user.name=T", "commit", "-m", "init"],
cwd=root, check=True, capture_output=True,
)
return root
def _capture(repo: pathlib.Path, out: pathlib.Path):
from ouroboros.headless import write_workspace_patch_artifacts
_artifacts, manifest = write_workspace_patch_artifacts(repo, out, task={})
return (out / "workspace.patch").read_text(encoding="utf-8"), manifest
def test_a_project_without_a_gitignore_keeps_its_build_output_in_the_patch(tmp_path):
repo = _genesis_repo(tmp_path / "repo")
(repo / "dist").mkdir()
(repo / "dist" / "widget.js").write_text("export const widget = 1;\n", encoding="utf-8")
patch, manifest = _capture(repo, tmp_path / "artifacts")
assert "dist/widget.js" in patch
assert "dist/widget.js" in manifest["untracked_included"]
assert manifest["exclude_rules_version"] == 3
assert not [item["path"] for item in manifest["untracked_excluded"]]
def test_a_project_that_declares_dist_ignored_keeps_it_out_of_the_patch(tmp_path):
repo = _genesis_repo(tmp_path / "repo", gitignore="dist/\n")
(repo / "dist").mkdir()
(repo / "dist" / "widget.js").write_text("export const widget = 1;\n", encoding="utf-8")
(repo / "src.js").write_text("export const src = 1;\n", encoding="utf-8")
patch, manifest = _capture(repo, tmp_path / "artifacts")
assert "dist/widget.js" not in patch
assert "src.js" in patch
# Git-ignored files are outside the capture universe: they are not carried
# and not reported as exclusions either.
assert "dist/widget.js" not in [item["path"] for item in manifest["untracked_excluded"]]
assert manifest["counts"]["untracked_excluded"] == 0
def test_the_remaining_host_vetoes_still_apply_to_generated_output(tmp_path, monkeypatch):
import ouroboros.headless as headless
repo = _genesis_repo(tmp_path / "repo")
(repo / "dist").mkdir()
(repo / "dist" / "widget.js").write_text("export const widget = 1;\n", encoding="utf-8")
(repo / "dist" / "app.bin").write_bytes(b"\x7fELF\x00\x01\x02\x03binary\x00blob")
(repo / "dist" / "build.log").write_text("noise\n", encoding="utf-8")
(repo / "dist" / "big.js").write_text("x" * 200, encoding="utf-8")
monkeypatch.setattr(headless, "_PATCH_MAX_UNTRACKED_FILE_BYTES", 100)
patch, manifest = _capture(repo, tmp_path / "artifacts")
excluded = {item["path"]: item["reason"] for item in manifest["untracked_excluded"]}
assert "dist/widget.js" in patch
assert "binary file" in excluded.get("dist/app.bin", "")
assert "junk artifact" in excluded.get("dist/build.log", "")
assert "size cap" in excluded.get("dist/big.js", "")
def test_the_snapshot_veto_agrees_with_the_patch_by_subtraction(tmp_path):
"""The delegated-run baseline snapshot asks the same predicate, so removing
the name rule keeps the snapshot and the patch from drifting apart."""
from ouroboros.headless import untracked_capture_veto_reason
repo = _genesis_repo(tmp_path / "repo")
(repo / "dist").mkdir()
(repo / "dist" / "widget.js").write_text("export const widget = 1;\n", encoding="utf-8")
(repo / "dist" / "run.log").write_text("noise\n", encoding="utf-8")
assert untracked_capture_veto_reason(repo, "dist/widget.js") == ""
assert untracked_capture_veto_reason(repo, "build/widget.js") == ""
assert "junk artifact" in untracked_capture_veto_reason(repo, "dist/run.log")
def test_the_benchmark_capture_script_owns_its_own_rule(tmp_path):
"""The comment that justified the host rule claimed consistency with the
benchmark script. That script never imported this module and uses an
any-depth rule of its own, so the two were never one rule."""
from ouroboros import workspace_patch_rules
script = pathlib.Path(__file__).resolve().parents[1] / "devtools" / "benchmarks" / "swe_bench_pro" / "capture_patch.sh"
if script.is_file():
text = script.read_text(encoding="utf-8")
assert "dist/" in text
assert "workspace_patch_rules" not in text
assert "dist" not in workspace_patch_rules._PATCH_JUNK_RE.pattern

View file

@ -77,6 +77,17 @@ function setMobileDrawerOpen(open, { sync = true } = {}) {
if (sync) syncNavigationState();
}
// The application's one client-side route: a `#<page>` fragment, honoured once
// on load and validated against the injected sections rather than against a
// duplicated page list — the DOM is the single source of truth, so an unknown
// fragment is ignored instead of painting a blank surface. Deliberately NOT
// written back on navigation: the desktop shell and the Telegram mini app have
// no address bar to read it from.
function pageFromHash() {
const name = String(window.location.hash || '').replace(/^#/, '').trim();
return name && document.getElementById(`page-${name}`) ? name : '';
}
async function showPage(name, options = {}) {
const pageName = String(name || '').trim();
if (!pageName) return false;
@ -713,6 +724,8 @@ initOnboardingOverlay();
initMatrixRain();
loadVersion();
syncNavigationState();
const hashPage = pageFromHash();
if (hashPage && hashPage !== state.activePage) showPage(hashPage);
// Mobile soft-keyboard handling: viewport shrink counts only while an editable
// owns focus. Drawer opening clears that state explicitly before navigation is

View file

@ -18,7 +18,7 @@
<body>
<div id="app">
<div id="nav-drawer-backdrop" class="nav-drawer-backdrop" hidden></div>
<aside id="primary-sidebar" class="primary-sidebar" aria-label="Ouroboros navigation">
<nav id="primary-sidebar" class="primary-sidebar" aria-label="Ouroboros navigation">
<div class="sidebar-scroll">
<button class="nav-row nav-row-main active" data-nav-page="chat" title="Main Ouroboros chat">
<svg width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round"><path d="M22 17a2 2 0 0 1-2 2H6.828a2 2 0 0 0-1.414.586l-2.202 2.202A.71.71 0 0 1 2 21.286V5a2 2 0 0 1 2-2h16a2 2 0 0 1 2 2z"/><path d="M7 11h10"/><path d="M7 15h6"/><path d="M7 7h8"/></svg>
@ -65,7 +65,7 @@
</div>
</div>
<span id="nav-version"></span>
</aside>
</nav>
<div id="divider"></div>
<main id="content">
<!-- Pages injected by JS -->

View file

@ -562,6 +562,12 @@
* @property {boolean=} project_mirror
*/
/** Additive /api/chat/history terminality projection.
* @typedef {Object} TaskOutcomeHistoryFields
* @property {"working"|"done"|"warn"|"error"|"cancelled"=} outcome_phase // canonical display phase; "working" is not terminal
* @property {boolean=} outcome_final // true only after the canonical task outcome settles; false marks a pre-finalization narrative
*/
/**
* Additive /api/chat/history row fields on `system_type: "skill_review"` rows:
* the exact-job reference the producer already writes into chat.jsonl. A row

View file

@ -2114,11 +2114,11 @@ export function createChatInstance({
return finishLiveCard(taskId, 'done');
}
let changed = false;
// Restore the coined task name even when history retained no progress row.
// Restore task name from history.
if (msg?.suggested_name) {
changed = applySuggestedName(taskId, msg.suggested_name) || changed;
}
const finalizing = msg?.task_phase === 'finalizing';
const finalizing = msg?.task_phase === 'finalizing' || msg?.outcome_final === false;
const projectedReviews = reviewGroupsFromTaskDetail(msg, taskId);
const hasAcceptanceReview = projectedReviews.length > 0;
const taskState = getTaskUiState(taskId, hasAcceptanceReview || finalizing);

View file

@ -360,7 +360,20 @@ export function taskReasonPhrase(code) {
export function taskReasonDetail(evt) {
// An owner-requested stop is a success and carries its own marker instead.
if (taskStoppedWithSummary(evt) || !evt?.reason_code) return '';
if (taskStoppedWithSummary(evt)) return '';
// A warning caused by REVIEW must not be explained by the execution reason
// that happens to sit beside it: the host's acceptance decision is the
// cause, and it speaks in its own stored words. A hard failure or a
// cancellation keeps explaining itself by its execution reason.
const record = normalizeTaskTerminalRecord(evt);
const decision = record.outcome_axes?.review?.acceptance_decision
?? record.review_status?.acceptance_decision;
const severity = taskOutcomeSeverity(evt);
if (severity !== 'error' && severity !== 'cancelled' && decision?.status && decision.status !== 'accepted') {
const rationale = String(decision.rationale || '').split(/\s+/).filter(Boolean).join(' ');
return `Acceptance: ${decision.status}${rationale ? ` — ${rationale}` : ''}`;
}
if (!evt?.reason_code) return '';
return `Reason: ${taskReasonPhrase(evt.reason_code)}`;
}
@ -487,6 +500,8 @@ function taskOutcomeMeta(evt) {
axes.lifecycle?.status ? `lifecycle ${axes.lifecycle.status}` : '',
axes.execution?.status ? `execution ${axes.execution.status}` : '',
axes.objective?.status ? `objective ${axes.objective.status}` : '',
axes.review?.status ? `review ${axes.review.status}` : '',
axes.review?.acceptance_decision?.status ? `acceptance ${axes.review.acceptance_decision.status}` : '',
].filter(Boolean);
}

View file

@ -422,3 +422,33 @@ test('chat bubble heading clamp is scoped in style.css', () => {
const richSource = readFileSync(new URL('../modules/chat_markdown.js', import.meta.url), 'utf8');
assert.match(richSource, /querySelectorAll\('h1, h2, h3, h4, h5, h6'\)[\s\S]{0,160}Math\.min\(Number\(heading\.tagName\.slice\(1\)\), 3\)/);
});
test('a host notice concludes a replayed card and keeps its markdown', async () => {
// A terminal text the host wrote alone is projected as role=system with NO
// system_type — exactly the shape replay needs to treat it as the task's
// last word — and, unlike the salvage receipt, it keeps its markdown so the
// host's own code spans do not render escaped.
assert.match(chatSource, /const plainUntypedFinal = !msg\.system_type && !msg\.msg_type;/);
assert.match(
chatSource,
/\(msg\.role === 'assistant' \|\| msg\.role === 'system'\)\n\s+&& \(positiveTaskTerminalFact\(msg\) \|\| plainUntypedFinal\)/,
);
const { prior, mount } = installDom();
let instance;
try {
const made = makeInstance(mount);
instance = made.instance;
made.handlers.get('chat')({
chat_id: 2, role: 'system', markdown: true, task_id: 'notice-task',
content: 'Task rejected line one\nline two',
ts: '2026-09-03T00:00:01Z',
});
const bubble = findBubble('system');
assert.match(bubble.innerHTML, /Task rejected line one<br>line two/);
assert.equal(bubble.getAttribute('data-chat-markdown-enhanced'), 'true');
} finally {
instance?.destroy();
restoreDom(prior);
}
});

View file

@ -0,0 +1,146 @@
{
"purpose": "One status-word family across the browser and the host. Every case is a terminal-or-not task record read by log_events.js (taskDoneIsTerminal + taskTerminalPhase + taskPresentation) and by ouroboros.project_dialogue (outcome_phase + OUTCOME_PHASE_HEADLINE + _completion_verdict). A new axis, reason or acceptance status is added to both sides in the same commit, with a row here.",
"cases": [
{
"name": "clean completed root",
"record": {"status": "completed", "outcome_axes": {"execution": {"status": "ok"}}},
"phase": "done",
"headline": "Done",
"acceptance_clause": ""
},
{
"name": "completed while post-task synthesis is still open reads Working on both sides",
"record": {
"status": "completed",
"outcome_axes": {"execution": {"status": "ok"}},
"root_phase_checkpoint": {"post_task_synthesis": "running"}
},
"phase": "working",
"headline": "Working",
"acceptance_clause": ""
},
{
"name": "legacy task_done frame carrying status done stays terminal",
"record": {"status": "done", "outcome_axes": {"execution": {"status": "ok"}}},
"phase": "done",
"headline": "Done",
"acceptance_clause": ""
},
{
"name": "cancelled outranks failure-shaped teardown side facts",
"record": {
"status": "cancelled",
"outcome_axes": {"lifecycle": {"status": "cancelled"}, "artifacts": {"status": "missing"}}
},
"phase": "cancelled",
"headline": "Cancelled",
"acceptance_clause": ""
},
{
"name": "legacy cancel_requested event spelling",
"record": {"status": "cancel_requested"},
"phase": "cancelled",
"headline": "Cancelled",
"acceptance_clause": ""
},
{
"name": "failed review is a failure",
"record": {
"status": "completed",
"outcome_axes": {"execution": {"status": "ok"}, "review": {"status": "fail"}}
},
"phase": "error",
"headline": "Failed",
"acceptance_clause": ""
},
{
"name": "owner-requested finalization is a success, not a warning",
"record": {
"status": "completed",
"reason_code": "owner_requested_finalization",
"outcome_axes": {"execution": {"status": "best_effort"}}
},
"phase": "done",
"headline": "Done",
"acceptance_clause": ""
},
{
"name": "scheduler duplicate rejection is a warning",
"record": {"status": "rejected_duplicate", "outcome_axes": {"lifecycle": {"status": "rejected_duplicate"}}},
"phase": "warn",
"headline": "Done with warnings",
"acceptance_clause": ""
},
{
"name": "degraded review with an unaccepted acceptance decision",
"record": {
"status": "completed",
"reason_code": "final_message",
"outcome_axes": {
"execution": {"status": "ok"},
"review": {
"status": "degraded",
"acceptance_decision": {
"status": "finalized_unaccepted",
"rationale": "Acceptance reviewers did not reach a valid quorum."
}
}
}
},
"phase": "warn",
"headline": "Done with warnings",
"acceptance_clause": "Acceptance: finalized_unaccepted — Acceptance reviewers did not reach a valid quorum."
},
{
"name": "an accepted decision leaves the execution reason standing",
"record": {
"status": "completed",
"reason_code": "final_message",
"outcome_axes": {
"execution": {"status": "ok"},
"review": {"status": "pass", "acceptance_decision": {"status": "accepted", "rationale": "Quorum reached."}}
}
},
"phase": "done",
"headline": "Done",
"acceptance_clause": ""
},
{
"name": "a revision request without a rationale states the status alone",
"record": {
"status": "completed",
"outcome_axes": {
"execution": {"status": "degraded"},
"review": {"status": "degraded", "acceptance_decision": {"status": "revision_requested"}}
}
},
"phase": "warn",
"headline": "Done with warnings",
"acceptance_clause": "Acceptance: revision_requested."
},
{
"name": "a hard failure explains itself by its execution reason, not by the decision",
"record": {
"status": "failed",
"reason_code": "delegated_custody_unreconciled",
"outcome_axes": {
"execution": {"status": "failed"},
"review": {"status": "degraded", "acceptance_decision": {"status": "finalized_unaccepted", "rationale": "No quorum."}}
}
},
"phase": "error",
"headline": "Failed",
"acceptance_clause": ""
},
{
"name": "an objective warning is a warning even when every status word is clean",
"record": {
"status": "completed",
"outcome_axes": {"execution": {"status": "ok"}, "objective": {"status": "pass", "warning": "partial coverage"}}
},
"phase": "warn",
"headline": "Done with warnings",
"acceptance_clause": ""
}
]
}

View file

@ -2,7 +2,9 @@ import assert from 'node:assert/strict';
import test from 'node:test';
import { readFileSync } from 'node:fs';
import { taskReasonDetail, taskReasonPhrase } from '../modules/log_events.js';
import {
taskDoneIsTerminal, taskPresentation, taskReasonDetail, taskReasonPhrase, taskTerminalPhase,
} from '../modules/log_events.js';
// A degraded delivery used to name one generic cause on every card. The record
// keeps the machine code; the card says what actually happened.
@ -46,3 +48,90 @@ test('every typed cause the loop can record has a sentence', () => {
);
}
});
test('one status-word family: the card phase matches the host over the shared fixture', () => {
// The same fixture is read by tests/test_project_plain_rows.py, so a
// divergence between this severity fold and the host's durable label word
// fails on both sides of the boundary.
const fixture = JSON.parse(
readFileSync(new URL('./fixtures/outcome_phase_parity.json', import.meta.url), 'utf8'),
);
assert.ok(fixture.cases.length >= 10);
for (const { name, record, phase, headline, acceptance_clause: clause } of fixture.cases) {
const resolved = taskDoneIsTerminal(record) ? taskTerminalPhase(record) : 'working';
assert.deepEqual(taskPresentation(resolved), { phase, headline }, name);
if (clause) {
// The host composes the same sentence for its durable prose rows and
// terminates it there; the card line adds no punctuation of its own.
const detail = taskReasonDetail(record);
assert.ok(clause === detail || clause === `${detail}.`, `${name}: ${detail} vs ${clause}`);
}
}
});
// A review-caused warning used to be explained by whatever execution reason sat
// beside it ('Reason: final_message'), which named the delivery step rather than
// the actual cause. The host's acceptance decision now speaks for itself.
const A4 = {
status: 'completed',
reason_code: 'final_message',
outcome_axes: {
execution: { status: 'ok' },
review: {
status: 'degraded',
acceptance_decision: {
status: 'finalized_unaccepted',
rationale: 'Acceptance reviewers did not reach a valid quorum.',
},
},
},
};
test('an unaccepted decision explains the warning in its own words', () => {
assert.equal(
taskReasonDetail(A4),
'Acceptance: finalized_unaccepted — Acceptance reviewers did not reach a valid quorum.',
);
assert.doesNotMatch(taskReasonDetail(A4), /final_message/);
});
test('an accepted decision leaves the execution reason line byte-identical', () => {
const accepted = {
...A4,
outcome_axes: {
execution: { status: 'ok' },
review: { status: 'pass', acceptance_decision: { status: 'accepted', rationale: 'Quorum reached.' } },
},
};
assert.equal(taskReasonDetail(accepted), 'Reason: final_message');
});
test('a decision without a rationale states its status alone', () => {
const record = {
outcome_axes: { review: { acceptance_decision: { status: 'revision_requested' } } },
status: 'completed',
};
assert.equal(taskReasonDetail(record), 'Acceptance: revision_requested');
});
test('a decision with no reason code still reaches the acceptance branch', () => {
// The old single early return swallowed this frame before the branch.
const record = { status: 'completed', review_status: { acceptance_decision: { status: 'revision_requested' } } };
assert.equal(taskReasonDetail(record), 'Acceptance: revision_requested');
});
test('a hard failure keeps explaining itself by its execution reason', () => {
const failed = { ...A4, status: 'failed', reason_code: 'delegated_custody_unreconciled' };
assert.equal(taskReasonDetail(failed), 'Reason: delegated_custody_unreconciled');
});
test('a multi-line rationale is flattened into one sentence', () => {
const noisy = {
status: 'completed',
outcome_axes: {
review: { acceptance_decision: { status: 'revision_requested', rationale: 'Two\n\nlines here.' } },
},
};
assert.equal(taskReasonDetail(noisy), 'Acceptance: revision_requested — Two lines here.');
});

View file

@ -251,7 +251,7 @@ test('history replay keeps open summaries live and terminal fallbacks factual',
chatSource.indexOf('function appendTaskSummaryToLiveCard'),
chatSource.indexOf('// child task_id'),
);
assert.match(summary, /const finalizing = msg\?\.task_phase === 'finalizing';/);
assert.match(summary, /const finalizing = msg\?\.task_phase === 'finalizing' \|\| msg\?\.outcome_final === false;/);
assert.match(summary, /terminal: !finalizing/);
assert.match(summary, /record\.finalizingHold = true/);
assert.match(summary, /if \(finalizing\) return changed;\s*changed = finishLiveCard/);
@ -371,3 +371,28 @@ test('header status has no terminal-attention state or writer', () => {
assert.doesNotMatch(chatSource, /lastTerminalAttention/);
assert.doesNotMatch(activitySource, /lastTerminalAttention|text: 'Attention'/);
});
test('a review-caused warning names the acceptance decision on the card and in Logs', () => {
// The execution reason beside it ('final_message') names the delivery step,
// not the cause; the card body and the Logs meta now say what happened.
const evt = {
type: 'task_done', status: 'completed', reason_code: 'final_message',
outcome_axes: {
execution: { status: 'ok' },
review: {
status: 'degraded',
acceptance_decision: {
status: 'finalized_unaccepted',
rationale: 'Acceptance reviewers did not reach a valid quorum.',
},
},
},
};
const live = summarizeChatLiveEvent(evt);
const replay = summarizeLogEvent(evt);
assert.deepEqual({ phase: live.phase, headline: live.headline }, { phase: 'warn', headline: 'Done with warnings' });
assert.match(live.body, /Acceptance: finalized_unaccepted — Acceptance reviewers did not reach a valid quorum\./);
assert.doesNotMatch(live.body, /final_message/);
assert.ok(replay.meta.includes('review degraded'));
assert.ok(replay.meta.includes('acceptance finalized_unaccepted'));
});

View file

@ -234,3 +234,9 @@ test('live structured delivery frames keep additive grouping and size fields', (
assert.match(chat, /enhanceMarkdown: enhanceMountedMarkdown/);
assert.match(contracts, /WS_MESSAGE_TYPES[\s\S]*?"links"/);
});
test('history typedef declares task outcome terminality fields', () => {
const types = moduleFile('api_types.js');
assert.match(types, /@property \{"working"\|"done"\|"warn"\|"error"\|"cancelled"=\} outcome_phase/);
assert.match(types, /@property \{boolean=\} outcome_final/);
});