mirror of
https://github.com/razzant/ouroboros.git
synced 2026-10-03 04:07:04 +00:00
Merge phase P5+P7 (terminal presentation: coherent depth tuple, settled result overrides early finality)
This commit is contained in:
commit
33d4124c36
46 changed files with 1720 additions and 177 deletions
|
|
@ -39,7 +39,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de
|
|||
│ ├── task_lifecycle.py ← Cancellation custody — the ONE settle owner of durable cancel intents: claim → capture → confirmed death → natural-completion re-check → owed delivery registration → settle → delivery/cleanup, plus the `sweep_cancel_intents` watchdog and the queue-owned root-budget admission fence; every custody rule it enforces is stated once in §10 (cancellation custody)
|
||||
│ ├── cancel_publication.py ← Cancellation settlement publication, re-imported by `task_lifecycle.py`: typed CANCEL_* outcome vocabulary, artifact-honest cancelled result fields, physical-ledger cost reconstruction, salvage adapter, owed-before-settle outbox registration, publication of the STORED terminal truth, capture-miss terminalization/delivery adapter
|
||||
│ ├── queue_transitions.py ← Queue-owned lifecycle transitions that are not cancellation custody: acceptance-fence open/inspect/seal, explicit budget resume, typed `stop_evolution_tasks` (per-task typed outcomes through the durable-intent ingress, never an in-place prune; an incomplete stop leaves the campaign OPEN via the durable `evolution_owner_stopped` flag, the settle-time backstop in events.py closes it when the last live evolution task settles, and both start ingresses clear the flag BEFORE minting a fresh campaign), and fenced Project deletion (cascade only lineage ROOTS — descendants fall with their trees, one cascade and one summary per tree; tombstone only after provable quiescence; a settled-but-LIVE root still mints the coordination intent, and wind-down defers and RE-CHECKS bounded instead of re-running the cancel pass over a settled-lingering set, which would deliver duplicate owner summaries); imports nothing from task_lifecycle; `supervisor.queue` re-exports these names
|
||||
│ ├── terminal_delivery.py ← Durable terminal-answer delivery seam: restart-surviving `delivery_id` dedupe + bounded PENDING outbox `state/terminal_deliveries.json` (owed before enqueue, cleared in the delivering write, replayed on boot and on the supervisor tick), shared by natural final answers (every non-ephemeral root registers at durable-result persistence), cancel salvage, cascade digest, and non-retry reap; the cascade digest enumerates descendants by ANCESTRY (parent-chain walk, never `root_task_id` equality); eviction past outbox capacity is disclosed via typed `terminal_delivery_exhausted`, never a silent pop; salvage messages carry a bounded preview plus a full-copy receipt (path, size, full 64-hex sha256 or an explicit marker) and route by lineage chat — no resolvable chat records a typed `terminal_delivery_handoff` row; reads/mutations are row-strict per §10; the delivery id digests only the STABLE part (task id + status framing + core answer) so a replay whose rebuilt note shrank dedups instead of double-sending
|
||||
│ ├── terminal_delivery.py ← Durable terminal-answer delivery seam: restart-surviving `delivery_id` dedupe + bounded PENDING outbox `state/terminal_deliveries.json` (owed before enqueue, cleared in the delivering write, replayed on boot and on the supervisor tick), shared by natural final answers (every non-ephemeral root registers at durable-result persistence), cancel salvage, cascade digest, and non-retry reap; the cascade digest enumerates descendants by ANCESTRY (parent-chain walk, never `root_task_id` equality); eviction past outbox capacity is disclosed via typed `terminal_delivery_exhausted`, never a silent pop; salvage messages carry a bounded preview plus a full-copy receipt (path, size, full 64-hex sha256 or an explicit marker) and route by lineage chat — no resolvable chat records a typed `terminal_delivery_handoff` row; reads/mutations are row-strict per §10; the delivery id digests only the STABLE part (task id + status framing + core answer) so a replay whose rebuilt note shrank dedups instead of double-sending; the per-origin projection is `host_salvage` receipt / `host_notice` own text kept as a System row with its markdown / `model_final` assistant projection
|
||||
│ ├── task_reaper.py ← Off-loop single-owner reaper thread: kill/join/archive/respawn a timed-out worker; STRICT fail-closed — an unconfirmed death holds the slot `reaping`, leaves the task RUNNING, emits `task_reaper_wedged` + an owner /restart hint (no terminal/retry/respawn while the worker may be alive); after confirmed death it reconciles the task's open delegated runs through the custody seam before the retry/respawn decision and discloses still-open runs; mints no cancel intents
|
||||
│ ├── owner_stop.py ← Owner graceful stop: `finalize_then_cancel` policy as an axis on the SAME durable cancel intent (monotonic — immediate HARDENS a pending graceful, never softens back; hardening revokes an unread control via mailbox revocation, and the loop revalidates durable policy at drain); one deterministic typed `finalize_now` control whose first line is the `owner_requested_finalization` literal, routed by the loop to its own rail (zero or one tool-less turn); descendants settle first with a bounded child projection fed to the root's final turn; the grace budget starts at the durable control DRAIN (first drain wins), bounded by `request + OWNER_STOP_OUTER_CAP_SEC`, and neither anchor is ever progress-extended; `running_owner_stop_tasks` bypasses only the generic idle/finalization-grace rails; a COMPLETED finalize root suppresses the redundant cascade summary
|
||||
│ ├── schedule_time.py ← Cron/timezone schedule time parsing helpers
|
||||
|
|
@ -72,7 +72,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de
|
|||
├── agent.py ← Task orchestrator; the dispatch-note pair lives in `subagent_dispatch_notes.py` (same-name re-exports)
|
||||
├── agent_startup_checks.py ← Worker-boot verification: dirty repo, version sync, budget, memory files, health checks
|
||||
├── agent_task_pipeline.py ← Task execution pipeline orchestration; freezes one shared non-final subtree-cost snapshot for summary/reflection before the terminal checkpoint records final spend; hands the summary and reflection prompts the commit/advisory review lens PLUS the task's own acceptance-panel projection, and an absence statement names the lens it describes; calls the swarm-efficiency rollup owned by task_finalization.py at pipeline end
|
||||
├── task_finalization.py ← Terminal delivery + sealed final ground truth: live final-answer delivery before blocking post-task (final event selected by the finalizing task's id; buffered copy retained under one `delivery_id`), the sealed final package (delivered text + the durable result's own artifact manifest) fed to summary/reflection as a prompt input, never a validator; owns the per-task `swarm_efficiency` rollup (subagent_count / wave_count / inter-wave latency / `lanes_requested` — a rollup built from pre-dispatch fanout events cannot truthfully report effective lanes, which are per-child dispatch facts; `planned` stays null, never inferred as 0 from absent events; host-attested Swarm intent is the typed metadata `force_plan_source == "swarm"`, never prompt inspection, and a Swarm task that fanned out nothing records a minimal `no_fanout_observed` block instead of silence)
|
||||
├── task_finalization.py ← Terminal delivery + sealed final ground truth: live final-answer delivery before blocking post-task (final event selected by the finalizing task's id; buffered copy retained under one `delivery_id`), the sealed final package (delivered text + the durable result's own artifact manifest) fed to summary/reflection as a prompt input, never a validator; owns the per-task `swarm_efficiency` rollup (subagent_count / wave_count / inter-wave latency / `lanes_requested` — a rollup built from pre-dispatch fanout events cannot truthfully report effective lanes, which are per-child dispatch facts; `planned` stays null, never inferred as 0 from absent events; host-attested Swarm intent is the typed metadata `force_plan_source == "swarm"`, never prompt inspection, and a Swarm task that fanned out nothing records a minimal `no_fanout_observed` block instead of silence); a fanned-out root additionally carries a `depth` block (`requested_depth`/`permitted_depth`/`attempted_depth`/`achieved_depth` plus a typed status, `host_visible_only`) built from the root contract and its subtree's depth provenance, so a root that never carried its own request recovers it from the children it scheduled, and the terminal task_summary row reports it only when a request exists
|
||||
├── mutation_attribution.py ← Root-task baseline capture in the existing task result; clean-at-baseline Git candidate projection; terminal projection includes the committed interval delta
|
||||
├── process_interpreters.py ← Interpreter resolvers for the user process launch surfaces: one-time pre-guard unversioned-Python resolver + post-gates Node ladder (PATH-first health probe, bundled fallback, attested child-env PATH prepend; the probe EXECUTES a candidate, so it runs only after the dispatch gates approve the call)
|
||||
├── post_task_checkpoint.py ← Durable root post-task phase/final-cost checkpoint shared by task finalization and Project naming recovery
|
||||
|
|
@ -108,7 +108,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de
|
|||
├── owner_hurry.py ← Owner "hurry": a typed TASK-LOCAL acceleration latch, never a chat message; the durable `owner_hurry` projection is written by `update_json_locked` touching only its own keys — never `write_task_result`, whose status-regression guard could drop concurrent terminal fields — keyed by the real attempt identity `task["_attempt"]`; while latched, the next acceptance panel is skipped with zero reviewer calls (`acceptance_skip_applied`), remaining improvement passes overlay to 0 through `effective_budget_profile` (the immutable task_contract is never rewritten), and force-plan becomes task-locally advisory; the effect DIES WITH THE ATTEMPT (`retry_reset` on every same-id requeue producer), a never-applied request is marked `not_applied_before_terminal`, and the non-chat `owner_hurry` events are hidden from chat by `log_events.js`
|
||||
├── owner_quiz.py ← Owner-quiz lifecycle projection: worker-side `record_asked`, request-id-idempotent first-answer-wins `record_answered` (option index validated against the STORED labels), structural-only `reconcile_terminal` (open → expired_terminal at task done; no host TTL), `quiz_states` replay; same locked-writer idiom as owner_hurry, touching only the `owner_quiz` key
|
||||
├── routing_wait.py ← Root-parameterized SSOT of the durable routing-receipt waits (`wait_for_promotion_admission`, `wait_for_routing_annotation`); tools/control.py keeps thin wrappers, so the gateway picker dispatcher confirms clicks through the SAME receipts the LLM routing tools poll
|
||||
├── outcomes.py ← Typed task-outcome and acceptance-decision authority keeping the lifecycle/execution/objective/review/artifact/verification/child-absorption axes separate; policy denials, cosmetic exits, and ignored outcomes never masquerade as genuine tool failures; receipt reconciliation lives in `_outcome_receipts.py`, trace classification in `_outcome_tool_errors.py`
|
||||
├── outcomes.py ← Typed task-outcome and acceptance-decision authority keeping the lifecycle/execution/objective/review/artifact/verification/child-absorption axes separate; policy denials, cosmetic exits, and ignored outcomes never masquerade as genuine tool failures; receipt reconciliation lives in `_outcome_receipts.py`, trace classification in `_outcome_tool_errors.py`; a verification ledger above the inline threshold rides as a stub whose `summary` is re-projected from the refreshed artifact file at finalization, and the stub is never a source for entries or outcome axes (the ledger embeds the task contract, which its entry count excludes, so a stub is the normal shape for a swarm root)
|
||||
├── outcome_receipt_store.py ← Durable verification-receipt path/append/read authority + exact-row union of forked-child and canonical replicas; owns the zero-run WRITE enum (`incomplete`/`unknown` only — a zero-run "complete" is unverifiable self-report); outcomes.py re-exports the compatibility names
|
||||
├── depth_evidence.py ← Pure requested/permitted/attempted/achieved depth projection for root acceptance; missing admitted permission stays evidence-unknown rather than reconstructed from mutable live config
|
||||
├── _outcome_receipts.py ← Receipt parsing and the ONE canonical receipt identity (`receipt_canonical_identity` → `ReceiptIdentity`; invariant in §10): three independent components — `criterion_id`; structurally canonical `check` text PAIRED with its `check_rendering` stamp (quoted shell punctuation is data, not syntax, and receipts from different renderings are never the same verification — the stored string alone cannot say which renderer wrote it); and the raw-sorted `canonical_path_set` (whitespace untouched — a leading space is a legal filename byte); `ReceiptIdentity.key` selects ONE typed (kind, value) and sameness is that key's equality, never a match across kinds — the parts are disclosures, never the comparison; the per-kind normalization answer lives in the closed `IDENTITY_KINDS`/`KIND_NORMALIZES_COMMAND_TEXT` table, so a fourth kind must state its own answer in its own row rather than inherit a default; the outstanding sets `unreconciled_failed`/`unreconciled_masked` scan every candidate against ALL later reconcilers and collapse repeated failures of one check onto the freshest receipt; the shared disclosed projections (`receipt_identity_projection`, `disclosed_list_projection`) make every bound explicit — exact omitted counts plus a hash over the injective serialization, string bounding via the SSOT `utils.truncate_review_artifact`, never a hand-rolled slice; `verification_receipt_ledger_row` splats that projection, so a new receipt key is dropped unless added there
|
||||
|
|
@ -445,7 +445,7 @@ Workspace binding changes the contextual repo, never the system repo for BIBLE/p
|
|||
|
||||
Workspace preflight snapshots git state (bounded porcelain rows), manifests/scripts, and tool availability into the full `workspace_preflight.json` artifact with a bounded summary in metadata; `tools_on_path`/`tools_missing_from_path` are named that way because `shutil.which` measures PATH presence, not executability, and the structured keys stay frozen because they ride durable, replaying task metadata; a collection failure is a disclosed error summary, never a fictitious full artifact.
|
||||
|
||||
Completion compares against the captured preflight base (task-local commits stay in the delta, not `git diff HEAD`); the patch is bound to `task_constraint.base_sha`; a moved HEAD fails closed only for `self_worktree` (a shared tree relies on reverse-patch verification); an unborn repo diffs against the canonical empty tree. Patch capture streams the tracked binary diff plus admitted untracked files, excluding scratch/cache/junk/oversized/binary-untracked/incidental-lockfile entries with per-file reasons; a sensitive-looking untracked credential is excluded per-file and disclosed as `sensitive_blocked`. `workspace_patch.json` is written for EVERY workspace finalization (including no-change and failed) and is the truth source for CLI strict-patch — it distinguishes omitted vs no-op vs failed; `workspace.patch` exists only for `ready_with_changes`.
|
||||
Completion compares against the captured preflight base (task-local commits stay in the delta, not `git diff HEAD`); the patch is bound to `task_constraint.base_sha`; a moved HEAD fails closed only for `self_worktree` (a shared tree relies on reverse-patch verification); an unborn repo diffs against the canonical empty tree. Patch capture streams the tracked binary diff plus admitted untracked files, excluding scratch/cache/junk/oversized/binary-untracked/incidental-lockfile entries with per-file reasons; generated output (`dist/`, `build/`) is governed by the project's own `.gitignore`, honoured through `--exclude-standard`, not by a host name rule — git-ignored files are outside the capture universe and are not listed as exclusions; a sensitive-looking untracked credential is excluded per-file and disclosed as `sensitive_blocked`. `workspace_patch.json` is written for EVERY workspace finalization (including no-change and failed) and is the truth source for CLI strict-patch — it distinguishes omitted vs no-op vs failed; `workspace.patch` exists only for `ready_with_changes`.
|
||||
|
||||
Forked/empty task state lives under `data/state/headless_tasks/<task_id>/data`: a forked drive copies `identity.md`, `WORLD.md`, `registry.md` (a project fork carries `memory/knowledge/patterns.md` and omits global knowledge; an empty drive starts blank); dialogue, scratchpad, mailbox, and history never cross. The child drive is execution state: the result copies back to the canonical root, declared artifacts rebase to `data/task_results/artifacts/<task_id>/` (missing source = copy failure; collisions get a deterministic suffix), and verification-receipt replicas union with exact-row de-dup. Once the canonical result is terminal, late copy-back and effective reads pass through the same pure field-custody projection — the parent-owned terminal marker and cost/round/token fields cannot be overwritten. `memory_export.json` is an explicit artifact, never merged automatically.
|
||||
|
||||
|
|
@ -601,7 +601,7 @@ One shared WebSocket serves the whole application; Projects open no independent
|
|||
|
||||
### Navigation and shared UI contracts
|
||||
|
||||
Primary navigation exposes Chat (Main), a collapsible Projects group, Files, Skills, Widgets, Dashboard, and Settings; About is a Settings sub-tab. `syncNavigationState()` is the single presentation state machine for active page, active Project, Projects expansion, mobile drawer, and panel backdrop — independent toggles must not leave multiple rows active or a hidden surface looking selected. Sidebar and Project panel widths are owner-local UI preferences, not runtime settings.
|
||||
Primary navigation exposes Chat (Main), a collapsible Projects group, Files, Skills, Widgets, Dashboard, and Settings; About is a Settings sub-tab. `syncNavigationState()` is the single presentation state machine for active page, active Project, Projects expansion, mobile drawer, and panel backdrop — independent toggles must not leave multiple rows active or a hidden surface looking selected. Sidebar and Project panel widths are owner-local UI preferences, not runtime settings. The application has exactly one client-side route: a `#<page>` fragment for the injected page ids, honoured once on load and validated against the existing `#page-<name>` section, so an unknown fragment is ignored rather than painting a blank surface. It is never written back on navigation — the packaged desktop shell and the Linux browser fallback have no address bar to read it from, and the Telegram mini app always loads at `/` — so a browser gains a shareable `/#widgets` link while no other surface changes. Projects are panels, not routes, and the sidebar is the document's one `<nav>` landmark.
|
||||
|
||||
Each active or deleting Project has a sidebar row with pointer- and keyboard-operable open/rename/delete; the backend owns the 80-character name limit and lifecycle truth. Unread Projects sort ahead of read, then by durable activity; a deleting Project becomes non-openable and stays visibly transitional until the server publishes authoritative registry state. On narrow screens navigation is an explicit drawer and the Project chat a full-width overlay; no gesture-only navigation layer competes with message scroll, text selection, or the software keyboard.
|
||||
|
||||
|
|
@ -628,7 +628,7 @@ Messages, media bubbles, and task-card roots order by raw numeric timestamps (ti
|
|||
|
||||
History reconciliation is two-pass: progress and system records first rebuild timestamped task-card state, messages and cards insert chronologically, and only then may terminal state seal a card — so terminal replay cannot discard earlier progress, and cards whose final summary was missed offline recover. Live echoes and history rows deduplicate on stable message identities. That rebuild is also the memory bound: cards are minted from progress rows, `task_summary` rows and subagent lineage, so the cap is relative to the population the last rebuild produced — once more than 200 live cards exist beyond it the next history sync replays durable history instead of folding into the existing cards, and that rebuild sets the new floor (`web/modules/chat.js`).
|
||||
|
||||
The Main chat receives ordinary main-thread dialogue plus exactly the two host-stamped Project lifecycle rows — `project_started` and the terminal `project_completion_summary`, each with the shared "Open Project" action; all other Project traffic stays in the Project thread. Lifecycle rows are plain dashboard text: the producer strips markdown once before durable write and live send, history normalizes older rows on read (`chat.jsonl` is never rewritten), and the renderer escapes any system row without `markdown: true` (except `skill_review`'s dedicated renderer). Project panels accept only their own registered `chat_id`; the complete Project chat-id set is separate from the sidebar's bounded summary, and a `projects_changed` frame adds a new chat id synchronously before the asynchronous state refresh, so an early frame cannot be misclassified as Main.
|
||||
The Main chat receives ordinary main-thread dialogue plus exactly the two host-stamped Project lifecycle rows — `project_started` and the terminal `project_completion_summary`, each with the shared "Open Project" action; all other Project traffic stays in the Project thread. Lifecycle rows are plain dashboard text: the producer strips markdown once before durable write and live send, history normalizes older rows on read (`chat.jsonl` is never rewritten), and the renderer escapes any system row without `markdown: true` (except `skill_review`'s dedicated renderer). Finality metadata is row-specific. Durable `task_summary` rows carry the SAME phase the browser paints: `project_dialogue.outcome_phase` mirrors `log_events.js` — the terminality gate plus the severity fold over normalized axes — and stamps it as `outcome_phase` beside the status word taken from `OUTCOME_PHASE_HEADLINE`, so there is no second status-word family. The pre-finalization authored row carries the phase with `outcome_final=false` but no host verdict clause; only terminal `task_summary` rows append the host verdict clause (`_completion_verdict`, the host twin of `taskReasonDetail`'s acceptance branch) when one exists. The Main `project_completion_summary` instead puts the shared headline and any verdict directly in its plain text; it does not carry `outcome_phase`. The start row carries neither outcome finality nor a verdict. A verdict is the execution reason on ordinary terminal summaries, or — when the host's acceptance decision is anything but accepted — the decision status with its stored rationale, markdown-stripped and flattened like every other lifecycle-row text; the latter leads the Main completion row so a host-authored row never presents an unaccepted claim as the whole story. History keeps older status words unrewritten, and the Telegram skill reads the stamped task-summary phase instead of deriving one of its own. Project panels accept only their own registered `chat_id`; the complete Project chat-id set is separate from the sidebar's bounded summary, and a `projects_changed` frame adds a new chat id synchronously before the asynchronous state refresh, so an early frame cannot be misclassified as Main.
|
||||
|
||||
The Main composer exposes one-shot Swarm planning, the owner Low/Max context choice, file attachment, and Send. Swarm places a structural `force_plan` fact on the next ordinary message and disarms after that send — never inferred from keywords. Low/Max uses the dedicated owner endpoint; a derived Low fit never masquerades as an owner-selected posture, and selecting Max may invoke the exact-route capability acknowledgement when provider metadata cannot prove the window. Project panels omit global Restart, Panic, evolution, consciousness, review, and budget controls — those belong to the one Ouroboros process, not a Project thread.
|
||||
|
||||
|
|
@ -640,7 +640,7 @@ Direct in-process chat turns and ephemeral decision turns create no supervisor q
|
|||
|
||||
Owner-message continuity is journal-backed: each locally sent owner row is kept (bounded, with its routing annotation) until a fetched history response returns the same `client_message_id`, and a full rebuild re-renders unconfirmed journal rows after the server response, so a stale history snapshot cannot erase a message the owner just sent. Task finalization is honest: a root's early final answer carries `task_phase="finalizing"` (the same fact replay derives from the open post-task checkpoint), the card holds a sticky `Finalizing…` phase, and `task_cost_finalized` is bookkeeping that never resolves a card. If an observed managed root disappears from a queue-authoritative snapshot whose request began after the observation, the state-refresh fan-out removes activity/cancel authority and starts one single-flight durable task-detail read; request generations prevent an older response from undoing a newer projection. Two liveness invariants close the stuck-`Working...` class: the history window is lineage-closed — subagent lineage older than the window's progress recency floor is kept only while the child still runs or finalizes, or its parent is REPRESENTED by the response (a row that proves the task's card: telemetry, its own message or summary, a folded review group or plan reference it owns — a typed photo/document/quiz delivery does NOT count), or the parent is alive, and an anchored child represents its own children in turn, so a nested swarm is kept or dropped whole; a FINAL row failing that test is emitted without lineage fields (`ouroboros/gateway/history.py`, one anchor set computed in the quota pass), so replay cannot mint an unfinishable parent card while a visible or live parent keeps its completed children and their executor receipts — and liveness reconciles over the CARD SET, not only the activity registry: an unvouched connected root card is finished ONLY by proven durable terminal detail, and a card with no durable result honestly keeps its state. The header badge has exactly one writer, the status reducer, with the panel-boot "Online" seed as the sole exception.
|
||||
|
||||
Task activity collapses into a live task card per root instead of flooding the transcript; `log_events.js` keeps the live task card and grouped task cards on one reducer across Chat and Dashboard Logs. Non-terminal LLM/tool/checkpoint failures stay inspectable timeline facts but do not promote the card; only authoritative terminal task truth changes terminal status, and unknown Chat event names do not acquire severity from keyword substrings. A failed child keeps a local `Failed` chip while its root continues. A card keeps a concise plain-text latest-activity line and an expandable timeline; server-truncated rows fetch the complete typed task record on demand into a bounded viewer. The compact card is a navigation projection, not the result authority; urgent toast/unread behavior stays confined to explicit `task_incident` facts, never inferred from severity.
|
||||
Task activity collapses into a live task card per root instead of flooding the transcript; `log_events.js` keeps the live task card and grouped task cards on one reducer across Chat and Dashboard Logs. Non-terminal LLM/tool/checkpoint failures stay inspectable timeline facts but do not promote the card; only authoritative terminal task truth changes terminal status, and unknown Chat event names do not acquire severity from keyword substrings. A failed child keeps a local `Failed` chip while its root continues. A card keeps a concise plain-text latest-activity line and an expandable timeline; server-truncated rows fetch the complete typed task record on demand into a bounded viewer. The compact card is a navigation projection, not the result authority; urgent toast/unread behavior stays confined to explicit `task_incident` facts, never inferred from severity. The ONE explanatory line under a terminal headline is `taskReasonDetail`: an owner-requested soft stop has none, a hard failure or a cancellation names its execution reason, a non-accepted host acceptance decision names the decision status plus its stored rationale, and everything else keeps the typed reason phrase (an unknown code stays raw); Logs meta additionally names `review <status>` and `acceptance <status>`.
|
||||
|
||||
Subagents render as distinct child cards keyed by their actual child task ids; parents keep lineage references without duplicating the child's final answer. Nested children collapse by default: headline `role · model` (short task id only when a sibling shares both; Logs keep the full form), status carried by the chip alone. Reviews are not task lineage: a reviewer run's execution receipt belongs inside the real owning task card, and its harness or neutral API mark identifies the delivery channel without minting a child card or proving execution. Selected `subagent_id`/configured snapshot, requested route, effective engine route/model/account, and terminal execution evidence stay separate facts — intent must not be redrawn as proof of where the run settled. Legacy lane/executor fields stay readable on historical cards only.
|
||||
|
||||
|
|
@ -936,7 +936,7 @@ The delivery-control protocol is resolved here and only here (DEVELOPMENT keeps
|
|||
|
||||
The child-absorption gate is an action gate: while undispositioned direct children remain, the loop HOLDS the candidate (`child_absorption_or_revision_required`) instead of arming the JSON-only control instruction — the hold-vs-arm split exists so the model never receives two contradictory instructions in one round. A typed keep cannot close the gate, and after the one bounded reminder the gate forces the best-effort `children_unabsorbed` rail with a current `id [status] sha256` listing. The absorption digest's `## child` header carries the child's typed custody debt (`delegated_runs_unreconciled`, the first ten ids plus an omitted count and a `get_task_result` pointer) — visibility only: the parent's authority over that patch is exactly the orphan rule, and a child's debt never relabels the root card (DESIGN §4, a failed child keeps a compact factual marker inside its parent while the root continues under its own authoritative status). The acceptance-subtree copy of that debt is a disclosed gap, deferred with the pre-finalization reminder. Finalizing over an UNDISPOSED OWN delegated patch is deliberately NOT gated: the consequence is disclosed where the decision is made — in the `integrate_delegated_patch` schema and in the apply receipts — and lands as the additive Done-with-warnings custody overlay rather than a hold. A pre-finalization reminder for that case is deferred.
|
||||
|
||||
Provider death is the one forced rail that is NOT a best-effort completion: `_handle_provider_unavailable` salvages the best available text but stamps `infra_failed`, so the task terminalizes `failed` with the typed `provider_unavailable` reason and an immediate "provider outage — NOT completed" owner notification; a waited-out transport outage reaches the same rail through its deterministic no-resend branch (`transport_unavailable_no_resend`). The rail makes its one forced model call only while a call can still land: a transport that spent its same-model retry wall stamps `_llm_retry_wall_exhausted`, which the rail reads as a no-call gate beside `context_overflow` and `provider_outcome_unknown`, shipping the salvage with no request — so the forced rail does not re-pay a second retry window over a proven-dead provider; the marker is a last-invocation bool on the shared usage dict, a disclosed residual. Terminal delivery preserves producer authorship: complete model answers appear as Ouroboros, host-authored incident receipts as System.
|
||||
Provider death is the one forced rail that is NOT a best-effort completion: `_handle_provider_unavailable` salvages the best available text but stamps `infra_failed`, so the task terminalizes `failed` with the typed `provider_unavailable` reason and an immediate "provider outage — NOT completed" owner notification; a waited-out transport outage reaches the same rail through its deterministic no-resend branch (`transport_unavailable_no_resend`). The rail makes its one forced model call only while a call can still land: a transport that spent its same-model retry wall stamps `_llm_retry_wall_exhausted`, which the rail reads as a no-call gate beside `context_overflow` and `provider_outcome_unknown`, shipping the salvage with no request — so the forced rail does not re-pay a second retry window over a proven-dead provider; the marker is a last-invocation bool on the shared usage dict, a disclosed residual. Terminal delivery preserves producer authorship. Every forced rail stamps one closed-vocabulary producer word at the single forced-finalization sink: `model_final` for complete model text (including a host disclosure glued inside it — plan review left open, deferred children — exactly as the ordinary no-tool final does), `host_notice` for a terminal text the host wrote alone (kept verbatim on every transport, with its markdown, and projected as a System row without a system_type), and `host_salvage` only on the provider-death rail, where the text is replaced by the outage receipt and the full bytes stay in task details; the provider-death wrapper upgrades a bare notice to salvage so a no-call/no-resend arm keeps its receipt. A missing origin identifies only rows written before that stamp existed.
|
||||
|
||||
Host-enforced task acceptance is a root-owned completion coach, not the P3 commit gate; its mechanism is stated once, here. `off` disables it. In `auto` and `required`, substantive queued, headless, and scheduled roots are eligible; direct chat becomes eligible only after an observable reviewable effect or a typed deliverable/criterion — pure conversation, read-only exploration, routing turns, and cognitive-memory updates create no eligibility. Child reviews are advisory evidence superseded by the root decision.
|
||||
|
||||
|
|
@ -1065,7 +1065,7 @@ Children coordinate through `tree_note` and `tree_read`; only the parent may use
|
|||
|
||||
An undisclosed spend contributes `0.0` to `accounted_usd` — inventing a conservative bound would fabricate a number the harness never gave (BIBLE P1) — so a `TOTAL_BUDGET` fence cannot stop spend it was never told about; the honest consequence is loss of finality, not a guessed charge.
|
||||
|
||||
**The nanny model.** An `agent_session` subagent is an ordinary recursive task-tree child acting as a **nanny** supervising at most one active bounded external leaf. The task node keeps lineage, authority, deadline, budget, acceptance, cancellation, and descendants; the harness process remains a non-recursive tool leaf — session rows never flatten the task tree into harness processes. The nanny is the host: verification receipts stay host-authored, and harness output is a claim to check, never proof. Harness-agnostic by construction: the row holds an opaque Claudexor target, Ouroboros asks for an access profile derived from task authority and lets Claudexor choose the mechanism; no harness-name branch selects a capability or fallback in core dispatch, and login-wire asymmetries stay presentation adapters in `gateway/claudexor_accounts.py`.
|
||||
**The nanny model.** An `agent_session` subagent is an ordinary recursive task-tree child acting as a **nanny** supervising at most one active bounded external leaf. The task node keeps lineage, authority, deadline, budget, acceptance, cancellation, and descendants; the harness process remains a non-recursive tool leaf — session rows never flatten the task tree into harness processes. The nanny is the host: verification receipts stay host-authored, and harness output is a claim to check, never proof. A nanny-to-nanny chain through `schedule_subagent` is the host-attested realization of a nested subscription swarm, each level one metered supervisor task plus one free harness run; `schedule_subagent.requested_depth` is the typed ABSOLUTE request counted from the root, recorded as telemetry that never narrows the configured caps, while the legacy `depth_remaining` envelope keeps its narrowing semantics — two semantics, both disclosed on the contract, and the root's `swarm_efficiency.depth` block plus its terminal summary row report requested, permitted and achieved with the typed status. Harness-agnostic by construction: the row holds an opaque Claudexor target, Ouroboros asks for an access profile derived from task authority and lets Claudexor choose the mechanism; no harness-name branch selects a capability or fallback in core dispatch, and login-wire asymmetries stay presentation adapters in `gateway/claudexor_accounts.py`.
|
||||
|
||||
**Transport.** `gateways/claudexor.py` is pure transport (descriptor read, `/v2` handshake, `config.CLAUDEXOR_MIN_VERSION` 3.2.0 floor). The daemon bearer token grants the entire `/v2` surface, so it never leaves this module, and the HTTP client runs `trust_env=False` so an ambient proxy cannot intercept the loopback control plane. Production starts obtain a handshaken owned gateway from `claudexor_daemon.ensure_owned_gateway` (exact reviewed engine/Node pins; never a PATH install); keeping lifecycle above transport keeps account status side-effect-free and harness mechanisms out of Ouroboros (`claudexor_daemon.py`/`claudexor_runtime.py` docstrings).
|
||||
|
||||
|
|
@ -1540,7 +1540,7 @@ otherwise the view is partial and the consumer remains non-final or abstains.
|
|||
| Surface | Canonical source and owner | Bounded projection | Actor-readable source/ref | Decision and retention rule |
|
||||
|---|---|---|---|---|
|
||||
| Owner authority and biography | Canonical `logs/chat.jsonl`, archive generations, and `memory/dialogue_blocks.json` owned by the canonical drive | Main/Project context sections and archive-aware history windows | Existing `chat_history`/archive readers with generation and gap metadata | A known gap is disclosed; summaries/blocks never replace exact current owner directives. Raw generations and durable blocks follow their existing retention owner. |
|
||||
| Execution evidence | Task results, observability call manifests/blobs, service logs, and process-custody records | Status cards, terminal rows, bounded tails, and compact child summaries | Exact artifact/blob/service-log refs carried by the task result or canonical promotion | A projection cannot certify a missing child/source. Referenced canonical artifacts are promoted before child-drive GC; disposable execution scratch follows unified GC. |
|
||||
| Execution evidence | Task results, observability call manifests/blobs, service logs, and process-custody records | Status cards, terminal rows, bounded tails, and compact child summaries | Exact artifact/blob/service-log refs carried by the task result or canonical promotion | A projection cannot certify a missing child/source. Referenced canonical artifacts are promoted before child-drive GC; disposable execution scratch follows unified GC. An omitted-to-artifact verification ledger stub carries only its re-projected `summary`; entries and axes are read from the artifact file it points at. |
|
||||
| Terminal task/project memory | Root terminal result plus existing task/project summary producers | Cognitive Main terminal summaries and the two Project-root UI lifecycle rows (started + terminal completion) | Task-result ID, project binding, and summary/source refs | Summary is a biography projection, not raw evidence. Terminal outcomes, including failed/cancelled/degraded, remain retained through their canonical result owner. |
|
||||
| Background Consciousness observations | `data/state/consciousness_observations.jsonl`, append-only enqueue/ACK rows owned by `BackgroundConsciousness` | Pending count/oldest metadata and a bounded recent observation rendering | `read_file(root='runtime_data', path='state/consciousness_observations.jsonl')` | Unacknowledged rows survive restart/overflow/error. Gaps block ACK and the existing direct identity rewrite; only a settled successful cycle appends ACK. |
|
||||
| Plan/review authority | Exact task-artifact/observability wave bodies, evidence selectors, reviewer route/thread receipts, and the bounded review hot index | Review status, latest wave, obligations, and compact findings | Exact artifact/source handle plus SHA/range/thread selectors | Missing or partial evidence is `DEGRADED`/`NOT_RUN`, never PASS. Exact artifacts remain bound to the reviewed candidate SHA; hot indexes may rotate only after the source is retained. |
|
||||
|
|
|
|||
|
|
@ -176,7 +176,9 @@ Status, owner action, and urgent notification are separate product concepts:
|
|||
|
||||
- **Status** states a fact about the affected object. It does not imply that the
|
||||
owner can or must act. Task status uses one factual word family: `Working`,
|
||||
`Done`, `Done with warnings`, `Failed`, `Cancelled`.
|
||||
`Done`, `Done with warnings`, `Failed`, `Cancelled`. The same five words are
|
||||
the host's durable label vocabulary for its own task rows in Main and the
|
||||
Project thread (`OUTCOME_PHASE_HEADLINE`), not only the browser's.
|
||||
- **Owner action** exists only when the responsible domain exposes a current
|
||||
concrete continuation, such as Resume, Retry, Connect, Repair, Grant access,
|
||||
or Restart. The action is a real adjacent control; severity alone never
|
||||
|
|
|
|||
|
|
@ -1320,7 +1320,11 @@ both critical. The imperatives:
|
|||
`integrate_subagent_patch` and runs its own `commit_reviewed`. The shared
|
||||
`external_workspace` surface verifies and records without re-applying; a
|
||||
genesis project is durable because the project directory IS the
|
||||
deliverable. The canonical/replica terminal field-custody projection is
|
||||
deliverable. A genesis project starts without a `.gitignore`, so its small
|
||||
text build output (`dist/`, `build/`) rides the `workspace.patch` record
|
||||
until the project declares one — a disclosed residual, bounded only by the
|
||||
per-file size cap and git's binary verdict, since there is no total-patch
|
||||
cap. The canonical/replica terminal field-custody projection is
|
||||
ONE pure reducer reused by copy-back and effective reads — every change
|
||||
adds a stale-replica regression at BOTH seams
|
||||
(`tests/test_available_subagents_runtime_review_fixes.py`). Do not broaden
|
||||
|
|
@ -1719,7 +1723,9 @@ by "Provider Independence" above. Call-site imperatives:
|
|||
Binding the complete-result SHA-256 means a parent cannot claim it
|
||||
integrated a result that later changed. `deferred` suppresses only the
|
||||
reminder and forces an honest degraded/best-effort terminal answer until
|
||||
resolved. A child wedged in the legacy `cancel_requested` latch is intent,
|
||||
resolved. That per-value consequence is carried by the
|
||||
`tree_note` payload schema itself, which is the SSOT for when to choose each
|
||||
value, and its enum reads the validator's own set. A child wedged in the legacy `cancel_requested` latch is intent,
|
||||
not outcome — it stays visible as cancel-pending until custody settles it.
|
||||
- Host task acceptance is root-only; eligibility uses structured facts
|
||||
(`outcomes.turn_has_reviewable_effects` plus a typed
|
||||
|
|
@ -1753,9 +1759,11 @@ by "Provider Independence" above. Call-site imperatives:
|
|||
resulting pair stays outside `_ACCEPTANCE_BLOCKED_TERMINAL_REASONS`.
|
||||
The reviewer verdict vocabulary `PASS|FAIL|DEGRADED` is NOT narrowable;
|
||||
`adaptive_quorum` applies, any contributing FAIL fails, DEGRADED abstains,
|
||||
and no quorum is a terminal HOST decision. Chat and Logs use the same
|
||||
severity reducer; degraded review or a best-effort objective must never
|
||||
render as green solved. Do not add task scope review or reuse the commit
|
||||
and no quorum is a terminal HOST decision. Chat, Logs, the
|
||||
durable Project lifecycle rows and the Telegram notifier use the same phase
|
||||
(`taskOutcomeSeverity`/`taskTerminalPhase` mirrored by
|
||||
`project_dialogue.outcome_phase`); degraded review or a best-effort
|
||||
objective must never render as green solved on any of them. Do not add task scope review or reuse the commit
|
||||
gate.
|
||||
- The acceptance improvement loop is a reviewer-authored DIALOGUE: obligation
|
||||
identity comes from the reviewer's typed
|
||||
|
|
@ -2009,7 +2017,14 @@ rules, not a copied color/radius/dimension inventory.
|
|||
- Task outcome truth stays in `log_events.js::taskOutcomeSeverity` and
|
||||
`taskTerminalPhase`; `taskPresentation` is the one compact factual
|
||||
projection consumed by chips, live completion, history replay, and child
|
||||
terminal presentation. A non-terminal diagnostic may add a timeline fact
|
||||
terminal presentation. Its host mirror is
|
||||
`project_dialogue.outcome_phase`, pinned to the browser by one shared
|
||||
fixture (`web/tests/fixtures/outcome_phase_parity.json`): a new axis, reason
|
||||
or acceptance status is added to both sides in the same commit, with a row
|
||||
in that fixture. The detail line under the headline comes from
|
||||
`taskReasonDetail` in the order soft stop, hard failure or cancellation
|
||||
reason, host acceptance decision (status plus stored rationale), typed
|
||||
reason phrase or raw code — never from a second producer. A non-terminal diagnostic may add a timeline fact
|
||||
but must not promote the whole task; unknown event names never acquire
|
||||
Chat severity from `error`/`crash`/`fail` keyword matching. The Chat
|
||||
header reports connection and server-authoritative activity only; failed
|
||||
|
|
|
|||
|
|
@ -1207,7 +1207,7 @@ def _run_task_summary(env, llm, task, usage, llm_trace, drive_logs, review_evide
|
|||
sealed_final=None):
|
||||
"""Generate a detailed task summary and inject it into chat.jsonl."""
|
||||
try:
|
||||
from ouroboros.project_dialogue import append_authored_task_summary, completion_status_label
|
||||
from ouroboros.project_dialogue import append_authored_task_summary, completion_status_label, outcome_phase
|
||||
from ouroboros.projects_registry import project_thread_note_for_task
|
||||
from ouroboros.consolidator import CONSOLIDATION_REASONING_EFFORT, _consolidation_route
|
||||
task_id = str(task.get("id") or "unknown")
|
||||
|
|
@ -1230,7 +1230,7 @@ def _run_task_summary(env, llm, task, usage, llm_trace, drive_logs, review_evide
|
|||
"summary_kind": "authored_root_summary", "summary_id": summary_id,
|
||||
"task_id": task_id, "parent_task_id": str(task.get("parent_task_id") or ""), "root_task_id": str(task.get("root_task_id") or task_id),
|
||||
"project_id": str(task.get("project_id") or ""), "chat_id": int(task.get("chat_id") or 0), "delegation_role": str(task.get("delegation_role") or ""), "role": str(task.get("role") or ""),
|
||||
"status": str(stored_result.get("status") or "completed"), "outcome": completion_status_label(stored_result, usage),
|
||||
"status": str(stored_result.get("status") or "completed"), "outcome": completion_status_label(stored_result, usage), "outcome_phase": outcome_phase(stored_result, usage),
|
||||
"outcome_final": False, "outcome_authority": "pre_finalization_narrative_context",
|
||||
"text": value, "tool_calls": n_tool_calls, "rounds": rounds, "outcome_axes": outcome_axes, "reason_code": reason_code,
|
||||
"result_ref": result_ref, "source_coverage": {"task_result": result_ref}, **_summary_row_cost_fields(usage), **presence_fields,
|
||||
|
|
|
|||
|
|
@ -82,11 +82,11 @@ def build_depth_summary(
|
|||
statuses = [row for row in subtree_statuses if isinstance(row, dict)]
|
||||
provenances = [task_depth_provenance(row) for row in statuses]
|
||||
provenances = [row for row in provenances if row]
|
||||
if requested is None:
|
||||
requested = next(
|
||||
(row.get("requested_depth") for row in provenances if row.get("requested_depth") is not None),
|
||||
None,
|
||||
)
|
||||
requested_values = [
|
||||
value for value in [requested, *(row.get("requested_depth") for row in provenances)]
|
||||
if value is not None
|
||||
]
|
||||
requested = max(requested_values) if requested_values else None
|
||||
permitted_values = [
|
||||
value
|
||||
for value in [
|
||||
|
|
@ -97,15 +97,57 @@ def build_depth_summary(
|
|||
]
|
||||
permitted = min(permitted_values) if permitted_values else None
|
||||
|
||||
def _maximum(key: str) -> Any:
|
||||
values = [row.get(key) for row in provenances if row.get(key) is not None]
|
||||
def _maximum(key: str, rows: list = provenances) -> Any:
|
||||
values = [row.get(key) for row in rows if row.get(key) is not None]
|
||||
if values:
|
||||
return max(values)
|
||||
return 0 if not statuses else None
|
||||
|
||||
attempted = _maximum("attempted_depth")
|
||||
achieved = _maximum("achieved_depth")
|
||||
if requested is None:
|
||||
|
||||
# Decision: rows are per-task depth facts of one tree. Rows sharing one
|
||||
# (request, permitted) pair form a chain whose achievement is its deepest
|
||||
# row; a tree with several chains reports the most-reduced chain first
|
||||
# (capability_reduced > evidence_unknown > chosen_shallower > achieved), so
|
||||
# the verdict and its coherent chain tuple never depend on child order.
|
||||
def _chain_status(ask: Any, cap: Any, rows: list) -> str:
|
||||
if ask is None:
|
||||
return "request_unknown"
|
||||
reached = [row.get("achieved_depth") for row in rows]
|
||||
tried = [row.get("attempted_depth") for row in rows]
|
||||
if cap is not None and cap < ask:
|
||||
return "capability_reduced"
|
||||
known = [value for value in reached if value is not None]
|
||||
if known and max(known) >= ask:
|
||||
return "achieved"
|
||||
if cap is None or None in reached or None in tried:
|
||||
return "evidence_unknown"
|
||||
return "chosen_shallower"
|
||||
|
||||
chains: Dict[Any, list] = {}
|
||||
for row in provenances:
|
||||
chains.setdefault((row.get("requested_depth"), row.get("permitted_depth")), []).append(row)
|
||||
if chains:
|
||||
order = ["request_unknown", "capability_reduced", "evidence_unknown", "chosen_shallower", "achieved"]
|
||||
decisions = [
|
||||
(
|
||||
_chain_status(ask, cap, rows), ask, cap,
|
||||
_maximum("attempted_depth", rows), _maximum("achieved_depth", rows),
|
||||
)
|
||||
for (ask, cap), rows in chains.items()
|
||||
]
|
||||
status, requested, permitted, attempted, achieved = min(
|
||||
decisions,
|
||||
key=lambda item: (
|
||||
order.index(item[0]),
|
||||
-(item[1] if item[1] is not None else -1),
|
||||
item[4] if item[4] is not None else -1,
|
||||
item[2] if item[2] is not None else -1,
|
||||
item[3] if item[3] is not None else -1,
|
||||
),
|
||||
)
|
||||
elif requested is None:
|
||||
status = "request_unknown"
|
||||
elif permitted is None or attempted is None or achieved is None:
|
||||
status = "evidence_unknown"
|
||||
|
|
|
|||
|
|
@ -484,11 +484,9 @@ def _copy_task_summary_metadata(rec: Dict[str, Any], entry: Dict[str, Any]) -> N
|
|||
rec["reason_code"] = str(entry.get("reason_code") or "")
|
||||
if isinstance(entry.get("review_projection"), dict):
|
||||
rec["review_projection"] = dict(entry.get("review_projection") or {})
|
||||
# v6.82 P1: the summary row now carries the flat task-scope cost snapshot
|
||||
# written by agent_task_pipeline; replay it so a reload still shows cost.
|
||||
# _annotate_terminal_task_truth later OVERRIDES these with the persisted
|
||||
# task_results values when the result file survives (row = fallback only).
|
||||
for key in _TASK_COST_META_FIELDS:
|
||||
# The summary row is the pruned-result fallback for cost and finality.
|
||||
# Persisted task_results values still override cost fields below.
|
||||
for key in ("outcome_phase", "outcome_final", *_TASK_COST_META_FIELDS):
|
||||
if key in entry:
|
||||
rec[key] = entry[key]
|
||||
|
||||
|
|
@ -546,6 +544,7 @@ def _annotate_terminal_task_truth(
|
|||
the pre-computed set — zero extra reads."""
|
||||
|
||||
try:
|
||||
from ouroboros.project_dialogue import outcome_phase
|
||||
from ouroboros.task_status import FINAL_STATUSES
|
||||
|
||||
cache = result_cache if result_cache is not None else {}
|
||||
|
|
@ -599,6 +598,7 @@ def _annotate_terminal_task_truth(
|
|||
terminal_status_by_task[task_id] = status
|
||||
terminal_truth: Dict[str, Any] = {
|
||||
"outcome_axes": normalize_outcome_axes(result),
|
||||
"outcome_phase": outcome_phase(result, {}), "outcome_final": True,
|
||||
}
|
||||
if result.get("reason_code"):
|
||||
terminal_truth["reason_code"] = str(result.get("reason_code") or "")
|
||||
|
|
|
|||
|
|
@ -832,10 +832,16 @@ def finalize_task_artifacts(parent_drive_root: pathlib.Path, task: Dict[str, Any
|
|||
item["size"] = len(data)
|
||||
item["sha256"] = sha256(data).hexdigest()
|
||||
item["status"] = ARTIFACT_STATUS_READY
|
||||
stub = fields.get("verification_ledger")
|
||||
if isinstance(stub, dict) and stub.get("omitted_to_artifact") and isinstance(refreshed_artifact_ledger.get("summary"), dict):
|
||||
fields["verification_ledger"] = {**stub, "summary": dict(refreshed_artifact_ledger["summary"])}
|
||||
except Exception:
|
||||
log.debug("Failed to refresh verification ledger artifact for task %s", task_id, exc_info=True)
|
||||
# The loop rewrote the ledger file in place, so the bundle computed
|
||||
# before it no longer describes the bytes on disk.
|
||||
fields["artifact_bundle"] = artifact_bundle_from_result(provisional)
|
||||
except Exception:
|
||||
pass
|
||||
log.warning("Artifact bundle/ledger refresh failed for task %s", task_id, exc_info=True)
|
||||
write_task_result(
|
||||
parent_drive_root,
|
||||
task_id,
|
||||
|
|
|
|||
|
|
@ -68,6 +68,7 @@ from ouroboros.usage_accounting import (
|
|||
last_physical_attempt_capture,
|
||||
)
|
||||
from ouroboros.task_finalization import (
|
||||
TERMINAL_ORIGIN_HOST_NOTICE,
|
||||
TERMINAL_ORIGIN_HOST_SALVAGE,
|
||||
TERMINAL_ORIGIN_MODEL_FINAL,
|
||||
)
|
||||
|
|
@ -2701,17 +2702,16 @@ def _handle_provider_unavailable(
|
|||
ctx: _RoundLimitContext, *, error_kind: str = "provider_unavailable",
|
||||
wait_cause: str = "", waited: bool = False, wait_eligible: bool = True,
|
||||
) -> Tuple[str, Dict[str, Any], Dict[str, Any]]:
|
||||
"""Provider-death rail wrapper: every arm carries a terminal provenance;
|
||||
``setdefault`` keeps explicit stamps authoritative. Not every exit is a
|
||||
provider death: the deadline grace arm can return a MODEL-AUTHORED final
|
||||
(``deadline_local``) and a scheduled swarm handoff pops its reason code —
|
||||
both keep their legacy shape."""
|
||||
"""Provider-death rail wrapper: every arm carries terminal provenance.
|
||||
The forced-finalization sink stamps ``host_notice``; retained/generated
|
||||
model candidates stamp ``model_final``."""
|
||||
text, usage, llm_trace = _provider_unavailable_result(
|
||||
ctx, error_kind=error_kind, wait_cause=wait_cause, waited=waited,
|
||||
wait_eligible=wait_eligible,
|
||||
)
|
||||
if str(usage.get("reason_code") or "") not in ("", "deadline_local"):
|
||||
usage.setdefault("terminal_origin", TERMINAL_ORIGIN_HOST_SALVAGE)
|
||||
if str(usage.get("reason_code") or "") not in ("", "deadline_local") and usage.get(
|
||||
"terminal_origin") in (None, TERMINAL_ORIGIN_HOST_NOTICE):
|
||||
usage["terminal_origin"] = TERMINAL_ORIGIN_HOST_SALVAGE
|
||||
return text, usage, llm_trace
|
||||
|
||||
|
||||
|
|
@ -2759,9 +2759,7 @@ def _provider_unavailable_result(
|
|||
)
|
||||
return text, usage, llm_trace
|
||||
if is_transport_wait:
|
||||
# No-resend terminal: salvage, no forced-final call over a dead
|
||||
# egress. Stamp BEFORE the composer (owner-stop pattern): a SCHEDULED
|
||||
# swarm handoff clears it; guard mirrors the sibling below.
|
||||
# No-resend terminal over dead egress.
|
||||
live_trace = getattr(ctx, "llm_trace", None)
|
||||
llm_trace = live_trace if isinstance(live_trace, dict) else {}
|
||||
ctx.accumulated_usage["execution_status"] = RESULT_INFRA_FAILED
|
||||
|
|
@ -2773,7 +2771,6 @@ def _provider_unavailable_result(
|
|||
if str(usage.get("reason_code") or "") == "provider_unavailable":
|
||||
usage["execution_status"] = RESULT_INFRA_FAILED
|
||||
return text, usage, llm_trace
|
||||
# No-call shapes; see provider_no_call_source
|
||||
no_call, wall = provider_no_call_source(ctx.accumulated_usage, is_deadline_exhausted)
|
||||
if no_call:
|
||||
if wall:
|
||||
|
|
@ -3404,6 +3401,7 @@ def _record_forced_finalization(
|
|||
# answer (`_forced_final_answer`) and the no-spend host-fallback fence
|
||||
# path (`_handle_budget_exceeded` -> `_forced_fallback_result`).
|
||||
_record_forced_acceptance_bypass(ctx, llm_trace, reason_code)
|
||||
ctx.accumulated_usage.setdefault("terminal_origin", TERMINAL_ORIGIN_HOST_NOTICE)
|
||||
binding = dict(candidate.acceptance_binding or {}) if candidate is not None else {}
|
||||
tools = getattr(ctx, "tools", None)
|
||||
current_fingerprint = str(
|
||||
|
|
@ -4429,8 +4427,7 @@ def _forced_fallback_result(
|
|||
candidate_reason: str = "",
|
||||
provider_terminal: bool = False,
|
||||
) -> Tuple[str, Dict[str, Any], Dict[str, Any]]:
|
||||
"""Return a current or fallback candidate."""
|
||||
|
||||
"""Compose fallback."""
|
||||
router_result = _forced_swarm_router_result(ctx, llm_trace, reason_code)
|
||||
if router_result is not None:
|
||||
return router_result
|
||||
|
|
@ -4451,11 +4448,10 @@ def _forced_fallback_result(
|
|||
candidate.model_text or candidate.full_text if provider_terminal else
|
||||
sanitize_tool_result_for_log(_compose_delivery_suffix(candidate.full_text, suffix))
|
||||
)
|
||||
if provider_terminal:
|
||||
ctx.accumulated_usage.update(
|
||||
terminal_origin=TERMINAL_ORIGIN_MODEL_FINAL,
|
||||
terminal_plan_review_open=bool(plan_suffix),
|
||||
)
|
||||
ctx.accumulated_usage.update(
|
||||
terminal_origin=TERMINAL_ORIGIN_MODEL_FINAL,
|
||||
terminal_plan_review_open=bool(plan_suffix),
|
||||
)
|
||||
if composed != candidate.full_text:
|
||||
candidate = _publish_model_forced_candidate(
|
||||
ctx, llm_trace, composed, reason_code,
|
||||
|
|
@ -4757,13 +4753,11 @@ def _forced_final_answer(
|
|||
_force_plan_disclosure(tools_ctx, llm_trace, forced_reason=reason_code)
|
||||
if tools_ctx is not None else ""
|
||||
)
|
||||
if provider_terminal:
|
||||
ctx.accumulated_usage["terminal_plan_review_open"] = bool(plan_suffix)
|
||||
ctx.accumulated_usage["terminal_plan_review_open"] = bool(plan_suffix)
|
||||
full_text = extracted if provider_terminal else _compose_delivery_suffix(
|
||||
extracted, plan_suffix + _forced_orphan_note(ctx),
|
||||
)
|
||||
if provider_terminal:
|
||||
ctx.accumulated_usage["terminal_origin"] = TERMINAL_ORIGIN_MODEL_FINAL
|
||||
ctx.accumulated_usage["terminal_origin"] = TERMINAL_ORIGIN_MODEL_FINAL
|
||||
candidate = _publish_model_forced_candidate(
|
||||
ctx, llm_trace, full_text, reason_code,
|
||||
degraded_reason=degraded,
|
||||
|
|
|
|||
|
|
@ -1376,6 +1376,11 @@ def refresh_verification_ledger_artifacts(
|
|||
|
||||
if not isinstance(ledger, dict):
|
||||
return ledger
|
||||
# An omitted-to-artifact stub is a PROJECTION of the artifact file, not a
|
||||
# source: it carries no entries, so rebuilding from it would mint "0
|
||||
# entries / no failures / execution ok" over the real ledger's summary.
|
||||
if ledger.get("omitted_to_artifact"):
|
||||
return ledger
|
||||
entries = [
|
||||
item for item in (ledger.get("entries") or [])
|
||||
if not (isinstance(item, dict) and item.get("kind") == "artifact_bundle")
|
||||
|
|
|
|||
|
|
@ -419,42 +419,51 @@ def routing_option_label(option: Any) -> str:
|
|||
return "Project" if option.get("project_id") and not option.get("task_id") else "Task"
|
||||
|
||||
|
||||
def completion_status_label(result: Dict[str, Any], event: Dict[str, Any]) -> str:
|
||||
from ouroboros.task_results import (
|
||||
STATUS_CANCELLED, STATUS_COMPLETED, STATUS_FAILED, STATUS_REJECTED_DUPLICATE,
|
||||
)
|
||||
OUTCOME_PHASE_HEADLINE = {"working": "Working", "done": "Done", "warn": "Done with warnings",
|
||||
"error": "Failed", "cancelled": "Cancelled"}
|
||||
|
||||
status = str(result.get("status") or event.get("status") or "").strip().lower()
|
||||
axes = {}
|
||||
for source in (event, result):
|
||||
value = source.get("outcome_axes")
|
||||
if isinstance(value, dict):
|
||||
axes.update({key: axis for key, axis in value.items() if isinstance(axis, dict)})
|
||||
axis_status = {key: str(axis.get("status") or "").lower() for key, axis in axes.items()}
|
||||
failed = (
|
||||
status == STATUS_FAILED
|
||||
or axis_status.get("lifecycle") == STATUS_FAILED
|
||||
or axis_status.get("execution") in {"failed", "infra_failed"}
|
||||
or axis_status.get("objective") == "fail"
|
||||
or axis_status.get("review") == "fail"
|
||||
or axis_status.get("artifacts") in {"failed", "missing"}
|
||||
or str(result.get("artifact_status") or event.get("artifact_status") or "").lower()
|
||||
in {"failed", "missing"}
|
||||
)
|
||||
degraded = any(value in {"degraded", "partial", "best_effort"}
|
||||
for value in axis_status.values())
|
||||
checkpoint = result.get("root_phase_checkpoint")
|
||||
degraded |= bool(isinstance(checkpoint, dict)
|
||||
and str(checkpoint.get("post_task_synthesis") or "").lower() == "degraded")
|
||||
if status == STATUS_CANCELLED:
|
||||
return "Cancelled"
|
||||
if failed:
|
||||
return "Failed"
|
||||
if status == STATUS_COMPLETED:
|
||||
return "Completed with limitations" if degraded else "Completed"
|
||||
if status == STATUS_REJECTED_DUPLICATE:
|
||||
return "Not started"
|
||||
return status.replace("_", " ").title() or "Finished"
|
||||
|
||||
def outcome_phase(result: Dict[str, Any], event: Dict[str, Any]) -> str:
|
||||
"""The host mirror of the browser's terminality gate and severity fold.
|
||||
|
||||
Durable host rows read exactly what ``log_events.js`` paints, over
|
||||
NORMALIZED axes; web/tests/fixtures/outcome_phase_parity.json pins both.
|
||||
"""
|
||||
from ouroboros.outcomes import REASON_OWNER_REQUESTED_FINALIZATION, normalize_outcome_axes
|
||||
from ouroboros.post_task_checkpoint import post_task_synthesis_is_open
|
||||
from ouroboros.task_status import FINAL_STATUSES
|
||||
|
||||
record = {**event, **{key: value for key, value in result.items() if value not in (None, "")}}
|
||||
sources = [s.get("outcome_axes") for s in (event, result) if isinstance(s.get("outcome_axes"), dict)]
|
||||
record["outcome_axes"] = {k: v for source in sources for k, v in source.items() if isinstance(v, dict)}
|
||||
axes = normalize_outcome_axes(record)
|
||||
axis = {k: str(v.get("status") or "").lower() for k, v in axes.items() if isinstance(v, dict)}
|
||||
status = str(record.get("task_terminal_status") or record.get("status") or "").strip().lower()
|
||||
checkpoint = record.get("root_phase_checkpoint")
|
||||
synthesis = checkpoint.get("post_task_synthesis") if isinstance(checkpoint, dict) else ""
|
||||
if not (status in {"done", "cancel_requested"} or (status in FINAL_STATUSES and not (
|
||||
status == "completed" and post_task_synthesis_is_open(synthesis)))):
|
||||
return "working"
|
||||
lifecycle = axis.get("lifecycle") or status
|
||||
if lifecycle in {"cancelled", "cancel_requested"}:
|
||||
return "cancelled"
|
||||
if (lifecycle == "failed" or axis.get("execution") in {"failed", "infra_failed"}
|
||||
or axis.get("objective") == "fail" or axis.get("review") == "fail"
|
||||
or {axis.get("artifacts"), str(record.get("artifact_status") or "").lower()} & {"failed", "missing"}):
|
||||
return "error"
|
||||
if str(record.get("reason_code") or "") == REASON_OWNER_REQUESTED_FINALIZATION:
|
||||
return "done"
|
||||
if (lifecycle == "rejected_duplicate" or bool((axes.get("objective") or {}).get("warning"))
|
||||
or axis.get("execution") in {"degraded", "best_effort"}
|
||||
or axis.get("objective") in {"degraded", "best_effort"}
|
||||
or axis.get("review") == "degraded"):
|
||||
return "warn"
|
||||
return "done"
|
||||
|
||||
|
||||
def completion_status_label(result: Dict[str, Any], event: Dict[str, Any]) -> str:
|
||||
"""The one owner-visible status word for a host-authored task row."""
|
||||
return OUTCOME_PHASE_HEADLINE[outcome_phase(result, event)]
|
||||
|
||||
|
||||
def append_canonical_task_summary(drive_root: Any, row: Dict[str, Any]) -> bool:
|
||||
|
|
@ -711,7 +720,8 @@ def _append_terminal_task_projection(
|
|||
project_id = resolve_project_id({**task, **effective})
|
||||
role = str(effective.get("role") or task.get("role") or ("root" if is_root else "child"))
|
||||
reason = str(effective.get("reason_code") or event.get("reason_code") or "")
|
||||
outcome = completion_status_label(effective, event)
|
||||
phase = outcome_phase(effective, event)
|
||||
outcome = OUTCOME_PHASE_HEADLINE[phase]
|
||||
excerpt = _completion_excerpt(effective)
|
||||
details = f'Details: get_task_result(task_id="{tid}")'
|
||||
text = (
|
||||
|
|
@ -720,8 +730,12 @@ def _append_terminal_task_projection(
|
|||
)
|
||||
if excerpt:
|
||||
text += f" {excerpt}"
|
||||
if reason:
|
||||
text += f" Reason: {reason}."
|
||||
verdict = _completion_verdict(effective, event)
|
||||
if verdict:
|
||||
text += f" {verdict}"
|
||||
depth = (effective.get("swarm_efficiency") or {}).get("depth") if isinstance(effective.get("swarm_efficiency"), dict) else None
|
||||
if isinstance(depth, dict) and depth.get("requested_depth") is not None:
|
||||
text += f" Depth requested={depth['requested_depth']}, permitted={depth.get('permitted_depth')}, achieved={depth.get('achieved_depth')} ({depth.get('status')})."
|
||||
result_ref = {"kind": "task_result", "task_id": tid, "reader": "get_task_result"}
|
||||
row = {
|
||||
"ts": str(event.get("ts") or effective.get("ts") or utc_now_iso()),
|
||||
|
|
@ -732,7 +746,7 @@ def _append_terminal_task_projection(
|
|||
"chat_id": int(event.get("chat_id") or task.get("chat_id") or 0),
|
||||
"delegation_role": str(effective.get("delegation_role") or task.get("delegation_role") or ""),
|
||||
"role": role, "status": str(effective.get("status") or status),
|
||||
"outcome": outcome, "outcome_final": True,
|
||||
"outcome": outcome, "outcome_phase": phase, "outcome_final": True,
|
||||
"outcome_authority": "canonical_task_result_after_finalization",
|
||||
"outcome_axes": effective.get("outcome_axes") or event.get("outcome_axes") or {},
|
||||
"reason_code": reason, "result_ref": result_ref,
|
||||
|
|
@ -787,6 +801,37 @@ def _completion_excerpt(result: Dict[str, Any]) -> str:
|
|||
return ""
|
||||
|
||||
|
||||
def _completion_verdict(result: Dict[str, Any], event: Dict[str, Any]) -> str:
|
||||
"""One TERMINATED host clause for BOTH lifecycle rows.
|
||||
|
||||
A host row must not present an unaccepted claim as the whole story: a
|
||||
non-accepted decision leads with its full upstream-bounded rationale;
|
||||
otherwise the execution reason stands. The Python twin of
|
||||
``taskReasonDetail``'s acceptance branch; callers add no punctuation.
|
||||
"""
|
||||
from ouroboros.outcomes import ACCEPTANCE_ACCEPTED, REASON_OWNER_REQUESTED_FINALIZATION
|
||||
|
||||
decision: Dict[str, Any] = {}
|
||||
for source in (event, result):
|
||||
axes = source.get("outcome_axes") if isinstance(source.get("outcome_axes"), dict) else {}
|
||||
for holder in (source.get("review_status"), axes.get("review")):
|
||||
if isinstance(holder, dict) and isinstance(holder.get("acceptance_decision"), dict):
|
||||
decision = holder["acceptance_decision"]
|
||||
status = str(decision.get("status") or "").strip()
|
||||
reason = str(result.get("reason_code") or event.get("reason_code") or "")
|
||||
if (reason != REASON_OWNER_REQUESTED_FINALIZATION and status != ACCEPTANCE_ACCEPTED
|
||||
and status and outcome_phase(result, event) in {"done", "warn"}):
|
||||
clause = f"Acceptance: {status}"
|
||||
rationale = " ".join(strip_markdown(str(decision.get("rationale") or "")).split())
|
||||
if rationale:
|
||||
clause += " — " + rationale
|
||||
elif reason and reason != REASON_OWNER_REQUESTED_FINALIZATION:
|
||||
clause = f"Reason: {reason}"
|
||||
else:
|
||||
return ""
|
||||
return clause if clause.endswith((".", "!", "?", "…")) else clause + "."
|
||||
|
||||
|
||||
def _run_lives_in_its_project(
|
||||
drive_root: Any, task_id: str, project_id: str, task: Dict[str, Any], result: Dict[str, Any],
|
||||
) -> bool:
|
||||
|
|
@ -867,11 +912,13 @@ def enqueue_project_completion_summary(
|
|||
# a Main row leading into an empty room.
|
||||
return False
|
||||
excerpt = _completion_excerpt(result)
|
||||
verdict = _completion_verdict(result, task_done_event)
|
||||
lead = f"{verdict} " if verdict else ""
|
||||
event = {
|
||||
"type": "send_message", "chat_id": 1, "task_id": tid,
|
||||
"text": (f"{snapshot['target_label']} · "
|
||||
f"{completion_status_label(result, task_done_event)}\n"
|
||||
f"{excerpt or 'Open the Project for details.'}"),
|
||||
f"{lead}{excerpt or 'Open the Project for details.'}"),
|
||||
"role": "system", "system_type": "project_completion_summary",
|
||||
"delivery_id": f"project-completion:{tid}",
|
||||
"progress_meta": {
|
||||
|
|
@ -942,6 +989,7 @@ __all__ = [
|
|||
"latest_chat_annotations",
|
||||
"enqueue_project_completion_summary",
|
||||
"completion_status_label",
|
||||
"outcome_phase",
|
||||
"owner_message_ref_is_valid",
|
||||
"project_origin_rows",
|
||||
"project_recent_dialogue",
|
||||
|
|
|
|||
|
|
@ -41,6 +41,17 @@ _SEALED_FINAL_TEXT_PROMPT_CHARS = 4000
|
|||
# never be inferred from result text or lifecycle status.
|
||||
TERMINAL_ORIGIN_MODEL_FINAL = "model_final"
|
||||
TERMINAL_ORIGIN_HOST_SALVAGE = "host_salvage"
|
||||
# A terminal text the HOST wrote alone (a budget rejection, a round-limit rail
|
||||
# with nothing to deliver, a scheduled swarm handoff). It is not salvage: its
|
||||
# own words ARE the answer, so they are published verbatim on every transport
|
||||
# instead of being replaced by the outage receipt.
|
||||
TERMINAL_ORIGIN_HOST_NOTICE = "host_notice"
|
||||
HOST_AUTHORED_TERMINAL_ORIGINS = frozenset({
|
||||
TERMINAL_ORIGIN_HOST_SALVAGE, TERMINAL_ORIGIN_HOST_NOTICE,
|
||||
})
|
||||
_STAMPED_TERMINAL_ORIGINS = frozenset({
|
||||
TERMINAL_ORIGIN_MODEL_FINAL, *HOST_AUTHORED_TERMINAL_ORIGINS,
|
||||
})
|
||||
TERMINAL_PLAN_REVIEW_NOTE = (
|
||||
"Plan review was still open when the outage forced finalization; "
|
||||
"its details remain in the task."
|
||||
|
|
@ -98,9 +109,7 @@ def prepare_terminal_send_event(
|
|||
# branch (supervisor/workers.py stamps task_terminal_status="failed")
|
||||
# and lets the live concludesTurn gate settle the activity.
|
||||
send_event.setdefault("progress_meta", {})["task_terminal_status"] = "completed"
|
||||
if ephemeral or presence or origin not in {
|
||||
TERMINAL_ORIGIN_MODEL_FINAL, TERMINAL_ORIGIN_HOST_SALVAGE,
|
||||
}:
|
||||
if ephemeral or presence or origin not in _STAMPED_TERMINAL_ORIGINS:
|
||||
return send_event
|
||||
canonical_root = pathlib.Path(task.get("budget_drive_root") or env_drive_root)
|
||||
preserved_path = ""
|
||||
|
|
@ -126,7 +135,7 @@ def terminal_result_fields(usage: Dict[str, Any]) -> Dict[str, Any]:
|
|||
"""Additive durable origin/full-copy fields; unknown producers stay legacy."""
|
||||
fields: Dict[str, Any] = {}
|
||||
origin = str(usage.get("terminal_origin") or "")
|
||||
if origin in {TERMINAL_ORIGIN_MODEL_FINAL, TERMINAL_ORIGIN_HOST_SALVAGE}:
|
||||
if origin in _STAMPED_TERMINAL_ORIGINS:
|
||||
fields["terminal_origin"] = origin
|
||||
path = str(usage.get("terminal_salvage_path") or "")
|
||||
if path:
|
||||
|
|
@ -388,6 +397,25 @@ def build_swarm_efficiency(env: Any, task: Dict[str, Any]) -> Dict[str, Any] | N
|
|||
"inter_wave_latency_sec_total": round(inter_wave_latency_total, 3),
|
||||
"lanes_requested": lanes,
|
||||
}
|
||||
try:
|
||||
# Depth is the one swarm fact the root could not see: its own
|
||||
# contract carries the request, the subtree carries what was
|
||||
# actually reached. Its own try/except — the enclosing one returns
|
||||
# None for the WHOLE rollup, and a subtree read failure must not
|
||||
# erase the fan-out numbers.
|
||||
from ouroboros.depth_evidence import build_depth_summary
|
||||
from ouroboros.task_status import find_child_tasks
|
||||
|
||||
canonical = pathlib.Path(task.get("budget_drive_root") or drive_root)
|
||||
rollup["depth"] = build_depth_summary(
|
||||
task.get("task_contract"),
|
||||
find_child_tasks(
|
||||
canonical, parent_task_id=task_id, root_task_id=task_id,
|
||||
scope="subtree", materialize_artifacts=False,
|
||||
),
|
||||
)
|
||||
except Exception:
|
||||
log.debug("swarm depth summary failed", exc_info=True)
|
||||
if swarm_intent:
|
||||
rollup["intent_source"] = "swarm"
|
||||
# The planned figure under its existing event name — the waves'
|
||||
|
|
|
|||
|
|
@ -426,10 +426,13 @@ def _subtask_outcome_summary(data: Dict[str, Any], receipts: list | None = None)
|
|||
if isinstance(data.get("artifact_bundle"), dict):
|
||||
summary["artifact_bundle"] = data.get("artifact_bundle")
|
||||
if ledger:
|
||||
# An omitted-to-artifact stub carries no entries; its summary is the
|
||||
# count authority, and for a full ledger the two always agree.
|
||||
ledger_summary = ledger.get("summary") if isinstance(ledger.get("summary"), dict) else {}
|
||||
summary["verification_ledger"] = {
|
||||
"schema_version": ledger.get("schema_version"),
|
||||
"summary": ledger.get("summary") if isinstance(ledger.get("summary"), dict) else {},
|
||||
"entry_count": len(ledger.get("entries") or []) if isinstance(ledger.get("entries"), list) else 0,
|
||||
"summary": ledger_summary,
|
||||
"entry_count": ledger_summary.get("entry_count", len(ledger.get("entries") or []) if isinstance(ledger.get("entries"), list) else 0),
|
||||
}
|
||||
if receipts:
|
||||
# W2: bounded per-receipt rows for the FULL single-child handoff ONLY
|
||||
|
|
@ -1704,6 +1707,10 @@ def schedule_subagent_properties() -> Dict[str, Any]:
|
|||
"may_mutate": {"type": "boolean", "default": False, "description": "Optional: grant this child the intent to spawn MUTATIVE (acting) descendants of its own. Still bounded by the usual mutative-subagent gating and depth/active caps."},
|
||||
"may_fan_out": {"type": "boolean", "default": True, "description": "Optional: whether this child may spawn MULTIPLE children (a wave). Bounded by the per-root active cap."},
|
||||
"max_children": {"type": "integer", "default": 0, "description": "Optional soft cap on this child's own direct children (0 = inherit / configured cap)."},
|
||||
"requested_depth": {
|
||||
"type": "integer", "default": 0,
|
||||
"description": "Optional: how deep, counted ABSOLUTELY FROM THE ROOT, you intend this branch to nest (root=0, direct children=1; asking for children, grandchildren and great-grandchildren is 3). Recorded as your attested request and reported back as requested/permitted/achieved on the root result; it never widens or narrows the configured caps. 0 or omitted = no request.",
|
||||
},
|
||||
"required_capabilities": {
|
||||
"type": "array",
|
||||
"items": {"type": "string", "enum": list(SUBAGENT_CAPABILITIES)},
|
||||
|
|
@ -2014,6 +2021,7 @@ def _schedule_task(ctx: ToolContext, internal: Dict[str, Any] | None = None, /,
|
|||
may_mutate=may_mutate, may_fan_out=params.get("may_fan_out", True),
|
||||
max_children=params.get("max_children", 0),
|
||||
intent_note=params.get("delegation_intent", ""),
|
||||
requested_depth=params.get("requested_depth", 0),
|
||||
)
|
||||
|
||||
child_contract = _build_child_subagent_contract({
|
||||
|
|
|
|||
|
|
@ -207,8 +207,14 @@ def admitted_depth_cap(parent_contract: Any, live_max_depth: Any) -> int:
|
|||
def depth_provenance_for_schedule(
|
||||
parent_budget: Dict[str, Any], *, new_depth: int, max_depth: int,
|
||||
achieved_depth: Any = None, use_remaining_envelope: bool = False,
|
||||
requested_depth: Any = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""Carry requested/permitted/attempted/achieved depth as additive facts."""
|
||||
"""Carry requested/permitted/attempted/achieved depth as additive facts.
|
||||
|
||||
A request is telemetry, never a cap: only the configured cap and the legacy
|
||||
remaining envelope narrow what a branch may do, so asking for less than the
|
||||
cap records the intent without silently binding the descendants to it.
|
||||
"""
|
||||
budget = parent_budget if isinstance(parent_budget, dict) else {}
|
||||
inherited = normalize_depth_provenance(budget.get("depth_provenance"))
|
||||
requested = inherited.get("requested_depth")
|
||||
|
|
@ -226,11 +232,22 @@ def depth_provenance_for_schedule(
|
|||
requested = max(0, int(budget.get("depth_remaining")))
|
||||
except (TypeError, ValueError):
|
||||
requested = None
|
||||
if requested is None:
|
||||
# An explicit request of record on THIS call stays unchanged. A
|
||||
# non-integer fails soft to "no request recorded": the
|
||||
# field is telemetry, and refusing the whole schedule over it would
|
||||
# trade a real capability for a bookkeeping nicety.
|
||||
try:
|
||||
explicit = max(0, int(requested_depth or 0))
|
||||
except (TypeError, ValueError):
|
||||
explicit = 0
|
||||
if explicit > 0:
|
||||
requested = explicit
|
||||
try:
|
||||
cap = min(MAX_SUBAGENT_DEPTH_HARD_CAP, max(0, int(max_depth)))
|
||||
except (TypeError, ValueError):
|
||||
cap = 0
|
||||
current_permitted = cap if requested is None else min(cap, max(0, int(requested)))
|
||||
current_permitted = cap
|
||||
remaining = budget.get("depth_remaining")
|
||||
if (
|
||||
use_remaining_envelope
|
||||
|
|
@ -337,12 +354,7 @@ def stamp_task_assignment_depth(
|
|||
requested = admitted.get("requested_depth")
|
||||
permitted = _bounded_permitted_depth(admitted.get("permitted_depth"))
|
||||
if permitted is None:
|
||||
permitted = min(
|
||||
min(MAX_SUBAGENT_DEPTH_HARD_CAP, max(0, int(max_depth))),
|
||||
min(MAX_SUBAGENT_DEPTH_HARD_CAP, max(0, int(requested)))
|
||||
if requested is not None
|
||||
else min(MAX_SUBAGENT_DEPTH_HARD_CAP, max(0, int(max_depth))),
|
||||
)
|
||||
permitted = min(MAX_SUBAGENT_DEPTH_HARD_CAP, max(0, int(max_depth)))
|
||||
provenance = {
|
||||
"requested_depth": requested,
|
||||
"permitted_depth": permitted,
|
||||
|
|
@ -418,6 +430,7 @@ def record_depth_limit_refusal(
|
|||
may_fan_out=params.get("may_fan_out", True),
|
||||
max_children=params.get("max_children", 0),
|
||||
intent_note=params.get("delegation_intent", ""),
|
||||
requested_depth=params.get("requested_depth", 0),
|
||||
)
|
||||
from ouroboros.contracts.task_contract import build_task_contract
|
||||
|
||||
|
|
@ -793,6 +806,7 @@ def child_budget_for_schedule(
|
|||
may_fan_out: bool,
|
||||
max_children: int,
|
||||
intent_note: str,
|
||||
requested_depth: Any = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""Resolve a child's delegation_budget at schedule time (C3.1): decrement
|
||||
depth_remaining one generation (falling back to the configured max_depth/new_depth
|
||||
|
|
@ -803,7 +817,7 @@ def child_budget_for_schedule(
|
|||
parent_budget = parent_budget if isinstance(parent_budget, dict) else {}
|
||||
provenance = depth_provenance_for_schedule(
|
||||
parent_budget, new_depth=new_depth, max_depth=max_depth,
|
||||
use_remaining_envelope=True,
|
||||
use_remaining_envelope=True, requested_depth=requested_depth,
|
||||
)
|
||||
permitted_remaining = max(
|
||||
0, int(provenance.get("permitted_depth") or 0) - int(new_depth),
|
||||
|
|
|
|||
|
|
@ -77,10 +77,13 @@ def _task_record(
|
|||
record["artifact_bundle"] = data.get("artifact_bundle")
|
||||
ledger = data.get("verification_ledger") if isinstance(data.get("verification_ledger"), dict) else {}
|
||||
if ledger:
|
||||
# An omitted-to-artifact stub carries no entries; its summary is the
|
||||
# count authority, and for a full ledger the two always agree.
|
||||
ledger_summary = ledger.get("summary") if isinstance(ledger.get("summary"), dict) else {}
|
||||
record["verification_ledger"] = {
|
||||
"schema_version": ledger.get("schema_version"),
|
||||
"summary": ledger.get("summary") if isinstance(ledger.get("summary"), dict) else {},
|
||||
"entry_count": len(ledger.get("entries") or []) if isinstance(ledger.get("entries"), list) else 0,
|
||||
"summary": ledger_summary,
|
||||
"entry_count": ledger_summary.get("entry_count", len(ledger.get("entries") or []) if isinstance(ledger.get("entries"), list) else 0),
|
||||
}
|
||||
if include_results:
|
||||
record["result"] = result
|
||||
|
|
|
|||
|
|
@ -84,7 +84,24 @@ def _tree_read(
|
|||
|
||||
|
||||
def get_tools() -> List[ToolEntry]:
|
||||
from ouroboros.task_tree_ledger import DELEGATION_CONSTRAINT_DIRECTIVES, LEDGER_KINDS
|
||||
from ouroboros.task_tree_ledger import (
|
||||
CHILD_RESULT_DISPOSITIONS, DELEGATION_CONSTRAINT_DIRECTIVES, LEDGER_KINDS,
|
||||
)
|
||||
|
||||
# The schema is where the model learns WHEN to choose each value, so the
|
||||
# per-value consequence belongs here rather than in prose elsewhere. Read
|
||||
# from the validator's own frozenset, sorted so the enum order is stable
|
||||
# across worker processes (a frozenset's iteration order is not).
|
||||
disposition_schema = {
|
||||
"type": "string",
|
||||
"enum": sorted(CHILD_RESULT_DISPOSITIONS),
|
||||
"description": (
|
||||
"integrated = consumed by this answer (item closes); irrelevant = not needed "
|
||||
"here (the child result stays as evidence; item closes); deferred = STILL OWED: "
|
||||
"your terminal answer reads degraded/best_effort until you re-disposition this "
|
||||
"exact child_result_sha256."
|
||||
),
|
||||
}
|
||||
|
||||
return [
|
||||
ToolEntry("tree_note", {
|
||||
|
|
@ -125,10 +142,7 @@ def get_tools() -> List[ToolEntry]:
|
|||
"properties": {
|
||||
"type": {"type": "string", "enum": ["child_result_disposition"]},
|
||||
"child_task_id": {"type": "string"},
|
||||
"disposition": {
|
||||
"type": "string",
|
||||
"enum": ["integrated", "irrelevant", "deferred"],
|
||||
},
|
||||
"disposition": disposition_schema,
|
||||
"child_result_sha256": {"type": "string"},
|
||||
"children": {
|
||||
"type": "array",
|
||||
|
|
@ -140,10 +154,7 @@ def get_tools() -> List[ToolEntry]:
|
|||
"type": "object",
|
||||
"properties": {
|
||||
"child_task_id": {"type": "string"},
|
||||
"disposition": {
|
||||
"type": "string",
|
||||
"enum": ["integrated", "irrelevant", "deferred"],
|
||||
},
|
||||
"disposition": disposition_schema,
|
||||
"child_result_sha256": {"type": "string"},
|
||||
},
|
||||
"required": [
|
||||
|
|
|
|||
|
|
@ -16,13 +16,15 @@ import pathlib
|
|||
import re
|
||||
from typing import List
|
||||
|
||||
# v6.35.0 (T7): bumped to 2 with binary + size + junk-artifact hygiene so the
|
||||
# real-usage workspace.patch (consumed by subagents / PR integration) never
|
||||
# carries a compiled `go build` binary, a Redis dump, or other untracked build
|
||||
# junk. Kept consistent with the bench capture_patch.sh JUNK_RE + numstat
|
||||
# binary detection. This is patch-transport hygiene only (artifact path/extension
|
||||
# + git's own binary verdict), never code/content inference (Bible P5).
|
||||
_PATCH_EXCLUDE_RULES_VERSION = 2
|
||||
# Version 3: generated output directories are no longer excluded by NAME. The
|
||||
# project's own `.gitignore` (honoured through `--exclude-standard`), the 5 MiB
|
||||
# per-file cap below, git's own binary verdict and the credential-name check
|
||||
# decide what a workspace.patch may carry; a project whose deliverable IS its
|
||||
# build output must not have it silently dropped. The benchmark capture script
|
||||
# owns its own any-depth rule and does not import this module. This is
|
||||
# patch-transport hygiene only (artifact path/extension + git's binary
|
||||
# verdict), never code/content inference (Bible P5).
|
||||
_PATCH_EXCLUDE_RULES_VERSION = 3
|
||||
_PATCH_MAX_UNTRACKED_FILE_BYTES = 5 * 1024 * 1024 # 5 MiB per untracked file
|
||||
_TOP_LEVEL_EXCLUDE_DIRS = {".ouroboros", ".venv", "venv", "env"}
|
||||
_ANY_SEGMENT_EXCLUDE_DIRS = {
|
||||
|
|
@ -37,11 +39,12 @@ _ANY_SEGMENT_EXCLUDE_DIRS = {
|
|||
"__pycache__",
|
||||
"node_modules",
|
||||
}
|
||||
# Junk file tails / build dirs the dir-sets above don't already cover; the same
|
||||
# JUNK_RE the bench capture_patch.sh uses (devtools/benchmarks/swe_bench_pro/).
|
||||
# Junk file tails and caches the dir-sets above don't already cover. Runtime
|
||||
# dumps, compiled bytecode and coverage output only; generated-output
|
||||
# directories are the project's own .gitignore decision, not a name rule here.
|
||||
_PATCH_JUNK_RE = re.compile(
|
||||
r"appendonlydir|\.rdb$|\.aof$|\.manifest$|\.log$|\.tmp$|\.pid$|\.sock$"
|
||||
r"|\.pyc$|\.pyo$|^(dist|build)/|\.DS_Store|(^|/)\.coverage$"
|
||||
r"|\.pyc$|\.pyo$|\.DS_Store|(^|/)\.coverage$"
|
||||
r"|coverage\.xml$|(^|/)htmlcov/"
|
||||
)
|
||||
_LOCKFILE_MANIFESTS = {
|
||||
|
|
|
|||
|
|
@ -137,11 +137,13 @@ def _summary_ids_in_tail(api, limit: int = 200) -> list:
|
|||
max_entries=limit,
|
||||
tail_bytes=256 * 1024,
|
||||
)
|
||||
ids = []
|
||||
summaries = {}
|
||||
for e in rows:
|
||||
if str(e.get("type") or "") == "task_summary" and e.get("task_id"):
|
||||
ids.append((str(e.get("task_id")), e))
|
||||
return ids
|
||||
if (str(e.get("type") or "") == "task_summary" and e.get("task_id")
|
||||
and e.get("outcome_final") is not False
|
||||
and str(e.get("outcome_phase") or "") != "working"):
|
||||
summaries[str(e.get("task_id"))] = e
|
||||
return list(summaries.items())
|
||||
|
||||
|
||||
async def _check_tasks_notify(
|
||||
|
|
@ -180,7 +182,14 @@ async def _check_tasks_notify(
|
|||
if outcome and outcome not in ("completed", "done"):
|
||||
parts.append(outcome)
|
||||
tail = (" · " + " · ".join(parts)) if parts else ""
|
||||
healthy = outcome in ("", "completed", "done") and not degraded
|
||||
# The host stamps one status phase on its own task rows; consume it
|
||||
# instead of re-deriving a third status ladder from the axes. Legacy
|
||||
# rows and pre-finalization "working" rows keep the axes rule.
|
||||
phase = str(e.get("outcome_phase") or "")
|
||||
if phase and phase != "working":
|
||||
healthy = phase == "done"
|
||||
else:
|
||||
healthy = outcome in ("", "completed", "done") and not degraded
|
||||
icon = "✅" if healthy else "⚠️"
|
||||
msg = (f"{icon} Задача {tid[:8]} готова{tail}" if lang == "ru" else f"{icon} Task {tid[:8]} done{tail}")
|
||||
send_outcome, exc = await _push_notification(api, chat_id, msg, trust_env=trust_env)
|
||||
|
|
|
|||
|
|
@ -34,6 +34,7 @@ from typing import Any, Dict, List, Optional
|
|||
|
||||
from ouroboros.utils import update_json_locked, utc_now_iso
|
||||
from ouroboros.task_finalization import (
|
||||
HOST_AUTHORED_TERMINAL_ORIGINS,
|
||||
TERMINAL_ORIGIN_HOST_SALVAGE,
|
||||
TERMINAL_ORIGIN_MODEL_FINAL,
|
||||
)
|
||||
|
|
@ -719,10 +720,13 @@ def project_terminal_result_event(
|
|||
"""Project one terminal event from producer-stamped origin.
|
||||
|
||||
``host_salvage`` becomes one short keyed plain System receipt (inherited
|
||||
``format``/``log_text`` dropped; the full bytes stay in task details);
|
||||
``model_final`` and a missing legacy origin keep the assistant projection
|
||||
untouched — no text/status/length inference. The delivery id always
|
||||
digests the stable core result rather than a mutable disclosure suffix.
|
||||
``format``/``log_text`` dropped; the full bytes stay in task details).
|
||||
``host_notice`` is a text the host wrote alone, so it keeps its OWN words
|
||||
and inherited markdown and becomes a System row WITHOUT a system_type,
|
||||
which is what lets a replayed card conclude on it. ``model_final`` and a
|
||||
missing legacy origin keep the assistant projection untouched — no
|
||||
text/status/length inference. The delivery id always digests the stable
|
||||
core result rather than a mutable disclosure suffix.
|
||||
"""
|
||||
tid = str(task_id or "")
|
||||
core_text = str(result_text or "")
|
||||
|
|
@ -732,19 +736,22 @@ def project_terminal_result_event(
|
|||
event.setdefault("chat_id", lineage_chat_id(drive_root, task or {}, tid))
|
||||
event["delivery_id"] = delivery_id_for(tid, core_text)
|
||||
origin = str(terminal_origin or "")
|
||||
if origin == TERMINAL_ORIGIN_HOST_SALVAGE:
|
||||
if origin in HOST_AUTHORED_TERMINAL_ORIGINS:
|
||||
salvage = origin == TERMINAL_ORIGIN_HOST_SALVAGE
|
||||
event.update({
|
||||
"text": _HOST_SALVAGE_RECEIPT,
|
||||
"text": _HOST_SALVAGE_RECEIPT if salvage else (event.get("text") or core_text),
|
||||
"role": "system",
|
||||
"system_type": "terminal_incident",
|
||||
"terminal_origin": TERMINAL_ORIGIN_HOST_SALVAGE,
|
||||
"terminal_origin": origin,
|
||||
**({"system_type": "terminal_incident"} if salvage else {}),
|
||||
})
|
||||
event.pop("log_text", None)
|
||||
event.pop("format", None)
|
||||
if salvage:
|
||||
event.pop("log_text", None)
|
||||
event.pop("format", None)
|
||||
return event
|
||||
if origin == TERMINAL_ORIGIN_MODEL_FINAL:
|
||||
event["terminal_origin"] = TERMINAL_ORIGIN_MODEL_FINAL
|
||||
# Missing origin remains absent so legacy rows replay byte-compatibly.
|
||||
# A missing origin now identifies only a row written before every forced
|
||||
# rail stamped one; it stays absent so legacy rows replay byte-compatibly.
|
||||
return event
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1526,3 +1526,22 @@ def test_select_subagent_constraint_read_only_token_is_readonly():
|
|||
constraint = _select_subagent_constraint(surface, "", False, [], "")
|
||||
assert isinstance(constraint, dict), surface
|
||||
assert constraint == baseline, surface
|
||||
|
||||
|
||||
def test_schedule_subagent_publishes_the_depth_request_and_the_handler_accepts_it():
|
||||
"""A nanny chain could describe its intended nesting only in prose: the tool
|
||||
refused ``requested_depth`` by name, so the root's own depth summary could
|
||||
never say what had been asked for."""
|
||||
from ouroboros.tools import control
|
||||
|
||||
props = control.schedule_subagent_properties()
|
||||
row = props["requested_depth"]
|
||||
assert row["type"] == "integer"
|
||||
assert row["default"] == 0
|
||||
# Absolute from the root, and telemetry rather than a cap: both semantics
|
||||
# have to be readable by the model that fills the field in.
|
||||
assert "ABSOLUTELY FROM THE ROOT" in row["description"]
|
||||
assert "never widens or narrows" in row["description"]
|
||||
# The handler's closed keyword set DERIVES from the same schema.
|
||||
assert "requested_depth" in control.schedule_subagent_param_names()
|
||||
assert "requested_depth" not in control.HIDDEN_LEGACY_SCHEDULE_PARAMS
|
||||
|
|
|
|||
|
|
@ -914,3 +914,25 @@ def test_full_blackboard_never_locks_out_validated_dispositions(tmp_path, monkey
|
|||
)
|
||||
assert "ledger is full" not in accepted
|
||||
assert not accepted.startswith("⚠️")
|
||||
|
||||
|
||||
def test_the_disposition_enum_and_its_cost_come_from_one_place():
|
||||
"""The enum was hand-written twice with no per-value meaning, although the
|
||||
consequence of ``deferred`` is real and host-enforced. Both payload sites
|
||||
now read the validator's own set, in a stable order, and the schema states
|
||||
what each value costs — the schema is where the model actually reads it."""
|
||||
from ouroboros.task_tree_ledger import CHILD_RESULT_DISPOSITIONS
|
||||
from ouroboros.tools import task_tree
|
||||
|
||||
entry = next(tool for tool in task_tree.get_tools() if tool.name == "tree_note")
|
||||
payload = entry.schema["parameters"]["properties"]["payload"]["properties"]
|
||||
single = payload["disposition"]
|
||||
batch = payload["children"]["items"]["properties"]["disposition"]
|
||||
|
||||
assert single is batch or single == batch
|
||||
for schema in (single, batch):
|
||||
assert schema["enum"] == sorted(CHILD_RESULT_DISPOSITIONS)
|
||||
assert schema["enum"] == ["deferred", "integrated", "irrelevant"]
|
||||
for value in CHILD_RESULT_DISPOSITIONS:
|
||||
assert value in schema["description"], value
|
||||
assert "degraded/best_effort" in schema["description"]
|
||||
|
|
|
|||
|
|
@ -4,6 +4,8 @@ task contract and be surfaced in the child's prompt — instead of being lost in
|
|||
freeform objective prose (the cyber-racing 'maximum subagents' request that
|
||||
collapsed into 3 flat research leaves)."""
|
||||
|
||||
from types import SimpleNamespace
|
||||
|
||||
|
||||
def test_delegation_budget_defaults_and_normalization():
|
||||
from ouroboros.contracts.task_contract import build_task_contract, normalize_delegation_budget
|
||||
|
|
@ -216,3 +218,117 @@ def test_child_budget_strict_boolean_parsing():
|
|||
)
|
||||
assert child2["may_mutate"] is True
|
||||
assert child2["may_fan_out"] is True
|
||||
|
||||
|
||||
def test_requested_depth_is_recorded_as_telemetry_and_never_narrows_the_cap():
|
||||
"""A nanny chain had no way to SAY how deep it meant to nest, so the root's
|
||||
depth summary read "request unknown" and a shallow outcome could not be told
|
||||
apart from a reduced capability. The request is now recorded — and recording
|
||||
it must not become a second cap: asking for less than the configured depth
|
||||
used to narrow what the descendants were permitted."""
|
||||
from ouroboros.depth_evidence import build_depth_summary
|
||||
from ouroboros.tools.control_delegation import child_budget_for_schedule
|
||||
|
||||
root = {"delegation_budget": {"may_delegate": True}}
|
||||
|
||||
def _schedule(**kwargs):
|
||||
return child_budget_for_schedule(
|
||||
root, current_depth=0, new_depth=1, max_depth=4, may_mutate=False,
|
||||
may_fan_out=True, max_children=0, intent_note="", **kwargs,
|
||||
)["depth_provenance"]
|
||||
|
||||
asked = _schedule(requested_depth=2)
|
||||
assert asked["requested_depth"] == 2
|
||||
assert asked["permitted_depth"] == 4, "a request is telemetry, not a cap"
|
||||
assert asked["attempted_depth"] == 1
|
||||
assert asked["achieved_depth"] is None
|
||||
|
||||
# Absent or zero means no request of record — never a recorded zero.
|
||||
assert _schedule()["requested_depth"] is None
|
||||
assert _schedule(requested_depth=0)["requested_depth"] is None
|
||||
# The immutable host ceiling bounds authority, not the attested request.
|
||||
over_cap = _schedule(requested_depth=99)
|
||||
assert over_cap["requested_depth"] == 99
|
||||
assert over_cap["permitted_depth"] == 4
|
||||
assert build_depth_summary(
|
||||
{"delegation_budget": {"depth_provenance": over_cap}}, [],
|
||||
)["status"] == "capability_reduced"
|
||||
# A non-integer fails SOFT: the handler validates argument names, not value
|
||||
# types, and refusing a whole schedule over a telemetry field would trade a
|
||||
# real capability for bookkeeping.
|
||||
assert _schedule(requested_depth="x")["requested_depth"] is None
|
||||
|
||||
|
||||
def test_an_inherited_request_of_record_survives_a_nannys_own_number():
|
||||
from ouroboros.tools.control_delegation import child_budget_for_schedule
|
||||
|
||||
parent = {"delegation_budget": {
|
||||
"may_delegate": True, "depth_provenance": {"requested_depth": 3, "permitted_depth": 4},
|
||||
}}
|
||||
provenance = child_budget_for_schedule(
|
||||
parent, current_depth=1, new_depth=2, max_depth=4, may_mutate=False,
|
||||
may_fan_out=True, max_children=0, intent_note="", requested_depth=1,
|
||||
)["depth_provenance"]
|
||||
assert provenance["requested_depth"] == 3
|
||||
assert provenance["permitted_depth"] == 4
|
||||
|
||||
|
||||
def test_assignment_stamp_without_a_permitted_fact_uses_the_configured_cap():
|
||||
"""The stamp used to fold the REQUEST into the permitted number, so a branch
|
||||
that asked for two lost the depth the configuration actually allowed."""
|
||||
from ouroboros.tools.control_delegation import stamp_task_assignment_depth
|
||||
|
||||
task = {
|
||||
"depth": 1,
|
||||
"task_contract": {"delegation_budget": {
|
||||
"depth_provenance": {"requested_depth": 2, "permitted_depth": "not-a-number"},
|
||||
}},
|
||||
}
|
||||
stamped = stamp_task_assignment_depth(task, max_depth=4)
|
||||
provenance = stamped["task_contract"]["delegation_budget"]["depth_provenance"]
|
||||
assert provenance["requested_depth"] == 2
|
||||
assert provenance["permitted_depth"] == 4
|
||||
|
||||
|
||||
def test_the_swarm_rollup_reports_requested_permitted_and_achieved_depth(tmp_path):
|
||||
from ouroboros.task_finalization import build_swarm_efficiency
|
||||
from ouroboros.task_results import STATUS_COMPLETED, write_task_result
|
||||
from ouroboros.utils import append_jsonl
|
||||
|
||||
root_id = "depth-root"
|
||||
(tmp_path / "logs").mkdir(parents=True, exist_ok=True)
|
||||
append_jsonl(tmp_path / "logs" / "events.jsonl", {
|
||||
"type": "swarm_fanout", "parent_task_id": root_id, "task_ids": ["c1"],
|
||||
"requested_count": 1, "ts": "2026-09-03T10:00:00Z",
|
||||
})
|
||||
write_task_result(
|
||||
tmp_path, "c1", STATUS_COMPLETED, result="done",
|
||||
parent_task_id=root_id, root_task_id=root_id, delegation_role="subagent",
|
||||
depth_provenance={
|
||||
"requested_depth": 2, "permitted_depth": 4,
|
||||
"attempted_depth": 1, "achieved_depth": 1,
|
||||
},
|
||||
)
|
||||
rollup = build_swarm_efficiency(
|
||||
SimpleNamespace(drive_root=tmp_path),
|
||||
{
|
||||
"id": root_id, "budget_drive_root": str(tmp_path),
|
||||
"task_contract": {"delegation_budget": {
|
||||
"depth_provenance": {"requested_depth": 2, "permitted_depth": 4},
|
||||
}},
|
||||
},
|
||||
)
|
||||
assert rollup["subagent_count"] == 1
|
||||
assert rollup["depth"]["requested_depth"] == 2
|
||||
assert rollup["depth"]["permitted_depth"] == 4
|
||||
assert rollup["depth"]["achieved_depth"] == 1
|
||||
assert rollup["depth"]["status"] == "chosen_shallower"
|
||||
assert rollup["depth"]["host_visible_only"] is True
|
||||
|
||||
|
||||
def test_a_plain_task_gets_no_depth_block(tmp_path):
|
||||
from ouroboros.task_finalization import build_swarm_efficiency
|
||||
|
||||
assert build_swarm_efficiency(
|
||||
SimpleNamespace(drive_root=tmp_path), {"id": "lonely"},
|
||||
) is None
|
||||
|
|
|
|||
|
|
@ -48,6 +48,15 @@ def test_architecture_mentions_shared_log_grouping_and_direct_provider_review_fa
|
|||
assert "claudeRuntimeHasError" not in arch
|
||||
|
||||
|
||||
def test_architecture_limits_finality_and_verdict_claims_to_actual_rows():
|
||||
arch = _read("docs/ARCHITECTURE.md")
|
||||
|
||||
assert "The start row carries neither outcome finality nor a verdict" in arch
|
||||
assert "The pre-finalization authored row carries the phase with `outcome_final=false`" in arch
|
||||
assert "only terminal `task_summary` rows append the host verdict clause" in arch
|
||||
assert "Both Main rows, the Project thread rows" not in arch
|
||||
|
||||
|
||||
def test_architecture_documents_skill_schedule_lifecycle_and_evolution_light_block():
|
||||
arch = _read("docs/ARCHITECTURE.md")
|
||||
|
||||
|
|
@ -246,6 +255,8 @@ def test_architecture_mirror_matches_the_split_axes_contracts():
|
|||
|
||||
# swarm_efficiency reports the REQUEST: lanes_requested, never lanes_used.
|
||||
assert "lanes_requested" in arch
|
||||
# A fanned-out root also reports the depth REQUEST beside those lanes.
|
||||
assert "`requested_depth`" in arch
|
||||
assert "lanes_used" not in arch
|
||||
# Task-group compaction left with the degenerate lane fan-out (v6.87.28).
|
||||
assert "task-group compaction" not in arch
|
||||
|
|
|
|||
|
|
@ -662,6 +662,67 @@ def test_chat_history_task_summary_row_passes_flat_cost_fields_through(tmp_path)
|
|||
assert "cost_usd" not in rec
|
||||
|
||||
|
||||
def test_chat_history_replays_task_summary_finality_without_task_result(tmp_path):
|
||||
logs = tmp_path / "logs"
|
||||
logs.mkdir()
|
||||
(logs / "chat.jsonl").write_text(
|
||||
json.dumps({
|
||||
"ts": "2026-09-03T00:00:00Z",
|
||||
"direction": "system",
|
||||
"type": "task_summary",
|
||||
"task_id": "open-summary",
|
||||
"chat_id": 1,
|
||||
"text": "Narrative written before finalization.",
|
||||
"outcome_phase": "warn",
|
||||
"outcome_final": False,
|
||||
}) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(logs / "progress.jsonl").write_text("", encoding="utf-8")
|
||||
|
||||
endpoint = make_chat_history_endpoint(tmp_path)
|
||||
response = asyncio.run(endpoint(SimpleNamespace(query_params={"limit": "10"})))
|
||||
payload = json.loads(response.body.decode("utf-8"))["messages"]
|
||||
|
||||
rec = next(item for item in payload if item.get("task_id") == "open-summary")
|
||||
assert rec["outcome_phase"] == "warn"
|
||||
assert rec["outcome_final"] is False
|
||||
assert "task_terminal_status" not in rec
|
||||
|
||||
|
||||
def test_chat_history_settled_result_overrides_pre_final_summary_finality(tmp_path):
|
||||
logs = tmp_path / "logs"
|
||||
logs.mkdir()
|
||||
(logs / "chat.jsonl").write_text(
|
||||
json.dumps({
|
||||
"ts": "2026-09-03T00:00:00Z",
|
||||
"direction": "system",
|
||||
"type": "task_summary",
|
||||
"task_id": "settled-summary",
|
||||
"chat_id": 1,
|
||||
"text": "Narrative written before finalization.",
|
||||
"outcome_phase": "working",
|
||||
"outcome_final": False,
|
||||
}) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(logs / "progress.jsonl").write_text("", encoding="utf-8")
|
||||
results = tmp_path / "task_results"
|
||||
results.mkdir()
|
||||
(results / "settled-summary.json").write_text(
|
||||
json.dumps({"task_id": "settled-summary", "status": "completed"}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
endpoint = make_chat_history_endpoint(tmp_path)
|
||||
response = asyncio.run(endpoint(SimpleNamespace(query_params={"limit": "10"})))
|
||||
payload = json.loads(response.body.decode("utf-8"))["messages"]
|
||||
|
||||
rec = next(item for item in payload if item.get("task_id") == "settled-summary")
|
||||
assert rec["outcome_phase"] == "done"
|
||||
assert rec["outcome_final"] is True
|
||||
|
||||
|
||||
def test_chat_history_attaches_terminal_cost_truth_from_task_result(tmp_path):
|
||||
"""v6.82 P1: a terminal task_results/<id>.json carries the final cost truth;
|
||||
it is attached to the surviving progress anchor on replay."""
|
||||
|
|
|
|||
|
|
@ -1472,7 +1472,12 @@ def test_workspace_patch_preserves_lockfile_when_other_changes_are_junk(tmp_path
|
|||
repo.mkdir()
|
||||
subprocess.run(["git", "init"], cwd=repo, check=True, capture_output=True)
|
||||
(repo / "README.md").write_text("base\n", encoding="utf-8")
|
||||
subprocess.run(["git", "add", "README.md"], cwd=repo, check=True, capture_output=True)
|
||||
# Version 3: generated output is excluded because the PROJECT declares it,
|
||||
# not because a host name rule guesses. Without the .gitignore the dist file
|
||||
# would ride the patch and, being a real change beside the lockfile, would
|
||||
# also stop the lockfile from reading as incidental.
|
||||
(repo / ".gitignore").write_text("dist/\n", encoding="utf-8")
|
||||
subprocess.run(["git", "add", "README.md", ".gitignore"], cwd=repo, check=True, capture_output=True)
|
||||
subprocess.run(
|
||||
["git", "-c", "user.email=t@example.com", "-c", "user.name=T", "commit", "-m", "init"],
|
||||
cwd=repo,
|
||||
|
|
@ -1489,7 +1494,9 @@ def test_workspace_patch_preserves_lockfile_when_other_changes_are_junk(tmp_path
|
|||
assert "package-lock.json" in patch
|
||||
assert "dist/out.txt" not in patch
|
||||
assert manifest["counts"]["untracked_included"] == 1
|
||||
assert manifest["counts"]["untracked_excluded"] == 1
|
||||
# A git-ignored file is outside the capture universe, so it is not listed
|
||||
# as an exclusion either.
|
||||
assert manifest["counts"]["untracked_excluded"] == 0
|
||||
|
||||
|
||||
def test_workspace_patch_excludes_binary_junk_and_oversize(tmp_path, monkeypatch):
|
||||
|
|
@ -1520,7 +1527,7 @@ def test_workspace_patch_excludes_binary_junk_and_oversize(tmp_path, monkeypatch
|
|||
|
||||
artifacts, manifest = write_workspace_patch_artifacts(repo, tmp_path / "artifacts", task={})
|
||||
|
||||
assert manifest["exclude_rules_version"] == 2
|
||||
assert manifest["exclude_rules_version"] == 3
|
||||
excluded = {item["path"]: item["reason"] for item in manifest["untracked_excluded"]}
|
||||
assert "binary file" in excluded.get("app", "")
|
||||
assert "binary file" in excluded.get("dump.rdb", "") or "junk artifact" in excluded.get("dump.rdb", "")
|
||||
|
|
|
|||
|
|
@ -18,6 +18,10 @@ def test_navigation_shell_dom_has_sidebar_drawer_and_project_slots():
|
|||
html = _read("web/index.html")
|
||||
chat_js = _read("web/modules/chat.js")
|
||||
assert 'id="primary-sidebar"' in html
|
||||
# The sidebar is the document's one navigation landmark.
|
||||
assert '<nav id="primary-sidebar"' in html
|
||||
assert '<aside id="primary-sidebar"' not in html
|
||||
assert '<aside id="project-panel"' in html
|
||||
assert 'data-mobile-nav-toggle' not in html
|
||||
assert 'id="nav-drawer-backdrop"' in html
|
||||
assert 'data-nav-page="chat"' in html
|
||||
|
|
@ -58,6 +62,13 @@ def test_navigation_state_and_mobile_drawer_are_first_class():
|
|||
assert ".nav-drawer-backdrop" in css
|
||||
assert ".mobile-nav-toggle" in css
|
||||
assert "grid-template-columns: var(--sidebar-width) minmax(0, 1fr) auto;" in css
|
||||
# One client-side route, read once on load and validated against the DOM
|
||||
# rather than a duplicated page list; never written back on navigation
|
||||
# (the desktop shell and the Telegram mini app have no address bar).
|
||||
assert "function pageFromHash()" in app_js
|
||||
assert "document.getElementById(`page-${name}`)" in app_js
|
||||
assert "hashchange" not in app_js
|
||||
assert "replaceState" not in app_js
|
||||
assert ".mobile-nav-toggle {\n position: fixed" not in css
|
||||
assert "flex: 0 0 44px;" in css
|
||||
|
||||
|
|
|
|||
|
|
@ -113,7 +113,7 @@ def test_depth_summary_reports_lower_cap_as_typed_reduction(monkeypatch):
|
|||
assert build_depth_summary(root_contract, statuses) == {
|
||||
"requested_depth": 3, "permitted_depth": 2,
|
||||
"attempted_depth": 2, "achieved_depth": 2,
|
||||
"status": "capability_reduced", "host_visible_only": True,
|
||||
"status": "capability_reduced", "host_visible_only": True
|
||||
}
|
||||
|
||||
|
||||
|
|
@ -140,24 +140,64 @@ def test_depth_summary_is_order_independent_and_allows_chosen_shallower():
|
|||
expected = {
|
||||
"requested_depth": 3, "permitted_depth": 2,
|
||||
"attempted_depth": 2, "achieved_depth": 2,
|
||||
"status": "capability_reduced", "host_visible_only": True,
|
||||
"status": "capability_reduced", "host_visible_only": True
|
||||
}
|
||||
assert build_depth_summary(root_contract, mixed) == expected
|
||||
assert build_depth_summary(root_contract, reversed(mixed)) == expected
|
||||
assert build_depth_summary(root_contract, [mixed[0]]) == {
|
||||
"requested_depth": 3, "permitted_depth": 3,
|
||||
"attempted_depth": 1, "achieved_depth": 1,
|
||||
"status": "chosen_shallower", "host_visible_only": True,
|
||||
"status": "chosen_shallower", "host_visible_only": True
|
||||
}
|
||||
|
||||
|
||||
def test_depth_summary_mixed_requests_uses_strongest_ask_and_branch_status():
|
||||
mixed = [
|
||||
{"depth_provenance": {
|
||||
"requested_depth": 2, "permitted_depth": 2,
|
||||
"attempted_depth": 2, "achieved_depth": 2,
|
||||
}},
|
||||
{"depth_provenance": {
|
||||
"requested_depth": 4, "permitted_depth": 4,
|
||||
"attempted_depth": 3, "achieved_depth": 3,
|
||||
}},
|
||||
]
|
||||
expected = {
|
||||
"requested_depth": 4, "permitted_depth": 4,
|
||||
"attempted_depth": 3, "achieved_depth": 3,
|
||||
"status": "chosen_shallower", "host_visible_only": True
|
||||
}
|
||||
assert build_depth_summary({}, mixed) == expected
|
||||
assert build_depth_summary({}, reversed(mixed)) == expected
|
||||
|
||||
|
||||
def test_depth_summary_reduced_chain_decides_over_deeper_achieved_chain():
|
||||
mixed = [
|
||||
{"depth_provenance": {
|
||||
"requested_depth": 5, "permitted_depth": 5,
|
||||
"attempted_depth": 5, "achieved_depth": 5,
|
||||
}},
|
||||
{"depth_provenance": {
|
||||
"requested_depth": 3, "permitted_depth": 2,
|
||||
"attempted_depth": 2, "achieved_depth": 2,
|
||||
}},
|
||||
]
|
||||
expected = {
|
||||
"requested_depth": 3, "permitted_depth": 2,
|
||||
"attempted_depth": 2, "achieved_depth": 2,
|
||||
"status": "capability_reduced", "host_visible_only": True,
|
||||
}
|
||||
assert build_depth_summary({}, mixed) == expected
|
||||
assert build_depth_summary({}, reversed(mixed)) == expected
|
||||
|
||||
|
||||
def test_depth_summary_never_recomputes_missing_history_from_live_settings(monkeypatch):
|
||||
root_contract = build_task_contract({"delegation_budget": {"depth_remaining": 3}})
|
||||
monkeypatch.setenv("OUROBOROS_MAX_SUBAGENT_DEPTH", "7")
|
||||
assert build_depth_summary(root_contract, []) == {
|
||||
"requested_depth": 3, "permitted_depth": None,
|
||||
"attempted_depth": 0, "achieved_depth": 0,
|
||||
"status": "evidence_unknown", "host_visible_only": True,
|
||||
"status": "evidence_unknown", "host_visible_only": True
|
||||
}
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
import gzip
|
||||
import json
|
||||
import os
|
||||
import pathlib
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from types import SimpleNamespace
|
||||
|
||||
|
|
@ -1238,3 +1239,76 @@ def test_has_failures_true_for_failed_receipt():
|
|||
{"status": "ready", "artifacts": [], "errors": []},
|
||||
)
|
||||
assert refreshed["summary"]["has_failures"] is True
|
||||
|
||||
|
||||
def test_omitted_ledger_stub_keeps_the_real_summary_across_artifact_finalization(tmp_path):
|
||||
"""A ledger above the inline threshold rides as a stub with no entries.
|
||||
|
||||
Artifact finalization rebuilt the ledger FROM that stub, so a task whose
|
||||
real ledger held nine entries and a failing verification receipt was
|
||||
rewritten to "0 entries / no failures / execution ok" on the record the
|
||||
owner and the parent read — the stub is a projection of the artifact file,
|
||||
never a source. The stub's summary is now re-projected from the refreshed
|
||||
file, and the recomputed bundle describes the bytes actually on disk.
|
||||
"""
|
||||
from hashlib import sha256
|
||||
|
||||
from tests.test_headless_cli import _init_repo_with_file
|
||||
|
||||
from ouroboros import headless
|
||||
from ouroboros.task_results import STATUS_COMPLETED, load_task_result, write_task_result
|
||||
|
||||
parent = tmp_path / "data"
|
||||
parent.mkdir()
|
||||
repo = tmp_path / "repo"
|
||||
_init_repo_with_file(repo)
|
||||
task_id = "stubledger"
|
||||
ledger = {
|
||||
"schema_version": 2, "created_at": "now", "task_id": task_id,
|
||||
"entries": [{"kind": "verification_receipt", "status": "fail", "detail": "x" * 1600}]
|
||||
+ [{"kind": "objective_outcome", "status": "not_evaluated", "detail": "y" * 1600} for _ in range(8)],
|
||||
"summary": {"entry_count": 9, "has_failures": True},
|
||||
}
|
||||
refs = maybe_write_verification_artifact(parent, task_id, ledger)
|
||||
assert refs["artifact"] is not None, "the fixture must exceed the inline threshold"
|
||||
assert refs["inline"]["omitted_to_artifact"] is True
|
||||
assert "entries" not in refs["inline"]
|
||||
|
||||
write_task_result(
|
||||
parent, task_id, STATUS_COMPLETED, result="done",
|
||||
verification_ledger=refs["inline"], artifacts=[refs["artifact"]],
|
||||
artifact_status="ready",
|
||||
outcome_axes={"lifecycle": {"status": STATUS_COMPLETED}, "execution": {"status": EXECUTION_OK}},
|
||||
)
|
||||
|
||||
for run in ("first", "second"):
|
||||
headless.finalize_task_artifacts(parent, {"id": task_id, "workspace_root": str(repo)})
|
||||
stored = load_task_result(parent, task_id)
|
||||
stub = stored["verification_ledger"]
|
||||
assert stub["omitted_to_artifact"] is True, run
|
||||
assert "entries" not in stub, run
|
||||
assert "outcome_axes" not in stub, run
|
||||
assert stub["summary"]["entry_count"] == 9, run
|
||||
assert stub["summary"]["has_failures"] is True, run
|
||||
|
||||
ledger_item = next(
|
||||
item for item in stored["artifact_bundle"]["artifacts"]
|
||||
if item.get("kind") == "verification_ledger"
|
||||
)
|
||||
data = pathlib.Path(ledger_item["path"]).read_bytes()
|
||||
assert ledger_item["size"] == len(data), run
|
||||
assert ledger_item["sha256"] == sha256(data).hexdigest(), run
|
||||
on_disk = json.loads(data.decode("utf-8"))
|
||||
assert on_disk["summary"]["entry_count"] == 9, run
|
||||
assert on_disk["summary"]["has_failures"] is True, run
|
||||
|
||||
|
||||
def test_refreshing_an_omitted_ledger_stub_is_an_identity():
|
||||
stub = {
|
||||
"schema_version": 1, "created_at": "now", "task_id": "t",
|
||||
"summary": {"entry_count": 9, "has_failures": True}, "omitted_to_artifact": True,
|
||||
}
|
||||
for status in ("ready", "ready_with_changes", "failed", "pending", "missing"):
|
||||
assert refresh_verification_ledger_artifacts(
|
||||
dict(stub), {"status": status, "artifacts": [], "errors": []},
|
||||
) == stub
|
||||
|
|
|
|||
|
|
@ -161,13 +161,13 @@ def test_missing_role_terminal_root_is_never_labeled_child(tmp_path):
|
|||
|
||||
|
||||
def test_terminal_child_projection_is_idempotent_and_honest_for_all_outcomes(tmp_path):
|
||||
from ouroboros.project_dialogue import append_terminal_task_projection
|
||||
from ouroboros.project_dialogue import OUTCOME_PHASE_HEADLINE, append_terminal_task_projection
|
||||
|
||||
cases = [
|
||||
("ok", "completed", {"execution": {"status": "ok"}}, "Completed"),
|
||||
("ok", "completed", {"execution": {"status": "ok"}}, "Done"),
|
||||
("failed", "failed", {"execution": {"status": "failed"}}, "Failed"),
|
||||
("cancelled", "cancelled", {"execution": {"status": "ok"}}, "Cancelled"),
|
||||
("degraded", "completed", {"execution": {"status": "best_effort"}}, "Completed with limitations"),
|
||||
("degraded", "completed", {"execution": {"status": "best_effort"}}, "Done with warnings"),
|
||||
]
|
||||
for suffix, status, axes, label in cases:
|
||||
task_id = f"child-{suffix}"
|
||||
|
|
@ -196,6 +196,7 @@ def test_terminal_child_projection_is_idempotent_and_honest_for_all_outcomes(tmp
|
|||
assert row["role"] == "reviewer"
|
||||
assert row["status"] == status
|
||||
assert row["outcome"] == label
|
||||
assert OUTCOME_PHASE_HEADLINE[row["outcome_phase"]] == label
|
||||
assert row["reason_code"] == f"reason-{suffix}"
|
||||
assert row["result_ref"] == {
|
||||
"kind": "task_result", "task_id": task_id, "reader": "get_task_result",
|
||||
|
|
@ -285,7 +286,7 @@ def test_terminal_root_fallback_covers_cancel_without_preempting_open_synthesis(
|
|||
)
|
||||
normal = next(row for row in _chat_rows(tmp_path) if row.get("task_id") == "normal-root")
|
||||
assert normal["summary_kind"] == "terminal_root_projection"
|
||||
assert normal["outcome"] == "Completed"
|
||||
assert normal["outcome"] == "Done"
|
||||
assert normal["outcome_final"] is True
|
||||
|
||||
|
||||
|
|
@ -313,7 +314,7 @@ def test_running_async_root_truth_survives_restart_degradation_once(tmp_path):
|
|||
assert stored["root_phase_checkpoint"]["post_task_synthesis"] == "degraded"
|
||||
rows = [row for row in _chat_rows(tmp_path) if row.get("task_id") == "restart-root"]
|
||||
assert len(rows) == 1
|
||||
assert rows[0]["outcome"] == "Completed with limitations"
|
||||
assert rows[0]["outcome"] == "Done"
|
||||
assert rows[0]["outcome"] == completion_status_label(stored, {})
|
||||
assert rows[0]["outcome_final"] is True
|
||||
assert recover_pending_root_post_task_synthesis(tmp_path, repo_dir=tmp_path / "repo") == 0
|
||||
|
|
|
|||
|
|
@ -93,7 +93,7 @@ def test_completion_summary_event_text_is_plain_and_fully_normalized(
|
|||
ctx.DRIVE_ROOT, {"status": "completed"}, "root-project", root, result, done,
|
||||
) is True
|
||||
assert queued[0]["text"] == (
|
||||
f"Launch 🚀 › Ship release · Completed\n{PLAIN_EXCERPT}"
|
||||
f"Launch 🚀 › Ship release · Done\n{PLAIN_EXCERPT}"
|
||||
)
|
||||
for marker in ("#", "**", "`"):
|
||||
assert marker not in queued[0]["text"]
|
||||
|
|
@ -136,6 +136,17 @@ def test_host_salvage_terminal_incident_drops_inherited_markdown_format(tmp_path
|
|||
)
|
||||
assert legacy["format"] == "markdown"
|
||||
|
||||
# A host NOTICE is not salvage: its own words are the answer, and its
|
||||
# markdown must survive or the host's own code spans render escaped.
|
||||
notice = project_terminal_result_event(
|
||||
tmp_path, {"chat_id": 7}, "terminal-a",
|
||||
result_text=raw, terminal_origin="host_notice", base_event=dict(base),
|
||||
)
|
||||
assert notice["format"] == "markdown"
|
||||
assert notice["role"] == "system"
|
||||
assert "system_type" not in notice
|
||||
assert notice["text"] == raw
|
||||
|
||||
|
||||
def test_history_normalizes_old_project_rows_on_read_without_rewriting_log(tmp_path):
|
||||
"""Bug report #10: rows persisted BEFORE the producer stripped markdown are
|
||||
|
|
@ -285,3 +296,259 @@ def test_cancel_receipt_rides_verbatim_live_durable_and_on_replay(
|
|||
)
|
||||
assert replayed["text"] == verbatim
|
||||
assert replayed["markdown"] is False
|
||||
|
||||
|
||||
def _parity_cases():
|
||||
import pathlib
|
||||
|
||||
fixture = pathlib.Path(__file__).resolve().parents[1] / "web" / "tests" / "fixtures" / "outcome_phase_parity.json"
|
||||
return json.loads(fixture.read_text(encoding="utf-8"))["cases"]
|
||||
|
||||
|
||||
def test_host_status_phase_mirrors_the_browser_over_the_shared_fixture():
|
||||
"""S5-03: one status-word family. The same fixture is read by
|
||||
``web/tests/reason_detail.test.js``, so a divergence between the browser's
|
||||
severity fold and the host's durable label fails on both sides."""
|
||||
from ouroboros.project_dialogue import completion_status_label, outcome_phase
|
||||
|
||||
cases = _parity_cases()
|
||||
assert len(cases) >= 10
|
||||
for case in cases:
|
||||
record = case["record"]
|
||||
assert outcome_phase(record, {}) == case["phase"], case["name"]
|
||||
assert completion_status_label(record, {}) == case["headline"], case["name"]
|
||||
# The event frame is the other half of the same merge: a record read
|
||||
# from the task_done event alone must resolve identically.
|
||||
assert outcome_phase({}, record) == case["phase"], case["name"]
|
||||
|
||||
|
||||
def test_host_status_phase_folds_the_legacy_partial_result_status_to_a_warning():
|
||||
"""The browser already shows a warning here (the gateway normalizes axes on
|
||||
read); the host label used to agree only by accident, through a 'partial'
|
||||
member of its own degraded set."""
|
||||
from ouroboros.project_dialogue import completion_status_label, outcome_phase
|
||||
|
||||
legacy = {"status": "completed", "result_status": "partial"}
|
||||
assert outcome_phase(legacy, {}) == "warn"
|
||||
assert completion_status_label(legacy, {}) == "Done with warnings"
|
||||
|
||||
|
||||
def test_owner_requested_stop_is_done_on_the_host_row_too():
|
||||
from ouroboros.project_dialogue import _completion_verdict, completion_status_label
|
||||
|
||||
stopped = {
|
||||
"status": "completed", "reason_code": "owner_requested_finalization",
|
||||
"outcome_axes": {"execution": {"status": "best_effort"}},
|
||||
}
|
||||
assert completion_status_label(stopped, {}) == "Done"
|
||||
assert _completion_verdict(stopped, {}) == ""
|
||||
|
||||
|
||||
A4_DECISION = {
|
||||
"status": "finalized_unaccepted",
|
||||
"rationale": "Acceptance reviewers did not reach a valid quorum.",
|
||||
}
|
||||
A4_CLAUSE = "Acceptance: finalized_unaccepted — Acceptance reviewers did not reach a valid quorum."
|
||||
|
||||
|
||||
def _a4_result(**overrides):
|
||||
result = {
|
||||
"task_id": "root-project", "status": "completed", "reason_code": "final_message",
|
||||
"project_id": "launch", "title": "Ship release", "result": "Release shipped.",
|
||||
"outcome_axes": {
|
||||
"execution": {"status": "ok"},
|
||||
"review": {"status": "degraded", "acceptance_decision": dict(A4_DECISION)},
|
||||
},
|
||||
}
|
||||
result.update(overrides)
|
||||
return result
|
||||
|
||||
|
||||
def test_host_verdict_states_an_unaccepted_acceptance_decision_in_its_own_words():
|
||||
"""S5-04: a warning caused by REVIEW used to be explained by the execution
|
||||
reason that happened to sit beside it (``Reason: final_message``), which
|
||||
named the delivery step rather than the cause."""
|
||||
from ouroboros.project_dialogue import _completion_verdict
|
||||
|
||||
assert _completion_verdict(_a4_result(), {}) == A4_CLAUSE
|
||||
# The stored rationale already ends in a period; the clause must not double it.
|
||||
assert not _completion_verdict(_a4_result(), {}).endswith("..")
|
||||
|
||||
|
||||
def test_host_verdict_keeps_the_execution_reason_when_acceptance_was_reached():
|
||||
from ouroboros.project_dialogue import _completion_verdict
|
||||
|
||||
accepted = _a4_result(outcome_axes={
|
||||
"execution": {"status": "ok"},
|
||||
"review": {"status": "pass", "acceptance_decision": {"status": "accepted"}},
|
||||
})
|
||||
assert _completion_verdict(accepted, {}) == "Reason: final_message."
|
||||
assert _completion_verdict({"status": "completed"}, {}) == ""
|
||||
# A hard failure explains itself by its execution reason, not by a decision.
|
||||
assert _completion_verdict(
|
||||
_a4_result(status="failed", reason_code="delegated_custody_unreconciled",
|
||||
outcome_axes={"execution": {"status": "failed"},
|
||||
"review": {"acceptance_decision": dict(A4_DECISION)}}),
|
||||
{},
|
||||
) == "Reason: delegated_custody_unreconciled."
|
||||
|
||||
|
||||
def test_host_verdict_flattens_and_strips_a_markdown_rationale():
|
||||
"""These are durable plain-text rows: the rationale is free owner-visible
|
||||
text up to 500 characters and may carry newlines and markdown markers."""
|
||||
from ouroboros.project_dialogue import _completion_verdict
|
||||
|
||||
noisy = _a4_result(outcome_axes={
|
||||
"execution": {"status": "ok"},
|
||||
"review": {"status": "degraded", "acceptance_decision": {
|
||||
"status": "revision_requested",
|
||||
"rationale": "## Verdict\n\nThe **tests** never ran with `pytest`",
|
||||
}},
|
||||
})
|
||||
verdict = _completion_verdict(noisy, {})
|
||||
assert verdict == "Acceptance: revision_requested — Verdict The tests never ran with pytest."
|
||||
for marker in ("#", "**", "`", "\n"):
|
||||
assert marker not in verdict
|
||||
|
||||
# A rationale that already terminates itself keeps its own punctuation: a
|
||||
# question mark is as terminal as a period, and appending one would render
|
||||
# "…did the tests run?." to the owner.
|
||||
asking = _a4_result(outcome_axes={
|
||||
"execution": {"status": "ok"},
|
||||
"review": {"status": "degraded", "acceptance_decision": {
|
||||
"status": "revision_requested", "rationale": "Did the tests ever run?",
|
||||
}},
|
||||
})
|
||||
assert _completion_verdict(asking, {}) == "Acceptance: revision_requested — Did the tests ever run?"
|
||||
|
||||
|
||||
def test_host_verdict_keeps_the_full_bounded_acceptance_rationale():
|
||||
from ouroboros.project_dialogue import _completion_verdict
|
||||
|
||||
rationale = ("Review evidence " + ("remains material and owner-visible. " * 9)).strip()
|
||||
result = _a4_result(outcome_axes={
|
||||
"execution": {"status": "ok"},
|
||||
"review": {"status": "degraded", "acceptance_decision": {
|
||||
"status": "revision_requested", "rationale": rationale,
|
||||
}},
|
||||
})
|
||||
|
||||
assert len(rationale) > 240
|
||||
assert _completion_verdict(result, {}) == f"Acceptance: revision_requested — {rationale}"
|
||||
|
||||
|
||||
def test_host_verdict_states_a_decision_without_a_rationale_alone():
|
||||
from ouroboros.project_dialogue import _completion_verdict
|
||||
|
||||
bare = _a4_result(outcome_axes={
|
||||
"execution": {"status": "degraded"},
|
||||
"review": {"status": "degraded", "acceptance_decision": {"status": "revision_requested"}},
|
||||
})
|
||||
assert _completion_verdict(bare, {}) == "Acceptance: revision_requested."
|
||||
|
||||
|
||||
def test_host_verdict_leads_both_lifecycle_rows(tmp_path, monkeypatch):
|
||||
"""The verdict must reach the owner where the owner looks: the Main row's
|
||||
second line and the Project thread's terminal row."""
|
||||
from ouroboros.project_dialogue import (
|
||||
append_terminal_task_projection, enqueue_project_completion_summary,
|
||||
)
|
||||
from ouroboros.projects_registry import bind_task_to_project, create_project
|
||||
|
||||
project = create_project(tmp_path, "launch", name="Launch 🚀")
|
||||
bind_task_to_project(
|
||||
tmp_path, "root-project", project["id"], project["chat_id"],
|
||||
origin={"absent": "system"},
|
||||
)
|
||||
queued = []
|
||||
monkeypatch.setattr(
|
||||
"supervisor.terminal_delivery.enqueue_terminal_delivery",
|
||||
lambda _root, event, **_kwargs: queued.append(dict(event)) or True,
|
||||
)
|
||||
root = {
|
||||
"id": "root-project", "project_id": "launch",
|
||||
"title": "Ship release", "chat_id": project["chat_id"],
|
||||
}
|
||||
result = _a4_result()
|
||||
done = {"status": "completed", "outcome_axes": result["outcome_axes"]}
|
||||
|
||||
assert enqueue_project_completion_summary(
|
||||
tmp_path, {"status": "completed"}, "root-project", root, result, done,
|
||||
) is True
|
||||
assert queued[0]["text"] == (
|
||||
f"Launch 🚀 › Ship release · Done with warnings\n{A4_CLAUSE} Release shipped."
|
||||
)
|
||||
assert "final_message" not in queued[0]["text"]
|
||||
|
||||
ordinary = _a4_result(
|
||||
reason_code="budget_exhausted",
|
||||
outcome_axes={"execution": {"status": "degraded"}},
|
||||
)
|
||||
assert enqueue_project_completion_summary(
|
||||
tmp_path, {"status": "completed"}, "root-project", root, ordinary,
|
||||
{"status": "completed", "outcome_axes": ordinary["outcome_axes"]},
|
||||
) is True
|
||||
assert queued[1]["text"] == (
|
||||
"Launch 🚀 › Ship release · Done with warnings\n"
|
||||
"Reason: budget_exhausted. Release shipped."
|
||||
)
|
||||
|
||||
assert append_terminal_task_projection(tmp_path, "root-project", root, result, done)
|
||||
rows = [
|
||||
json.loads(line)
|
||||
for line in (tmp_path / "logs" / "chat.jsonl").read_text(encoding="utf-8").splitlines()
|
||||
if line.strip()
|
||||
]
|
||||
projection = next(row for row in rows if row.get("summary_kind") == "terminal_root_projection")
|
||||
assert A4_CLAUSE in projection["text"]
|
||||
assert "Reason: final_message" not in projection["text"]
|
||||
assert projection["reason_code"] == "final_message"
|
||||
assert projection["outcome"] == "Done with warnings"
|
||||
assert projection["text"].endswith('Details: get_task_result(task_id="root-project")')
|
||||
|
||||
|
||||
def test_host_verdict_and_the_card_line_compose_the_same_sentence():
|
||||
"""The shared fixture is the only place the two languages agree; a clause
|
||||
present there must be produced by the host verdict as well."""
|
||||
from ouroboros.project_dialogue import _completion_verdict
|
||||
|
||||
for case in _parity_cases():
|
||||
clause = case.get("acceptance_clause") or ""
|
||||
if clause:
|
||||
assert _completion_verdict(case["record"], {}) == clause, case["name"]
|
||||
|
||||
|
||||
def test_terminal_row_reports_the_depth_request_only_when_one_exists(tmp_path):
|
||||
"""S3-05: the owner-visible row is where a nested swarm's depth becomes
|
||||
checkable — numbers first, and no line at all on swarms nobody asked to
|
||||
nest, so an ordinary task's row stays byte-unchanged."""
|
||||
from ouroboros.project_dialogue import append_terminal_task_projection
|
||||
|
||||
task = {"id": "swarm-root", "chat_id": 3, "role": "root"}
|
||||
result = {
|
||||
"task_id": "swarm-root", "status": "completed", "result": "Shipped.",
|
||||
"outcome_axes": {"execution": {"status": "ok"}},
|
||||
"swarm_efficiency": {
|
||||
"subagent_count": 3,
|
||||
"depth": {
|
||||
"requested_depth": 2, "permitted_depth": 4, "attempted_depth": 2,
|
||||
"achieved_depth": 2, "status": "achieved", "host_visible_only": True,
|
||||
},
|
||||
},
|
||||
}
|
||||
done = {"chat_id": 3, "status": "completed", "outcome_axes": result["outcome_axes"]}
|
||||
assert append_terminal_task_projection(tmp_path, "swarm-root", task, result, done)
|
||||
|
||||
flat = {"id": "flat-root", "chat_id": 3, "role": "root"}
|
||||
flat_result = {**result, "task_id": "flat-root", "swarm_efficiency": {"subagent_count": 3}}
|
||||
assert append_terminal_task_projection(tmp_path, "flat-root", flat, flat_result, done)
|
||||
|
||||
rows = {
|
||||
json.loads(line)["task_id"]: json.loads(line)
|
||||
for line in (tmp_path / "logs" / "chat.jsonl").read_text(encoding="utf-8").splitlines()
|
||||
if line.strip()
|
||||
}
|
||||
assert "Depth requested=2, permitted=4, achieved=2 (achieved)." in rows["swarm-root"]["text"]
|
||||
assert "Depth" not in rows["flat-root"]["text"]
|
||||
for marker in ("#", "**", "`"):
|
||||
assert marker not in rows["swarm-root"]["text"]
|
||||
|
|
|
|||
|
|
@ -856,7 +856,7 @@ def test_project_completion_enqueues_once_for_root_and_never_for_child_or_epheme
|
|||
"type": "send_message",
|
||||
"chat_id": 1,
|
||||
"task_id": "root-project",
|
||||
"text": "Launch 🚀 › Ship release · Completed\nRelease shipped.",
|
||||
"text": "Launch 🚀 › Ship release · Done\nRelease shipped.",
|
||||
"role": "system",
|
||||
"system_type": "project_completion_summary",
|
||||
"delivery_id": "project-completion:root-project",
|
||||
|
|
@ -893,15 +893,15 @@ def test_project_completion_enqueues_once_for_root_and_never_for_child_or_epheme
|
|||
(
|
||||
{"status": "completed"},
|
||||
{"outcome_axes": {"execution": {"status": "degraded"}}},
|
||||
"Completed with limitations",
|
||||
"Done with warnings",
|
||||
),
|
||||
(
|
||||
{"status": "completed", "outcome_axes": {"execution": {"status": "degraded"}}},
|
||||
{},
|
||||
"Completed with limitations",
|
||||
"Done with warnings",
|
||||
),
|
||||
(
|
||||
{"status": "completed", "outcome_axes": {"objective": {"status": "fail"}}},
|
||||
{"status": "completed", "outcome_axes": {"objective": {"status": "fail", "source": "task_acceptance_review"}}},
|
||||
{},
|
||||
"Failed",
|
||||
),
|
||||
|
|
|
|||
|
|
@ -1521,6 +1521,39 @@ def test_recent_tasks_includes_outcome_contract_and_ledger(tmp_path):
|
|||
assert record["artifact_bundle"]["status"] == "ready_no_changes"
|
||||
assert record["verification_ledger"]["entry_count"] == 1
|
||||
|
||||
# A ledger above the inline threshold rides as a stub with NO entries: its
|
||||
# own summary is the count authority, so the row must not report zero.
|
||||
write_task_result(
|
||||
tmp_path,
|
||||
"recent2",
|
||||
STATUS_COMPLETED,
|
||||
result="done",
|
||||
verification_ledger={
|
||||
"schema_version": 1, "omitted_to_artifact": True,
|
||||
"summary": {"entry_count": 9, "has_failures": True},
|
||||
},
|
||||
)
|
||||
payload = json.loads(_handle_recent_tasks(SimpleNamespace(drive_root=tmp_path), limit=2))
|
||||
stub_record = next(row for row in payload["tasks"] if row["task_id"] == "recent2")
|
||||
assert stub_record["verification_ledger"]["entry_count"] == 9
|
||||
assert stub_record["verification_ledger"]["summary"]["has_failures"] is True
|
||||
|
||||
|
||||
def test_subtask_outcome_summary_reports_a_stub_ledger_count(tmp_path):
|
||||
"""The same count authority on the parent-visible child summary: a stub
|
||||
ledger used to report ``0 entries / no failures`` to its parent."""
|
||||
from ouroboros.tools.control import _subtask_outcome_summary
|
||||
|
||||
summary = json.loads(_subtask_outcome_summary({
|
||||
"status": "completed", "result": "done",
|
||||
"verification_ledger": {
|
||||
"schema_version": 1, "omitted_to_artifact": True,
|
||||
"summary": {"entry_count": 9, "has_failures": True},
|
||||
},
|
||||
}))
|
||||
assert summary["verification_ledger"]["entry_count"] == 9
|
||||
assert summary["verification_ledger"]["summary"]["has_failures"] is True
|
||||
|
||||
|
||||
def test_effective_status_keeps_workspace_finalization_nonterminal_without_child_drive(tmp_path):
|
||||
from ouroboros.headless import ARTIFACT_STATUS_FINALIZING
|
||||
|
|
|
|||
|
|
@ -126,6 +126,41 @@ def test_task_summary_scan_uses_bounded_live_tail(tmp_path, monkeypatch):
|
|||
assert seen == {"path": path, "max_entries": 123, "tail_bytes": 256 * 1024}
|
||||
|
||||
|
||||
def test_tasks_notify_waits_for_latest_final_summary_for_each_task(tmp_path, monkeypatch):
|
||||
nt = _load(); api, data = _api(tmp_path); _Rec.sent = []
|
||||
monkeypatch.setattr(nt, "TelegramClient", _Rec)
|
||||
chat = data / "logs" / "chat.jsonl"
|
||||
working = {
|
||||
"type": "task_summary", "task_id": "same1", "outcome_final": False,
|
||||
"outcome_phase": "working",
|
||||
"outcome_axes": {"lifecycle": {"status": "completed"},
|
||||
"execution": {"status": "ok"}},
|
||||
}
|
||||
chat.write_text(json.dumps(working) + "\n", encoding="utf-8")
|
||||
state = {"notified_task_ids": []}
|
||||
|
||||
asyncio.run(nt._check_tasks_notify(
|
||||
api, {"TELEGRAM_NOTIFY_TASKS": "on"}, 42, state, "en",
|
||||
))
|
||||
assert _Rec.sent == []
|
||||
assert state["notified_task_ids"] == []
|
||||
|
||||
terminal = {
|
||||
**working, "outcome_final": True, "outcome_phase": "warn",
|
||||
"outcome_axes": {"lifecycle": {"status": "completed"},
|
||||
"execution": {"status": "ok"},
|
||||
"review": {"status": "degraded"}},
|
||||
}
|
||||
with open(chat, "a", encoding="utf-8") as f:
|
||||
f.write(json.dumps(terminal) + "\n")
|
||||
|
||||
asyncio.run(nt._check_tasks_notify(
|
||||
api, {"TELEGRAM_NOTIFY_TASKS": "on"}, 42, state, "en",
|
||||
))
|
||||
assert _Rec.sent == [(42, "⚠️ Task same1 done")]
|
||||
assert state["notified_task_ids"] == ["same1"]
|
||||
|
||||
|
||||
def test_notify_disabled_is_silent(tmp_path, monkeypatch):
|
||||
nt = _load(); api, data = _api(tmp_path); _Rec.sent = []
|
||||
monkeypatch.setattr(nt, "TelegramClient", _Rec)
|
||||
|
|
@ -540,3 +575,42 @@ def test_tasks_notify_reads_lifecycle_status_and_severity_icon(tmp_path, monkeyp
|
|||
]
|
||||
# The container is never stringified into an owner-visible push.
|
||||
assert all("{'status'" not in text for _chat, text in _Rec.sent)
|
||||
|
||||
|
||||
def test_tasks_notify_consumes_the_host_status_phase_when_the_row_carries_one(
|
||||
tmp_path, monkeypatch,
|
||||
):
|
||||
"""S5-03: the host stamps ONE phase on its own task rows, so the notifier
|
||||
consumes it instead of re-deriving a third status ladder.
|
||||
|
||||
Two owner-visible flips follow from the card's rule: a task whose execution
|
||||
was clean but whose acceptance review degraded now warns (it used to read
|
||||
✅), and an owner-requested stop is a success (it used to warn because the
|
||||
stop leaves a best_effort execution axis). Legacy rows without the field
|
||||
keep the axes rule; a pre-finalization ``working`` row is not a candidate.
|
||||
"""
|
||||
nt = _load(); api, data = _api(tmp_path); _Rec.sent = []
|
||||
monkeypatch.setattr(nt, "TelegramClient", _Rec)
|
||||
rows = [
|
||||
{"task_id": "review1", "outcome_phase": "warn",
|
||||
"outcome_axes": {"lifecycle": {"status": "completed"},
|
||||
"execution": {"status": "ok"},
|
||||
"review": {"status": "degraded"}}},
|
||||
{"task_id": "stop1", "outcome_phase": "done",
|
||||
"outcome_axes": {"lifecycle": {"status": "completed"},
|
||||
"execution": {"status": "best_effort"}}},
|
||||
{"task_id": "open1", "outcome_phase": "working",
|
||||
"outcome_axes": {"lifecycle": {"status": "completed"},
|
||||
"execution": {"status": "degraded"}}},
|
||||
]
|
||||
with open(data / "logs" / "chat.jsonl", "w", encoding="utf-8") as f:
|
||||
for row in rows:
|
||||
f.write(json.dumps({"type": "task_summary", **row}) + "\n")
|
||||
state = {"notified_task_ids": []}
|
||||
asyncio.run(nt._check_tasks_notify(api, {"TELEGRAM_NOTIFY_TASKS": "on"}, 42, state, "en"))
|
||||
|
||||
assert [text for _chat, text in _Rec.sent] == [
|
||||
"⚠️ Task review1 done",
|
||||
"✅ Task stop1 done",
|
||||
]
|
||||
assert state["notified_task_ids"] == ["review1", "stop1"]
|
||||
|
|
|
|||
|
|
@ -304,10 +304,12 @@ def test_deadline_grace_final_is_never_stamped_host_salvage(monkeypatch):
|
|||
assert usage.get("terminal_origin") != L.TERMINAL_ORIGIN_HOST_SALVAGE
|
||||
|
||||
|
||||
def test_budget_and_round_limit_rails_keep_legacy_missing_origin(monkeypatch):
|
||||
"""The wrapper deviation's whole point (fable lane B gap #2): budget and
|
||||
round-limit rails never enter the wrapper, so their terminals still carry
|
||||
NO terminal_origin (missing = legacy shape)."""
|
||||
def test_budget_and_round_limit_rails_stamp_host_notice(monkeypatch):
|
||||
"""INTENTIONAL behaviour change, not a bug fix of the test: these rails used
|
||||
to leave terminal_origin absent because they never enter the provider-death
|
||||
wrapper, so "missing" meant both "written before the stamp existed" and "the
|
||||
host wrote this text alone". The forced-finalization sink now stamps
|
||||
host_notice, and a missing origin identifies only legacy rows."""
|
||||
import pathlib
|
||||
from types import SimpleNamespace
|
||||
|
||||
|
|
@ -324,4 +326,131 @@ def test_budget_and_round_limit_rails_keep_legacy_missing_origin(monkeypatch):
|
|||
)
|
||||
monkeypatch.setattr(L, "call_llm_with_retry", lambda *a, **k: (None, 0.0))
|
||||
_t, usage, _tr = L._handle_round_limit(ctx)
|
||||
assert "terminal_origin" not in usage
|
||||
assert usage.get("terminal_origin") == L.TERMINAL_ORIGIN_HOST_NOTICE
|
||||
|
||||
|
||||
def _rail_ctx(task_id="t-rail", round_idx=11):
|
||||
import pathlib
|
||||
from types import SimpleNamespace
|
||||
|
||||
return SimpleNamespace(
|
||||
messages=[{"role": "user", "content": "do"}],
|
||||
llm=None, active_model="m", active_effort="low", max_retries=1,
|
||||
drive_logs=pathlib.Path("/tmp"), task_id=task_id, round_idx=round_idx,
|
||||
event_queue=None, accumulated_usage={},
|
||||
task_type="", active_use_local=False, max_rounds=10, deadline_ts=None,
|
||||
drive_root=None, budget_drive_root=None, root_task_id="", tools=None,
|
||||
llm_trace={},
|
||||
)
|
||||
|
||||
|
||||
def test_budget_rejection_before_any_work_is_a_host_notice(monkeypatch):
|
||||
"""The host wrote this text alone, so it must be attributable — and it must
|
||||
be published verbatim rather than replaced by the outage receipt."""
|
||||
import ouroboros.loop as L
|
||||
|
||||
ctx = _rail_ctx(task_id="t-budget", round_idx=1)
|
||||
result = L._check_budget_limits(ctx, 0.0)
|
||||
assert result is not None
|
||||
text, usage, _trace = result
|
||||
assert text.startswith("🚫 Task rejected")
|
||||
assert usage["terminal_origin"] == L.TERMINAL_ORIGIN_HOST_NOTICE
|
||||
assert usage["reason_code"] == "budget_exhausted"
|
||||
|
||||
|
||||
def test_a_host_notice_publishes_its_own_words_with_its_markdown(tmp_path):
|
||||
"""A notice is NOT salvage: replacing its text with the outage receipt would
|
||||
name the wrong cause, and dropping its markdown would render the host's own
|
||||
code spans as escaped plain text. It also carries no system_type, which is
|
||||
what lets a replayed task card conclude on it."""
|
||||
from supervisor.terminal_delivery import project_terminal_result_event
|
||||
|
||||
text = "🚫 Task rejected. Total budget exhausted.\n\nPlan review left open: `plan_review_advisory`."
|
||||
base = {
|
||||
"type": "send_message", "chat_id": 7, "task_id": "terminal-n",
|
||||
"text": text, "format": "markdown", "log_text": "kept",
|
||||
}
|
||||
notice = project_terminal_result_event(
|
||||
tmp_path, {"chat_id": 7}, "terminal-n",
|
||||
result_text=text, terminal_origin="host_notice", base_event=dict(base),
|
||||
)
|
||||
assert notice["text"] == text
|
||||
assert "model-provider outage" not in notice["text"]
|
||||
assert notice["role"] == "system"
|
||||
assert notice["terminal_origin"] == "host_notice"
|
||||
assert "system_type" not in notice
|
||||
assert notice["format"] == "markdown"
|
||||
assert notice["log_text"] == "kept"
|
||||
|
||||
|
||||
def test_the_durable_result_carries_the_third_producer_word():
|
||||
from ouroboros.task_finalization import terminal_result_fields
|
||||
|
||||
assert terminal_result_fields({"terminal_origin": "host_notice"})["terminal_origin"] == "host_notice"
|
||||
assert terminal_result_fields({"terminal_origin": "model_final"})["terminal_origin"] == "model_final"
|
||||
# An unknown producer stays legacy rather than acquiring a word.
|
||||
assert "terminal_origin" not in terminal_result_fields({"terminal_origin": "something_new"})
|
||||
|
||||
|
||||
def test_provider_death_arms_are_not_downgraded_by_the_forced_sink(monkeypatch):
|
||||
"""Regression for the sink's interaction with the provider rail: the sink
|
||||
stamps host_notice FIRST on the three no-call/no-resend provider-death arms,
|
||||
so a plain setdefault in the wrapper would have become a no-op and a
|
||||
provider-death salvage would have published verbatim."""
|
||||
import ouroboros.loop as L
|
||||
|
||||
monkeypatch.setattr(L, "call_llm_with_retry", lambda *a, **k: (None, 0.0))
|
||||
for kind, wait_cause in (
|
||||
("provider_outcome_unknown", ""),
|
||||
("provider_unavailable", "transport_unavailable"),
|
||||
("context_overflow", ""),
|
||||
):
|
||||
ctx = _rail_ctx(task_id=f"t-{kind}")
|
||||
ctx.accumulated_usage = {"terminal_origin": L.TERMINAL_ORIGIN_HOST_NOTICE}
|
||||
_text, usage, _trace = L._handle_provider_unavailable(
|
||||
ctx, error_kind=kind, wait_cause=wait_cause,
|
||||
)
|
||||
assert usage.get("terminal_origin") == L.TERMINAL_ORIGIN_HOST_SALVAGE, kind
|
||||
|
||||
|
||||
def test_a_notice_keeps_its_completion_excerpt_unlike_a_salvage():
|
||||
"""Only salvage hides its text behind the neutral details copy."""
|
||||
from ouroboros.project_dialogue import _completion_excerpt
|
||||
|
||||
assert _completion_excerpt({"result": "x", "terminal_origin": "host_notice"}) == "x"
|
||||
assert _completion_excerpt({"result": "x", "terminal_origin": "host_salvage"}) == ""
|
||||
|
||||
|
||||
def test_a_non_provider_rail_with_a_complete_candidate_stays_model_final(tmp_path, monkeypatch):
|
||||
"""The candidate branch only stamped provenance on the provider rail, so a
|
||||
round-limit stop that delivered the model's own complete answer went out
|
||||
unattributed. The ordinary no-tool final composes the same disclosure glue
|
||||
and stays model_final; this rail now says the same thing, and the glue is
|
||||
still appended to the model's text."""
|
||||
from tests.test_delivery_forced_finalization import _forced_test_context
|
||||
|
||||
loop, registry, limit_ctx, trace = _forced_test_context(tmp_path)
|
||||
answer = "Complete current model answer."
|
||||
loop._replace_delivery_candidate(registry, limit_ctx, trace, answer, control="replace")
|
||||
monkeypatch.setattr(
|
||||
loop, "_force_plan_disclosure",
|
||||
lambda *_a, **_k: "\n\nPlan review was left open.",
|
||||
)
|
||||
|
||||
text, usage, _returned_trace = loop._handle_round_limit(limit_ctx)
|
||||
|
||||
assert text.startswith(answer)
|
||||
assert text.endswith("Plan review was left open.")
|
||||
assert usage["terminal_origin"] == loop.TERMINAL_ORIGIN_MODEL_FINAL
|
||||
assert usage["terminal_plan_review_open"] is True
|
||||
|
||||
|
||||
def test_a_non_provider_rail_without_a_candidate_is_a_host_notice(tmp_path, monkeypatch):
|
||||
from tests.test_delivery_forced_finalization import _forced_test_context
|
||||
|
||||
loop, _registry, limit_ctx, _trace = _forced_test_context(tmp_path)
|
||||
monkeypatch.setattr(loop, "call_llm_with_retry", lambda *_a, **_k: (None, 0.0))
|
||||
|
||||
_text, usage, _returned_trace = loop._handle_round_limit(limit_ctx)
|
||||
|
||||
assert usage["terminal_origin"] == loop.TERMINAL_ORIGIN_HOST_NOTICE
|
||||
|
|
|
|||
|
|
@ -1042,7 +1042,9 @@ def test_expired_local_deadline_does_not_dispatch_a_paid_final_call(monkeypatch,
|
|||
|
||||
assert result is not None
|
||||
assert result[0].startswith("⚠️ Task reached its deadline")
|
||||
assert ctx.accumulated_usage == {
|
||||
usage = dict(ctx.accumulated_usage)
|
||||
assert usage.pop("terminal_origin") == "host_notice"
|
||||
assert usage == {
|
||||
"execution_status": "failed",
|
||||
"reason_code": "deadline_local",
|
||||
}
|
||||
|
|
|
|||
|
|
@ -914,3 +914,37 @@ def test_ui_smoke_module_widget_temporal_convergence(direct_server_with_data, br
|
|||
if "Executable doesn't exist" in str(exc) or "playwright install" in str(exc).lower():
|
||||
pytest.skip(str(exc))
|
||||
raise
|
||||
|
||||
|
||||
@pytest.mark.ui_browser
|
||||
def test_ui_smoke_widgets_page_has_a_typed_address_and_one_nav_landmark(
|
||||
direct_server_with_data,
|
||||
):
|
||||
"""A page could not be linked to and the sidebar was not a landmark.
|
||||
|
||||
Opening the application with a `#widgets` fragment must land on Widgets
|
||||
(browsers gain a shareable link; the desktop shell and the Linux fallback
|
||||
get the same load-time route with no address bar to break), and an
|
||||
automation or assistive client must be able to resolve the navigation by
|
||||
its landmark instead of by an id.
|
||||
"""
|
||||
pytest.importorskip("playwright.sync_api", reason="Playwright is not installed")
|
||||
from playwright.sync_api import Error as PlaywrightError
|
||||
from playwright.sync_api import sync_playwright
|
||||
|
||||
url = direct_server_with_data["url"]
|
||||
try:
|
||||
with sync_playwright() as pw:
|
||||
browser = pw.chromium.launch(headless=True)
|
||||
page = browser.new_page(viewport={"width": 1280, "height": 800})
|
||||
try:
|
||||
page.goto(f"{url}/#widgets", wait_until="domcontentloaded", timeout=30_000)
|
||||
page.wait_for_selector("nav#primary-sidebar", timeout=15_000)
|
||||
page.wait_for_selector("#page-widgets.active", timeout=15_000)
|
||||
assert page.locator('[data-nav-page="widgets"].active').count() == 1
|
||||
finally:
|
||||
browser.close()
|
||||
except PlaywrightError as exc:
|
||||
if "Executable doesn't exist" in str(exc) or "playwright install" in str(exc).lower():
|
||||
pytest.skip(str(exc))
|
||||
raise
|
||||
|
|
|
|||
117
tests/test_workspace_patch_rules.py
Normal file
117
tests/test_workspace_patch_rules.py
Normal file
|
|
@ -0,0 +1,117 @@
|
|||
"""Who decides that a workspace change is junk.
|
||||
|
||||
A skill whose deliverable IS its build output (a bundled widget under
|
||||
``dist/``) had that file silently dropped from ``workspace.patch``, because the
|
||||
capture rules excluded generated-output directories by NAME — a rule inherited
|
||||
from a benchmark capture script that owns its own copy and never imported this
|
||||
module. The project's own ``.gitignore`` is the authority instead, and the
|
||||
remaining host rules stay: the per-file size cap, git's binary verdict, and the
|
||||
credential-shaped name check.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pathlib
|
||||
import subprocess
|
||||
|
||||
|
||||
def _genesis_repo(root: pathlib.Path, gitignore: str = "") -> pathlib.Path:
|
||||
"""A freshly created project: one commit, optionally a .gitignore."""
|
||||
root.mkdir()
|
||||
subprocess.run(["git", "init"], cwd=root, check=True, capture_output=True)
|
||||
(root / "README.md").write_text("base\n", encoding="utf-8")
|
||||
tracked = ["README.md"]
|
||||
if gitignore:
|
||||
(root / ".gitignore").write_text(gitignore, encoding="utf-8")
|
||||
tracked.append(".gitignore")
|
||||
subprocess.run(["git", "add", *tracked], cwd=root, check=True, capture_output=True)
|
||||
subprocess.run(
|
||||
["git", "-c", "user.email=t@example.com", "-c", "user.name=T", "commit", "-m", "init"],
|
||||
cwd=root, check=True, capture_output=True,
|
||||
)
|
||||
return root
|
||||
|
||||
|
||||
def _capture(repo: pathlib.Path, out: pathlib.Path):
|
||||
from ouroboros.headless import write_workspace_patch_artifacts
|
||||
|
||||
_artifacts, manifest = write_workspace_patch_artifacts(repo, out, task={})
|
||||
return (out / "workspace.patch").read_text(encoding="utf-8"), manifest
|
||||
|
||||
|
||||
def test_a_project_without_a_gitignore_keeps_its_build_output_in_the_patch(tmp_path):
|
||||
repo = _genesis_repo(tmp_path / "repo")
|
||||
(repo / "dist").mkdir()
|
||||
(repo / "dist" / "widget.js").write_text("export const widget = 1;\n", encoding="utf-8")
|
||||
|
||||
patch, manifest = _capture(repo, tmp_path / "artifacts")
|
||||
|
||||
assert "dist/widget.js" in patch
|
||||
assert "dist/widget.js" in manifest["untracked_included"]
|
||||
assert manifest["exclude_rules_version"] == 3
|
||||
assert not [item["path"] for item in manifest["untracked_excluded"]]
|
||||
|
||||
|
||||
def test_a_project_that_declares_dist_ignored_keeps_it_out_of_the_patch(tmp_path):
|
||||
repo = _genesis_repo(tmp_path / "repo", gitignore="dist/\n")
|
||||
(repo / "dist").mkdir()
|
||||
(repo / "dist" / "widget.js").write_text("export const widget = 1;\n", encoding="utf-8")
|
||||
(repo / "src.js").write_text("export const src = 1;\n", encoding="utf-8")
|
||||
|
||||
patch, manifest = _capture(repo, tmp_path / "artifacts")
|
||||
|
||||
assert "dist/widget.js" not in patch
|
||||
assert "src.js" in patch
|
||||
# Git-ignored files are outside the capture universe: they are not carried
|
||||
# and not reported as exclusions either.
|
||||
assert "dist/widget.js" not in [item["path"] for item in manifest["untracked_excluded"]]
|
||||
assert manifest["counts"]["untracked_excluded"] == 0
|
||||
|
||||
|
||||
def test_the_remaining_host_vetoes_still_apply_to_generated_output(tmp_path, monkeypatch):
|
||||
import ouroboros.headless as headless
|
||||
|
||||
repo = _genesis_repo(tmp_path / "repo")
|
||||
(repo / "dist").mkdir()
|
||||
(repo / "dist" / "widget.js").write_text("export const widget = 1;\n", encoding="utf-8")
|
||||
(repo / "dist" / "app.bin").write_bytes(b"\x7fELF\x00\x01\x02\x03binary\x00blob")
|
||||
(repo / "dist" / "build.log").write_text("noise\n", encoding="utf-8")
|
||||
(repo / "dist" / "big.js").write_text("x" * 200, encoding="utf-8")
|
||||
monkeypatch.setattr(headless, "_PATCH_MAX_UNTRACKED_FILE_BYTES", 100)
|
||||
|
||||
patch, manifest = _capture(repo, tmp_path / "artifacts")
|
||||
|
||||
excluded = {item["path"]: item["reason"] for item in manifest["untracked_excluded"]}
|
||||
assert "dist/widget.js" in patch
|
||||
assert "binary file" in excluded.get("dist/app.bin", "")
|
||||
assert "junk artifact" in excluded.get("dist/build.log", "")
|
||||
assert "size cap" in excluded.get("dist/big.js", "")
|
||||
|
||||
|
||||
def test_the_snapshot_veto_agrees_with_the_patch_by_subtraction(tmp_path):
|
||||
"""The delegated-run baseline snapshot asks the same predicate, so removing
|
||||
the name rule keeps the snapshot and the patch from drifting apart."""
|
||||
from ouroboros.headless import untracked_capture_veto_reason
|
||||
|
||||
repo = _genesis_repo(tmp_path / "repo")
|
||||
(repo / "dist").mkdir()
|
||||
(repo / "dist" / "widget.js").write_text("export const widget = 1;\n", encoding="utf-8")
|
||||
(repo / "dist" / "run.log").write_text("noise\n", encoding="utf-8")
|
||||
|
||||
assert untracked_capture_veto_reason(repo, "dist/widget.js") == ""
|
||||
assert untracked_capture_veto_reason(repo, "build/widget.js") == ""
|
||||
assert "junk artifact" in untracked_capture_veto_reason(repo, "dist/run.log")
|
||||
|
||||
|
||||
def test_the_benchmark_capture_script_owns_its_own_rule(tmp_path):
|
||||
"""The comment that justified the host rule claimed consistency with the
|
||||
benchmark script. That script never imported this module and uses an
|
||||
any-depth rule of its own, so the two were never one rule."""
|
||||
from ouroboros import workspace_patch_rules
|
||||
|
||||
script = pathlib.Path(__file__).resolve().parents[1] / "devtools" / "benchmarks" / "swe_bench_pro" / "capture_patch.sh"
|
||||
if script.is_file():
|
||||
text = script.read_text(encoding="utf-8")
|
||||
assert "dist/" in text
|
||||
assert "workspace_patch_rules" not in text
|
||||
assert "dist" not in workspace_patch_rules._PATCH_JUNK_RE.pattern
|
||||
13
web/app.js
13
web/app.js
|
|
@ -77,6 +77,17 @@ function setMobileDrawerOpen(open, { sync = true } = {}) {
|
|||
if (sync) syncNavigationState();
|
||||
}
|
||||
|
||||
// The application's one client-side route: a `#<page>` fragment, honoured once
|
||||
// on load and validated against the injected sections rather than against a
|
||||
// duplicated page list — the DOM is the single source of truth, so an unknown
|
||||
// fragment is ignored instead of painting a blank surface. Deliberately NOT
|
||||
// written back on navigation: the desktop shell and the Telegram mini app have
|
||||
// no address bar to read it from.
|
||||
function pageFromHash() {
|
||||
const name = String(window.location.hash || '').replace(/^#/, '').trim();
|
||||
return name && document.getElementById(`page-${name}`) ? name : '';
|
||||
}
|
||||
|
||||
async function showPage(name, options = {}) {
|
||||
const pageName = String(name || '').trim();
|
||||
if (!pageName) return false;
|
||||
|
|
@ -713,6 +724,8 @@ initOnboardingOverlay();
|
|||
initMatrixRain();
|
||||
loadVersion();
|
||||
syncNavigationState();
|
||||
const hashPage = pageFromHash();
|
||||
if (hashPage && hashPage !== state.activePage) showPage(hashPage);
|
||||
|
||||
// Mobile soft-keyboard handling: viewport shrink counts only while an editable
|
||||
// owns focus. Drawer opening clears that state explicitly before navigation is
|
||||
|
|
|
|||
|
|
@ -18,7 +18,7 @@
|
|||
<body>
|
||||
<div id="app">
|
||||
<div id="nav-drawer-backdrop" class="nav-drawer-backdrop" hidden></div>
|
||||
<aside id="primary-sidebar" class="primary-sidebar" aria-label="Ouroboros navigation">
|
||||
<nav id="primary-sidebar" class="primary-sidebar" aria-label="Ouroboros navigation">
|
||||
<div class="sidebar-scroll">
|
||||
<button class="nav-row nav-row-main active" data-nav-page="chat" title="Main Ouroboros chat">
|
||||
<svg width="20" height="20" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round"><path d="M22 17a2 2 0 0 1-2 2H6.828a2 2 0 0 0-1.414.586l-2.202 2.202A.71.71 0 0 1 2 21.286V5a2 2 0 0 1 2-2h16a2 2 0 0 1 2 2z"/><path d="M7 11h10"/><path d="M7 15h6"/><path d="M7 7h8"/></svg>
|
||||
|
|
@ -65,7 +65,7 @@
|
|||
</div>
|
||||
</div>
|
||||
<span id="nav-version"></span>
|
||||
</aside>
|
||||
</nav>
|
||||
<div id="divider"></div>
|
||||
<main id="content">
|
||||
<!-- Pages injected by JS -->
|
||||
|
|
|
|||
|
|
@ -562,6 +562,12 @@
|
|||
* @property {boolean=} project_mirror
|
||||
*/
|
||||
|
||||
/** Additive /api/chat/history terminality projection.
|
||||
* @typedef {Object} TaskOutcomeHistoryFields
|
||||
* @property {"working"|"done"|"warn"|"error"|"cancelled"=} outcome_phase // canonical display phase; "working" is not terminal
|
||||
* @property {boolean=} outcome_final // true only after the canonical task outcome settles; false marks a pre-finalization narrative
|
||||
*/
|
||||
|
||||
/**
|
||||
* Additive /api/chat/history row fields on `system_type: "skill_review"` rows:
|
||||
* the exact-job reference the producer already writes into chat.jsonl. A row
|
||||
|
|
|
|||
|
|
@ -2114,11 +2114,11 @@ export function createChatInstance({
|
|||
return finishLiveCard(taskId, 'done');
|
||||
}
|
||||
let changed = false;
|
||||
// Restore the coined task name even when history retained no progress row.
|
||||
// Restore task name from history.
|
||||
if (msg?.suggested_name) {
|
||||
changed = applySuggestedName(taskId, msg.suggested_name) || changed;
|
||||
}
|
||||
const finalizing = msg?.task_phase === 'finalizing';
|
||||
const finalizing = msg?.task_phase === 'finalizing' || msg?.outcome_final === false;
|
||||
const projectedReviews = reviewGroupsFromTaskDetail(msg, taskId);
|
||||
const hasAcceptanceReview = projectedReviews.length > 0;
|
||||
const taskState = getTaskUiState(taskId, hasAcceptanceReview || finalizing);
|
||||
|
|
|
|||
|
|
@ -360,7 +360,20 @@ export function taskReasonPhrase(code) {
|
|||
|
||||
export function taskReasonDetail(evt) {
|
||||
// An owner-requested stop is a success and carries its own marker instead.
|
||||
if (taskStoppedWithSummary(evt) || !evt?.reason_code) return '';
|
||||
if (taskStoppedWithSummary(evt)) return '';
|
||||
// A warning caused by REVIEW must not be explained by the execution reason
|
||||
// that happens to sit beside it: the host's acceptance decision is the
|
||||
// cause, and it speaks in its own stored words. A hard failure or a
|
||||
// cancellation keeps explaining itself by its execution reason.
|
||||
const record = normalizeTaskTerminalRecord(evt);
|
||||
const decision = record.outcome_axes?.review?.acceptance_decision
|
||||
?? record.review_status?.acceptance_decision;
|
||||
const severity = taskOutcomeSeverity(evt);
|
||||
if (severity !== 'error' && severity !== 'cancelled' && decision?.status && decision.status !== 'accepted') {
|
||||
const rationale = String(decision.rationale || '').split(/\s+/).filter(Boolean).join(' ');
|
||||
return `Acceptance: ${decision.status}${rationale ? ` — ${rationale}` : ''}`;
|
||||
}
|
||||
if (!evt?.reason_code) return '';
|
||||
return `Reason: ${taskReasonPhrase(evt.reason_code)}`;
|
||||
}
|
||||
|
||||
|
|
@ -487,6 +500,8 @@ function taskOutcomeMeta(evt) {
|
|||
axes.lifecycle?.status ? `lifecycle ${axes.lifecycle.status}` : '',
|
||||
axes.execution?.status ? `execution ${axes.execution.status}` : '',
|
||||
axes.objective?.status ? `objective ${axes.objective.status}` : '',
|
||||
axes.review?.status ? `review ${axes.review.status}` : '',
|
||||
axes.review?.acceptance_decision?.status ? `acceptance ${axes.review.acceptance_decision.status}` : '',
|
||||
].filter(Boolean);
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -422,3 +422,33 @@ test('chat bubble heading clamp is scoped in style.css', () => {
|
|||
const richSource = readFileSync(new URL('../modules/chat_markdown.js', import.meta.url), 'utf8');
|
||||
assert.match(richSource, /querySelectorAll\('h1, h2, h3, h4, h5, h6'\)[\s\S]{0,160}Math\.min\(Number\(heading\.tagName\.slice\(1\)\), 3\)/);
|
||||
});
|
||||
|
||||
test('a host notice concludes a replayed card and keeps its markdown', async () => {
|
||||
// A terminal text the host wrote alone is projected as role=system with NO
|
||||
// system_type — exactly the shape replay needs to treat it as the task's
|
||||
// last word — and, unlike the salvage receipt, it keeps its markdown so the
|
||||
// host's own code spans do not render escaped.
|
||||
assert.match(chatSource, /const plainUntypedFinal = !msg\.system_type && !msg\.msg_type;/);
|
||||
assert.match(
|
||||
chatSource,
|
||||
/\(msg\.role === 'assistant' \|\| msg\.role === 'system'\)\n\s+&& \(positiveTaskTerminalFact\(msg\) \|\| plainUntypedFinal\)/,
|
||||
);
|
||||
|
||||
const { prior, mount } = installDom();
|
||||
let instance;
|
||||
try {
|
||||
const made = makeInstance(mount);
|
||||
instance = made.instance;
|
||||
made.handlers.get('chat')({
|
||||
chat_id: 2, role: 'system', markdown: true, task_id: 'notice-task',
|
||||
content: 'Task rejected line one\nline two',
|
||||
ts: '2026-09-03T00:00:01Z',
|
||||
});
|
||||
const bubble = findBubble('system');
|
||||
assert.match(bubble.innerHTML, /Task rejected line one<br>line two/);
|
||||
assert.equal(bubble.getAttribute('data-chat-markdown-enhanced'), 'true');
|
||||
} finally {
|
||||
instance?.destroy();
|
||||
restoreDom(prior);
|
||||
}
|
||||
});
|
||||
|
|
|
|||
146
web/tests/fixtures/outcome_phase_parity.json
vendored
Normal file
146
web/tests/fixtures/outcome_phase_parity.json
vendored
Normal file
|
|
@ -0,0 +1,146 @@
|
|||
{
|
||||
"purpose": "One status-word family across the browser and the host. Every case is a terminal-or-not task record read by log_events.js (taskDoneIsTerminal + taskTerminalPhase + taskPresentation) and by ouroboros.project_dialogue (outcome_phase + OUTCOME_PHASE_HEADLINE + _completion_verdict). A new axis, reason or acceptance status is added to both sides in the same commit, with a row here.",
|
||||
"cases": [
|
||||
{
|
||||
"name": "clean completed root",
|
||||
"record": {"status": "completed", "outcome_axes": {"execution": {"status": "ok"}}},
|
||||
"phase": "done",
|
||||
"headline": "Done",
|
||||
"acceptance_clause": ""
|
||||
},
|
||||
{
|
||||
"name": "completed while post-task synthesis is still open reads Working on both sides",
|
||||
"record": {
|
||||
"status": "completed",
|
||||
"outcome_axes": {"execution": {"status": "ok"}},
|
||||
"root_phase_checkpoint": {"post_task_synthesis": "running"}
|
||||
},
|
||||
"phase": "working",
|
||||
"headline": "Working",
|
||||
"acceptance_clause": ""
|
||||
},
|
||||
{
|
||||
"name": "legacy task_done frame carrying status done stays terminal",
|
||||
"record": {"status": "done", "outcome_axes": {"execution": {"status": "ok"}}},
|
||||
"phase": "done",
|
||||
"headline": "Done",
|
||||
"acceptance_clause": ""
|
||||
},
|
||||
{
|
||||
"name": "cancelled outranks failure-shaped teardown side facts",
|
||||
"record": {
|
||||
"status": "cancelled",
|
||||
"outcome_axes": {"lifecycle": {"status": "cancelled"}, "artifacts": {"status": "missing"}}
|
||||
},
|
||||
"phase": "cancelled",
|
||||
"headline": "Cancelled",
|
||||
"acceptance_clause": ""
|
||||
},
|
||||
{
|
||||
"name": "legacy cancel_requested event spelling",
|
||||
"record": {"status": "cancel_requested"},
|
||||
"phase": "cancelled",
|
||||
"headline": "Cancelled",
|
||||
"acceptance_clause": ""
|
||||
},
|
||||
{
|
||||
"name": "failed review is a failure",
|
||||
"record": {
|
||||
"status": "completed",
|
||||
"outcome_axes": {"execution": {"status": "ok"}, "review": {"status": "fail"}}
|
||||
},
|
||||
"phase": "error",
|
||||
"headline": "Failed",
|
||||
"acceptance_clause": ""
|
||||
},
|
||||
{
|
||||
"name": "owner-requested finalization is a success, not a warning",
|
||||
"record": {
|
||||
"status": "completed",
|
||||
"reason_code": "owner_requested_finalization",
|
||||
"outcome_axes": {"execution": {"status": "best_effort"}}
|
||||
},
|
||||
"phase": "done",
|
||||
"headline": "Done",
|
||||
"acceptance_clause": ""
|
||||
},
|
||||
{
|
||||
"name": "scheduler duplicate rejection is a warning",
|
||||
"record": {"status": "rejected_duplicate", "outcome_axes": {"lifecycle": {"status": "rejected_duplicate"}}},
|
||||
"phase": "warn",
|
||||
"headline": "Done with warnings",
|
||||
"acceptance_clause": ""
|
||||
},
|
||||
{
|
||||
"name": "degraded review with an unaccepted acceptance decision",
|
||||
"record": {
|
||||
"status": "completed",
|
||||
"reason_code": "final_message",
|
||||
"outcome_axes": {
|
||||
"execution": {"status": "ok"},
|
||||
"review": {
|
||||
"status": "degraded",
|
||||
"acceptance_decision": {
|
||||
"status": "finalized_unaccepted",
|
||||
"rationale": "Acceptance reviewers did not reach a valid quorum."
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"phase": "warn",
|
||||
"headline": "Done with warnings",
|
||||
"acceptance_clause": "Acceptance: finalized_unaccepted — Acceptance reviewers did not reach a valid quorum."
|
||||
},
|
||||
{
|
||||
"name": "an accepted decision leaves the execution reason standing",
|
||||
"record": {
|
||||
"status": "completed",
|
||||
"reason_code": "final_message",
|
||||
"outcome_axes": {
|
||||
"execution": {"status": "ok"},
|
||||
"review": {"status": "pass", "acceptance_decision": {"status": "accepted", "rationale": "Quorum reached."}}
|
||||
}
|
||||
},
|
||||
"phase": "done",
|
||||
"headline": "Done",
|
||||
"acceptance_clause": ""
|
||||
},
|
||||
{
|
||||
"name": "a revision request without a rationale states the status alone",
|
||||
"record": {
|
||||
"status": "completed",
|
||||
"outcome_axes": {
|
||||
"execution": {"status": "degraded"},
|
||||
"review": {"status": "degraded", "acceptance_decision": {"status": "revision_requested"}}
|
||||
}
|
||||
},
|
||||
"phase": "warn",
|
||||
"headline": "Done with warnings",
|
||||
"acceptance_clause": "Acceptance: revision_requested."
|
||||
},
|
||||
{
|
||||
"name": "a hard failure explains itself by its execution reason, not by the decision",
|
||||
"record": {
|
||||
"status": "failed",
|
||||
"reason_code": "delegated_custody_unreconciled",
|
||||
"outcome_axes": {
|
||||
"execution": {"status": "failed"},
|
||||
"review": {"status": "degraded", "acceptance_decision": {"status": "finalized_unaccepted", "rationale": "No quorum."}}
|
||||
}
|
||||
},
|
||||
"phase": "error",
|
||||
"headline": "Failed",
|
||||
"acceptance_clause": ""
|
||||
},
|
||||
{
|
||||
"name": "an objective warning is a warning even when every status word is clean",
|
||||
"record": {
|
||||
"status": "completed",
|
||||
"outcome_axes": {"execution": {"status": "ok"}, "objective": {"status": "pass", "warning": "partial coverage"}}
|
||||
},
|
||||
"phase": "warn",
|
||||
"headline": "Done with warnings",
|
||||
"acceptance_clause": ""
|
||||
}
|
||||
]
|
||||
}
|
||||
|
|
@ -2,7 +2,9 @@ import assert from 'node:assert/strict';
|
|||
import test from 'node:test';
|
||||
import { readFileSync } from 'node:fs';
|
||||
|
||||
import { taskReasonDetail, taskReasonPhrase } from '../modules/log_events.js';
|
||||
import {
|
||||
taskDoneIsTerminal, taskPresentation, taskReasonDetail, taskReasonPhrase, taskTerminalPhase,
|
||||
} from '../modules/log_events.js';
|
||||
|
||||
// A degraded delivery used to name one generic cause on every card. The record
|
||||
// keeps the machine code; the card says what actually happened.
|
||||
|
|
@ -46,3 +48,90 @@ test('every typed cause the loop can record has a sentence', () => {
|
|||
);
|
||||
}
|
||||
});
|
||||
|
||||
test('one status-word family: the card phase matches the host over the shared fixture', () => {
|
||||
// The same fixture is read by tests/test_project_plain_rows.py, so a
|
||||
// divergence between this severity fold and the host's durable label word
|
||||
// fails on both sides of the boundary.
|
||||
const fixture = JSON.parse(
|
||||
readFileSync(new URL('./fixtures/outcome_phase_parity.json', import.meta.url), 'utf8'),
|
||||
);
|
||||
assert.ok(fixture.cases.length >= 10);
|
||||
for (const { name, record, phase, headline, acceptance_clause: clause } of fixture.cases) {
|
||||
const resolved = taskDoneIsTerminal(record) ? taskTerminalPhase(record) : 'working';
|
||||
assert.deepEqual(taskPresentation(resolved), { phase, headline }, name);
|
||||
if (clause) {
|
||||
// The host composes the same sentence for its durable prose rows and
|
||||
// terminates it there; the card line adds no punctuation of its own.
|
||||
const detail = taskReasonDetail(record);
|
||||
assert.ok(clause === detail || clause === `${detail}.`, `${name}: ${detail} vs ${clause}`);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
// A review-caused warning used to be explained by whatever execution reason sat
|
||||
// beside it ('Reason: final_message'), which named the delivery step rather than
|
||||
// the actual cause. The host's acceptance decision now speaks for itself.
|
||||
|
||||
const A4 = {
|
||||
status: 'completed',
|
||||
reason_code: 'final_message',
|
||||
outcome_axes: {
|
||||
execution: { status: 'ok' },
|
||||
review: {
|
||||
status: 'degraded',
|
||||
acceptance_decision: {
|
||||
status: 'finalized_unaccepted',
|
||||
rationale: 'Acceptance reviewers did not reach a valid quorum.',
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
test('an unaccepted decision explains the warning in its own words', () => {
|
||||
assert.equal(
|
||||
taskReasonDetail(A4),
|
||||
'Acceptance: finalized_unaccepted — Acceptance reviewers did not reach a valid quorum.',
|
||||
);
|
||||
assert.doesNotMatch(taskReasonDetail(A4), /final_message/);
|
||||
});
|
||||
|
||||
test('an accepted decision leaves the execution reason line byte-identical', () => {
|
||||
const accepted = {
|
||||
...A4,
|
||||
outcome_axes: {
|
||||
execution: { status: 'ok' },
|
||||
review: { status: 'pass', acceptance_decision: { status: 'accepted', rationale: 'Quorum reached.' } },
|
||||
},
|
||||
};
|
||||
assert.equal(taskReasonDetail(accepted), 'Reason: final_message');
|
||||
});
|
||||
|
||||
test('a decision without a rationale states its status alone', () => {
|
||||
const record = {
|
||||
outcome_axes: { review: { acceptance_decision: { status: 'revision_requested' } } },
|
||||
status: 'completed',
|
||||
};
|
||||
assert.equal(taskReasonDetail(record), 'Acceptance: revision_requested');
|
||||
});
|
||||
|
||||
test('a decision with no reason code still reaches the acceptance branch', () => {
|
||||
// The old single early return swallowed this frame before the branch.
|
||||
const record = { status: 'completed', review_status: { acceptance_decision: { status: 'revision_requested' } } };
|
||||
assert.equal(taskReasonDetail(record), 'Acceptance: revision_requested');
|
||||
});
|
||||
|
||||
test('a hard failure keeps explaining itself by its execution reason', () => {
|
||||
const failed = { ...A4, status: 'failed', reason_code: 'delegated_custody_unreconciled' };
|
||||
assert.equal(taskReasonDetail(failed), 'Reason: delegated_custody_unreconciled');
|
||||
});
|
||||
|
||||
test('a multi-line rationale is flattened into one sentence', () => {
|
||||
const noisy = {
|
||||
status: 'completed',
|
||||
outcome_axes: {
|
||||
review: { acceptance_decision: { status: 'revision_requested', rationale: 'Two\n\nlines here.' } },
|
||||
},
|
||||
};
|
||||
assert.equal(taskReasonDetail(noisy), 'Acceptance: revision_requested — Two lines here.');
|
||||
});
|
||||
|
|
|
|||
|
|
@ -251,7 +251,7 @@ test('history replay keeps open summaries live and terminal fallbacks factual',
|
|||
chatSource.indexOf('function appendTaskSummaryToLiveCard'),
|
||||
chatSource.indexOf('// child task_id'),
|
||||
);
|
||||
assert.match(summary, /const finalizing = msg\?\.task_phase === 'finalizing';/);
|
||||
assert.match(summary, /const finalizing = msg\?\.task_phase === 'finalizing' \|\| msg\?\.outcome_final === false;/);
|
||||
assert.match(summary, /terminal: !finalizing/);
|
||||
assert.match(summary, /record\.finalizingHold = true/);
|
||||
assert.match(summary, /if \(finalizing\) return changed;\s*changed = finishLiveCard/);
|
||||
|
|
@ -371,3 +371,28 @@ test('header status has no terminal-attention state or writer', () => {
|
|||
assert.doesNotMatch(chatSource, /lastTerminalAttention/);
|
||||
assert.doesNotMatch(activitySource, /lastTerminalAttention|text: 'Attention'/);
|
||||
});
|
||||
|
||||
test('a review-caused warning names the acceptance decision on the card and in Logs', () => {
|
||||
// The execution reason beside it ('final_message') names the delivery step,
|
||||
// not the cause; the card body and the Logs meta now say what happened.
|
||||
const evt = {
|
||||
type: 'task_done', status: 'completed', reason_code: 'final_message',
|
||||
outcome_axes: {
|
||||
execution: { status: 'ok' },
|
||||
review: {
|
||||
status: 'degraded',
|
||||
acceptance_decision: {
|
||||
status: 'finalized_unaccepted',
|
||||
rationale: 'Acceptance reviewers did not reach a valid quorum.',
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
const live = summarizeChatLiveEvent(evt);
|
||||
const replay = summarizeLogEvent(evt);
|
||||
assert.deepEqual({ phase: live.phase, headline: live.headline }, { phase: 'warn', headline: 'Done with warnings' });
|
||||
assert.match(live.body, /Acceptance: finalized_unaccepted — Acceptance reviewers did not reach a valid quorum\./);
|
||||
assert.doesNotMatch(live.body, /final_message/);
|
||||
assert.ok(replay.meta.includes('review degraded'));
|
||||
assert.ok(replay.meta.includes('acceptance finalized_unaccepted'));
|
||||
});
|
||||
|
|
|
|||
|
|
@ -234,3 +234,9 @@ test('live structured delivery frames keep additive grouping and size fields', (
|
|||
assert.match(chat, /enhanceMarkdown: enhanceMountedMarkdown/);
|
||||
assert.match(contracts, /WS_MESSAGE_TYPES[\s\S]*?"links"/);
|
||||
});
|
||||
|
||||
test('history typedef declares task outcome terminality fields', () => {
|
||||
const types = moduleFile('api_types.js');
|
||||
assert.match(types, /@property \{"working"\|"done"\|"warn"\|"error"\|"cancelled"=\} outcome_phase/);
|
||||
assert.match(types, /@property \{boolean=\} outcome_final/);
|
||||
});
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue