From 7b7e4e6fa189f3a0209726949b343a97f6dc26a1 Mon Sep 17 00:00:00 2001 From: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> Date: Wed, 23 Sep 2026 01:55:08 +0300 Subject: [PATCH 01/15] fix: aggregate read-only release diagnostics on the selected Git source --- docs/architecture/06-agent-core.md | 6 +- .../05-review-and-commit-protocol.md | 4 +- ouroboros/commit_admission.py | 210 ++++++----- ouroboros/tools/claude_advisory_review.py | 34 +- ouroboros/tools/commit_gate.py | 6 +- ouroboros/tools/release_sync.py | 36 ++ ouroboros/tools/review.py | 107 +----- tests/test_advisory_inline_freshness.py | 2 +- tests/test_advisory_preflight.py | 4 + tests/test_commit_review_task_evidence.py | 2 +- tests/test_git_review_preflight_gate.py | 28 +- tests/test_reference_book_budgets.py | 4 +- tests/test_release_metadata_diagnostics.py | 353 ++++++++++++++++++ 13 files changed, 566 insertions(+), 230 deletions(-) create mode 100644 tests/test_release_metadata_diagnostics.py diff --git a/docs/architecture/06-agent-core.md b/docs/architecture/06-agent-core.md index 2c6f776fc..31cd56bcb 100644 --- a/docs/architecture/06-agent-core.md +++ b/docs/architecture/06-agent-core.md @@ -343,7 +343,7 @@ Disclosed delegated-isolation residuals (deliberate): a live top-level task with `tools/git.py` owns repository writes, staging, reviewed commit, rollback or restore, tags, push and CI follow-up; its leaves are `git_plumbing.py`, `git_repo_edit.py`, `git_vcs_ops.py`, `git_review_cycle.py` (staging plus the advisory/triad/scope review and the reviewed-material binding the commit gate consumes) and `git_evolution.py` (campaign authority at the reviewed-commit and publication boundaries). File-edit tools validate their own atomic write shape. `mutation_attribution.py` captures the root-task baseline and projects the clean-at-baseline system-repository delta plus an explicitly adopted predecessor's exact retained changes — a changed pre-existing dirty path, a stale or missing baseline, or a failed scan blocks automatic staging; `commit_reviewed(paths=None)` stages only that attributed candidate, explicit paths must be a subset, an empty candidate returns `GIT_NO_ATTRIBUTED_CHANGES`, and managed update transactions keep their separate typed whole-tree authority. At startup a later independently initiated task may adopt exact retained candidates from its host-validated predecessor source in the same repository. `predecessor_adoption` preserves the observed dirty baseline and original source; terminal quiescence, unchanged path content and unchanged per-path base are required, while unrelated dirty work stays excluded. Missing fingerprints or size-only observations establish no transfer, and adoption grants no review approval or automatic new task. Commit preparation verifies the exact local working-branch ref before an unambiguous checkout: a missing ref refuses without changing the current branch, index or files (remote guessing and implicit branch creation are disabled), detached work retains the `checkout -B HEAD` recovery, and managed assisted merges retain transaction-owned precommit verification. -A reviewed commit is bound to one staged fingerprint. A cheap LLM-first advisory pass may run before the expensive gates; it is advisory, and skipping it never skips independently applicable tests, triad, applicable scope review, aggregation or exact-SHA binding. The hermetic preflight runs the candidate in a disposable worktree and data root; triad and scope inspect the same staged snapshot, aggregation preserves actor evidence and obligations, and any mutation stales the binding. Managed exception: a managed-update resolution commit reviews the declared M0→S subject (`tools/review_subject.py`) and the commit gate binds S to the exact index write-tree the fingerprint pins. External review wrappers report readiness but do not grant commit authority. The exact binding includes the `git write-tree` SHA, ordered `HEAD` and `MERGE_HEAD` parents, indexed VERSION, expected `v{VERSION}` tag, any existing tag target, and the binary staged-diff hash; after commit, tree, parents, VERSION and tag target are re-read before success or push is recorded, and an existing release tag is never silently accepted or retargeted. `release_sync.sync_release_metadata` (version carriers) runs in commit-admission preflight; `VERSION_CARRIER_SPANS`/`substitute_carrier_spans` is the ONE span primitive the managed-update resolver and the commit-triad pack cut share, and `carrier_only_change` names a carrier changed only inside its declared version spans. Durable review state keeps attempts, obligations, readiness debt, raw actor evidence, and the final commit or tag binding; raw advisory output lives in `state/advisory_review.json` (selected by `snapshot_hash`/`ts`), not in `review_status(include_raw=true)`, which exposes raw triad and scope attempt evidence; `commit_gate.py` classifies review blocks, refuses an identical verdict, counts paid cycles against the ceiling and fingerprints the review contract. BIBLE supplies review authority, CHECKLISTS supplies criteria, and Development supplies the procedure; snapshot identity, advisory coverage or audited-skip evidence, deterministic results, actor evidence, and final Git identity must all describe the same material. +A reviewed commit is bound to one staged fingerprint. A cheap LLM-first advisory pass may run before the expensive gates; it is advisory, and skipping it never skips independently applicable tests, triad, applicable scope review, aggregation or exact-SHA binding. The hermetic preflight runs the candidate in a disposable worktree and data root; triad and scope inspect the same staged snapshot, aggregation preserves actor evidence and obligations, and any mutation stales the binding. Managed exception: a managed-update resolution commit reviews the declared M0→S subject (`tools/review_subject.py`) and the commit gate binds S to the exact index write-tree the fingerprint pins. External review wrappers report readiness but do not grant commit authority. The exact binding includes the `git write-tree` SHA, ordered `HEAD` and `MERGE_HEAD` parents, indexed VERSION, expected `v{VERSION}` tag, any existing tag target, and the binary staged-diff hash; after commit, tree, parents, VERSION and tag target are re-read before success or push is recorded, and an existing release tag is never silently accepted or retargeted. `release_sync.sync_release_metadata` projects version carriers during ordinary commit preparation; `VERSION_CARRIER_SPANS`/`substitute_carrier_spans` is the ONE span primitive the managed-update resolver and the commit-triad pack cut share, and `carrier_only_change` names a carrier changed only inside its declared version spans. Durable review state keeps attempts, obligations, readiness debt, raw actor evidence, and the final commit or tag binding; raw advisory output lives in `state/advisory_review.json` (selected by `snapshot_hash`/`ts`), not in `review_status(include_raw=true)`, which exposes raw triad and scope attempt evidence; `commit_gate.py` classifies review blocks, refuses an identical verdict, counts paid cycles against the ceiling and fingerprints the review contract. BIBLE supplies review authority, CHECKLISTS supplies criteria, and Development supplies the procedure; snapshot identity, advisory coverage or audited-skip evidence, deterministic results, actor evidence, and final Git identity must all describe the same material. Material ordinary Advisory commit review returns the complete outcome and exact `review_reference` before commit, tag or push. The author may correct, accept unchanged bytes, request another permitted review or stop. Explicit `commit_reviewed`/`vcs_commit_reviewed` continuation with that reference and `author_disposition` binds the current attributed candidate, reruns independent required preflight/tests and exact Git checks, and dispatches no critic. The author record references the original attempt instead of rewriting its subject or paid facts. Settled handback is `reviewed/review_only`; actual pending custody remains `reviewing/late_wait` and collectible. Clean supported review keeps its one-call path; Blocking gets no author override. Evolution receipts record actual `triad_scope_status`, never infer PASS from a successful Git commit. @@ -357,6 +357,10 @@ The hermetic runner (`preflight_runner.py`) alone mints `ctx._preflight_test_pro #### Commit advisory cycle +`preflight_review(deterministic_only=True, source="worktree" | "index")` returns a release-only diagnostic report before sync, staging, custody/state access, provider checks, tests or fingerprints. `commit_admission.release_metadata_diagnostics` owns source acquisition and reports all independent applicable `findings` plus `unavailable` sources; `release_sync.release_metadata_findings` reuses the carrier validators, release grammar and P9 history counters. The report identifies its source and status (`clean`, `blocked`, `unavailable` or `not_applicable`), gives no review freshness, and writes no review history. It checks the actual selected VERSION, never an imagined next release. VERSION/README absence and failed reads are unavailable evidence; an absent optional older carrier stays optional, while a readable malformed carrier is a finding. Coverage is release metadata only, not syntax, structural size, tests or critic review; size policy stays warning-only locally. + +Ordinary standalone advisory retains automatic carrier sync and reads worktree release metadata. Prepared advisory and the commit gate read index blobs, including partially staged files; the version-neutral staged lane and standalone documentation-only carve keep their existing applicability. The Optional[str] preflight wrappers format the same complete release findings and distinguish unavailable evidence from candidate defects. Author continuation uses the shared name-status formatter before these same staged checks. Changelog prose, history trimming and version allocation remain deliberate edits. + Advisory availability is evaluated from the current configured slot and route, never inferred from a stale stored verdict: a disabled advisory slot is an audited bypass, an `api_chat` row requires provider credentials for its RESOLVED model, an `agent_session` row a resolvable session route. `claude_advisory_review.preflight_review` (callable `advisory_review` alias) owns that admission policy: an `api_chat` row rides the native tool-round episode (§6 Review stack), an `agent_session` row the session review executor; a native episode that ends on its own transcript bound — keyed on the structured `native_transcript_cap_exceeded` code, never on message text — reaches the caller as the typed non-blocking `ADVISORY_SKIPPED` with reason `native_transcript_bound_exceeded`, carrying the bound, refused chars and paid rounds, and every episode exception keeps `failure_custody()` as the advisory meta's `usage`, never an empty `{}`. The retrieving advisory brief carries a touched-path manifest instead of duplicated file bodies and applies the shared span-only release-carrier cut over HEAD→working-tree, with the same `PACK EXCLUSION NOTE`; omitted bodies remain readable through its own tools. Governance comes from the shared tiers (§6 Governance delivery). If the commit advisory is unavailable, the commit gate runs its compensating hermetic preflight only when tests remain independently applicable (not explicitly skipped, diff not documentation-only). Readiness projections receive only an exact repo/hash-matched advisory record and keep its failed status and freshness. The reviewed-commit cycle (`git_review_cycle`) checks authorization and unresolved prior work before mechanical preparation and staging, binds the exact candidate before free-cycle/budget admission, runs a needed preflight inline with the full rebuttal and applicable tests, and revalidates the index and worktree snapshot before triad/scope; it owns entry-point resets and pending/blocked finalization, and its custody check joins current reviewer facts with the strict durable advisory record before index cleanup, an interruption before local metadata was updated included. `AdvisoryRunRecord.execution_pending` preserves physical custody for history retention and the external wrapper; `blocks_preflight` separately tracks logical admission. An explicit audited bypass releases that admission while retaining the original task, invocation, source and unknown physical outcome; late results update their original history row without superseding the newer bypass. Free advisory replay still checks freshness and buys no automatic preflight; stale coverage requires an explicit audited skip with applicable compensating tests preserved. diff --git a/docs/development/05-review-and-commit-protocol.md b/docs/development/05-review-and-commit-protocol.md index 09f6cc4d3..80546a2da 100644 --- a/docs/development/05-review-and-commit-protocol.md +++ b/docs/development/05-review-and-commit-protocol.md @@ -1,6 +1,6 @@ # Review & Commit Protocol -This chapter owns the three stages of a reviewed commit — prepared preflight, the authoritative gate, and publication binding — together with the shared paid-cycle cap, the free-replay rules, the external-review evidence contract and the release-sync rule a pull request must obey. It exists because technical failure and commit permission are separate facts, and every rule here keeps a missing review from becoming a PASS. +This chapter owns the three stages of a reviewed commit — prepared preflight, the authoritative gate, and publication binding — together with the shared paid-cycle cap, the free-replay rules, the external-review evidence contract and diagnostic release preflight/release-sync rules a pull request must obey. It exists because technical failure and commit permission are separate facts, and every rule here keeps a missing review from becoming a PASS. Keep optional task evidence outside the stable governance prefix and shrink its excerpt before reducing existing review material; a source pointer gives a packet-only model no retrieval capability; rejoin preserves the original hash and project-local view while any physical reviewer may still read it; removing an ignored view never deletes the canonical source. No separate notes corpus, blanket ToolResult metadata or mandatory whole-history read belongs to this evidence. @@ -32,7 +32,7 @@ The authoring agent freezes the final committed base-to-head range and gives it ### Release sync -A pull request into `ouroboros` leaves every version carrier byte-identical to its target (the carrier list and the one projection that writes them: ARCHITECTURE §10, invariant 2). At integration, `ouroboros/tools/release_sync.py::sync_release_metadata()` projects the chosen version and `version_carrier_desyncs()` verifies the file carriers (the history row is pinned by the packaging-sync test); changelog prose remains a deliberate maintainer edit. The installer filename templates, the immutable exact-tag download links and the stable promotion of `main` are ARCHITECTURE §8 "Build scripts". +A pull request into `ouroboros` leaves every version carrier byte-identical to its target (the carrier list and the one projection that writes them: ARCHITECTURE §10, invariant 2). Use `preflight_review(deterministic_only=True, source="worktree" | "index")` for release diagnostics before review spend; it grants no freshness. Keep source failures separate from candidate findings and verify partial staging plus unchanged files/index/state (`tests/test_release_metadata_diagnostics.py`; source/applicability contracts: ARCHITECTURE §6 "Commit advisory cycle"). At integration, `release_sync.sync_release_metadata()` projects the chosen version; its shared evaluator checks carriers, the current README row and P9 limits. Changelog prose and trimming remain deliberate edits; diagnostics add no autofix. Hermetic preflight uses a disposable worktree, temporary data/settings/pycache, and a scrubbed runtime/secret-class environment. Tests must rebind imported process-global roots and fail closed on the live data root; setting only `OUROBOROS_DATA_DIR` is insufficient. A reviewed local commit is the durability boundary; an `origin` push and CI are follow-up signals, not prerequisites for local self-modification survival. diff --git a/ouroboros/commit_admission.py b/ouroboros/commit_admission.py index 381f92773..8b494df74 100644 --- a/ouroboros/commit_admission.py +++ b/ouroboros/commit_admission.py @@ -20,7 +20,6 @@ import json import logging import os import pathlib -import re import subprocess from typing import List, NamedTuple, Optional @@ -31,20 +30,24 @@ log = logging.getLogger("ouroboros.commit_admission") def changed_worktree_paths( - repo_dir: pathlib.Path, paths: list[str] | None = None + repo_dir: pathlib.Path, paths: list[str] | None = None, *, strict: bool = False ) -> list[str]: - """Changed paths from ``git status --porcelain`` (empty on any git error).""" + """Changed worktree paths; admission uses strict errors instead of empty-on-error.""" from ouroboros.tools.review_helpers import parse_changed_paths_from_porcelain path_args = (["--"] + [str(p) for p in paths]) if paths else [] try: result = subprocess.run( - ["git", "status", "--porcelain"] + path_args, + ["git", "--no-optional-locks", "status", "--porcelain"] + path_args, cwd=str(repo_dir), capture_output=True, text=True, timeout=10, ) except Exception: + if strict: + raise return [] if result.returncode != 0: + if strict: + raise RuntimeError("git status failed") return [] return parse_changed_paths_from_porcelain(result.stdout) @@ -84,108 +87,109 @@ def auto_sync_release_metadata_if_needed( return [] +def read_release_file(repo_dir, path: str, *, source: str) -> str | None: + """Read exact worktree/index text; absent optional carriers differ from failed reads.""" + if source == "worktree": + try: + return (pathlib.Path(repo_dir) / path).read_text(encoding="utf-8") + except FileNotFoundError: + return None + # Establish absence independently: a failed git-show is never empty content. + present = subprocess.run( + ["git", "ls-files", "--error-unmatch", "--", path], cwd=str(repo_dir), + capture_output=True, timeout=10, + ) + if present.returncode == 1: + return None + present.check_returncode() + result = subprocess.run( + ["git", "show", f":{path}"], cwd=str(repo_dir), capture_output=True, + encoding="utf-8", timeout=10, check=True, + ) + return result.stdout + + +def release_metadata_diagnostics( + repo_dir, paths: list[str] | None = None, *, source: str = "worktree", read_text=None, +) -> dict: + """Read-only release report; an index reader may supply already-classified active paths. + + Source acquisition failures are unavailable evidence, independent of candidate + findings. Optional carriers absent from older trees remain optional; VERSION + and README are required when release checks apply. No review state is read. + """ + from ouroboros.tools.release_sync import CARRIER_SPAN_PATHS, release_metadata_findings + + report = {"source": source, "status": "clean", "findings": [], "unavailable": []} + findings, unavailable = report["findings"], report["unavailable"] + if source not in ("worktree", "index"): + report.update(status="unavailable", unavailable=["source must be worktree or index"]) + return report + touched = set(paths or []) if source == "worktree" or read_text else set() + try: + if source == "worktree": + touched.update(changed_worktree_paths(repo_dir, paths=paths, strict=True)) + elif read_text is None: + result = subprocess.run( + ["git", "diff", "--cached", "--name-only", "--diff-filter=d", "-z"] + + (["--", *paths] if paths else []), cwd=str(repo_dir), + capture_output=True, encoding="utf-8", timeout=10, check=True, + ) + touched.update(filter(None, result.stdout.split("\0"))) + except Exception as exc: + unavailable.append(f"Changed {source} paths could not be read ({type(exc).__name__}).") + + version_in_scope = "VERSION" in touched + if touched and not version_in_scope and source == "worktree": + # Same doc-only carve as the commit gate. Code-bearing standalone + # advisory still requires VERSION; the version-neutral index lane does not. + from ouroboros.tools.git_review_cycle import _diff_is_doc_only + if not _diff_is_doc_only(sorted(touched)): + findings.append( + "Changed files are present but VERSION is not in scope. " + "BIBLE.md P9 requires every commit to bump VERSION and sync release artifacts. " + "Stage or include VERSION plus its release carriers before advisory review. " + f"Currently changed/in-scope: {', '.join(sorted(touched))}" + ) + if version_in_scope or findings or unavailable: + if source == "index" and version_in_scope and "README.md" not in touched: + findings.append("Missing from staged: README.md (badge + changelog). Stage all related files together.") + texts = {} + for path in sorted(CARRIER_SPAN_PATHS): + try: + content = read_text(path) if read_text else read_release_file(repo_dir, path, source=source) + if content is not None: + texts[path] = content + elif path in ("VERSION", "README.md"): + unavailable.append(f"{source}:{path} is missing; release checks require this source.") + except Exception as exc: + unavailable.append(f"{source}:{path} could not be read ({type(exc).__name__}).") + findings.extend(release_metadata_findings(texts)) + else: + report["status"] = "not_applicable" + if unavailable: + report["status"] = "unavailable" + elif findings: + report["status"] = "blocked" + return report + + +def format_release_metadata_preflight(report: dict) -> Optional[str]: + """Compatibility error text without collapsing unavailable evidence into a defect.""" + if not report["findings"] and not report["unavailable"]: + return None + code = "PREFLIGHT_UNAVAILABLE" if report["unavailable"] else "PREFLIGHT_BLOCKED" + return (f"⚠️ {code}: Release metadata diagnostics ({report['source']}).\n" + + "".join(f" - {message}\n" for message in report["findings"]) + + "".join(f" - Unavailable: {message}\n" for message in report["unavailable"])) + + def release_metadata_preflight( - repo_dir: pathlib.Path, - commit_message: str, - paths: list[str] | None, + repo_dir: pathlib.Path, commit_message: str, paths: list[str] | None, + *, source: str = "worktree", ) -> Optional[str]: """Cheap deterministic P9/release checks before any paid review spend.""" - touched = set(str(p) for p in (paths or []) if str(p).strip()) | set( - changed_worktree_paths(repo_dir, paths=paths)) - version_in_scope = "VERSION" in touched - if touched and not version_in_scope: - # Doc-only carve (finding W3A-F1). The commit gate ALREADY exempts a - # doc-only diff from its compensating preflight; this admission blocked - # the same diff outright, so on every install a doc-only change could - # never obtain a fresh advisory verdict at all — the standard - # preflight_review -> commit_reviewed flow degraded to the AUDITED - # BYPASS for every doc-only change, and hardest for the two commit - # classes BIBLE P9 exempts from the bump (a version-neutral external - # contribution, a forensic recovery snapshot), which have no VERSION to - # name by construction. Same classifier as the commit gate, read from - # its owner module: one detector, so the two gates cannot drift. Narrow - # on purpose, and NARROWER than those two classes — a code-bearing diff - # without VERSION still blocks here whatever its provenance, and every - # carrier-coherence check below still runs the moment VERSION IS in - # scope. - from ouroboros.tools.git_review_cycle import _diff_is_doc_only - - if _diff_is_doc_only(sorted(touched)): - return None - return ( - "⚠️ PREFLIGHT_BLOCKED: Changed files are present but VERSION is not in scope.\n" - " BIBLE.md P9 requires every commit to bump VERSION and sync release artifacts.\n" - " Stage or include VERSION plus pyproject.toml, web/package.json, README.md, and docs/ARCHITECTURE.md before advisory review.\n" - f" Currently changed/in-scope: {', '.join(sorted(touched)) or '(none)'}" - ) - if not version_in_scope: - return None - try: - from ouroboros.tools.release_sync import ( - check_history_limit, - is_release_version, - version_carrier_desyncs, - ) - version_path = repo_dir / "VERSION" - readme_path = repo_dir / "README.md" - pyproject_path = repo_dir / "pyproject.toml" - uv_lock_path = repo_dir / "uv.lock" - web_package_path = repo_dir / "web" / "package.json" - web_package_lock_path = repo_dir / "web" / "package-lock.json" - arch_path = repo_dir / "docs" / "ARCHITECTURE.md" - api_types_path = repo_dir / "web" / "modules" / "api_types.js" - site_install_path = repo_dir / "site" / "install" / "index.html" - docs_install_path = repo_dir / "docs" / "install" / "index.html" - version_str = version_path.read_text(encoding="utf-8").strip() - if not is_release_version(version_str): - return None - pyproject_text = pyproject_path.read_text(encoding="utf-8") if pyproject_path.exists() else "" - uv_lock_text = uv_lock_path.read_text(encoding="utf-8") if uv_lock_path.exists() else "" - web_package_text = web_package_path.read_text(encoding="utf-8") if web_package_path.exists() else "" - web_package_lock_text = ( - web_package_lock_path.read_text(encoding="utf-8") if web_package_lock_path.exists() else "" - ) - readme_text = readme_path.read_text(encoding="utf-8") if readme_path.exists() else "" - arch_text = arch_path.read_text(encoding="utf-8") if arch_path.exists() else "" - api_types_text = api_types_path.read_text(encoding="utf-8") if api_types_path.exists() else "" - desync = version_carrier_desyncs( - version_str, - pyproject_text=pyproject_text, - uv_lock_text=uv_lock_text, - web_package_text=web_package_text, - web_package_lock_text=web_package_lock_text, - readme_text=readme_text, - arch_text=arch_text, - api_types_text=api_types_text, - download_readme_text=readme_text, - site_install_text=(site_install_path.read_text(encoding="utf-8") if site_install_path.exists() else ""), - docs_install_text=(docs_install_path.read_text(encoding="utf-8") if docs_install_path.exists() else ""), - detailed=True, - ) - if readme_text: - if not re.search(r'\|\s*' + re.escape(version_str) + r'\s*\|', readme_text): - return ( - f"⚠️ PREFLIGHT_BLOCKED: VERSION is {version_str} but README.md " - "changelog has no table row for this version.\n" - " Add a changelog entry in the Version History table in README.md before advisory review." - ) - limit_warnings = check_history_limit(readme_text) - if limit_warnings: - return ( - "⚠️ PREFLIGHT_BLOCKED: README.md Version History exceeds BIBLE.md P9 limits.\n" - + "".join(f" - {w}\n" for w in limit_warnings) - + " Trim the oldest entry in the over-limit category before advisory review." - ) - if desync: - return ( - f"⚠️ PREFLIGHT_BLOCKED: VERSION file says {version_str} but " - "the following worktree files have a different version value:\n" - + "".join(f" - {d}\n" for d in desync) - + "Run release metadata sync before advisory review." - ) - except Exception: - return None - return None + return format_release_metadata_preflight(release_metadata_diagnostics(repo_dir, paths, source=source)) def syntax_preflight_staged_py_files( diff --git a/ouroboros/tools/claude_advisory_review.py b/ouroboros/tools/claude_advisory_review.py index 660e9d038..ae62b6cd7 100644 --- a/ouroboros/tools/claude_advisory_review.py +++ b/ouroboros/tools/claude_advisory_review.py @@ -832,7 +832,7 @@ def _next_step_guidance(latest: Optional["AdvisoryRunRecord"], state: "AdvisoryR problem = "syntax preflight: a staged .py file has a SyntaxError" fix = "See raw_result for file:line:msg, fix it, and re-run preflight_review." elif reason_kind == "release_metadata": - problem = "release metadata preflight: version/README release carriers failed the deterministic check" + problem = "release metadata preflight: inspect all findings with preflight_review(deterministic_only=True, source=worktree or index)" fix = "See raw_result for the exact carrier mismatch, fix it, and re-run preflight_review." else: problem = "a deterministic preflight check (see raw_result for the exact cause)" @@ -948,6 +948,7 @@ def _advisory_pre_sdk_gate( paths: Optional[List[str]], skip_tests: bool, review_rebuttal: str = "", + prepared: bool = False, ): """Run cheap pre-SDK gates and return warnings/status/early JSON exit.""" repo_key = make_repo_key(repo_dir) @@ -1012,16 +1013,19 @@ def _advisory_pre_sdk_gate( ), }) - release_preflight_err = _release_metadata_preflight(repo_dir, commit_message, paths) + release_preflight_err = (_release_metadata_preflight(repo_dir, commit_message, paths, source="index") + if prepared else _release_metadata_preflight(repo_dir, commit_message, paths)) if release_preflight_err: + unavailable = release_preflight_err.startswith("⚠️ PREFLIGHT_UNAVAILABLE:") + status = "error" if unavailable else "preflight_blocked" ctx.emit_progress_fn(release_preflight_err) _persist_preflight_record( ctx=ctx, snapshot_hash=snapshot_hash, commit_message=commit_message, record={ - "status": "preflight_blocked", - "reason_kind": "release_metadata", + "status": status, + "reason_kind": "release_metadata_unavailable" if unavailable else "release_metadata", "raw_result": release_preflight_err, "paths": paths, "duration_sec": 0.0, @@ -1029,7 +1033,7 @@ def _advisory_pre_sdk_gate( }, ) return readiness_warnings, changed_files, _json_response({ - "status": "preflight_blocked", + "status": status, "snapshot_hash": snapshot_hash, "error": release_preflight_err, "readiness_warnings": readiness_warnings, @@ -1105,8 +1109,17 @@ def _handle_advisory_pre_review( skip_tests: bool = False, review_rebuttal: str = "", prepared: bool = False, + deterministic_only: bool = False, + source: str = "", ) -> str: - """Run an advisory pre-commit review through the configured read-only route.""" + """Run release diagnostics or advisory review through the configured read-only route.""" + if deterministic_only: + from ouroboros.commit_admission import release_metadata_diagnostics + if source not in ("worktree", "index"): + return _json_response({"status": "error", "failure_code": "PREFLIGHT_SOURCE_REQUIRED", + "message": "deterministic_only requires explicit source=worktree or source=index."}) + return _json_response({**release_metadata_diagnostics(ctx.repo_dir, paths, source=source), + "deterministic_only": True, "review_freshness": False}) skip_advisory_pre_review = bool(skip_advisory_review or skip_advisory_pre_review) repo_dir = pathlib.Path(ctx.repo_dir) drive_root = pathlib.Path(ctx.drive_root) @@ -1195,7 +1208,7 @@ def _handle_advisory_pre_review( commit_message=commit_message, paths=paths, skip_tests=skip_tests, - review_rebuttal=review_rebuttal, + review_rebuttal=review_rebuttal, prepared=prepared, ) if early_exit is not None: return early_exit @@ -1433,6 +1446,8 @@ def _preflight_review_params() -> dict: "scope": _schema_param("string", "Declared scope boundary. Issues outside scope are advisory-only."), "review_rebuttal": _schema_param("string", "Counter-argument to previous review findings, delivered in full to this preflight reviewer."), "paths": _schema_param("array", "Explicit list of changed file paths. Auto-detected from git status if omitted.", items={"type": "string"}), + "deterministic_only": _schema_param("boolean", "Only diagnose release metadata; requires explicit source. No sync, staging, tests, providers, review-state reads/writes or freshness. Returns all applicable findings and unavailable-source errors separately. Default: False.", default=False), + "source": _schema_param("string", "Required for deterministic_only: worktree reads current files; index reads staged blobs. Ignored for ordinary review (standalone uses worktree; prepared uses index).", enum=["worktree", "index"]), "skip_tests": _schema_param("boolean", "Skip the preflight pytest run. Default: False (tests run by default). Use True only for intentionally incomplete WIP code where test failures are expected. Tests are run before the paid critic call — in a hermetic worktree, as the same two passes CI runs (parallel 'not serial' then serial) — to catch broken code early and avoid wasting review budget.", default=False), }, "required": ["commit_message"], @@ -1463,7 +1478,8 @@ def get_tools() -> list: "description": ( "Run the preflight pre-commit review (formerly `advisory_review`) " "through the configured read-only route. " - "Returns structured JSON findings; any edit afterward makes the result stale. " + "Use deterministic_only=True with explicit source=worktree or index for release diagnostics without effects or review freshness. " + "Ordinary review returns structured JSON findings; any edit afterward makes the result stale. " f"{ADVISORY_REVIEW_CHOICE_GUIDANCE} " f"{_identical_diff_cap_note()}" ), @@ -1491,7 +1507,7 @@ def get_tools() -> list: schema={ "name": "review_status", "description": ( - "Show recent advisory pre-review run history. Read-only diagnostic — use to check advisory freshness before commit_reviewed. Also shows: last commit attempt state (reviewing/blocked/succeeded/failed) with block reason and actionable guidance; whether advisory is stale because of a worktree edit; open obligations from previous blocking rounds; open commit-readiness debt (durable repo-scoped anti-thrashing signal with fields `commit_readiness_debts`, `commit_readiness_debts_count`); `repo_commit_ready` (an advisory-readiness projection only: a fresh/bypassed/skipped advisory and no open advisory obligations or debt, not the full commit gate); `retry_anchor` (non-null, currently `commit_readiness_debt`, when debt is open — start the next retry from that record instead of patching one obligation at a time); and a concrete next_step recommendation. " + "Show recent advisory pre-review run history. Read-only diagnostic — use to check advisory freshness before commit_reviewed; deterministic-only release diagnostics confer no freshness and create no history. Also shows: last commit attempt state (reviewing/blocked/succeeded/failed) with block reason and actionable guidance; whether advisory is stale because of a worktree edit; open obligations from previous blocking rounds; open commit-readiness debt (durable repo-scoped anti-thrashing signal with fields `commit_readiness_debts`, `commit_readiness_debts_count`); `repo_commit_ready` (an advisory-readiness projection only: a fresh/bypassed/skipped advisory and no open advisory obligations or debt, not the full commit gate); `retry_anchor` (non-null, currently `commit_readiness_debt`, when debt is open — start the next retry from that record instead of patching one obligation at a time); and a concrete next_step recommendation. " f"{ADVISORY_REVIEW_CHOICE_GUIDANCE} " "Pass include_raw=true to surface the full per-actor evidence (triad_raw_results, scope_raw_result) for the targeted attempt." ), diff --git a/ouroboros/tools/commit_gate.py b/ouroboros/tools/commit_gate.py index 0002bd21f..1d679523d 100644 --- a/ouroboros/tools/commit_gate.py +++ b/ouroboros/tools/commit_gate.py @@ -1218,15 +1218,15 @@ def bind_author_commit_candidate(ctx: ToolContext, commit_message: str, pre_fing from ouroboros.tools import git as git_mod author_source = ctx._author_commit_source from ouroboros.review_records import build_author_disposition_from_mapping - from ouroboros.tools.review import _preflight_check + from ouroboros.tools.review import _preflight_check, format_name_status_for_preflight from ouroboros.config import get_review_enforcement author = build_author_disposition_from_mapping(ctx._author_commit_decision, subject_hash=pre_fingerprint["fingerprint"], reviewer_signal=author_source.block_reason or author_source.status, enforcement=get_review_enforcement()) author["review_reference"] = ctx._author_commit_reference ctx._author_commit_record = author - preflight = _preflight_check(commit_message, git_mod.run_cmd(["git", "diff", "--cached", "--name-status"], cwd=ctx.repo_dir), ctx.repo_dir) - return preflight + staged = git_mod.run_cmd(["git", "diff", "--cached", "--name-status"], cwd=ctx.repo_dir) + return _preflight_check(commit_message, format_name_status_for_preflight(staged), ctx.repo_dir) def record_bound_commit_success(ctx: ToolContext, commit_message: str, started_at: float, before: dict, after: dict) -> None: diff --git a/ouroboros/tools/release_sync.py b/ouroboros/tools/release_sync.py index 10cff9dd5..7fd055bf4 100644 --- a/ouroboros/tools/release_sync.py +++ b/ouroboros/tools/release_sync.py @@ -513,6 +513,42 @@ def version_carrier_desyncs( return desync +def release_metadata_findings(texts: dict[str, str]) -> List[str]: + """All independently decidable release findings over one source's readable files. + + Reuse carrier grammar and P9 counters; unavailable files are absent from this + mapping and are reported by the reader. Never invent a future release version. + """ + findings: List[str] = [] + version = texts.get("VERSION", "").strip() + if "VERSION" in texts and not is_release_version(version): + findings.append("VERSION is empty or malformed; use a supported release version.") + readme = texts.get("README.md") + if readme is not None: + if is_release_version(version) and not re.search(r'\|\s*' + re.escape(version) + r'\s*\|', readme): + findings.append(f"VERSION is {version} but README.md changelog has no table row for this version. " + "Add a changelog entry in the Version History table in README.md.") + limits = check_history_limit(readme) + if limits: + findings.append("README.md Version History exceeds BIBLE.md P9 limits.") + findings.extend(limits) + # A present empty carrier is malformed, not an absent optional older carrier. + findings.extend(f"{path} is empty; restore its release metadata." + for path, text in texts.items() if path != "VERSION" and not text.strip()) + findings.extend(version_carrier_desyncs( + version, + pyproject_text=texts.get("pyproject.toml", ""), + uv_lock_text=texts.get("uv.lock", ""), + web_package_text=texts.get("web/package.json", ""), + web_package_lock_text=texts.get("web/package-lock.json", ""), + readme_text=readme or "", arch_text=texts.get("docs/ARCHITECTURE.md", ""), + api_types_text=texts.get("web/modules/api_types.js", ""), + download_readme_text=readme or "", site_install_text=texts.get("site/install/index.html", ""), + docs_install_text=texts.get("docs/install/index.html", ""), detailed=True, + )) + return findings + + def check_worktree_version_sync(repo_dir) -> str: """Return a non-fatal warning when release version carriers disagree. diff --git a/ouroboros/tools/review.py b/ouroboros/tools/review.py index 6ae56be67..58b5e3b5d 100644 --- a/ouroboros/tools/review.py +++ b/ouroboros/tools/review.py @@ -478,20 +478,10 @@ def _parse_review_json(raw: str) -> Optional[list]: return extract_json_array(raw, normalize=True) -def _git_show_staged(repo_dir, path: str) -> str: - """Return staged index content via ``git show :PATH`` or ``""``.""" - import subprocess - try: - result = subprocess.run( - ["git", "show", f":{path}"], - cwd=str(repo_dir), - capture_output=True, - text=True, - timeout=10, - ) - return result.stdout if result.returncode == 0 else "" - except Exception: - return "" +def _git_show_staged(repo_dir, path: str) -> Optional[str]: + """Return indexed text (None only for absence); propagate failed reads.""" + from ouroboros.commit_admission import read_release_file + return read_release_file(repo_dir, path, source="index") def _preflight_check(commit_message: str, staged_files: str, @@ -508,7 +498,6 @@ def _preflight_check(commit_message: str, staged_files: str, the semantic checklist: docs/CHECKLISTS.md item 6 (tests_affected) and item 8 (version_bump). """ - import re import string as _string # Accept either name-status lines ("A path") or plain filenames. @@ -537,17 +526,13 @@ def _preflight_check(commit_message: str, staged_files: str, active_staged = {path for status, path in file_status if status != "D"} # Added/Copied count as new modules; renames do not. new_files = {path for status, path in file_status if status in ("A", "C")} - version_staged = "VERSION" in active_staged - - # VERSION staged but README missing. - if version_staged and "README.md" not in active_staged: - return ( - "⚠️ PREFLIGHT_BLOCKED: Staged diff is incomplete — fix before review.\n" - " Missing from staged: README.md (badge + changelog)\n" - f" Currently staged: {', '.join(sorted(staged_set)) or '(none)'}\n\n" - "Stage all related files together. Use write_file for all files first,\n" - "then commit_reviewed to stage and commit everything in one diff." - ) + from ouroboros.commit_admission import release_metadata_diagnostics, format_release_metadata_preflight + release_error = format_release_metadata_preflight(release_metadata_diagnostics( + repo_dir, sorted(active_staged), source="index", + read_text=lambda path: _git_show_staged(repo_dir, path), + )) + if release_error: + return release_error # The version-reference and tests-required lexical heuristics were removed # here (false blocks: a "conversion" commit told to bump VERSION; a @@ -580,74 +565,6 @@ def _preflight_check(commit_message: str, staged_files: str, f" Currently staged: {', '.join(sorted(staged_set)) or '(none)'}" ) - # VERSION changes must keep staged version carriers synchronized. - if version_staged: - try: - from ouroboros.tools.release_sync import ( - is_release_version, - version_carrier_desyncs, - ) - version_str = _git_show_staged(repo_dir, "VERSION").strip() - if is_release_version(version_str): - desync = version_carrier_desyncs( - version_str, - pyproject_text=_git_show_staged(repo_dir, "pyproject.toml"), - uv_lock_text=_git_show_staged(repo_dir, "uv.lock"), - web_package_text=_git_show_staged(repo_dir, "web/package.json"), - web_package_lock_text=_git_show_staged(repo_dir, "web/package-lock.json"), - readme_text=_git_show_staged(repo_dir, "README.md"), - arch_text=_git_show_staged(repo_dir, "docs/ARCHITECTURE.md"), - api_types_text=_git_show_staged(repo_dir, "web/modules/api_types.js"), - download_readme_text=_git_show_staged(repo_dir, "README.md"), - site_install_text=_git_show_staged(repo_dir, "site/install/index.html"), - docs_install_text=_git_show_staged(repo_dir, "docs/install/index.html"), - detailed=True, - ) - if desync: - return ( - f"⚠️ PREFLIGHT_BLOCKED: VERSION file says {version_str} but " - "the following staged files have a different version value:\n" - + "".join(f" - {d}\n" for d in desync) - + "Update all version references to match VERSION before committing.\n" - f" Currently staged: {', '.join(sorted(staged_set)) or '(none)'}" - ) - except Exception: - pass # Non-fatal: LLM reviewers handle version sync - - # VERSION changes need a staged README changelog row, and the staged README - # must respect P9 history limits. - if version_staged: - try: - from ouroboros.tools.release_sync import is_release_version - version_str = _git_show_staged(repo_dir, "VERSION").strip() - if is_release_version(version_str): - readme_text = _git_show_staged(repo_dir, "README.md") - if readme_text and not re.search(r'\|\s*' + re.escape(version_str) + r'\s*\|', readme_text): - return ( - f"⚠️ PREFLIGHT_BLOCKED: VERSION is {version_str} but README.md " - "changelog has no table row for this version.\n" - " Add a changelog entry in the Version History table in README.md.\n" - f" Currently staged: {', '.join(sorted(staged_set)) or '(none)'}" - ) - except Exception: - pass # Non-fatal - try: - readme_staged = _git_show_staged(repo_dir, "README.md") - if readme_staged: - from ouroboros.tools.release_sync import check_history_limit - limit_warnings = check_history_limit(readme_staged) - if limit_warnings: - return ( - "⚠️ PREFLIGHT_BLOCKED: README.md Version History exceeds BIBLE.md P9 limits.\n" - + "".join(f" - {w}\n" for w in limit_warnings) - + " Trim the oldest entry in the over-limit category before committing.\n" - + " Quick check: python -c \"from ouroboros.tools.release_sync import " - "check_history_limit; print(check_history_limit(open('README.md').read()))\"\n" - + f" Currently staged: {', '.join(sorted(staged_set)) or '(none)'}" - ) - except Exception: - pass # Non-fatal: LLM reviewers handle P9 limits as advisory fallback - # conftest.py must not contain collectable module-level tests. conftest_files = [f for f in active_staged if pathlib.Path(f).name == "conftest.py"] if conftest_files: @@ -1048,7 +965,7 @@ def _prepare_unified_review(ctx: ToolContext, commit_message: str, preflight_err = _preflight_check(commit_message, preflight_staged, target_repo) if preflight_err: - ctx._last_review_block_reason = "preflight" + ctx._last_review_block_reason = ("infra_failure" if preflight_err.startswith("⚠️ PREFLIGHT_UNAVAILABLE:") else "preflight") result = _handle_review_block_or_warning( ctx, blocking_review, preflight_err, "Review enforcement=Advisory: preflight warning did not block commit. ", diff --git a/tests/test_advisory_inline_freshness.py b/tests/test_advisory_inline_freshness.py index 57daa97ba..3277d6e21 100644 --- a/tests/test_advisory_inline_freshness.py +++ b/tests/test_advisory_inline_freshness.py @@ -30,7 +30,7 @@ def candidate(tmp_path, monkeypatch): monkeypatch.setattr(advisory, "advisory_review_route", lambda: "agent_session") monkeypatch.setattr(advisory, "advisory_slot_enabled", lambda: True) monkeypatch.setattr(advisory, "check_worktree_readiness", lambda *a, **kw: []) - monkeypatch.setattr(advisory, "_release_metadata_preflight", lambda *a: None) + monkeypatch.setattr(advisory, "_release_metadata_preflight", lambda *a, **kw: None) monkeypatch.setattr(advisory, "_check_worktree_version_sync_shared", lambda *a: "") monkeypatch.setattr(git, "advisory_gate_unavailable", lambda: False) monkeypatch.setattr(git, "_managed_candidate_needs_proof", lambda ctx: False) diff --git a/tests/test_advisory_preflight.py b/tests/test_advisory_preflight.py index 0d26a89f0..bbb643a76 100644 --- a/tests/test_advisory_preflight.py +++ b/tests/test_advisory_preflight.py @@ -340,6 +340,7 @@ class TestHandleAdvisoryPreReviewSurfacesPreflightBlocked: adv, "_check_worktree_version_sync_shared", lambda *args, **kwargs: "", ) + monkeypatch.setattr(adv, "_release_metadata_preflight", lambda *a, **kw: None) monkeypatch.setattr( adv, "compute_snapshot_hash", lambda *args, **kwargs: "deadbeef", ) @@ -408,6 +409,7 @@ class TestHandleAdvisoryPreReviewSurfacesPreflightBlocked: from ouroboros.tools import claude_advisory_review as adv repo = _make_agent_repo(tmp_path) + _init_git_repo(repo) _write_release_files(repo, version="5.99.0-rc.1") (repo / "uv.lock").write_text( '[[package]]\nname = "ouroboros"\nversion = "5.98.0"\n' @@ -432,6 +434,7 @@ class TestHandleAdvisoryPreReviewSurfacesPreflightBlocked: from ouroboros.tools import claude_advisory_review as adv repo = _make_agent_repo(tmp_path) + _init_git_repo(repo) _write_release_files(repo, version="5.99.0-rc.1") (repo / "web").mkdir() (repo / "web" / "package.json").write_text( @@ -657,6 +660,7 @@ class TestPreflightBlockedPersistence: adv, "_check_worktree_version_sync_shared", lambda *args, **kwargs: "", ) + monkeypatch.setattr(adv, "_release_metadata_preflight", lambda *a, **kw: None) monkeypatch.setattr( adv, "compute_snapshot_hash", lambda *args, **kwargs: "preflight-test-hash", diff --git a/tests/test_commit_review_task_evidence.py b/tests/test_commit_review_task_evidence.py index d530f22d7..fed31031d 100644 --- a/tests/test_commit_review_task_evidence.py +++ b/tests/test_commit_review_task_evidence.py @@ -483,7 +483,7 @@ def test_pending_preflight_selection_survives_reconciliation_then_stage_dispatch monkeypatch.setattr(advisory, "advisory_review_route", lambda: "agent_session") monkeypatch.setattr(advisory, "advisory_slot_enabled", lambda: True) monkeypatch.setattr(advisory, "check_worktree_readiness", lambda *a, **kw: []) - monkeypatch.setattr(advisory, "_release_metadata_preflight", lambda *a: None) + monkeypatch.setattr(advisory, "_release_metadata_preflight", lambda *a, **kw: None) monkeypatch.setattr(advisory, "_check_worktree_version_sync_shared", lambda *a: "") monkeypatch.setattr("ouroboros.delegate_custody.invocation_record", lambda *a: {"request": {"prompt": "ORIGINAL_PREFLIGHT_PROMPT"}}) sent = [] diff --git a/tests/test_git_review_preflight_gate.py b/tests/test_git_review_preflight_gate.py index 3d645ba17..c4a54d483 100644 --- a/tests/test_git_review_preflight_gate.py +++ b/tests/test_git_review_preflight_gate.py @@ -106,9 +106,12 @@ _PREFLIGHT_CASES = [ _PREFLIGHT_CASES, ids=[c[0] for c in _PREFLIGHT_CASES], ) -def test_preflight_check(case_id, message, staged_files, expected): +def test_preflight_check(case_id, message, staged_files, expected, monkeypatch, tmp_path): review = _get_review_module() - result = review._preflight_check(message, staged_files, "/tmp") + values = {"VERSION": "3.24.0", "README.md": + "[![Version 3.24.0](https://img.shields.io/badge/version-3.24.0-green.svg)]\n| 3.24.0 | release |"} + monkeypatch.setattr(review, "_git_show_staged", lambda repo, path: values.get(path)) + result = review._preflight_check(message, staged_files, tmp_path) if expected is None: assert result is None, f"expected pass, got: {result!r}" else: @@ -142,7 +145,7 @@ class TestPreflightCheck7P9Limits: return 'version = "4.99.0"' if path == "docs/ARCHITECTURE.md": return "# Ouroboros v4.99.0 — " - return "" + return None monkeypatch.setattr(review, "_git_show_staged", _fake_git_show) staged = f"M VERSION\nM README.md\nM tests/test_foo.py\n{extra_staged}".strip() @@ -234,7 +237,7 @@ class TestPreflightCheck7P9Limits: def _fake_git_show(repo_dir, path: str) -> str: if path == "README.md": return bloated_readme - return "" + return None monkeypatch.setattr(review, "_git_show_staged", _fake_git_show) # Only README staged — no VERSION, no ouroboros/*.py. @@ -262,7 +265,7 @@ class TestPreflightCheck7P9Limits: "README.md": readme, "docs/ARCHITECTURE.md": "# Ouroboros v4.99.0 — Architecture", } - return values.get(path, "") + return values.get(path) monkeypatch.setattr(review, "_git_show_staged", _fake_git_show) result = review._preflight_check( @@ -301,7 +304,7 @@ class TestPreflightCheck7P9Limits: "README.md": readme, "docs/ARCHITECTURE.md": "# Ouroboros v4.99.0 — Architecture", } - return values.get(path, "") + return values.get(path) monkeypatch.setattr(review, "_git_show_staged", _fake_git_show) result = review._preflight_check( @@ -314,21 +317,18 @@ class TestPreflightCheck7P9Limits: assert result is not None and "PREFLIGHT_BLOCKED" in result assert 'web/package-lock.json (expected both root "version" entries = "4.99.0")' in result - def test_check7_passes_when_readme_not_staged(self, monkeypatch): - """VERSION staged but README not staged → check 7 silently skips - (git show returns empty string for an un-staged README).""" + def test_missing_readme_reports_staging_and_source_problems(self, monkeypatch): + """Missing indexed README retains the staging finding beside source unavailability.""" review = _get_review_module() def _fake_git_show(repo_dir, path: str) -> str: if path == "VERSION": return "4.99.0" - return "" # README absent from staged index + return None # README absent from staged index monkeypatch.setattr(review, "_git_show_staged", _fake_git_show) result = review._preflight_check( "v4.99.0 bump", "M VERSION\nM tests/test_foo.py", "/repo" ) - # Check 1 fires first (README.md missing from staged when VERSION staged). - # This is acceptable — the missing README is caught by check 1, not check 7. - # Either result is valid here; we just verify no crash. - assert result is None or "PREFLIGHT_BLOCKED" in result + assert result is not None and "Missing from staged: README.md" in result + assert "PREFLIGHT_UNAVAILABLE" in result diff --git a/tests/test_reference_book_budgets.py b/tests/test_reference_book_budgets.py index eb78edf20..54fbb3d3d 100644 --- a/tests/test_reference_book_budgets.py +++ b/tests/test_reference_book_budgets.py @@ -94,7 +94,9 @@ CHAPTER_BYTE_BUDGETS: dict[str, int] = { # FOCUS_SOURCE_UNRESOLVED) and a settled root's focus is dropped — new # contract facts of the cross-focus paragraph, not a restatement; +250 for # the digest-selected historical read and the reader admission rule. - "docs/architecture/06-agent-core.md": 295650, + # 295650 -> 297250: document the new diagnostic-only source/coverage contract, + # unavailable evidence and no-effects ordering without removing review/custody rules. + "docs/architecture/06-agent-core.md": 297250, "docs/architecture/07-configuration.md": 36991, # 18947 -> 19287: CI failure collection now documents diagnostic desktop builds while release remains gated. "docs/architecture/08-git-branching-ci-and-build.md": 19287, diff --git a/tests/test_release_metadata_diagnostics.py b/tests/test_release_metadata_diagnostics.py new file mode 100644 index 000000000..608246959 --- /dev/null +++ b/tests/test_release_metadata_diagnostics.py @@ -0,0 +1,353 @@ +"""Release diagnostics read a chosen Git source and never grant review authority.""" +from __future__ import annotations + +import json +import subprocess +from types import SimpleNamespace + +import pytest + +from ouroboros import commit_admission as admission +from ouroboros.tools import claude_advisory_review as advisory, review +from ouroboros.tools.release_sync import release_metadata_findings + +pytestmark = pytest.mark.serial # real Git processes and isolated runtime globals + + +def _git(repo, *args): + return subprocess.check_output(["git", *args], cwd=repo, text=True).strip() + + +def _release(version="1.2.3"): + return { + "VERSION": version + "\n", + "README.md": f"[![Version {version}](https://img.shields.io/badge/version-{version}-green.svg)]\n" + f"## Version History\n| {version} | today | release |\n", + "pyproject.toml": f'[project]\nversion = "{version}"\n', + "docs/ARCHITECTURE.md": f"# Ouroboros v{version}\n", + "web/package.json": json.dumps({"version": version}), + "uv.lock": f'[[package]]\nname = "ouroboros"\nversion = "{version}"\nsource = {{ editable = "." }}\n', + "web/modules/api_types.js": f"GATEWAY_CONTRACT_VERSION = '{version}'\n", + } + + +def _write(repo, files): + for name, value in files.items(): + path = repo / name + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(value, encoding="utf-8") + + +@pytest.fixture +def candidate(tmp_path, monkeypatch): + from ouroboros import config + from supervisor import git_ops, queue, state, workers + + repo, drive = tmp_path / "repo", tmp_path / "data" + repo.mkdir() + drive.mkdir() + # Environment is scrubbed before imports by the test invocation; rebind all + # imported mutable roots as well. Never initialize or contact a live launcher. + for key, value in {"OUROBOROS_APP_ROOT": tmp_path, "OUROBOROS_REPO_DIR": repo, + "OUROBOROS_DATA_DIR": drive, "OUROBOROS_SETTINGS_PATH": drive / "settings.json", + "OUROBOROS_SUBAGENT_PROJECTS_ROOT": tmp_path / "projects"}.items(): + monkeypatch.setenv(key, str(value)) + for module, attrs in ((config, {"APP_ROOT": tmp_path, "REPO_DIR": repo, "DATA_DIR": drive, + "SETTINGS_PATH": drive / "settings.json"}), + (git_ops, {"REPO_DIR": repo, "DRIVE_ROOT": drive}), + (workers, {"REPO_DIR": repo, "DRIVE_ROOT": drive, "DATA_DIR": drive}), + (state, {"DRIVE_ROOT": drive, "DATA_DIR": drive, + "STATE_PATH": drive / "state/state.json", + "STATE_LAST_GOOD_PATH": drive / "state/state.last_good.json", + "STATE_LOCK_PATH": drive / "locks/state.lock"}), + (queue, {"DRIVE_ROOT": drive, "DATA_DIR": drive, + "QUEUE_SNAPSHOT_PATH": drive / "state/queue_snapshot.json"})): + for name, value in attrs.items(): + monkeypatch.setattr(module, name, value, raising=False) + _git(repo, "init") + _git(repo, "config", "user.name", "Fixture") + _git(repo, "config", "user.email", "fixture@example.invalid") + _write(repo, {**_release(), "change.py": "value = 1\n"}) + _git(repo, "add", ".") + _git(repo, "commit", "-qm", "fixture") + return SimpleNamespace(repo_dir=repo, drive_root=drive, task_id="diagnostic-test", + emit_progress_fn=lambda *_: None) + + +def _broken(repo): + files = _release("1.2.4") + files["README.md"] = _release()["README.md"] + "".join( + f"| 1.1.{patch} | today | old patch |\n" for patch in range(1, 7)) + files["pyproject.toml"] = 'version = "0.0.0"\n' + _write(repo, files) + + +def _snapshot(root): + """Bytes and mtimes catch index refreshes and unchanged-content rewrites too.""" + return {p.relative_to(root).as_posix(): (p.read_bytes(), p.stat().st_mtime_ns) + for p in root.rglob("*") if p.is_file()} + + +def _diagnose(ctx, source, **kwargs): + return json.loads(advisory._handle_advisory_pre_review( + ctx, "release", deterministic_only=True, source=source, **kwargs)) + + +@pytest.mark.parametrize("source", ["worktree", "index"]) +@pytest.mark.parametrize("outcome", ["clean", "blocked", "unavailable"]) +def test_diagnostic_has_no_effects_even_with_pending_managed_review(candidate, monkeypatch, source, outcome): + ctx = candidate + if outcome == "blocked": + _broken(ctx.repo_dir) + else: + _write(ctx.repo_dir, _release("1.2.4")) + if outcome == "unavailable": + (ctx.repo_dir / "pyproject.toml").write_bytes(b"\xff") + _git(ctx.repo_dir, "add", ".") + # Actual durable bytes are present: a dry run must neither read nor rewrite + # pending custody, obligations or review freshness, even for a managed caller. + _write(ctx.drive_root, {"state/advisory_review.json": '{"pending":true,"obligations":["keep"]}', + "logs/events.jsonl": '{"type":"keep"}\n', + "state/delegate_custody.jsonl": '{"invocation":"pending"}\n'}) + ctx.task_metadata = {"managed_update": {"transaction_id": "pending"}} + ctx._preflight_test_proof = object() + before, context_before = _snapshot(ctx.repo_dir.parent), dict(vars(ctx)) + + def forbidden(*args, **kwargs): + pytest.fail("diagnostics crossed into review, preparation or test work") + + for name in ("pending_advisory_execution", "_auto_sync_release_metadata_if_needed", + "compute_snapshot_hash", "load_state", "update_state", "make_repo_key", + "_record_bypass", "_persist_preflight_record", "advisory_review_route", + "advisory_slot_enabled", "_advisory_pre_sdk_gate", "_run_claude_advisory", + "_run_advisory_tests", "check_worktree_readiness", "append_jsonl"): + monkeypatch.setattr(advisory, name, forbidden) + monkeypatch.setattr("ouroboros.tools.release_sync.sync_release_metadata", forbidden) + monkeypatch.setattr("ouroboros.provider_models.model_has_credentials", forbidden) + monkeypatch.setattr("ouroboros.tools.review._fingerprint_staged_diff", forbidden, raising=False) + monkeypatch.setattr("ouroboros.tools.registry._authorized_managed_update_resolver", forbidden) + real_run = admission.subprocess.run + + def only_reads(argv, *args, **kwargs): + assert argv[0] == "git" and not set(argv[1:]) & {"add", "write-tree", "commit", "reset", "update-index"} + return real_run(argv, *args, **kwargs) + + monkeypatch.setattr(admission.subprocess, "run", only_reads) + result = _diagnose(ctx, source, prepared=True, skip_advisory_review=True, skip_tests=False) + assert result["status"] == outcome, result + assert result["source"] == source and result["review_freshness"] is False + assert "snapshot_hash" not in result and "review_reference" not in result + assert _snapshot(ctx.repo_dir.parent) == before + assert vars(ctx) == context_before + + +@pytest.mark.parametrize("source", ["worktree", "index"]) +def test_reports_simultaneous_missing_row_history_overflow_and_desync(candidate, source): + _broken(candidate.repo_dir) + _git(candidate.repo_dir, "add", ".") + result = _diagnose(candidate, source) + assert result["status"] == "blocked" and result["unavailable"] == [] + findings = "\n".join(result["findings"]) + assert "no table row" in findings + assert "patch rows (limit 5)" in findings + assert "pyproject.toml" in findings and "README.md badge" in findings + staged = review.format_name_status_for_preflight(_git(candidate.repo_dir, "diff", "--cached", "--name-status")) + legacy = review._preflight_check("release", staged, candidate.repo_dir) if source == "index" else admission.release_metadata_preflight(candidate.repo_dir, "release", None) + assert all(finding in legacy for finding in result["findings"]) + + +@pytest.mark.parametrize("clean_source", ["worktree", "index"]) +def test_partial_staging_uses_selected_source(candidate, clean_source): + repo = candidate.repo_dir + _write(repo, _release("1.2.4")) if clean_source == "index" else _broken(repo) + _git(repo, "add", ".") + _write(repo, _release("1.2.4")) if clean_source == "worktree" else _broken(repo) + for source in ("worktree", "index"): + report = _diagnose(candidate, source) + assert report["status"] == ("clean" if source == clean_source else "blocked"), report + assert (admission.release_metadata_preflight(repo, "release", None) is None) == (clean_source == "worktree") + assert (admission.release_metadata_preflight(repo, "release", None, source="index") is None) == (clean_source == "index") + + +@pytest.mark.parametrize("version", ["", "not-a-release"]) +def test_malformed_version_does_not_hide_independent_history_failure(candidate, version): + _broken(candidate.repo_dir) + (candidate.repo_dir / "VERSION").write_text(version, encoding="utf-8") + result = _diagnose(candidate, "worktree") + findings = "\n".join(result["findings"]) + assert result["status"] == "blocked" + assert "empty or malformed" in findings and "patch rows (limit 5)" in findings + assert "no table row" not in findings # no invented version or dependent finding + + +@pytest.mark.parametrize("source", ["worktree", "index"]) +@pytest.mark.parametrize("path", ["VERSION", "README.md", "pyproject.toml"]) +def test_unreadable_source_preserves_independent_findings(candidate, monkeypatch, source, path): + _broken(candidate.repo_dir) + _git(candidate.repo_dir, "add", ".") + read = admission.read_release_file + + def fail_one(repo, relative, **kwargs): + if relative == path: + raise PermissionError("fixture unavailable") + return read(repo, relative, **kwargs) + + monkeypatch.setattr(admission, "read_release_file", fail_one) + result = _diagnose(candidate, source) + assert result["status"] == "unavailable" + assert any(f"{source}:{path}" in message for message in result["unavailable"]) + assert result["findings"] # independent readable metadata still diagnosed + assert "PREFLIGHT_UNAVAILABLE" in admission.format_release_metadata_preflight(result) + if path == "VERSION": + assert not any("no table row" in message for message in result["findings"]) + + +def test_index_read_failure_and_unmerged_entries_are_unavailable(candidate): + repo = candidate.repo_dir + _write(repo, _release("1.2.4")) + _git(repo, "add", ".") + blob = _git(repo, "rev-parse", ":pyproject.toml") + # Install an unmerged optional carrier in this disposable fixture index. + subprocess.run(["git", "update-index", "--index-info"], cwd=repo, check=True, text=True, + input=f"0 {'0' * 40}\tpyproject.toml\n100644 {blob} 1\tpyproject.toml\n100644 {blob} 2\tpyproject.toml\n") + report = _diagnose(candidate, "index") + assert report["status"] == "unavailable" + assert any("pyproject.toml" in item for item in report["unavailable"]) + + +@pytest.mark.parametrize("source", ["worktree", "index"]) +def test_git_discovery_failure_is_never_clean(candidate, monkeypatch, source): + real = admission.subprocess.run + + def fail_discovery(argv, **kwargs): + if "status" in argv or "diff" in argv: + return subprocess.CompletedProcess(argv, 128, stdout="", stderr="fixture failed") + return real(argv, **kwargs) + + # The real run(check=True) raises for index discovery; emulate that contract. + def checked(argv, **kwargs): + result = fail_discovery(argv, **kwargs) + if kwargs.get("check"): + result.check_returncode() + return result + + monkeypatch.setattr(admission.subprocess, "run", checked) + report = _diagnose(candidate, source) + assert report["status"] == "unavailable" and report["unavailable"] + + +def test_scope_preserves_contributor_doc_only_and_empty_paths(candidate): + repo = candidate.repo_dir + assert _diagnose(candidate, "index")["status"] == "not_applicable" + _write(repo, {"change.py": "value = 2\n", "docs/notes.md": "notes\n"}) + _git(repo, "add", ".") + # Staged contribution code without VERSION still passes; standalone code + # still requests VERSION. A docs-only selection keeps the existing carve. + assert _diagnose(candidate, "index")["status"] == "not_applicable" + report = _diagnose(candidate, "worktree", paths=["change.py"]) + assert len(report["findings"]) == 1 and "VERSION is not in scope" in report["findings"][0] + assert "no table row" not in str(report) + assert _diagnose(candidate, "worktree", paths=["docs/notes.md"])["status"] == "not_applicable" + assert _diagnose(candidate, "index", paths=["VERSION"])["status"] == "not_applicable" + + +@pytest.mark.parametrize("source", ["", "other"]) +def test_explicit_source_is_required_before_any_context_access(source): + result = _diagnose(object(), source) + assert result["failure_code"] == "PREFLIGHT_SOURCE_REQUIRED" + + +def test_schema_and_alias_offer_the_same_diagnostic_contract(): + entries = {entry.name: entry for entry in advisory.get_tools()} + canonical, alias = entries["preflight_review"], entries["advisory_review"] + assert canonical.schema["parameters"] == alias.schema["parameters"] + params = canonical.schema["parameters"]["properties"] + assert params["deterministic_only"]["default"] is False + assert params["source"]["enum"] == ["worktree", "index"] + assert "freshness" in params["deterministic_only"]["description"] + + +@pytest.mark.parametrize("prepared", [False, True]) +def test_normal_preflight_selects_worktree_or_prepared_index(candidate, monkeypatch, prepared): + repo = candidate.repo_dir + _broken(repo) + _git(repo, "add", ".") + _write(repo, _release("1.2.4")) + monkeypatch.setattr(advisory, "check_worktree_readiness", lambda *a, **kw: []) + monkeypatch.setattr(advisory, "_get_changed_file_list", lambda *a, **kw: "VERSION\nREADME.md") + warnings, changed, result = advisory._advisory_pre_sdk_gate( + candidate, repo, candidate.drive_root, "fixture", "release", ["VERSION", "README.md"], + skip_tests=True, prepared=prepared) + assert bool(result) == prepared + if prepared: + assert json.loads(result)["status"] == "preflight_blocked" + assert "no table row" in json.loads(result)["error"] + + +def test_normal_preflight_records_unavailable_as_failure_not_candidate_defect(candidate, monkeypatch): + _broken(candidate.repo_dir) + monkeypatch.setattr(advisory, "check_worktree_readiness", lambda *a, **kw: []) + monkeypatch.setattr(advisory, "_get_changed_file_list", lambda *a, **kw: "VERSION\nREADME.md") + (candidate.repo_dir / "VERSION").write_bytes(b"\xff") + _, _, result = advisory._advisory_pre_sdk_gate( + candidate, candidate.repo_dir, candidate.drive_root, "fixture", "release", ["VERSION"], True) + assert json.loads(result)["status"] == "error" + record = advisory.load_state(candidate.drive_root).advisory_runs[-1] + assert record.status == "error" and record.reason_kind == "release_metadata_unavailable" + assert not advisory.load_state(candidate.drive_root).is_fresh("fixture") + + +def test_author_continuation_formats_real_name_status_and_keeps_checks(candidate): + from ouroboros.tools.commit_gate import bind_author_commit_candidate + + repo = candidate.repo_dir + _broken(repo) + _git(repo, "add", ".") + candidate._author_commit_source = SimpleNamespace(block_reason="critical_findings", status="reviewed") + candidate._author_commit_decision = {"disposition": "accepted", "rationale": "Inspected original findings."} + candidate._author_commit_reference = {"attempt": 1} + result = bind_author_commit_candidate(candidate, "release", {"fingerprint": "current"}) + assert "no table row" in result and "patch rows (limit 5)" in result and "pyproject.toml" in result + _write(repo, _release("1.2.4")) + _git(repo, "add", ".") + assert bind_author_commit_candidate(candidate, "release", {"fingerprint": "corrected"}) is None + assert candidate._author_commit_record["subject_hash"] == "corrected" + assert candidate._author_commit_record["reviewer_signal"] == "critical_findings" + + +def test_release_validators_keep_optional_absence_and_detect_present_malformed_carriers(): + values = _release() + assert release_metadata_findings(values) == [] + values["web/package.json"] = "malformed" + values["uv.lock"] = "" + findings = "\n".join(release_metadata_findings(values)) + assert "web/package.json" in findings and "uv.lock is empty" in findings + + +@pytest.mark.parametrize("source", ["worktree", "index"]) +def test_missing_required_readme_is_unavailable_while_optional_older_carriers_are_absent(candidate, source): + repo = candidate.repo_dir + (repo / "VERSION").write_text("1.2.4\n", encoding="utf-8") + (repo / "README.md").unlink() + _git(repo, "add", "-A") + result = _diagnose(candidate, source) + assert result["status"] == "unavailable" + assert any("README.md is missing" in item for item in result["unavailable"]) + assert not any("package-lock.json" in item for item in result["unavailable"]) + assert any("pyproject.toml" in item for item in result["findings"]) + + +@pytest.mark.parametrize("path,content,expected", [ + ("ouroboros/new.py", "value = 1\n", "Architecture book"), + ("tests/conftest.py", "def test_misplaced(): pass\n", "contains test functions"), +]) +def test_author_formatter_also_preserves_non_release_staged_checks(candidate, path, content, expected): + from ouroboros.tools.commit_gate import bind_author_commit_candidate + + candidate._author_commit_source = SimpleNamespace(block_reason="critical_findings", status="reviewed") + candidate._author_commit_decision = {"disposition": "accepted", "rationale": "Read the review."} + candidate._author_commit_reference = {"attempt": 1} + _write(candidate.repo_dir, {path: content}) + _git(candidate.repo_dir, "add", ".") + result = bind_author_commit_candidate(candidate, "candidate", {"fingerprint": "current"}) + assert expected in result From d643dfd44f0491b62d6050cdce438d8220f4b35f Mon Sep 17 00:00:00 2001 From: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:12:20 +0300 Subject: [PATCH 02/15] fix: preserve release diagnostic classification and consumer fixtures --- docs/architecture/06-agent-core.md | 2 +- ouroboros/commit_admission.py | 15 +++++ ouroboros/tools/claude_advisory_review.py | 3 +- ouroboros/tools/git_review_cycle.py | 8 +-- ouroboros/tools/review.py | 5 +- tests/test_advisory_observability.py | 5 +- tests/test_e2e_live_runner.py | 5 ++ tests/test_git_review_enforcement.py | 58 +++++++++++++++++-- tests/test_release_metadata_diagnostics.py | 18 ++++++ .../test_tool_classification_differential.py | 6 ++ tests/tool_classification_corpus.py | 1 + 11 files changed, 113 insertions(+), 13 deletions(-) diff --git a/docs/architecture/06-agent-core.md b/docs/architecture/06-agent-core.md index 31cd56bcb..e2ac35370 100644 --- a/docs/architecture/06-agent-core.md +++ b/docs/architecture/06-agent-core.md @@ -359,7 +359,7 @@ The hermetic runner (`preflight_runner.py`) alone mints `ctx._preflight_test_pro `preflight_review(deterministic_only=True, source="worktree" | "index")` returns a release-only diagnostic report before sync, staging, custody/state access, provider checks, tests or fingerprints. `commit_admission.release_metadata_diagnostics` owns source acquisition and reports all independent applicable `findings` plus `unavailable` sources; `release_sync.release_metadata_findings` reuses the carrier validators, release grammar and P9 history counters. The report identifies its source and status (`clean`, `blocked`, `unavailable` or `not_applicable`), gives no review freshness, and writes no review history. It checks the actual selected VERSION, never an imagined next release. VERSION/README absence and failed reads are unavailable evidence; an absent optional older carrier stays optional, while a readable malformed carrier is a finding. Coverage is release metadata only, not syntax, structural size, tests or critic review; size policy stays warning-only locally. -Ordinary standalone advisory retains automatic carrier sync and reads worktree release metadata. Prepared advisory and the commit gate read index blobs, including partially staged files; the version-neutral staged lane and standalone documentation-only carve keep their existing applicability. The Optional[str] preflight wrappers format the same complete release findings and distinguish unavailable evidence from candidate defects. Author continuation uses the shared name-status formatter before these same staged checks. Changelog prose, history trimming and version allocation remain deliberate edits. +Standalone advisory retains automatic carrier sync and worktree reads. Prepared advisory now follows the commit gate's index applicability, including partial staging and the version-neutral carve, rather than its former worktree rule; standalone documentation-only scope stays exempt. A present malformed VERSION is now a finding, not a silent skip. Optional[str] wrappers format the complete release findings, separating unavailable evidence from defects. Author continuation uses the shared name-status formatter and unavailable classification. Changelog prose, history trimming and version allocation remain deliberate. Advisory availability is evaluated from the current configured slot and route, never inferred from a stale stored verdict: a disabled advisory slot is an audited bypass, an `api_chat` row requires provider credentials for its RESOLVED model, an `agent_session` row a resolvable session route. `claude_advisory_review.preflight_review` (callable `advisory_review` alias) owns that admission policy: an `api_chat` row rides the native tool-round episode (§6 Review stack), an `agent_session` row the session review executor; a native episode that ends on its own transcript bound — keyed on the structured `native_transcript_cap_exceeded` code, never on message text — reaches the caller as the typed non-blocking `ADVISORY_SKIPPED` with reason `native_transcript_bound_exceeded`, carrying the bound, refused chars and paid rounds, and every episode exception keeps `failure_custody()` as the advisory meta's `usage`, never an empty `{}`. The retrieving advisory brief carries a touched-path manifest instead of duplicated file bodies and applies the shared span-only release-carrier cut over HEAD→working-tree, with the same `PACK EXCLUSION NOTE`; omitted bodies remain readable through its own tools. Governance comes from the shared tiers (§6 Governance delivery). If the commit advisory is unavailable, the commit gate runs its compensating hermetic preflight only when tests remain independently applicable (not explicitly skipped, diff not documentation-only). Readiness projections receive only an exact repo/hash-matched advisory record and keep its failed status and freshness. diff --git a/ouroboros/commit_admission.py b/ouroboros/commit_admission.py index 8b494df74..964aa9d19 100644 --- a/ouroboros/commit_admission.py +++ b/ouroboros/commit_admission.py @@ -184,6 +184,21 @@ def format_release_metadata_preflight(report: dict) -> Optional[str]: + "".join(f" - Unavailable: {message}\n" for message in report["unavailable"])) +def preflight_evidence_unavailable(message: Optional[str]) -> bool: + """Whether a preflight message reports unavailable evidence, not a candidate defect. + + The two admission gates need that split (an unreadable source is an infra + failure, a bad carrier is the candidate's). Ask the one tool-result + classifier for the code the agent will see, so neither gate grows a second + private reading of the same warning text. + """ + from ouroboros.tools.tool_result import LegacyTextResultAdapter + + return bool(message) and LegacyTextResultAdapter.from_text( + "preflight_review", message, + ).status == "unavailable" + + def release_metadata_preflight( repo_dir: pathlib.Path, commit_message: str, paths: list[str] | None, *, source: str = "worktree", diff --git a/ouroboros/tools/claude_advisory_review.py b/ouroboros/tools/claude_advisory_review.py index ae62b6cd7..7ec3b3182 100644 --- a/ouroboros/tools/claude_advisory_review.py +++ b/ouroboros/tools/claude_advisory_review.py @@ -1016,7 +1016,8 @@ def _advisory_pre_sdk_gate( release_preflight_err = (_release_metadata_preflight(repo_dir, commit_message, paths, source="index") if prepared else _release_metadata_preflight(repo_dir, commit_message, paths)) if release_preflight_err: - unavailable = release_preflight_err.startswith("⚠️ PREFLIGHT_UNAVAILABLE:") + from ouroboros.commit_admission import preflight_evidence_unavailable + unavailable = preflight_evidence_unavailable(release_preflight_err) status = "error" if unavailable else "preflight_blocked" ctx.emit_progress_fn(release_preflight_err) _persist_preflight_record( diff --git a/ouroboros/tools/git_review_cycle.py b/ouroboros/tools/git_review_cycle.py index 13cacaff0..f06c530c4 100644 --- a/ouroboros/tools/git_review_cycle.py +++ b/ouroboros/tools/git_review_cycle.py @@ -687,16 +687,16 @@ def _run_reviewed_stage_cycle( # Free-cycle identity runs before advisory freshness and any paid dispatch. author_source = getattr(ctx, "_author_commit_source", None) gate_outcome = None if author_source is not None else _git()._free_cycle_gate( - ctx, commit_message, commit_start, - pre_fingerprint=pre_fingerprint, review_rebuttal=review_rebuttal, - goal=goal, scope=scope, + ctx, commit_message, commit_start, pre_fingerprint=pre_fingerprint, + review_rebuttal=review_rebuttal, goal=goal, scope=scope, ) advisory_replay: Optional[Dict[str, Any]] = None if author_source is not None: from ouroboros.tools.commit_gate import bind_author_commit_candidate preflight = bind_author_commit_candidate(ctx, commit_message, pre_fingerprint) if preflight: - return {"status": "blocked", "message": preflight, "block_reason": "preflight"} + from ouroboros.commit_admission import preflight_evidence_unavailable + return {"status": "blocked", "message": preflight, "block_reason": "infra_failure" if preflight_evidence_unavailable(preflight) else "preflight"} advisory_replay = {"advisory_replay": "Explicit current-author continuation; original reviewer facts retained.", "replay_reason": "author_finish"} skip_advisory_pre_review = True if gate_outcome is not None: diff --git a/ouroboros/tools/review.py b/ouroboros/tools/review.py index 58b5e3b5d..27d615adc 100644 --- a/ouroboros/tools/review.py +++ b/ouroboros/tools/review.py @@ -965,7 +965,10 @@ def _prepare_unified_review(ctx: ToolContext, commit_message: str, preflight_err = _preflight_check(commit_message, preflight_staged, target_repo) if preflight_err: - ctx._last_review_block_reason = ("infra_failure" if preflight_err.startswith("⚠️ PREFLIGHT_UNAVAILABLE:") else "preflight") + from ouroboros.commit_admission import preflight_evidence_unavailable + ctx._last_review_block_reason = ( + "infra_failure" if preflight_evidence_unavailable(preflight_err) else "preflight" + ) result = _handle_review_block_or_warning( ctx, blocking_review, preflight_err, "Review enforcement=Advisory: preflight warning did not block commit. ", diff --git a/tests/test_advisory_observability.py b/tests/test_advisory_observability.py index dba1b7348..9d956c3b6 100644 --- a/tests/test_advisory_observability.py +++ b/tests/test_advisory_observability.py @@ -1131,7 +1131,10 @@ class TestAdvisoryCleanSentinel: lambda repo_dir, commit_message, ctx, **kwargs: (adv._parse_advisory_output(raw_text), raw_text, "opus", 10), ) # Release-metadata preflight (BIBLE P9) runs before the SDK branch under - # test, so the fake change set must carry the release artifacts. + # test. These cases pin the sentinel verdict, not release admission, and + # ``tmp_path`` is no Git worktree — isolate the admission read rather than + # let its honest "source unavailable" answer stand in for a verdict. + monkeypatch.setattr(adv, "_release_metadata_preflight", lambda *a, **kw: None) monkeypatch.setattr(adv, "_get_staged_diff", lambda repo_dir, paths=None: "diff --git a/x.py b/x.py") monkeypatch.setattr( adv, "_get_changed_file_list", diff --git a/tests/test_e2e_live_runner.py b/tests/test_e2e_live_runner.py index a0844b879..9aa786a05 100644 --- a/tests/test_e2e_live_runner.py +++ b/tests/test_e2e_live_runner.py @@ -1114,6 +1114,11 @@ def test_sm1_stub_bumps_the_release_carriers_through_the_sync_ssot(tmp_path): bumped = carriers["VERSION"].strip() assert scenarios.version_is_bumped(seed, bumped) and f"| {bumped} |" in carriers["README.md"] root = tmp_path / "carriers" + root.mkdir() + # The gate under test is release admission on a Git worktree: it reads the + # candidate's own change scope from Git, so the carrier set is materialized + # in a real (disposable) worktree rather than a bare directory. + subprocess.run(["git", "init", "-q"], cwd=str(root), check=True) for rel in sorted(CARRIER_SPAN_PATHS): if (REPO_ROOT / rel).is_file(): (root / rel).parent.mkdir(parents=True, exist_ok=True) diff --git a/tests/test_git_review_enforcement.py b/tests/test_git_review_enforcement.py index 2d5ca58c0..ae1160591 100644 --- a/tests/test_git_review_enforcement.py +++ b/tests/test_git_review_enforcement.py @@ -275,6 +275,31 @@ class TestReviewEnforcementModes: for w in ctx._review_advisory ) + @pytest.mark.parametrize("message,expected", [ + ("⚠️ PREFLIGHT_BLOCKED: Release metadata diagnostics (index).\n" + " - Missing from staged: README.md (badge + changelog).\n", "preflight"), + ("⚠️ PREFLIGHT_UNAVAILABLE: Release metadata diagnostics (index).\n" + " - Unavailable: index:README.md could not be read (CalledProcessError).\n", + "infra_failure"), + ]) + def test_preflight_block_reason_separates_unavailable_evidence_from_a_defect( + self, review_ctx, monkeypatch, message, expected + ): + """A release source the gate could not read is an infra failure, not the + candidate's own defect. The split is read off the one tool-result + classifier, so this gate cannot drift from the code the agent is shown.""" + review, ctx = review_ctx + self._mock_staged(monkeypatch, review, changed_files="VERSION") + monkeypatch.setattr(review, "_preflight_check", lambda *a, **kw: message) + monkeypatch.setenv("OUROBOROS_REVIEW_ENFORCEMENT", "blocking") + monkeypatch.setattr( + review, "_handle_multi_model_review", + lambda *a, **k: (_ for _ in ()).throw(AssertionError("no reviewer may run")), + ) + result = review._run_unified_review(ctx, "v1.0.0: bump version", repo_dir=ctx.repo_dir) + assert result is not None and message in result + assert ctx._last_review_block_reason == expected + def test_triad_one_pass_fit_removes_only_duplicated_context(self, review_ctx, monkeypatch): """Oversized triad evidence is compacted before its single dispatch.""" review, ctx = review_ctx @@ -522,14 +547,33 @@ class TestReviewEnforcementModes: # No version-ref in commit message, so no preflight block expected assert result is None + @staticmethod + def _mock_indexed_release(monkeypatch, review, *, readme: bool, version: str = "3.24.0"): + """Give the name-status checks an index of their own. + + ``_preflight_check`` reads its release carriers out of the real Git + index, and an index it cannot read is honestly reported as unavailable + evidence. These cases pin the lexical name-status handling, so they + supply the indexed carriers instead of leaving admission to fail on a + directory that is not a repository. + """ + indexed = {"VERSION": f"{version}\n"} + if readme: + indexed["README.md"] = ( + f"[![Version {version}](https://img.shields.io/badge/version-{version}-green.svg)]\n" + f"| {version} | release |\n" + ) + monkeypatch.setattr(review, "_git_show_staged", lambda repo_dir, path: indexed.get(path)) + def test_rename_of_readme_counts_as_present(self, tmp_path, monkeypatch): """If README.md appears as a rename destination, preflight sees it as staged.""" review = _get_review_module() + self._mock_indexed_release(monkeypatch, review, readme=True) # Simulate: VERSION staged + README.md arrived via rename result = review._preflight_check( "v1.0.0: rename readme", "M VERSION\nR README.md", - "/tmp", + tmp_path, ) # Both VERSION and README.md present → no check 1 block # No ouroboros .py → no check 3 block @@ -593,17 +637,21 @@ class TestReviewEnforcementModes: assert "PREFLIGHT_BLOCKED" in result assert "ARCHITECTURE.md" in result - def test_deleted_readme_does_not_satisfy_check1(self): + def test_deleted_readme_does_not_satisfy_check1(self, tmp_path, monkeypatch): """Deleting README.md while VERSION is staged triggers check 1.""" review = _get_review_module() + self._mock_indexed_release(monkeypatch, review, readme=False) result = review._preflight_check( "v1.0.0: bump version", "M VERSION\nD README.md", - "/tmp", + tmp_path, ) assert result is not None - assert "PREFLIGHT_BLOCKED" in result - assert "README.md" in result + assert "Missing from staged: README.md" in result + # The deleted README is also the release source the carrier checks need, + # so its absence is reported as unavailable evidence beside the finding + # rather than collapsing the two into one candidate defect. + assert "PREFLIGHT_UNAVAILABLE" in result def test_copied_module_triggers_via_run_unified_review(self, tmp_path, monkeypatch): """Check 4 fires for C-status copy via _run_unified_review, but source NOT treated as deleted.""" diff --git a/tests/test_release_metadata_diagnostics.py b/tests/test_release_metadata_diagnostics.py index 608246959..dd3cc3acf 100644 --- a/tests/test_release_metadata_diagnostics.py +++ b/tests/test_release_metadata_diagnostics.py @@ -351,3 +351,21 @@ def test_author_formatter_also_preserves_non_release_staged_checks(candidate, pa _git(candidate.repo_dir, "add", ".") result = bind_author_commit_candidate(candidate, "candidate", {"fingerprint": "current"}) assert expected in result + + +@pytest.mark.parametrize("message,reason", [ + ("⚠️ PREFLIGHT_UNAVAILABLE: index evidence unavailable", "infra_failure"), + ("⚠️ PREFLIGHT_BLOCKED: malformed carrier", "preflight"), +]) +def test_author_stage_cycle_preserves_preflight_failure_kind(candidate, monkeypatch, message, reason): + from ouroboros.tools import commit_gate, git_review_cycle + + facade = git_review_cycle._git() + candidate._author_commit_source = object() + monkeypatch.setattr(facade, "_stage_candidate_for_review", lambda *a, **kw: ([], [], None)) + monkeypatch.setattr(facade, "protected_paths_in", lambda paths: []) + monkeypatch.setattr(facade, "_current_runtime_mode", lambda: "pro") + monkeypatch.setattr(facade, "_fingerprint_staged_diff", lambda repo: {"ok": True}) + monkeypatch.setattr(commit_gate, "bind_author_commit_candidate", lambda *a: message) + result = git_review_cycle._run_reviewed_stage_cycle(candidate, "release", 0.0) + assert result == {"status": "blocked", "message": message, "block_reason": reason} diff --git a/tests/test_tool_classification_differential.py b/tests/test_tool_classification_differential.py index 169b1d3dd..9e7c00244 100644 --- a/tests/test_tool_classification_differential.py +++ b/tests/test_tool_classification_differential.py @@ -381,6 +381,12 @@ CURRENT_PRODUCER_CONTRACTS = { "SCOPE_UNCONFIRMED": (True, "tool_reported_failure"), "TOOL_ERROR": (True, "error"), "native:TOOL_REPORTED_FAILURE:TOOL_ERROR": (True, "tool_reported_failure"), + # Release admission split its one PREFLIGHT_BLOCKED text in two: a source it + # could not read is unavailable evidence, not a candidate defect. The new + # identifier reaches its text through the `code` variable, so it is declared + # in the corpus' interpolated list and answered live here — the retired pair + # never saw a tree that emitted it. + "PREFLIGHT_UNAVAILABLE": (True, "unavailable"), } diff --git a/tests/tool_classification_corpus.py b/tests/tool_classification_corpus.py index 48ade83e2..621d6c186 100644 --- a/tests/tool_classification_corpus.py +++ b/tests/tool_classification_corpus.py @@ -93,6 +93,7 @@ _CODE_RE = re.compile(r"^[A-Z][A-Z0-9_]*$") _INTERPOLATED_IDENTIFIERS = ( "APPLY_PATCH_BLOCKED", # tools/edit_ops.py::apply_patch error_tag "EDIT_BATCH_BLOCKED", # tools/edit_ops.py::edit_batch error_tag + "PREFLIGHT_UNAVAILABLE", # commit_admission.py::format_release_metadata_preflight code "READ_FILE_BLOCKED", # tools/core_file_tools.py::_local_readonly_resource_block action "SCRIPT_CWD_BLOCKED", # tools/tool_resolution.py::_binding_error_text prefixes "SEARCH_BLOCKED", # tools/core.py search_code, same action argument From 9e09eba789c9e879a302b25ac13d78ab69f91b0e Mon Sep 17 00:00:00 2001 From: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> Date: Wed, 23 Sep 2026 03:49:04 +0300 Subject: [PATCH 03/15] fix: preserve exported release checks and unavailable-source guidance --- devtools/e2e_live/scenarios.py | 2 ++ .../01-high-level-architecture.md | 2 +- docs/architecture/06-agent-core.md | 4 ++-- .../05-review-and-commit-protocol.md | 2 +- docs/inventories/DATA_LAYOUT_INVENTORY.md | 2 +- ouroboros/commit_admission.py | 4 ++-- ouroboros/review_state_records.py | 3 ++- ouroboros/tools/claude_advisory_review.py | 8 +++++-- ouroboros/tools/commit_gate.py | 4 +++- tests/test_capinv447_ws2e.py | 23 +++++++++++++------ tests/test_e2e_live_runner.py | 8 ++++--- tests/test_release_metadata_diagnostics.py | 9 ++++++++ 12 files changed, 50 insertions(+), 21 deletions(-) diff --git a/devtools/e2e_live/scenarios.py b/devtools/e2e_live/scenarios.py index 5379c1169..486a83a88 100644 --- a/devtools/e2e_live/scenarios.py +++ b/devtools/e2e_live/scenarios.py @@ -488,6 +488,8 @@ def release_carriers_desync_at(clone: pathlib.Path, rev: str) -> str: with tempfile.TemporaryDirectory(prefix="sm1_carriers_") as tmp: root = pathlib.Path(tmp) + # Admission reads Git change scope; the export needs its own disposable repository. + subprocess.run(["git", "init", "-q"], cwd=str(root), check=True, capture_output=True) for rel in sorted(CARRIER_SPAN_PATHS): text = _git_show(clone, rev, rel) if text: diff --git a/docs/architecture/01-high-level-architecture.md b/docs/architecture/01-high-level-architecture.md index daabe4964..5620a8620 100644 --- a/docs/architecture/01-high-level-architecture.md +++ b/docs/architecture/01-high-level-architecture.md @@ -437,7 +437,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de │ ├── process_facts.py ← Per-call selected environment, secret egress masking and typed process/runtime facts consumed by loop_tool_execution; the regex harvest stays a read fallback │ ├── write_shape.py ← Retained interpreter/non-interpreter syntax helpers; process permission is owned by the task/resource and Supervisor contract (§6 Safety and runtime mode) │ ├── extension_dispatch.py ← Extension tool dispatch (contracts preserved; discovery stays in registry.py) - │ ├── release_sync.py ← `sync_release_metadata` (version carriers) used by commit-admission preflight; `_preflight_check` uses `check_history_limit`; the carrier-span SSOT (`VERSION_CARRIER_SPANS`, `substitute_carrier_spans`, the `carrier_only_change` predicate) shared by the managed-update resolver and the commit-triad pack cut (§10 invariant 2) + │ ├── release_sync.py ← `sync_release_metadata` (version carriers) used by commit-admission preflight; shared `release_metadata_findings` evaluator; the carrier-span SSOT (`VERSION_CARRIER_SPANS`, `substitute_carrier_spans`, the `carrier_only_change` predicate) shared by the managed-update resolver and the commit-triad pack cut (§10 invariant 2) │ ├── review_synthesis.py ← Shared synthesis helpers; the parser/aggregator lives in plan_spec.py │ ├── ci.py ← CI trigger/monitoring │ ├── claude_advisory_review.py ← `preflight_review` (callable `advisory_review` alias) and its admission policy: api_chat rows ride review_native_episode, agent_session rows the AgentSessionReviewExecutor; a native episode ending on its transcript bound is the typed non-blocking `ADVISORY_SKIPPED: native_transcript_bound_exceeded` (keyed on the structured `native_transcript_cap_exceeded` code), and every episode exception keeps `failure_custody()` as the advisory `usage`; the MANDATORY FULL READ corpus is measured at prompt build (`preflight_review_prompt._mandatory_read_corpus_chars`) and declared to the episode, which never raises the send bound — a shortfall shows in the prompt's MANDATORY READ budget and the usage as `native_multiple_windows_required` (§6 Commit advisory cycle, Native tool-round episode) diff --git a/docs/architecture/06-agent-core.md b/docs/architecture/06-agent-core.md index e2ac35370..0e51c333c 100644 --- a/docs/architecture/06-agent-core.md +++ b/docs/architecture/06-agent-core.md @@ -357,9 +357,9 @@ The hermetic runner (`preflight_runner.py`) alone mints `ctx._preflight_test_pro #### Commit advisory cycle -`preflight_review(deterministic_only=True, source="worktree" | "index")` returns a release-only diagnostic report before sync, staging, custody/state access, provider checks, tests or fingerprints. `commit_admission.release_metadata_diagnostics` owns source acquisition and reports all independent applicable `findings` plus `unavailable` sources; `release_sync.release_metadata_findings` reuses the carrier validators, release grammar and P9 history counters. The report identifies its source and status (`clean`, `blocked`, `unavailable` or `not_applicable`), gives no review freshness, and writes no review history. It checks the actual selected VERSION, never an imagined next release. VERSION/README absence and failed reads are unavailable evidence; an absent optional older carrier stays optional, while a readable malformed carrier is a finding. Coverage is release metadata only, not syntax, structural size, tests or critic review; size policy stays warning-only locally. +`preflight_review(commit_message="...", deterministic_only=True, source="worktree" | "index")` returns a release-only diagnostic report before sync, staging, custody/state access, provider checks, tests or fingerprints. `commit_admission.release_metadata_diagnostics` owns source acquisition and reports all independent applicable `findings` plus `unavailable` sources; `release_sync.release_metadata_findings` reuses the carrier validators, release grammar and P9 history counters. The report identifies its source and status (`clean`, `blocked`, `unavailable` or `not_applicable`), gives no review freshness, and writes no review history. It checks the actual selected VERSION, never an imagined next release. VERSION/README absence and failed reads are unavailable evidence; an absent optional older carrier stays optional, while a readable malformed carrier is a finding. Coverage is release metadata only, not syntax, structural size, tests or critic review; size policy stays warning-only locally. -Standalone advisory retains automatic carrier sync and worktree reads. Prepared advisory now follows the commit gate's index applicability, including partial staging and the version-neutral carve, rather than its former worktree rule; standalone documentation-only scope stays exempt. A present malformed VERSION is now a finding, not a silent skip. Optional[str] wrappers format the complete release findings, separating unavailable evidence from defects. Author continuation uses the shared name-status formatter and unavailable classification. Changelog prose, history trimming and version allocation remain deliberate. +Standalone advisory retains automatic carrier sync and worktree reads. Prepared advisory follows the commit gate's whole-index applicability, regardless of a narrower paths hint, including partial staging and the version-neutral carve; standalone documentation-only scope stays exempt. A present malformed VERSION is now a finding, not a silent skip. Optional[str] wrappers format the complete release findings, separating unavailable evidence from defects. Author continuation uses the shared name-status formatter and unavailable classification. Changelog prose, history trimming and version allocation remain deliberate. Advisory availability is evaluated from the current configured slot and route, never inferred from a stale stored verdict: a disabled advisory slot is an audited bypass, an `api_chat` row requires provider credentials for its RESOLVED model, an `agent_session` row a resolvable session route. `claude_advisory_review.preflight_review` (callable `advisory_review` alias) owns that admission policy: an `api_chat` row rides the native tool-round episode (§6 Review stack), an `agent_session` row the session review executor; a native episode that ends on its own transcript bound — keyed on the structured `native_transcript_cap_exceeded` code, never on message text — reaches the caller as the typed non-blocking `ADVISORY_SKIPPED` with reason `native_transcript_bound_exceeded`, carrying the bound, refused chars and paid rounds, and every episode exception keeps `failure_custody()` as the advisory meta's `usage`, never an empty `{}`. The retrieving advisory brief carries a touched-path manifest instead of duplicated file bodies and applies the shared span-only release-carrier cut over HEAD→working-tree, with the same `PACK EXCLUSION NOTE`; omitted bodies remain readable through its own tools. Governance comes from the shared tiers (§6 Governance delivery). If the commit advisory is unavailable, the commit gate runs its compensating hermetic preflight only when tests remain independently applicable (not explicitly skipped, diff not documentation-only). Readiness projections receive only an exact repo/hash-matched advisory record and keep its failed status and freshness. diff --git a/docs/development/05-review-and-commit-protocol.md b/docs/development/05-review-and-commit-protocol.md index 80546a2da..1b231e8ae 100644 --- a/docs/development/05-review-and-commit-protocol.md +++ b/docs/development/05-review-and-commit-protocol.md @@ -32,7 +32,7 @@ The authoring agent freezes the final committed base-to-head range and gives it ### Release sync -A pull request into `ouroboros` leaves every version carrier byte-identical to its target (the carrier list and the one projection that writes them: ARCHITECTURE §10, invariant 2). Use `preflight_review(deterministic_only=True, source="worktree" | "index")` for release diagnostics before review spend; it grants no freshness. Keep source failures separate from candidate findings and verify partial staging plus unchanged files/index/state (`tests/test_release_metadata_diagnostics.py`; source/applicability contracts: ARCHITECTURE §6 "Commit advisory cycle"). At integration, `release_sync.sync_release_metadata()` projects the chosen version; its shared evaluator checks carriers, the current README row and P9 limits. Changelog prose and trimming remain deliberate edits; diagnostics add no autofix. +A pull request into `ouroboros` leaves every version carrier byte-identical to its target (the carrier list and the one projection that writes them: ARCHITECTURE §10, invariant 2). Use `preflight_review(commit_message="...", deterministic_only=True, source="worktree" | "index")` for release diagnostics before review spend; it grants no freshness. Keep source failures separate from candidate findings and verify partial staging plus unchanged files/index/state (`tests/test_release_metadata_diagnostics.py`; source/applicability contracts: ARCHITECTURE §6 "Commit advisory cycle"). At integration, `release_sync.sync_release_metadata()` projects the chosen version; its shared evaluator checks carriers, the current README row and P9 limits. Changelog prose and trimming remain deliberate; diagnostics add no autofix. Packaging: ARCHITECTURE §8 “Build scripts”. Hermetic preflight uses a disposable worktree, temporary data/settings/pycache, and a scrubbed runtime/secret-class environment. Tests must rebind imported process-global roots and fail closed on the live data root; setting only `OUROBOROS_DATA_DIR` is insufficient. A reviewed local commit is the durability boundary; an `origin` push and CI are follow-up signals, not prerequisites for local self-modification survival. diff --git a/docs/inventories/DATA_LAYOUT_INVENTORY.md b/docs/inventories/DATA_LAYOUT_INVENTORY.md index 40488726d..ccfae73b4 100644 --- a/docs/inventories/DATA_LAYOUT_INVENTORY.md +++ b/docs/inventories/DATA_LAYOUT_INVENTORY.md @@ -2,7 +2,7 @@ Machine extraction of the `docs/ARCHITECTURE.md` "Data layout (`~/Ouroboros/`)" tree — the durable-file orientation carrier (this tree's counterpart of the reference PERSISTENCE_OWNERS derivation checklist) — regenerated by `python scripts/regenerate_inventories.py`. Do not edit. Every entry is probed against reality: repo entries must exist as tracked paths; data-plane entries must appear as a literal in the runtime sources that construct them. A durable file renamed or removed in code while its tree row survives = red (`tests/test_generated_inventories.py`). -Source: `docs/architecture/01-high-level-architecture.md`, physical LF lines 596-685; UTF-8 SHA-256 `6aac224db5ce90908de24832bb3bbebe21be6ae87bbe8e6a35cff606615e31fd`. +Source: `docs/architecture/01-high-level-architecture.md`, physical LF lines 596-685; UTF-8 SHA-256 `73e06bc0d2f0d0c68930ca59a26f7436de44ada0e8c82b9e6043f49797f8b4e0`. - entries: **79** (code-ref: 72, repo-dir: 6, repo-path: 1) diff --git a/ouroboros/commit_admission.py b/ouroboros/commit_admission.py index 964aa9d19..0f4577d2b 100644 --- a/ouroboros/commit_admission.py +++ b/ouroboros/commit_admission.py @@ -131,8 +131,8 @@ def release_metadata_diagnostics( touched.update(changed_worktree_paths(repo_dir, paths=paths, strict=True)) elif read_text is None: result = subprocess.run( - ["git", "diff", "--cached", "--name-only", "--diff-filter=d", "-z"] - + (["--", *paths] if paths else []), cwd=str(repo_dir), + ["git", "--no-optional-locks", "diff", "--cached", "--name-only", "--diff-filter=d", "-z"], + cwd=str(repo_dir), capture_output=True, encoding="utf-8", timeout=10, check=True, ) touched.update(filter(None, result.stdout.split("\0"))) diff --git a/ouroboros/review_state_records.py b/ouroboros/review_state_records.py index e8ef19f07..1e9b5c252 100644 --- a/ouroboros/review_state_records.py +++ b/ouroboros/review_state_records.py @@ -248,7 +248,8 @@ class AdvisoryRunRecord: snapshot_summary: str = "" raw_result: str = "" # Typed cause for status="preflight_blocked" rows: "syntax" (a staged .py - # failed compile) or "release_metadata" (deterministic release preflight). + # failed compile), "release_metadata" (defect); status="error" also carries + # "release_metadata_unavailable" when the required source could not be read. # "" = unknown/legacy — guidance must then point at raw_result instead of # asserting a specific problem class (H4, capinv-447). reason_kind: str = "" diff --git a/ouroboros/tools/claude_advisory_review.py b/ouroboros/tools/claude_advisory_review.py index 7ec3b3182..0099edfe5 100644 --- a/ouroboros/tools/claude_advisory_review.py +++ b/ouroboros/tools/claude_advisory_review.py @@ -819,7 +819,8 @@ def _next_step_guidance(latest: Optional["AdvisoryRunRecord"], state: "AdvisoryR if not effective_is_fresh: status = str(getattr(latest, "status", "") or "") - if latest and status in {"tests_preflight_blocked", "preflight_blocked"} and not stale_from_edit: + if latest and not stale_from_edit and (status in {"tests_preflight_blocked", "preflight_blocked"} or + latest.reason_kind == "release_metadata_unavailable"): if status == "tests_preflight_blocked": problem = "test preflight: pytest failed before the paid critic call" fix = "Fix the failing tests and re-run preflight_review. Use preflight_review(skip_tests=True) only for intentional WIP code." @@ -831,8 +832,11 @@ def _next_step_guidance(latest: Optional["AdvisoryRunRecord"], state: "AdvisoryR if reason_kind == "syntax": problem = "syntax preflight: a staged .py file has a SyntaxError" fix = "See raw_result for file:line:msg, fix it, and re-run preflight_review." + elif reason_kind == "release_metadata_unavailable": + problem = "unavailable release metadata evidence, not a candidate verdict" + fix = "Restore access to the sources named in raw_result and re-run preflight_review." elif reason_kind == "release_metadata": - problem = "release metadata preflight: inspect all findings with preflight_review(deterministic_only=True, source=worktree or index)" + problem = "release metadata preflight: inspect all findings with preflight_review(commit_message='...', deterministic_only=True, source='worktree' or 'index')" fix = "See raw_result for the exact carrier mismatch, fix it, and re-run preflight_review." else: problem = "a deterministic preflight check (see raw_result for the exact cause)" diff --git a/ouroboros/tools/commit_gate.py b/ouroboros/tools/commit_gate.py index 1d679523d..842994629 100644 --- a/ouroboros/tools/commit_gate.py +++ b/ouroboros/tools/commit_gate.py @@ -1091,7 +1091,8 @@ def _check_advisory_freshness(ctx: ToolContext, commit_message: str, "Or bypass: commit_reviewed(commit_message='...', skip_advisory_review=True) (audited)." ) - if matching_run and matching_run.status == "preflight_blocked": + if matching_run and (matching_run.status == "preflight_blocked" or + matching_run.reason_kind == "release_metadata_unavailable"): preflight_detail = (matching_run.raw_result or "").strip() # H4 (capinv-447): the status is shared by several deterministic checks; # name the problem class only when the typed cause is recorded. @@ -1099,6 +1100,7 @@ def _check_advisory_freshness(ctx: ToolContext, commit_message: str, cause = { "syntax": "The advisory delivery was skipped because a staged `.py` file has a SyntaxError.", "release_metadata": "The advisory delivery was skipped because the deterministic release metadata preflight failed.", + "release_metadata_unavailable": "Release metadata evidence could not be read; this is not a reviewer verdict or proof of a changed snapshot.", }.get(reason_kind, "The advisory delivery was skipped by a deterministic preflight check (exact cause below).") return ( f"⚠️ ADVISORY_PRE_REVIEW_REQUIRED: Last advisory run for this snapshot " diff --git a/tests/test_capinv447_ws2e.py b/tests/test_capinv447_ws2e.py index bc29b15cb..7be2e07b2 100644 --- a/tests/test_capinv447_ws2e.py +++ b/tests/test_capinv447_ws2e.py @@ -115,14 +115,14 @@ def test_module_load_failure_recorded_and_survives_schema_rebuilds(tmp_path, mon # H4 — typed preflight_blocked reason_kind # --------------------------------------------------------------------------- -def _guidance_for(reason_kind: str) -> str: +def _guidance_for(reason_kind: str, status: str = "preflight_blocked") -> str: from ouroboros.review_state import AdvisoryReviewState, AdvisoryRunRecord from ouroboros.tools.claude_advisory_review import _next_step_guidance latest = AdvisoryRunRecord( snapshot_hash="cafe" * 4, commit_message="m", - status="preflight_blocked", + status=status, ts="2026-09-01T00:00:00Z", raw_result="detail text", reason_kind=reason_kind, @@ -140,13 +140,20 @@ def test_release_metadata_block_never_claims_syntax_error(): assert "release metadata" in guidance +def test_unavailable_release_guidance_preserves_the_failure_kind(): + guidance = _guidance_for("release_metadata_unavailable", status="error") + assert "unavailable release metadata evidence" in guidance + assert "Restore access" in guidance + + def test_untyped_preflight_block_stays_generic(): guidance = _guidance_for("") assert "SyntaxError" not in guidance assert "raw_result" in guidance -def test_commit_gate_block_message_branches_on_reason_kind(tmp_path, monkeypatch): +@pytest.mark.parametrize("unavailable", [False, True]) +def test_commit_gate_block_message_branches_on_reason_kind(tmp_path, monkeypatch, unavailable): from ouroboros.review_state import AdvisoryRunRecord, compute_snapshot_hash, load_state, make_repo_key, save_state from ouroboros.tools.commit_gate import _check_advisory_freshness @@ -160,16 +167,18 @@ def test_commit_gate_block_message_branches_on_reason_kind(tmp_path, monkeypatch state = load_state(tmp_path) state.add_run(AdvisoryRunRecord( snapshot_hash=snapshot_hash, commit_message="msg", - status="preflight_blocked", ts="2026-09-01T00:00:00Z", - raw_result="⚠️ PREFLIGHT_BLOCKED: VERSION is 1.0 but README says 0.9", - reason_kind="release_metadata", repo_key=make_repo_key(repo), + status="error" if unavailable else "preflight_blocked", ts="2026-09-01T00:00:00Z", + raw_result="exact release source diagnostic", + reason_kind="release_metadata_unavailable" if unavailable else "release_metadata", repo_key=make_repo_key(repo), )) save_state(tmp_path, state) message = _check_advisory_freshness(ctx, "msg") assert message is not None assert "SyntaxError" not in message - assert "release metadata preflight failed" in message + assert "exact release source diagnostic" in message + assert "Snapshot changed" not in message + assert ("evidence could not be read" if unavailable else "release metadata preflight failed") in message # --------------------------------------------------------------------------- diff --git a/tests/test_e2e_live_runner.py b/tests/test_e2e_live_runner.py index 9aa786a05..e887050bd 100644 --- a/tests/test_e2e_live_runner.py +++ b/tests/test_e2e_live_runner.py @@ -1115,15 +1115,17 @@ def test_sm1_stub_bumps_the_release_carriers_through_the_sync_ssot(tmp_path): assert scenarios.version_is_bumped(seed, bumped) and f"| {bumped} |" in carriers["README.md"] root = tmp_path / "carriers" root.mkdir() - # The gate under test is release admission on a Git worktree: it reads the - # candidate's own change scope from Git, so the carrier set is materialized - # in a real (disposable) worktree rather than a bare directory. + # Release admission reads Git scope, so materialize carriers in a disposable + # repository rather than a bare directory. subprocess.run(["git", "init", "-q"], cwd=str(root), check=True) for rel in sorted(CARRIER_SPAN_PATHS): if (REPO_ROOT / rel).is_file(): (root / rel).parent.mkdir(parents=True, exist_ok=True) (root / rel).write_text(carriers.get(rel) or (REPO_ROOT / rel).read_text(encoding="utf-8"), encoding="utf-8") assert release_metadata_preflight(root, scenarios.SM1_COMMIT_MESSAGE, ["VERSION"]) is None + assert scenarios.release_carriers_desync_at(root, _commit(root, "coherent release")) == "" + (root / "pyproject.toml").write_text('[project]\nversion = "0.0.0"\n', encoding="utf-8") + assert "pyproject.toml" in scenarios.release_carriers_desync_at(root, _commit(root, "broken carrier")) assert scenarios.sm1_next_version("7.0.0-rc.14") == "7.0.0-rc.15" and scenarios.sm1_next_version("7.0.0") == "7.0.1" # A seed cloned from an older ref carries the newer tags: the stub skips taken versions. assert scenarios.sm1_next_version("7.0.0-rc.14", {"v7.0.0-rc.15", "v7.0.0-rc.16"}) == "7.0.0-rc.17" diff --git a/tests/test_release_metadata_diagnostics.py b/tests/test_release_metadata_diagnostics.py index dd3cc3acf..9cee659b6 100644 --- a/tests/test_release_metadata_diagnostics.py +++ b/tests/test_release_metadata_diagnostics.py @@ -369,3 +369,12 @@ def test_author_stage_cycle_preserves_preflight_failure_kind(candidate, monkeypa monkeypatch.setattr(commit_gate, "bind_author_commit_candidate", lambda *a: message) result = git_review_cycle._run_reviewed_stage_cycle(candidate, "release", 0.0) assert result == {"status": "blocked", "message": message, "block_reason": reason} + + +def test_index_checks_the_whole_staged_candidate_even_with_narrow_paths(candidate): + repo = candidate.repo_dir + _write(repo, _release("1.2.4")) + _git(repo, "add", ".") + report = admission.release_metadata_diagnostics(repo, ["VERSION"], source="index") + assert report["status"] == "clean" + assert report["findings"] == [] From db13aba405c56a5eccdb7743d0a45ef86cfa97e2 Mon Sep 17 00:00:00 2001 From: Ouroboros Date: Wed, 23 Sep 2026 04:52:18 +0300 Subject: [PATCH 04/15] Expose failed Presence turns to regular consciousness context Keep terminal authorship and cumulative memory scopes explicit in the architecture and adapter guidance. Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> --- docs/CREATING_SKILLS.md | 7 ++-- docs/architecture/06-agent-core.md | 10 +++--- ...12-host-service-companions-and-chat-ids.md | 2 +- ouroboros/consciousness_wake.py | 23 +++++++++++-- tests/test_consciousness_wake.py | 32 +++++++++++++++++++ 5 files changed, 64 insertions(+), 10 deletions(-) diff --git a/docs/CREATING_SKILLS.md b/docs/CREATING_SKILLS.md index 1a1496aa6..15dbd5b55 100644 --- a/docs/CREATING_SKILLS.md +++ b/docs/CREATING_SKILLS.md @@ -753,8 +753,11 @@ message/deferred body retains the later model-answer path. Native inline turns return their persisted result before optional post-task cognition; transport outbox custody still owns actual delivery. If a parent fails after work was scheduled, its handoff remains deferred with -the current failure text, so the adapter retains the late result's custody; -this does not turn the failed parent into successful execution. +the work reference and any current model-authored reply. Host diagnostics and +status notices stay in the owner task; an empty deferred body sends nothing but +still requires polling. Cached and late results preserve that empty body rather +than substituting the task diagnostic. This does not turn failure into success. +Ordinary implicit replies and genuine authored best-effort answers remain valid. `GET /identity` advertises `presence_delivery_version: 1` on supporting hosts. Only then request `delivery_reporting_version: 1` alongside `binding_id` and diff --git a/docs/architecture/06-agent-core.md b/docs/architecture/06-agent-core.md index 2c6f776fc..f6c8fe68c 100644 --- a/docs/architecture/06-agent-core.md +++ b/docs/architecture/06-agent-core.md @@ -14,13 +14,13 @@ Between the sends of one loop execution the transcript is append-only — each s `DeliveryCandidate` is retained before verification or review so a later notice, reviewer failure, deadline or provider outage cannot erase a useful answer. `outcomes.py` combines the execution, objective, review, artifact and child-absorption axes without converting one into another — the terminal custody overlay (`outcomes.custody_debt_axes`) is an instance of that rule, not an exception; verify-before-done receipts and exact artifact references are host-attested evidence, and declarations or answer prose are no substitute. Individual failed tool calls alone do not degrade a delivered answer: unresolved calls retain their typed signal/exit evidence in `execution.unresolved_tool_errors`, cosmetic errors keep their separate bucket, and either kind adds `residual_tool_errors_without_review` when the canonical objective is `not_evaluated`, including after delivery/child-state normalization. Host acceptance or qualified Advisory author completion establishes objective success; unresolved FAIL/DEGRADED verdicts, empty or failed execution, provider failure, deadlines, incomplete delivery, deferred children and failed artifact verification keep their own outcome rules. A forced exit may publish the best current candidate only with its typed rail and evidence-freshness disclosure, and lifecycle may remain `completed` while the objective or review axis records a best-effort or unaccepted result. Custody debt heals from the write side while the stored `reason_code` may not be rewritten, so the owner-facing Reason line resolves the code at RENDER time — the custody warning only while the row's own `delegated_runs_unreconciled` list is non-empty, else the execution reason, both when both are real; a healed debt is never restored, and the debt is a warning beside the rail cause, never a replacement for it — as is every standing limitation of the same answer (deferred child results; a plan review still open at delivery, worded by the outcome class recorded on `execution.plan_review` when there is one), which the normal rail types on the result exactly as the forced rails do, so one clause per fact reaches the card, the durable row and Ouroboros's own memory. -Host plan/orphan disclosures ride beside the model answer as `terminal_host_notice`; delivery sends the model answer alone and the disclosure stays a field of the result (an owed pre-upgrade notice row still replays as the untyped System row it was), and the custody audit is its own typed row (`terminal_custody_notice` on the send event → `system_type="custody_notice"`, `card_row="timeline"`, its own owed delivery id), so the open-delegation fact is a row of the task's card and the unreconciled-runs note never enters the assistant text. CLI and single-body Presence get one host-labelled status (`terminal_host_notice_text`) without changing stored answer bytes: the answer keeps its hash and its real PASS or FAIL when only the notice changes, and equal text cannot revive a superseded verdict. Every reader that hands a result on — synthesis, parent handoff, `get_task_result`/`wait_task`/`wait_tasks`, the `task:` plan-evidence reader — carries the notice as a separate field, and the child-result and plan-evidence hashes include it, so a changed child limitation invalidates an old parent disposition or an old review. A host-salvaged terminal is labelled, not hidden: the durable row carries `Preserved intermediate output (not a final answer):` plus the bounded excerpt every terminal row uses, the untruncated copy stays with `get_task_result` and the stop receipt, and only a row written to the chat the stop receipt itself reached (`cancel_receipt.delivered_chat_id`, recorded after a successful send) reduces to its label beside the pointer; the two durable writes are not atomic, and no writer trades its only pointer for the label. One event is disclosed once, at the layer that owns it: the forced orphan note leaves a child to its own terminal row only where that row reached the reader the note addresses, and a rail that ended a routing turn is named by that turn's own row, never inferred from another layer's stamp. +Host plan/orphan disclosures ride beside the model answer as `terminal_host_notice`; delivery sends the model answer alone and the disclosure stays a field of the result (an owed pre-upgrade notice row still replays as the untyped System row it was), and the custody audit is its own typed row (`terminal_custody_notice` on the send event → `system_type="custody_notice"`, `card_row="timeline"`, its own owed delivery id), so the open-delegation fact is a row of the task's card and the unreconciled-runs note never enters the assistant text. CLI gets one host-labelled status (`terminal_host_notice_text`); external Presence speech excludes host notices (chapter 12). Stored answer bytes do not change: the answer keeps its hash and its real PASS or FAIL when only the notice changes, and equal text cannot revive a superseded verdict. Every reader that hands a result on — synthesis, parent handoff, `get_task_result`/`wait_task`/`wait_tasks`, the `task:` plan-evidence reader — carries the notice as a separate field, and the child-result and plan-evidence hashes include it, so a changed child limitation invalidates an old parent disposition or an old review. A host-salvaged terminal is labelled, not hidden: the durable row carries `Preserved intermediate output (not a final answer):` plus the bounded excerpt every terminal row uses, the untruncated copy stays with `get_task_result` and the stop receipt, and only a row written to the chat the stop receipt itself reached (`cancel_receipt.delivered_chat_id`, recorded after a successful send) reduces to its label beside the pointer; the two durable writes are not atomic, and no writer trades its only pointer for the label. One event is disclosed once, at the layer that owns it: the forced orphan note leaves a child to its own terminal row only where that row reached the reader the note addresses, and a rail that ended a routing turn is named by that turn's own row, never inferred from another layer's stamp. The delivery-control protocol is resolved here and only here (DEVELOPMENT keeps the rule and points here). The candidate carries sticky loop-local provenance that its lineage has seen a host-issued delivery-control episode; without one, exact JSON is ordinary text. In a marked lineage both the ordinary and the forced resolver intercept recognizable whole-body envelopes and balanced trailing protocol attempts — valid `keep` resolves to the retained candidate, valid `replace` to `full_answer`, anything malformed preserves the retained candidate. Both strip one whole-body fence and treat a balanced protocol object at the very END of prose as a protocol attempt (`utils.extract_trailing_json_object` + `loop_delivery._parse_delivery_control_body`); the trailing-object rule deliberately refuses substring scanning, because quoted protocol literals mid-prose are legitimate text, so a control object quoted mid-prose stays prose and a truncated trailing fragment remains prose. During an ordinary acceptance continuation, `finalization_control="acceptance_feedback"` makes a complete revised answer ordinary prose and keeps `keep`/`replace`/`pending_review` optional. Prose resets the pending-review choice to `wait`; it never means `finish`. An outstanding effect, owner-revision or child-action control retains its stricter rule, and unread owner source still requires acknowledgement. Empty or recognizable malformed control bodies retain the candidate rather than becoming its replacement. The ordinary resolver takes one repair round then degraded-preserve; the forced resolver resolves purely and never re-loops — malformation preserves the retained candidate with the typed `delivery_control_degraded` reason, which is this forced rail's own code, while the ordinary repair path records `invalid_delivery_control_after_repair`. Every degradation carries the cause it computed: `outcomes.derive_loop_outcome` falls back to `delivery_control_degraded` only for a degradation that reports no cause and publishes the `loop_outcome.degraded`/`degraded_reason` pair the benchmark ledgers read. A malformed attempt in a marked lineage never leaks JSON, even after the transient latch clears. The child-absorption gate is an action gate: while undispositioned direct children remain, the loop HOLDS the candidate (`child_absorption_or_revision_required`) instead of arming the JSON-only control instruction — the hold-vs-arm split exists so the model never receives two contradictory instructions in one round. A typed keep cannot close the gate; after the one bounded reminder it forces the best-effort `children_unabsorbed` rail with a current `id [status] sha256` listing. The absorption digest's `## child` header carries the child's typed custody debt (`delegated_runs_unreconciled`, bounded, with a `get_task_result` pointer) as visibility only — the parent's authority over that patch is exactly the orphan rule, and a child's debt never relabels the root card (DESIGN §4). Finalizing over an UNDISPOSED OWN delegated patch is deliberately NOT gated: the consequence is disclosed where the decision is made (the `integrate_delegated_patch` schema, the apply receipts) and lands as the additive Done-with-warnings custody overlay rather than a hold; a pre-finalization reminder and propagation of child custody debt into the acceptance-subtree snapshot remain disclosed deferred gaps. -Provider death is the one forced rail that is NOT a best-effort completion: it salvages the best available text but stamps `infra_failed`, so the task terminalizes `failed` with the typed `provider_unavailable` reason and an immediate "provider outage — NOT completed" owner notification; a waited-out transport outage reaches the same rail through its deterministic no-resend branch (`transport_unavailable_no_resend`, or `provider_outcome_unknown_no_resend` when the round still holds a transport-death repeat record). The rail makes its one forced model call only while a call can still land: a transport that spent its same-model retry wall stamps `_llm_retry_wall_exhausted`, a no-call gate beside `context_overflow` and `provider_outcome_unknown`, so the forced rail never re-pays a second retry window over a proven-dead provider (disclosed residual: the marker is a last-invocation bool on the shared usage dict). Terminal delivery preserves producer authorship: `terminal_provider_notice` is host presentation beside the raw result, and receipts and secondary System incidents carry the wait duration, finalization cause and unknown-outcome warning and never recommend a blind rerun. Every forced rail stamps one closed-vocabulary producer word at the single forced-finalization sink: `model_final` for complete model text (host-authored notices stay separate), `host_notice` for a terminal text the host wrote alone (verbatim on every transport, a System row without a `system_type`), and `host_salvage` only on the provider-death rail, where managed/direct delivery replaces the text with the enriched outage receipt and the full bytes stay in task details. A missing origin identifies only rows written before the stamp existed. +Provider death is the one forced rail that is NOT a best-effort completion: it salvages the best available text but stamps `infra_failed`, so the task terminalizes `failed` with the typed `provider_unavailable` reason and an immediate "provider outage — NOT completed" owner notification; a waited-out transport outage reaches the same rail through its deterministic no-resend branch (`transport_unavailable_no_resend`, or `provider_outcome_unknown_no_resend` when the round still holds a transport-death repeat record). The rail makes its one forced model call only while a call can still land: a transport that spent its same-model retry wall stamps `_llm_retry_wall_exhausted`, a no-call gate beside `context_overflow` and `provider_outcome_unknown`, so the forced rail never re-pays a second retry window over a proven-dead provider (disclosed residual: the marker is a last-invocation bool on the shared usage dict). Terminal delivery preserves producer authorship: `terminal_provider_notice` is host presentation beside the raw result, and receipts and secondary System incidents carry the wait duration, finalization cause and unknown-outcome warning and never recommend a blind rerun. Every forced rail stamps one closed-vocabulary producer word at the single forced-finalization sink: `model_final` for complete model text (host-authored notices stay separate), `host_notice` for host-written terminal text (an owner System row, never automatic external Presence speech), and `host_salvage` only on the provider-death rail, where managed/direct delivery replaces the text with the enriched outage receipt and the full bytes stay in task details. A missing origin remains unknown. Host text replacements in `agent.py` reset origin instead of inheriting the replaced answer's authorship. Finalization controls are typed owner-mailbox entries rather than injected owner prose. The supervisor may request one bounded tool-less answer, salvage the last persisted assistant text, and retain a full canonical copy when a preview would truncate it. A grace episode has one durable control and can be revoked atomically when the task itself resumes; descendant activity does not count as the task's own progress. A process that cannot be killed remains visibly running, and custody checks prevent another runtime instance from reaping work it does not own. @@ -150,7 +150,7 @@ Prompt caching is stable-first: governance and task-stable contracts precede mut Background Consciousness is Ouroboros taking a turn in its own Main chat when nobody wrote to it. There is no second mode: a wake-up is an ORDINARY Main direct turn (`_is_direct_chat`) with the same system prompt, context builder, tool loop, chat log, activity block and usage ledger, so consciousness structurally cannot count its context, its money or its tools differently from Main. Only three things are new: an alarm clock, a wake-up MESSAGE that stands where the owner's words would be (`prompts/CONSCIOUSNESS.md` rendered by `consciousness_wake.py` from existing readers, never silently cut), and the three owner settings beside the existing interval bounds. The alarm (`consciousness.py`) owns no thread: the supervisor pass calls `tick(now)`, whose ordered typed outcomes are disabled → a live wake → a live owner direct turn → not yet due → the rolling-24h allowance (an unreadable ledger is the disclosed skip `allowance_unknown`, never a silent block) → an owner chat must be bound → launch through `supervisor.workers.handle_wake_direct`. The next wake is last finish + interval (`set_next_wakeup` or the default, clamped into the owner's [min, max]; a failed or unadmitted wake doubles it up to max), `notify(reason)` pulls it forward to one shared floor, an owner message never wakes it, and a wake's own finish never re-arms it. Version awareness stays ambient: every task's and wake-up's Runtime context carries `official_update` (running version versus the official target at the last fetching check, plus the update letter, `update_letter.py`); no check forces a wake, and mentioning an update to the owner is the mind's judgment. -The rendered wake message puts the fresh cause and settled facts before the bounded outstanding-card view. Cards in `open` and `expired_terminal` remain answerable by the shared late-answer contract, but their state, age and bounded question preview are explicit; they are not presented as fresh events merely because they remain in the durable store. The `consciousness_last_wake_at` state key preserves the factual boundary used by the next wake after restart; the in-memory pending reason remains a best-effort trigger label and may honestly fall back to heartbeat when a process ended before launch. +The rendered wake message puts the fresh cause and settled facts before outstanding cards. Failed/cancelled or host-authored Presence terminals stay visible even on the direct-turn lane, with their stored outcome, task pointer and deferred work reference; ordinary owner direct turns and the wake's own turn stay excluded. This exposes an external failure without starting another wake or prescribing recovery: consciousness inspects the task and actual delivery through existing tools. Cards in `open` and `expired_terminal` retain late-answer semantics, explicit state, age and bounded preview. `consciousness_last_wake_at` preserves the factual boundary across restart; the best-effort in-memory trigger may fall back to heartbeat. Observe retains a narrow positive path. `consciousness_authority.OBSERVE_ARGUMENT_NARROWED` is the set of mutating-table names Observe still SEES, because each has a genuine read-only or own-work use that withholding the name would destroy along with the mutation: `schedule_subagent` and `delegate_start` for read-only research, `write_file`/`edit_text`/`journal_write`/`workpad_write` for its own notes, and `cancel_task` for the children it started. What such a name may be ASKED to do is decided by one predicate, `observe_argument_refusal`, which the dispatch guard consults directly — no fail-open wrapper, because a policy that cannot be consulted must not read as permission. It refuses a write root outside `task_drive`/`artifact_store`, a mutative `write_surface`/`may_mutate`, and a delegated session asked for `workspace_write` or a payload selector; an OMITTED access is the read-only shape (`_derive_authority` pins it) and stays allowed. `cancel_task` is the one name whose narrowing needs durable lineage rather than arguments, so it rides the tool's EXISTING own-child custody check (`join_ledger._cancel_task`), the same seam a delegated task meets: Observe may stop the children it started, never the owner's running work. A read-only child may delegate further read-only descendants under the inherited depth cap but never passes a mutating delegation budget onward (`delegation_may_mutate`). `manage_schedules` (`tools/followup.py`, beside `schedule_followup` — one module for the two tools over one table) is argument-aware rather than name-wide: the root turn may list and change, a delegated task may only list, and a Presence conversation gets neither — Presence authority is not expanded here. @@ -541,7 +541,9 @@ Every IMPLICIT claim — the UI conversion, that admission, the reaper's retry a `context.py` assembles static governance, semi-stable memory, and dynamic task evidence without treating truncation as forgetting; the recent-activity sections are each task's OWN newest rows (progress 50 rendered; tools 20 selected, 10 rendered and 20 scanned for review markers; events 200 counted by type) through the bounded reader `jsonl_tail.py` (`Memory.read_task_recent`: a doubling live tail plus at most three newest archives), never a global tail filtered afterwards (issue #131), and their header's coverage line names the rows, the window and any unopened archives while `read_file` pages the rest; a subagent child gets the same three windows beside its `## Working sources` block, its tools and events read from its own execution drive (its worker rows; host-side rows such as waits stay in the canonical log, as the header says) and progress from the canonical log; the Development context matrix and `context_layout.py` own which reference form is resident. When the rendered scratchpad exceeds `SCRATCHPAD_SECTION_BUDGET_CHARS`, `context.py` keeps the newest whole blocks that fit and drops the oldest behind an in-band gap marker naming `memory/scratchpad.md` as the live source; no block is retired by a context build, and scratchpad replacement keeps its explicit summary and source-journal provenance. -`consolidator.py` publishes dialogue summaries only after every part of a logical block succeeds, preserving raw chat generations and the captured generation cursor; an unfindable generation appends a loud durable `[MEMORY GAP]` block, never a silent offset reset. Each full Light request is measured with `context_fit` against fresh role/account capacity evidence and calibrated prompt density (local output reservation: `llm_local.local_context_limits`, the wire's normalization; missing or stale evidence stays unknown). Each room of a logical chunk is summarized from its own source bytes alone (`room_consolidation.py`: one Light draft, then one source-grounded Light correction of that draft against the same bytes), and the typed room sections are assembled deterministically into one shared block — no cross-room LLM recombine; a failed, empty or output-truncated room withholds the whole chunk (no block, no cursor advance, no nominations). Oversized source is split without clipping, including within one entry; continuation parts carry the source room context outside the original bytes. A real context refusal requires strictly fewer input bytes on the same route: a genuine refusal immediately saves its source hash and route bound in `dialogue_meta.json` (`consolidation_retry`) so a later cycle starts smaller, and a changed source, route, capacity or output reserve invalidates that bound. Era compression compresses each typed room section across the blocks it spans and reassembles the sections deterministically, so provenance survives repeated generations; a legacy block without typed sections is carried as one section of explicitly unknown room provenance, never labelled by guess. Any era stage can still fail or overflow, in which case the old blocks stay intact. Ordinary failures and empty summaries retain `last_consolidation_error` and an advance without a new failure clears it; a knowledge-nomination batch not fully published leaves `last_unpublished_nominations` in `dialogue_meta.json`; unknown spend stays nullable, and control/resource/unknown model errors preserve `propagate_model_error` semantics. +`consolidator.py` publishes a logical dialogue block only after every part succeeds, preserving raw generations and their captured cursor; an unfindable generation appends `[MEMORY GAP]`, never silently resets the offset. `context_fit` measures full Light requests against fresh route/account capacity and calibrated density (`llm_local.local_context_limits` owns local output reservation; missing/stale evidence remains unknown). `room_consolidation.py` drafts and corrects each room separately, then deterministically assembles the sections. Episodic text is grounded in that room's source; cumulative knowledge replacements require the complete current note plus the episode in BOTH stages. Corrected entries bind to the corrector's complete delivered read, never the draft's revision credit. A narrow or older episode cannot negate prior facts or later receipts; supported corrections and removals remain model judgment. Range reads, authored views, CAS and old/new history remain the publication path; no new stage or store. Failed, empty or truncated correction withholds the chunk, cursor and nominations. + +Oversized source splits without clipping, including inside an entry; continuation context stays outside source bytes. A real refusal records its source hash and strictly smaller same-route byte bound in `dialogue_meta.json` (`consolidation_retry`); changed source, route, capacity or output reserve invalidates it. Era compression regroups each recorded room across blocks and reassembles deterministically; legacy untyped blocks remain explicitly unknown provenance. Failed/overflowed eras preserve old blocks. Failures retain `last_consolidation_error`, cleared by an advance without a new failure; incomplete knowledge publication retains `last_unpublished_nominations`. Unknown spend remains nullable and control/resource/unknown model errors retain `propagate_model_error` semantics. Consolidation labels every chronological source message through `dialogue_provenance.RoomLabelResolver`, using the actual `chat_id`, never lineage `project_id`; one read-only registry snapshot supplies the window. Main is named only for the actual Main id, a resolved project uses its current registry name and stable chat id, and missing, unknown or ambiguous rooms stay explicit. Ephemeral formatter offsets carry the original room/author/direction/transport header into split continuations without parsing message bodies or duplicating their bytes. Room draft and correction prompts require meaningful decisions, approvals, outcomes and unresolved commitments of that room, retaining source distinctions (who decided, what was authorized, what stays owed) and one first-person Ouroboros voice. Length adapts to content within the existing output-token ceiling; no per-room word quota, semantic gate or absent room is imposed. Labels establish provenance, not summary success. The mixed Main recent view opts into the same labels; focused Project rendering, membership and explicit `chat_history` retain their existing behavior and bytes. diff --git a/docs/architecture/12-host-service-companions-and-chat-ids.md b/docs/architecture/12-host-service-companions-and-chat-ids.md index 03d36513a..5de716e4c 100644 --- a/docs/architecture/12-host-service-companions-and-chat-ids.md +++ b/docs/architecture/12-host-service-companions-and-chat-ids.md @@ -14,7 +14,7 @@ Operation correlation: a named injected message has `operation_ref=:= since_iso: cost = row.get("accounted_upper_bound_usd", row.get("cost_usd")) cost_text = f", ${float(cost):.2f}" if isinstance(cost, (int, float)) else "" title = _clip_preview(row.get("description") or row.get("text") or row.get("result"), 80) - settled.append((stamp, task_id, f"- task {task_id} {status}{cost_text}: {title}".rstrip(": "))) + detail = "" + if presence_failure: + outcome = str(metadata.get("presence_outcome") or "unknown") + work = str(metadata.get("presence_work_ref") or "") + detail = f"; Presence outcome={outcome}; see get_task_result" + if work: + detail += f"; deferred work={work}" + settled.append((stamp, task_id, f"- task {task_id} {status}{cost_text}: {title}".rstrip(": ") + detail)) trigger, trigger_task_id = _trigger_line(root, reason, rows, now=now) if trigger: lines.append(trigger) diff --git a/tests/test_consciousness_wake.py b/tests/test_consciousness_wake.py index 0f1ab2eb1..9bf883092 100644 --- a/tests/test_consciousness_wake.py +++ b/tests/test_consciousness_wake.py @@ -90,6 +90,38 @@ def test_events_list_settled_tasks_open_cards_and_owner_messages_since_the_last_ assert wake.wake_events(tmp_path / "missing", since=since, now=T0) == [] +@pytest.mark.parametrize("status, origin", [("failed", "host_notice"), ("cancelled", ""), + ("completed", "host_notice"), ("failed", "model_final")]) +def test_failed_inline_presence_is_visible_on_regular_wake_without_reviving_owner_turns(tmp_path, status, origin): + from ouroboros.presence_runner import _build_task + from ouroboros.task_results import write_task_result + from tests.test_presence_runner import _admission, _event + + task = _build_task(_admission(), _event(), drive_root=tmp_path, staged_files=()) + assert task["_is_direct_chat"] is True + metadata = {**task["metadata"], "presence_outcome": "deferred", "presence_result_text": "", + "presence_work_ref": "still-running-child"} + write_task_result(tmp_path, task["id"], status, _is_direct_chat=True, metadata=metadata, + terminal_origin=origin, result="Host diagnostic remains available in the task.") + _result(tmp_path, "owner-failed", ts=T0, direct=True, status="failed") + _result(tmp_path, "successful-presence", ts=T0, direct=True) + path = tmp_path / "task_results" / "successful-presence.json" + successful = json.loads(path.read_text(encoding="utf-8")) + successful.update(metadata={"presence": {}, "presence_outcome": "message"}, terminal_origin="model_final") + path.write_text(json.dumps(successful), encoding="utf-8") + + lines = wake.wake_events(tmp_path, since=0, now=T0, reason="heartbeat") + fact = next(line for line in lines if line.startswith(f"- task {task['id']} ")) + assert f" {status}" in fact and "Presence outcome=deferred" in fact + assert "get_task_result" in fact and "deferred work=still-running-child" in fact + assert not any("owner-failed" in line or "successful-presence" in line for line in lines) + assert not any(task["id"] in line for line in wake.wake_events( + tmp_path, since=0, now=T0, reason="heartbeat", exclude_task_id=task["id"])) + metadata["initiator"] = "consciousness" + write_task_result(tmp_path, task["id"], status, _is_direct_chat=True, metadata=metadata, terminal_origin=origin) + assert not any(task["id"] in line for line in wake.wake_events(tmp_path, since=0, now=T0, reason="heartbeat")) + + def test_project_digest_pins_human_project_and_related_task_before_cards(tmp_path): since = T0 - 3600 (tmp_path / "state").mkdir() From 3f88f79ca43f89c4936691ec2645ceefc54d2143 Mon Sep 17 00:00:00 2001 From: Ouroboros Date: Wed, 23 Sep 2026 04:53:04 +0300 Subject: [PATCH 05/15] fix(memory): ground corrected knowledge in its own current reads Separate episodic summary grounding from cumulative note maintenance and bind corrected nominations only to sources delivered to the correcting operation. Preserve complete range reads, revision conflicts, legitimate revisions and durable unpublished outcomes. Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> --- ouroboros/consolidator.py | 17 ++- ouroboros/room_consolidation.py | 40 +++--- tests/test_knowledge_consolidation.py | 8 +- tests/test_knowledge_working_view.py | 35 +++-- tests/test_room_knowledge_correction.py | 166 ++++++++++++++++++++++++ tests/test_room_provenance_delta.py | 4 +- 6 files changed, 233 insertions(+), 37 deletions(-) create mode 100644 tests/test_room_knowledge_correction.py diff --git a/ouroboros/consolidator.py b/ouroboros/consolidator.py index 3ea97cbf5..8c7d12785 100644 --- a/ouroboros/consolidator.py +++ b/ouroboros/consolidator.py @@ -310,8 +310,8 @@ def _run_block_consolidation( for i in range(chunks_to_process): chunk = new_entries[i * BLOCK_SIZE : (i + 1) * BLOCK_SIZE] # The host partitions by the actual chat id BEFORE any model call: each - # room is summarized from its own exact chronological bytes and no call - # ever mixes rooms; the identity of every section is a host fact. + # room's episodic summary uses its own chronological bytes; cumulative + # knowledge can span rooms. Each section's identity is a host fact. rooms = room_consolidation.partition_entries(chunk, room_resolver) first_ts = str(chunk[0].get("ts", "unknown")) last_ts = str(chunk[-1].get("ts", "unknown")) @@ -675,10 +675,15 @@ class KnowledgeReadContext: KNOWLEDGE_MAINTENANCE_PROMPT = """ You may use knowledge_list and knowledge_read to understand existing notes before -nominating a durable revision. Keep the original episode below as evidence, read -the complete CURRENT note before replacing it, and preserve its sources, uncertainty, -unknown metadata and useful links. A new observation may correct an old interpretation; -do not merely repeat fragments. New topics may be created without a prior read. +nominating a durable revision. An episodic summary describes only its supplied source; +a knowledge note is cumulative understanding, grounded in the complete CURRENT note +you read in this operation together with the new episode. Absence from this episode +does not refute prior knowledge; an earlier episode cutoff does not undo later known +events. Preserve useful established facts, sources, uncertainty, unknown metadata and +links. Correct, remove or reorganize stale or unsupported understanding when the +evidence warrants it; memory is revisable, not append-only. Read the whole current +note before replacing it, rather than merely repeating fragments. New topics may be +created without a prior read. Understanding of the people involved — preferences, recurring reactions, shared history, tentative interpretations with their source — is ordinary knowledge to nominate in global scope; a pattern across several moments is worth more than one; revise the existing note rather than minting a rule, diff --git a/ouroboros/room_consolidation.py b/ouroboros/room_consolidation.py index 8338d2e5e..aeffe578b 100644 --- a/ouroboros/room_consolidation.py +++ b/ouroboros/room_consolidation.py @@ -5,9 +5,10 @@ exact chronological source, and every successful summary unit is compared once against the same complete bytes it was written from before anything is kept. The host owns room identity: it partitions by the actual ``chat_id`` before any model call, stamps the typed ``rooms`` sections and their deterministic -Markdown projection, and never parses generated text for labels. Cross-room -facts never share a model call, an era regroups the same recorded room across -blocks, and a legacy record without sections stays one explicitly +Markdown projection, and never parses generated text for labels. Episodic +claims stay grounded in that room's source; cumulative knowledge revisions +also require the correcting operation's own complete current-note read. +An era regroups the same recorded room across blocks, and a legacy record stays one explicitly unknown-provenance section. The Light transport (route, fit, retained sources, typed failures) stays with ``consolidator.py``; this module receives it as one ``call`` function. @@ -96,7 +97,7 @@ def room_draft_prompt( return f"""{knowledge_instruction}You are the memory consolidator of Ouroboros, a self-modifying AI agent. Write the episodic memory of one room's messages inside the dialogue block {block_range_text}. Room: {room_label}. This room contributed {message_count} messages; other rooms of the block are written separately and the host assembles them. -The source may be one contiguous part of the room; summarize only the supplied part. +The source may be one contiguous part of the room; the episodic summary covers only the supplied part. Existing knowledge may inform a cumulative note update, but must not become an event or approval in this episode. ## Rules 1. No block or room headers; the host writes them. Start with the memory itself. @@ -114,16 +115,22 @@ The source may be one contiguous part of the room; summarize only the supplied p def correction_prompt( draft: str, source: str, *, room_label: str, scope: str, - identity_text: str = "", continuation_note: str = "", + identity_text: str = "", continuation_note: str = "", knowledge_instruction: str = "", ) -> str: room_label = json.dumps(str(room_label), ensure_ascii=False) + knowledge_check = ("If the draft has a `KNOWLEDGE_ENTRIES_JSON:` block, check those cumulative updates " + "against each complete current note you read YOURSELF and this episode. The draft is a " + "proposal, not a source. Return corrected nominations for the same topics after the " + "episodic memory, or drop them. Unsupported episode claims must not survive in a note; " + "independently established knowledge may remain without becoming an event or approval " + "in this episode.\n" + knowledge_instruction if knowledge_instruction else "") return f"""Compare this draft memory of Ouroboros against its complete source and return the corrected memory. -Scope: {scope}; room: {room_label}. The draft was written from exactly this source; the host assembles rooms and headers separately. +Scope: {scope}; room: {room_label}. The source below is complete for this episodic summary, not for cumulative knowledge; the host assembles rooms and headers separately. Check sentence by sentence. Fix misattributed actors or approvals; decisions moved between rooms, tasks or people; invented, dropped or altered budgets, deadlines, checkpoints, boundaries and obligations; completion, review, verification or publication the source does not show; anything called approved that the source shows proposed, asked or rejected. Keep the first-person Ouroboros voice, quotes, task_ids and everything the draft got right; adapt length to the content. {FIDELITY_RULES} Return only the corrected memory text: no headers, commentary or diff. If the draft is already faithful, return it unchanged. -If the draft ends with a `KNOWLEDGE_ENTRIES_JSON:` block, it is subject to the same check: return it after the corrected memory with the same shape, dropping or fixing every entry the source does not support; a claim the correction removed from the memory must not survive as an entry. +{knowledge_check} {_identity_section(identity_text)} {continuation_note}## Draft memory {draft} @@ -194,8 +201,8 @@ def summarize_source( response withholds the whole source. A real refusal lowers the same route's byte limit for remaining parts and the next cycle. Knowledge nominations are released only from the CORRECTED response: the draft's trailing block - travels into the correction under the same source check, so a false claim - the correction removed from the memory cannot survive as a durable entry. + travels into the correction, where cumulative revisions use its own note + reads alongside the episode rather than inheriting the draft's read credit. """ pending, summaries, usages = [(0, len(text))], [], [] entries: List[Dict[str, Any]] = [] @@ -228,7 +235,7 @@ def summarize_source( while pending: start, end = pending.pop() part, note = text[start:end], source_continuation_note(spans, start, end) - draft, usage, draft_knowledge = call(draft_prompt(part, note), "Room summary", fixed_prompt=draft_prompt("", note), + draft, usage, _draft_knowledge = call(draft_prompt(part, note), "Room summary", fixed_prompt=draft_prompt("", note), input_limit=input_limit, call_type="memory_consolidation") usages.append(usage) if draft.strip(): @@ -241,10 +248,9 @@ def summarize_source( # only with its corrected text: a draft whose correction failed # (and was then split) never entered the block, and a draft # nomination the correction dropped was never source-checked. - # The corrected block may DROP or FIX the draft's entries, never add - # topics: only the draft call read the notes it nominates against, so - # its recorded reads attest exactly those topics' revisions, and an - # entry the correction invented has no read behind it. + # Keep this correction scoped to the draft's nominated topics. + # Only its OWN complete reads can authorize existing-note updates; + # draft read credit says nothing about the correction's evidence. from ouroboros.reflection import _extract_trailing_json _, draft_raw = _extract_trailing_json(draft, "KNOWLEDGE_ENTRIES_JSON:") @@ -253,8 +259,7 @@ def summarize_source( corrected, raw = _extract_trailing_json(corrected, "KNOWLEDGE_ENTRIES_JSON:") kept = [e for e in (raw if isinstance(raw, list) else []) if isinstance(e, dict) and (str(e.get("topic") or ""), str(e.get("scope") or "")) in draft_topics] - binder = draft_knowledge if draft_knowledge is not None else knowledge - entries.extend(binder.bind_entries(kept) if binder is not None and kept else []) + entries.extend(knowledge.bind_entries(kept) if knowledge is not None and kept else []) summaries.append(corrected.strip()) continue failure = usage["_consolidation_errors"][-1] @@ -302,7 +307,8 @@ def summarize_block( def correct(draft_text: str, part: str, note: str, room: RoomSource = room) -> str: return correction_prompt(draft_text, part, room_label=room.label, scope=f"dialogue block {range_text}", - identity_text=identity_text, continuation_note=note) + identity_text=identity_text, continuation_note=note, + knowledge_instruction=knowledge_instruction) content, usage = summarize_source(call, room.text, room.spans, draft, correct, input_limit=input_limit, on_refusal=on_refusal) diff --git a/tests/test_knowledge_consolidation.py b/tests/test_knowledge_consolidation.py index a5c248887..36de700ea 100644 --- a/tests/test_knowledge_consolidation.py +++ b/tests/test_knowledge_consolidation.py @@ -35,9 +35,9 @@ class MemoryLLM: def chat(self, **kwargs): self.calls.append(deepcopy(kwargs)) if kwargs["messages"][0]["content"].startswith("Compare this draft memory"): - # The correction answers with the checked memory text and carries the - # draft's nomination block through the same source check (the - # corrected block is the one released). + if kwargs["messages"][-1]["role"] != "tool": + return {"content": "", "tool_calls": [_call()]}, {"cost": 0.01} + # Corrected existing-note replacements require this operation's read. prompt = kwargs["messages"][0]["content"] block = prompt.split("## Draft memory", 1)[1].split("\n\n", 1)[0] if "## Draft memory" in prompt else "" nominations = block[block.index("KNOWLEDGE_ENTRIES_JSON:"):] if "KNOWLEDGE_ENTRIES_JSON:" in block else "" @@ -143,7 +143,7 @@ def test_dialogue_consolidation_retains_nominations_and_commits_shared_note(tmp_ llm = MemoryLLM(answer) ctx = ToolContext(repo_dir=tmp_path, drive_root=tmp_path, task_id="dialogue-memory") usage = c.consolidate(chat, blocks, meta, llm, knowledge_context=ctx) - assert usage["cost"] == pytest.approx(0.05) # read, draft answer, correction + assert usage["cost"] == pytest.approx(0.06) # draft read/answer, correction read/answer block = json.loads(blocks.read_text())[0] assert "KNOWLEDGE_ENTRIES_JSON" not in block["content"] assert block["rooms"][0]["content"] == "Checked interpretation." diff --git a/tests/test_knowledge_working_view.py b/tests/test_knowledge_working_view.py index 4b49e3445..1d0fdbbbc 100644 --- a/tests/test_knowledge_working_view.py +++ b/tests/test_knowledge_working_view.py @@ -3,6 +3,8 @@ import json from copy import deepcopy +import pytest + from ouroboros import consolidator as c, knowledge as k from ouroboros.context_fit import estimate_context_prompt_tokens from ouroboros.tools.registry import ToolContext @@ -64,12 +66,19 @@ def test_tool_result_projection_never_credits_undelivered_body_or_header(tmp_pat assert not reads.reads -def test_light_reads_large_note_in_multiple_windows_then_publishes_its_revision(tmp_path, fit): +@pytest.mark.parametrize("room_correction", [False, True]) +def test_light_reads_large_note_in_multiple_windows_then_publishes_its_revision(tmp_path, fit, room_correction): fit.window = 50000 ctx, original, reads = _setup(tmp_path, "Original account of events. " * 7000 + "DECISIVE LAST EVENT.") chunk = 40000 total = len(original.text) assert estimate_context_prompt_tokens([{"role": "user", "content": original.text}], reads.tools) + 16384 > fit.window + episode = "ORIGINAL EPISODE: reconsider the complete account." + revised = "A coherent revised account preserving the decisive last event." + nomination = [{"topic": "large", "scope": "global", "content": revised}] + answer = "I read the full account and retained what changed." + if room_correction: + answer += "\nKNOWLEDGE_ENTRIES_JSON: " + json.dumps(nomination) class Reader: def __init__(self): @@ -78,18 +87,21 @@ def test_light_reads_large_note_in_multiple_windows_then_publishes_its_revision( def chat(self, **kwargs): self.calls.append(deepcopy(kwargs)) assert estimate_context_prompt_tokens(kwargs["messages"], kwargs["tools"]) + kwargs["max_tokens"] <= fit.window - assert kwargs["messages"][0]["content"] == "ORIGINAL EPISODE: reconsider the complete account." + assert episode in kwargs["messages"][0]["content"] + if room_correction and not kwargs["messages"][0]["content"].startswith("Compare this draft memory"): + # No draft read credit: the correction must earn its own, across views. + return {"content": "Draft.\nKNOWLEDGE_ENTRIES_JSON: " + json.dumps(nomination)}, {"cost": 0.01} if self.stage == "read": if self.next_start >= total: assert "DECISIVE LAST EVENT." in kwargs["messages"][-1]["content"] - return {"content": "I read the full account and retained what changed."}, {"cost": 0.01} + return {"content": answer}, {"cost": 0.01} start, end = self.next_start, min(total, self.next_start + chunk) self.next_start, self.stage = end, "inspect" call = _call("knowledge_read", {"topic": "large", "scope": "global", "start_char": start, "end_char": end}, f"read-{start}") elif self.stage == "inspect": if self.next_start >= total: assert "DECISIVE LAST EVENT." in kwargs["messages"][-1]["content"] - return {"content": "I read the full account and retained what changed."}, {"cost": 0.01} + return {"content": answer}, {"cost": 0.01} call, self.stage = _call("compact_context", {"inspect": True}, f"inspect-{self.next_start}"), "compact" else: observed = json.loads(kwargs["messages"][-1]["content"]) @@ -101,12 +113,19 @@ def test_light_reads_large_note_in_multiple_windows_then_publishes_its_revision( return {"content": "", "tool_calls": [call]}, {"cost": 0.01} llm = Reader() - content, usage = c._call_consolidation_llm(llm, "ORIGINAL EPISODE: reconsider the complete account.", - "multiwindow", knowledge=reads) + if room_correction: + from ouroboros import room_consolidation as rc + room = rc.RoomSource("1", "Main", [{"text": episode}], episode) + content, usage = rc.summarize_block(c._light_call(llm, ctx, {}), [room], + first_ts="2026-09-01T10:00:00Z", last_ts="2026-09-01T10:01:00Z", + knowledge_instruction=c.KNOWLEDGE_MAINTENANCE_PROMPT) + bound = usage["_knowledge_entries"] + else: + content, usage = c._call_consolidation_llm(llm, episode, "multiwindow", knowledge=reads) + bound = reads.bind_entries(nomination) assert content, usage assert len(llm.revisions) > 2 - assert reads.reads[("global", "large")] == original.revision - bound = reads.bind_entries([{"topic": "large", "scope": "global", "content": "A coherent revised account preserving the decisive last event."}]) + assert bound[0]["expected_revision"] == original.revision assert c._write_knowledge_entries(original.address.shelf, bound, context=ctx)[0]["ok"] assert k.read_knowledge_note(original.address).text.endswith("decisive last event.") assert list((tmp_path / "task_results" / "artifacts" / "memory-view" / "source_handles").rglob("*.json")) diff --git a/tests/test_room_knowledge_correction.py b/tests/test_room_knowledge_correction.py new file mode 100644 index 000000000..69664e507 --- /dev/null +++ b/tests/test_room_knowledge_correction.py @@ -0,0 +1,166 @@ +"""Cumulative replacements belong to the correction's own delivered sources.""" + +from copy import deepcopy +import json + +import pytest + +from ouroboros import consolidator as c, knowledge as k, room_consolidation as rc +from ouroboros.tools.registry import ToolContext +from tests import test_consolidator_context_fit as fit_helpers + +fit = fit_helpers.fit +TOPIC = "shared-understanding" + + +def _answer(content): + return "I recorded the episode.\nKNOWLEDGE_ENTRIES_JSON: " + json.dumps([ + {"topic": TOPIC, "scope": "global", "content": content}]) + + +class CorrectionLLM: + """Scripted responses; all read, delivery, binding and publication are real.""" + + def __init__(self, draft, corrected, *, correction_read="complete", draft_read=True, + before_correction=None, after_correction_read=None, expected_note=None): + self.draft, self.corrected = draft, corrected + self.correction_read, self.draft_read = correction_read, draft_read + self.before_correction = before_correction + self.after_correction_read = after_correction_read + self.expected_note = expected_note + self.calls = [] + + def chat(self, **kwargs): + self.calls.append(deepcopy(kwargs)) + messages = kwargs["messages"] + correction = messages[0]["content"].startswith("Compare this draft memory") + if messages[-1]["role"] != "tool": + if correction and self.before_correction: + self.before_correction() + mode = self.correction_read if correction else ("complete" if self.draft_read else "none") + if mode != "none": + args = {"topic": TOPIC, "scope": "global"} + if mode == "partial": + args.update(start_char=0, end_char=12) + return {"tool_calls": [{"id": "read-note", "type": "function", "function": { + "name": "knowledge_read", "arguments": json.dumps(args)}}]}, {"cost": 0.01} + if correction and self.correction_read == "complete": + # The scripted answer is permitted only after actual source delivery. + assert self.expected_note in messages[-1]["content"] + if self.after_correction_read: + self.after_correction_read() + return {"content": _answer(self.corrected if correction else self.draft)}, {"cost": 0.01} + + +def _setup(tmp_path, initial): + ctx = ToolContext(repo_dir=tmp_path, drive_root=tmp_path, task_id="room-memory") + address = k.resolve_knowledge_address(tmp_path, TOPIC, "global") + original = k.write_knowledge_note(address, initial).current if initial else None + return ctx, address, original + + +def _consolidate(ctx, llm, episode): + room = rc.RoomSource("1", "Main", [{"text": episode}], episode) + block, usage = rc.summarize_block( + c._light_call(llm, ctx, {}), [room], first_ts="2026-09-01T10:00:00Z", + last_ts="2026-09-01T10:01:00Z", knowledge_instruction=c.KNOWLEDGE_MAINTENANCE_PROMPT) + assert block and block["rooms"][0]["content"] == "I recorded the episode." + return usage["_knowledge_entries"] + + +CASES = [ + ("Alex maintains Alpha and prefers email.", "Alex sent a meeting invitation.", + "Alex maintains Alpha and prefers email. Alex sent a meeting invitation; attendance is unknown."), + ("Delivery D1 of report R1 was confirmed at 10:05. Reply is unknown.", + "At 10:00 I was asked to send report R1.", + "Delivery D1 of report R1 was confirmed at 10:05 after the 10:00 request. Reply is unknown."), + ("Alpha owns ingestion; Beta owns storage; Gamma owns search.", "Alpha now owns validation too.", + "Alpha owns ingestion and validation; Beta owns storage; Gamma owns search."), +] + + +@pytest.mark.parametrize("initial,episode,updated", CASES) +@pytest.mark.parametrize("correction_read", ["none", "partial", "complete"]) +def test_correction_cannot_borrow_draft_read_and_can_preserve_cumulative_knowledge( + tmp_path, fit, initial, episode, updated, correction_read): + ctx, address, original = _setup(tmp_path, initial) + # Without the full note the failure-shaped response reduces it to the episode. + content = updated if correction_read == "complete" else episode + llm = CorrectionLLM(updated, content, correction_read=correction_read, + expected_note=original.text) + entries = _consolidate(ctx, llm, episode) + outcome = c._write_knowledge_entries(address.shelf, entries, context=ctx)[0] + current = k.read_knowledge_note(address) + if correction_read == "complete": + assert entries[0]["expected_revision"] == original.revision + assert outcome["ok"] and current.text.endswith(updated) + else: + assert entries[0]["expected_revision"] is None + assert outcome["reason"] == "revision_required" and not outcome["ok"] + assert current.raw == original.raw + # Both authoring stages receive the cumulative scope and range-read support. + for call in (llm.calls[0], next(call for call in llm.calls + if call["messages"][0]["content"].startswith("Compare this draft memory"))): + assert c.KNOWLEDGE_MAINTENANCE_PROMPT in call["messages"][0]["content"] + + +def test_own_complete_read_can_revise_and_remove_obsolete_facts_without_draft_read(tmp_path, fit): + ctx, address, original = _setup(tmp_path, "Alex owns Alpha. Old office: Building 7. Contact by email.") + updated = "Alex now owns Beta. Contact by email." + llm = CorrectionLLM(original.text, updated, draft_read=False, expected_note=original.text) + entries = _consolidate(ctx, llm, "Alex moved from Alpha to Beta and asked to remove the obsolete office address.") + assert entries[0]["expected_revision"] == original.revision + assert c._write_knowledge_entries(address.shelf, entries, context=ctx)[0]["ok"] + current = k.read_knowledge_note(address) + assert current.text.endswith(updated) + assert "Building 7" not in current.text and "owns Alpha" not in current.text + history = (tmp_path / "memory" / "knowledge_history.jsonl").read_text(encoding="utf-8") + assert "Building 7" in history and "now owns Beta" in history + + +@pytest.mark.parametrize("change_after_read", [False, True]) +def test_correction_binds_its_current_revision_and_preserves_later_concurrent_changes(tmp_path, fit, change_after_read): + ctx, address, original = _setup(tmp_path, "Draft-era understanding.") + latest = "A newer established observation." + + def replace(): + assert k.write_knowledge_note(address, latest, expected_revision=original.revision).ok + + llm = CorrectionLLM(original.text, "A newer established observation with the new episode.", + before_correction=None if change_after_read else replace, + after_correction_read=replace if change_after_read else None, + expected_note=original.text if change_after_read else latest) + entries = _consolidate(ctx, llm, "A new episode.") + current = k.read_knowledge_note(address) + assert entries[0]["expected_revision"] == (original.revision if change_after_read else current.revision) + result = c._write_knowledge_entries(address.shelf, entries, context=ctx)[0] + if change_after_read: + assert not result["ok"] and result["reason"] == "revision_conflict" + assert k.read_knowledge_note(address).raw == current.raw + else: + assert result["ok"] + assert k.read_knowledge_note(address).text.endswith("with the new episode.") + + +def test_correction_can_create_a_new_nominated_note_without_reading(tmp_path, fit): + ctx, address, _ = _setup(tmp_path, None) + llm = CorrectionLLM("New observation.", "Corrected new observation.", + draft_read=False, correction_read="none") + entries = _consolidate(ctx, llm, "A new observation.") + assert entries[0]["expected_revision"] is None + assert c._write_knowledge_entries(address.shelf, entries, context=ctx)[0]["ok"] + assert k.read_knowledge_note(address).text.endswith("Corrected new observation.") + + +def test_unread_correction_failure_remains_visible_after_dialogue_publication(tmp_path, fit): + ctx, address, original = _setup(tmp_path, "An established body of knowledge.") + chat, blocks, meta = fit_helpers._paths(tmp_path) + fit_helpers._write_chat(chat, text_size=0) + llm = CorrectionLLM(original.text, "Only this episode.", correction_read="none") + c.consolidate(chat, blocks, meta, llm, knowledge_context=ctx) + stored = json.loads(blocks.read_text(encoding="utf-8"))[0] + assert stored["knowledge_writes"][0]["reason"] == "revision_required" + state = json.loads(meta.read_text(encoding="utf-8")) + assert state["last_consolidated_offset"] == 100 + assert state["last_unpublished_nominations"]["failed"] == 1 + assert k.read_knowledge_note(address).raw == original.raw diff --git a/tests/test_room_provenance_delta.py b/tests/test_room_provenance_delta.py index 6b7ed38b9..d142fc676 100644 --- a/tests/test_room_provenance_delta.py +++ b/tests/test_room_provenance_delta.py @@ -365,8 +365,8 @@ def test_nominations_come_from_the_corrected_response_not_the_draft(): lambda draft, part, note: rc.correction_prompt(draft, part, room_label="Main", scope="block r", continuation_note=note), ) assert content == "corrected memory" - # The draft's block reaches the correction under the same source check... + # The draft's proposed block reaches the correction... assert "KNOWLEDGE_ENTRIES_JSON" in prompts[1][1] and "owner approved" in prompts[1][1] # ...and only the corrected block is released: the draft's topic with the - # corrected content, never a topic the correction invented without a read. + # corrected content, never an additional topic outside this correction's scope. assert [(e["topic"], e["content"]) for e in usage["_knowledge_entries"]] == [("leak", "owner asked")] From ac51e950c0faf220a3e66b100a3eaa5e523cfa3b Mon Sep 17 00:00:00 2001 From: Ouroboros Date: Wed, 23 Sep 2026 04:55:22 +0300 Subject: [PATCH 06/15] Preserve terminal authorship across Presence delivery and replay Keep host diagnostics in durable task evidence, deliver current model-authored speech, and retain scheduled child custody without inventing an external reply. Share cached and deferred projection, preserve intentional empty bodies, and recover only exactly recorded historical host composition. Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> --- ouroboros/agent.py | 11 +- ouroboros/agent_task_pipeline.py | 2 +- ouroboros/gateway/host_service.py | 16 +- ouroboros/presence_runner.py | 72 ++++--- ouroboros/task_finalization.py | 4 +- tests/test_presence_completion.py | 13 +- tests/test_presence_failed_handoff.py | 11 +- tests/test_presence_handoff.py | 2 +- tests/test_presence_terminal_authorship.py | 221 +++++++++++++++++++++ tests/test_presence_terminal_latency.py | 6 +- tests/test_provider_terminal_notice.py | 5 +- tests/test_provider_wall_rail_contract.py | 8 +- tests/test_terminal_host_notice.py | 5 +- 13 files changed, 312 insertions(+), 64 deletions(-) create mode 100644 tests/test_presence_terminal_authorship.py diff --git a/ouroboros/agent.py b/ouroboros/agent.py index cb70ab503..163351e1a 100644 --- a/ouroboros/agent.py +++ b/ouroboros/agent.py @@ -46,6 +46,7 @@ from ouroboros.agent_startup_checks import ( from ouroboros.agent_task_pipeline import ( emit_task_results, build_review_context, ) +from ouroboros.task_finalization import TERMINAL_ORIGIN_HOST_NOTICE from ouroboros.task_results import STATUS_RUNNING, write_task_result from ouroboros.contracts.task_constraint import normalize_task_constraint from ouroboros.consciousness_authority import apply_consciousness_authority @@ -86,7 +87,7 @@ def _authority_source_terminal(refusal: Dict[str, Any]): ).strip() usage = { "execution_status": "infra_failed", "reason_code": "authority_source_unavailable", - "authority_source_unavailable": refusal, + "authority_source_unavailable": refusal, "terminal_origin": TERMINAL_ORIGIN_HOST_NOTICE, } return text, usage, {"reasoning_notes": ["authority_source_unavailable"], "tool_calls": []} @@ -104,7 +105,8 @@ def _task_exception_terminal(env: Any, task: Dict[str, Any], exc: Exception, dri llm_trace = captured_trace if isinstance(captured_trace, dict) else { "reasoning_notes": [], "tool_calls": [], "loop_evidence_unavailable": True, } - usage.update(execution_status="infra_failed", reason_code="task_exception") + usage.update(execution_status="infra_failed", reason_code="task_exception", + terminal_origin=TERMINAL_ORIGIN_HOST_NOTICE) text = f"⚠️ Error during processing: {type(exc).__name__}: {exc}" append_jsonl(drive_logs / "events.jsonl", { "ts": utc_now_iso(), "type": "task_error", "task_id": task.get("id"), @@ -1020,6 +1022,10 @@ class OuroborosAgent: ) if not isinstance(text, str) or (not text.strip() and not intentional_empty): text = "⚠️ Model returned an empty response. Try rephrasing your request." + usage["terminal_origin"] = TERMINAL_ORIGIN_HOST_NOTICE + usage.pop("presence_completion_outcome", None) + if ctx is not None: + ctx._presence_completion_accepted = False # A task that scoped ITSELF mid-run (ensure_project_scope) set the scope on # ctx, but persistence/finalization read the task dict — sync it back so the @@ -1087,6 +1093,7 @@ class OuroborosAgent: usage = { "execution_status": "failed", "reason_code": "budget_exhausted", + "terminal_origin": TERMINAL_ORIGIN_HOST_NOTICE, "resource_limit": resource_limit, } llm_trace = { diff --git a/ouroboros/agent_task_pipeline.py b/ouroboros/agent_task_pipeline.py index d08582998..8f139a7fc 100644 --- a/ouroboros/agent_task_pipeline.py +++ b/ouroboros/agent_task_pipeline.py @@ -561,7 +561,7 @@ def emit_task_results( send_event["progress_meta"] = dict(_message_meta) send_event = prepare_terminal_send_event(env.drive_root, task, text, usage, send_event, presence=_presence) pending_events.append(build_presence_result_event( - task, text, ctx, provider_notice=terminal_notice_text(usage), + task, text, ctx, terminal_origin=str(usage.get("terminal_origin") or ""), retain_scheduled_handoff=failed_or_forced, ) if _presence else send_event) duration_sec = round(time.time() - start_time, 3) diff --git a/ouroboros/gateway/host_service.py b/ouroboros/gateway/host_service.py index 7c8c7504a..60c74cd15 100644 --- a/ouroboros/gateway/host_service.py +++ b/ouroboros/gateway/host_service.py @@ -813,20 +813,16 @@ async def _api_presence_work(request: Request) -> JSONResponse: if status not in {"completed", "failed", "cancelled"}: return JSONResponse({"ok": True, "status": "pending", "work_ref": work_ref, "delivery_reporting_version": presence.get("delivery_reporting_version", 0)}, status_code=202) - outcome = str(metadata.get("presence_outcome") or "message") - if outcome not in {"message", "silent", "tool_delivered", "deferred"}: - outcome = "message" + from ouroboros.presence_runner import presence_result_from_stored + + result = presence_result_from_stored(stored, work_ref) return JSONResponse({ "ok": True, "status": status, - "outcome": outcome, - "text": ( - str(metadata.get("presence_result_text") or stored.get("result") or "") - if outcome in {"message", "deferred"} - else "" - ), + "outcome": result.outcome, + "text": result.text, "work_ref": work_ref, - "delivery_reporting_version": presence.get("delivery_reporting_version", 0), + "delivery_reporting_version": result.delivery_reporting_version, }) except Exception as exc: code = str(getattr(exc, "code", "")) diff --git a/ouroboros/presence_runner.py b/ouroboros/presence_runner.py index 4a165c0a7..8984a04cd 100644 --- a/ouroboros/presence_runner.py +++ b/ouroboros/presence_runner.py @@ -15,6 +15,10 @@ from ouroboros.artifacts import stage_task_attachments from ouroboros.contracts.task_contract import attach_task_contract from ouroboros.presence_admission import PresenceAdmission from ouroboros.presence_authority import presence_ceiling_payload +from ouroboros.task_finalization import ( + HOST_AUTHORED_TERMINAL_ORIGINS, TERMINAL_ORIGIN_MODEL_FINAL, + provider_terminal_body, terminal_notice_text, +) from ouroboros.task_results import load_task_result from ouroboros.utils import append_jsonl, read_json_dict, utc_now_iso @@ -59,7 +63,42 @@ class PresenceTurnResult: delivery_reporting_version: int = 0 -def build_presence_result_event(task: dict[str, Any], text: str, ctx: Any, *, provider_notice: str = "", +def _presence_delivery(outcome: str, text: str, terminal_origin: str, *, legacy: bool = False) -> tuple[str, str]: + """Project speech from producer facts, never from its wording or task status.""" + if outcome not in {"message", "silent", "tool_delivered", "deferred"}: + outcome = "message" + if outcome not in {"message", "deferred"}: + return outcome, "" + # Unknown origin remains explicit stored-row compatibility, not evidence + # authorizing new speech. Even an old frozen body cannot override known host provenance. + authored = terminal_origin == TERMINAL_ORIGIN_MODEL_FINAL + if not authored and not (legacy and terminal_origin not in HOST_AUTHORED_TERMINAL_ORIGINS): + return ("deferred" if outcome == "deferred" else "silent"), "" + return outcome, text + + +def presence_result_from_stored(stored: Mapping[str, Any], task_id: str) -> PresenceTurnResult: + """One replay projection for cached turns and completed delegated work.""" + metadata = stored.get("metadata") if isinstance(stored.get("metadata"), dict) else {} + text = metadata["presence_result_text"] if "presence_result_text" in metadata else stored.get("result") + origin = str(stored.get("terminal_origin") or "") + if origin == TERMINAL_ORIGIN_MODEL_FINAL and text and "result" in stored: + raw, notice = str(stored["result"] or ""), terminal_notice_text(stored) + # Undo only the recorded host composition. Older explicit reply bodies + # could differ from the raw final; neither they nor deliberate emptiness are guessed away. + if notice and text == provider_terminal_body(raw, notice): + text = raw + outcome, text = _presence_delivery( + str(metadata.get("presence_outcome") or "message"), str(text or ""), origin, legacy=True, + ) + return PresenceTurnResult( + outcome=outcome, text=text, task_id=task_id, + work_ref=str(metadata.get("presence_work_ref") or ""), + delivery_reporting_version=int((metadata.get("presence") or {}).get("delivery_reporting_version") == 1), + ) + + +def build_presence_result_event(task: dict[str, Any], text: str, ctx: Any, *, terminal_origin: str = "", retain_scheduled_handoff: bool = False) -> dict[str, Any]: """Freeze typed delivery metadata before the ordinary durable result write.""" @@ -68,8 +107,6 @@ def build_presence_result_event(task: dict[str, Any], text: str, ctx: Any, *, pr isinstance(completion, dict) and getattr(ctx, "_presence_completion_accepted", False) ) else {} outcome = str(completion.get("outcome") or "message").strip() - if outcome not in {"message", "silent", "tool_delivered", "deferred"}: - outcome = "message" handoff = getattr(ctx, "_swarm_handoff_attempt", None) handoff = handoff if isinstance(handoff, dict) else {} work_ref = ( @@ -81,17 +118,12 @@ def build_presence_result_event(task: dict[str, Any], text: str, ctx: Any, *, pr outcome = "message" if retain_scheduled_handoff and work_ref: # A failed/forced parent still owes an already admitted child's result. - # Transports poll only deferred outcomes; the current body remains true. + # Transports poll only deferred outcomes, even when no reply was authored. outcome = "deferred" - result_text = str(text or "") - if outcome in {"message", "deferred"} and provider_notice: - from ouroboros.task_finalization import provider_terminal_body - - result_text = provider_terminal_body(result_text, provider_notice) + outcome, result_text = _presence_delivery(outcome, str(text or ""), terminal_origin) metadata = task.get("metadata") if isinstance(task.get("metadata"), dict) else {} metadata["presence_outcome"] = outcome - if outcome in {"message", "deferred"}: - metadata["presence_result_text"] = result_text + metadata["presence_result_text"] = result_text if work_ref: metadata["presence_work_ref"] = work_ref task["metadata"] = metadata @@ -99,7 +131,7 @@ def build_presence_result_event(task: dict[str, Any], text: str, ctx: Any, *, pr "type": "presence_result", "task_id": str(task.get("id") or ""), "outcome": outcome, - "text": result_text if outcome in {"message", "deferred"} else "", + "text": result_text, "work_ref": work_ref, "ts": utc_now_iso(), } @@ -219,21 +251,7 @@ def _cached_result(drive_root: Path, task_id: str) -> PresenceTurnResult | None: stored = load_task_result(drive_root, task_id) or {} if str(stored.get("status") or "") not in {"completed", "failed"}: return None - metadata = stored.get("metadata") if isinstance(stored.get("metadata"), dict) else {} - outcome = str(metadata.get("presence_outcome") or "message") - if outcome not in {"message", "silent", "tool_delivered", "deferred"}: - outcome = "message" - return PresenceTurnResult( - outcome=outcome, - text=( - str(metadata.get("presence_result_text") or stored.get("result") or "") - if outcome in {"message", "deferred"} - else "" - ), - task_id=task_id, - work_ref=str(metadata.get("presence_work_ref") or ""), - delivery_reporting_version=int((metadata.get("presence") or {}).get("delivery_reporting_version") == 1), - ) + return presence_result_from_stored(stored, task_id) def _log_dialogue( diff --git a/ouroboros/task_finalization.py b/ouroboros/task_finalization.py index 79e4f90a0..76d94b99f 100644 --- a/ouroboros/task_finalization.py +++ b/ouroboros/task_finalization.py @@ -45,8 +45,8 @@ TERMINAL_ORIGIN_MODEL_FINAL = "model_final" TERMINAL_ORIGIN_HOST_SALVAGE = "host_salvage" # A terminal text the HOST wrote alone (a budget rejection, a round-limit rail # with nothing to deliver, a scheduled swarm handoff). It is not salvage: its -# own words ARE the answer, so they are published verbatim on every transport -# instead of being replaced by the outage receipt. +# own words remain an owner-facing System diagnostic. Presence never turns +# host-authored text into external speech. TERMINAL_ORIGIN_HOST_NOTICE = "host_notice" HOST_AUTHORED_TERMINAL_ORIGINS = frozenset({ TERMINAL_ORIGIN_HOST_SALVAGE, TERMINAL_ORIGIN_HOST_NOTICE, diff --git a/tests/test_presence_completion.py b/tests/test_presence_completion.py index 0c60df951..fa6a67f72 100644 --- a/tests/test_presence_completion.py +++ b/tests/test_presence_completion.py @@ -72,7 +72,7 @@ def test_explicit_finish_uses_one_model_round_and_real_tool_batch(turn, outcome, assert usage["presence_completion_outcome"] == outcome assert usage["terminal_origin"] == "model_final" assert len(trace["tool_calls"]) == 2 - result = build_presence_result_event({"id": "parent1"}, text, registry._ctx) + result = build_presence_result_event({"id": "parent1"}, text, registry._ctx, terminal_origin=usage.get("terminal_origin", "")) assert result["outcome"] == outcome assert result["text"] == (message if outcome in {"message", "deferred"} else "") assert derive_loop_outcome(text, usage, trace)["outcome_axes"]["execution"]["status"] == "ok" @@ -101,7 +101,7 @@ def test_review_hold_drops_old_outcome_and_uses_replacement(turn, monkeypatch): assert len(calls) == 2 and reviews == ["Old answer", "Revised answer"] assert registry._ctx._presence_completion is None assert "presence_completion_outcome" not in usage - result = build_presence_result_event({"id": "parent1"}, text, registry._ctx) + result = build_presence_result_event({"id": "parent1"}, text, registry._ctx, terminal_origin=usage.get("terminal_origin", "")) assert (result["outcome"], result["text"]) == ("message", "Revised answer") @@ -164,7 +164,7 @@ def test_owner_followup_invalidates_finish_before_or_during_final_gate(turn, tmp assert len(calls) == 2 and text == "With the new detail" assert any("new detail" in str(row.get("content")) for row in calls[-1]) assert "presence_completion_outcome" not in usage - assert build_presence_result_event({"id": "parent1"}, text, registry._ctx)["outcome"] == "message" + assert build_presence_result_event({"id": "parent1"}, text, registry._ctx, terminal_origin=usage.get("terminal_origin", ""))["outcome"] == "message" @pytest.mark.parametrize("reason", ["cancel", "budget"]) @@ -194,8 +194,9 @@ def test_control_or_budget_tail_precedes_pending_finish(turn, tmp_path, monkeypa assert "presence_completion_outcome" not in usage assert registry._ctx._presence_completion_accepted is False assert usage["execution_status"] == "failed" - result = build_presence_result_event({"id": "parent1"}, text, registry._ctx) - assert result["outcome"] == "message" and result["text"] != "Old answer" + result = build_presence_result_event({"id": "parent1"}, text, registry._ctx, terminal_origin=usage.get("terminal_origin", "")) + assert result["outcome"] == "silent" and result["text"] == "" + assert usage["terminal_origin"] == "host_notice" if reason == "budget": assert tail == ["budget"] and len(calls) == 1 else: @@ -223,4 +224,4 @@ def test_pending_children_still_require_absorption(turn, tmp_path): assert len(calls) > 1 assert usage["reason_code"] == "children_unabsorbed" assert "presence_completion_outcome" not in usage - assert build_presence_result_event({"id": "parent1"}, text, _registry._ctx)["outcome"] == "message" + assert build_presence_result_event({"id": "parent1"}, text, _registry._ctx, terminal_origin=usage.get("terminal_origin", ""))["outcome"] == "message" diff --git a/tests/test_presence_failed_handoff.py b/tests/test_presence_failed_handoff.py index 40acb7809..7c116b64d 100644 --- a/tests/test_presence_failed_handoff.py +++ b/tests/test_presence_failed_handoff.py @@ -19,6 +19,7 @@ def _failed_parent(tmp_path, *, outcome="deferred", admission="scheduled", usage created = [] text = "Current partial result after the parent stopped." terminal_usage = {"execution_status": "infra_failed", "reason_code": "provider_unavailable", + "terminal_origin": "model_final", "terminal_provider_notice": "Provider unavailable; child work remains admitted."} if usage is None else usage class Agent: @@ -52,7 +53,8 @@ def test_failed_parent_and_cached_result_keep_admitted_child_pollable(tmp_path, result, stored, text = _failed_parent(tmp_path, outcome=outcome, request_text=request_text) assert result.outcome == "deferred" and result.work_ref == "managed-work" assert result.delivery_reporting_version == 1 - assert result.text.startswith(text) and result.text.count("[Host status]") == 1 + assert result.text == text + assert stored["terminal_provider_notice"] == "Provider unavailable; child work remains admitted." assert "Outdated proposed reply" not in result.text assert stored["metadata"]["presence_outcome"] == "deferred" assert stored["metadata"]["presence_result_text"] == result.text @@ -62,7 +64,8 @@ def test_failed_parent_and_cached_result_keep_admitted_child_pollable(tmp_path, @pytest.mark.parametrize("admission", ["scheduled", "unconfirmed", "rejected", ""]) def test_forced_best_effort_requires_positive_admission_and_keeps_current_body(tmp_path, admission): - usage = {"execution_status": "failed", "reason_code": "round_limit", "_best_effort_extracted": True} + usage = {"execution_status": "failed", "reason_code": "round_limit", "_best_effort_extracted": True, + "terminal_origin": "model_final"} result, stored, text = _failed_parent(tmp_path, outcome="silent", admission=admission, usage=usage, accepted=True) assert result.outcome == ("deferred" if admission == "scheduled" else "message") @@ -74,7 +77,7 @@ def test_forced_best_effort_requires_positive_admission_and_keeps_current_body(t @pytest.mark.parametrize("outcome", ["deferred", "silent", "tool_delivered"]) def test_successful_replacement_answer_does_not_inherit_old_outcome(tmp_path, outcome): - result, stored, text = _failed_parent(tmp_path, outcome=outcome, usage={}) + result, stored, text = _failed_parent(tmp_path, outcome=outcome, usage={"terminal_origin": "model_final"}) assert result.outcome == "message" and result.text == text assert result.work_ref == "managed-work" assert stored["status"] == "completed" and stored["outcome_axes"]["execution"]["status"] == "ok" @@ -82,6 +85,6 @@ def test_successful_replacement_answer_does_not_inherit_old_outcome(tmp_path, ou @pytest.mark.parametrize("outcome", ["silent", "tool_delivered"]) def test_accepted_successful_nonmessage_remains_nonmessage(tmp_path, outcome): - result, stored, _text = _failed_parent(tmp_path, outcome=outcome, usage={}, accepted=True) + result, stored, _text = _failed_parent(tmp_path, outcome=outcome, usage={"terminal_origin": "model_final"}, accepted=True) assert result.outcome == outcome and result.text == "" assert stored["status"] == "completed" diff --git a/tests/test_presence_handoff.py b/tests/test_presence_handoff.py index 115774730..b948042c2 100644 --- a/tests/test_presence_handoff.py +++ b/tests/test_presence_handoff.py @@ -34,7 +34,7 @@ def test_presence_handoff_writer_feeds_terminal_consumer_and_preserves_first_rec assert ctx._swarm_handoff_attempt is first assert first["task_id"] == "managed-presence-work" and first["status"] == status task = {"id": "presence-turn", "metadata": dict(ctx.task_metadata)} - terminal = build_presence_result_event(task, "Work continues.", ctx) + terminal = build_presence_result_event(task, "Work continues.", ctx, terminal_origin="model_final") assert terminal["work_ref"] == (event["task_id"] if status == "scheduled" else "") assert terminal["outcome"] == ("deferred" if status == "scheduled" else "message") diff --git a/tests/test_presence_terminal_authorship.py b/tests/test_presence_terminal_authorship.py new file mode 100644 index 000000000..33309e38b --- /dev/null +++ b/tests/test_presence_terminal_authorship.py @@ -0,0 +1,221 @@ +"""External speech follows terminal authorship through execution and replay.""" + +import queue +from types import SimpleNamespace + +import pytest +from starlette.testclient import TestClient + +from ouroboros import agent as agent_module, agent_task_pipeline as pipeline, loop +from ouroboros.gateway.host_service import create_host_service_app +from ouroboros.presence_authority import presence_ceiling_payload +from ouroboros.presence_runner import _cached_result, build_presence_result_event +from ouroboros.task_results import load_task_result, write_task_result +from ouroboros.task_finalization import provider_terminal_body +from ouroboros.tools.registry import ToolRegistry +from tests.test_host_service_api import _seed_presence_behavior, _seed_token +from tests.test_presence_completion import _call +from tests.test_presence_failed_handoff import _failed_parent +from tests.test_presence_runner import _admission + + +def _run_loop(root, monkeypatch, responses, *, held=False): + monkeypatch.setenv("OUROBOROS_TASK_REVIEW_MODE", "off") + monkeypatch.setenv("OUROBOROS_MAX_ROUNDS", "1") + registry = ToolRegistry(repo_dir=root, drive_root=root) + registry._ctx.is_direct_chat = True + registry._ctx.task_metadata = {"inline_max_rounds": 1} + registry._ctx.task_contract = {"capability_ceiling": presence_ceiling_payload(_admission().capability_ceiling)} + registry.override_handler("chat_history", lambda *_a, **_kw: "Synthetic history") + calls, held_origins = [], [] + responses = iter(responses) + + def respond(*_a, **_kw): + calls.append(1) + return next(responses), 0.0 + + def review(**_kw): + if held: + held_origins.append(registry._ctx._accumulated_usage.get("terminal_origin")) + registry._ctx._owner_directives = [{"content": "A new material constraint"}] + return held + + monkeypatch.setattr(loop, "call_llm_with_retry", respond) + monkeypatch.setattr(loop, "_run_task_acceptance_review_once", review) + text, usage, trace = loop.run_llm_loop( + [{"role": "user", "content": "Please help"}], registry, + SimpleNamespace(default_model=lambda: "test-model"), root / "logs", + lambda *_a, **_kw: None, queue.Queue(), task_id="presence-loop", drive_root=root, + ) + task = {"id": "presence-loop", "type": "presence", "_presence_turn": True, + "chat_id": 7, "text": "Please help", "_skip_post_task_synthesis": True} + events = [] + pipeline.emit_task_results(SimpleNamespace(drive_root=root, repo_dir=root), None, None, + events, task, text, usage, trace, 0.0, root / "logs", ctx=registry._ctx) + result = next(row for row in events if row["type"] == "presence_result") + return result, load_task_result(root, task["id"]), calls, held_origins + + +def _read_response(): + return {"role": "assistant", "content": None, "tool_calls": [{ + "id": "read", "type": "function", "function": {"name": "chat_history", "arguments": "{}"}, + }]} + + +@pytest.mark.parametrize("authored", [False, True]) +def test_real_round_limit_delivers_only_the_current_authored_final(tmp_path, monkeypatch, authored): + reply = "I found the record; the remaining check is incomplete." if authored else "" + result, stored, calls, _held = _run_loop(tmp_path, monkeypatch, [_read_response(), {"content": reply}]) + assert len(calls) == 2 and stored["reason_code"] == "round_limit" + assert stored["terminal_origin"] == ("model_final" if authored else "host_notice") + assert result["outcome"] == ("message" if authored else "silent") + assert result["text"] == reply + assert _cached_result(tmp_path, result["task_id"]).text == reply + if not authored: + assert "MAX_ROUNDS" in stored["result"] + else: + assert stored["result"] == reply + + +def test_exact_host_diagnostic_is_deliverable_when_the_model_authors_it(tmp_path, monkeypatch): + _result, host, _calls, _held = _run_loop(tmp_path / "host", monkeypatch, [_read_response(), {"content": ""}]) + result, authored, calls, _held = _run_loop(tmp_path / "author", monkeypatch, [{"content": host["result"]}]) + assert len(calls) == 1 # ordinary implicit final, no presence_finish required + assert authored["terminal_origin"] == "model_final" + assert result["outcome"] == "message" and result["text"] == host["result"] + + +def test_held_model_candidate_cannot_author_a_later_host_fallback(tmp_path, monkeypatch): + result, stored, _calls, origins = _run_loop(tmp_path, monkeypatch, + [{"content": "Not yet a final answer"}, {"content": ""}], held=True) + assert origins == [None] + assert stored["terminal_origin"] == "host_notice" + assert result["outcome"] == "silent" and result["text"] == "" + + +@pytest.mark.parametrize("origin", ["host_notice", "host_salvage", ""]) +@pytest.mark.parametrize("admission", ["scheduled", "unconfirmed", "rejected"]) +def test_answerless_failed_parent_keeps_only_admitted_child_custody(tmp_path, origin, admission): + result, stored, raw = _failed_parent(tmp_path, admission=admission, accepted=True, usage={ + "execution_status": "infra_failed", "reason_code": "provider_unavailable", "terminal_origin": origin, + }) + assert result.outcome == ("deferred" if admission == "scheduled" else "silent") + assert result.work_ref == ("managed-work" if admission == "scheduled" else "") + assert result.text == "" and stored["result"] == raw + assert stored["status"] == "failed" + + +@pytest.mark.parametrize("origin,frozen,outcome,expected", [ + ("host_notice", "Old frozen diagnostic", "message", ""), + ("host_salvage", "Old frozen partial", "deferred", ""), + ("host_notice", None, "message", ""), + ("model_final", "", "deferred", ""), + ("", "", "deferred", ""), + ("model_final", "Exact authored reply", "message", "Exact authored reply"), + ("model_final", None, "message", "Raw retained text"), + ("", None, "message", "Raw retained text"), # explicit legacy unknown-origin compatibility + ("", "Legacy frozen reply", "message", "Legacy frozen reply"), + ("model_final", provider_terminal_body("Raw retained text", "Recorded host detail"), "message", "Raw retained text"), + ("model_final", provider_terminal_body("Older explicit reply", "Recorded host detail"), "message", + provider_terminal_body("Older explicit reply", "Recorded host detail")), + ("", provider_terminal_body("Raw retained text", "Recorded host detail"), "message", + provider_terminal_body("Raw retained text", "Recorded host detail")), + ("model_final", "Unused text", "silent", ""), + ("model_final", "Unused text", "tool_delivered", ""), +]) +def test_cached_and_deferred_api_share_authorship_and_keyed_empty_body(tmp_path, origin, frozen, outcome, expected): + _seed_token(tmp_path, skill="telegram-bot", token="presence-token", + permissions=["presence"], manifest_permissions=["presence"]) + binding_id = _seed_presence_behavior(tmp_path) + metadata = {"presence": {"binding_id": binding_id, "delivery_reporting_version": 1}, + "presence_outcome": outcome, "presence_work_ref": "later-child"} + if frozen is not None: + metadata["presence_result_text"] = frozen + write_task_result(tmp_path, "completed-child", "completed", result="Raw retained text", + terminal_origin=origin, terminal_host_notice="Recorded host detail", metadata=metadata) + cached = _cached_result(tmp_path, "completed-child") + with TestClient(create_host_service_app(tmp_path)) as client: + response = client.get("/presence/work/completed-child", params={"binding_id": binding_id}, + headers={"X-Skill-Token": "presence-token"}) + assert response.status_code == 200 + body = response.json() + assert body["text"] == cached.text == expected + assert body["outcome"] == cached.outcome + assert body["status"] == "completed" # producer lifecycle is independent of speech + assert body["work_ref"] == "completed-child" and cached.work_ref == "later-child" + assert body["delivery_reporting_version"] == cached.delivery_reporting_version == 1 + if origin in {"host_notice", "host_salvage"}: + assert body["outcome"] == ("deferred" if outcome == "deferred" else "silent") + + +def test_fresh_missing_origin_never_borrows_legacy_authorship(): + task = {"id": "new"} + result = build_presence_result_event(task, "Unknown producer", SimpleNamespace()) + assert result["outcome"] == "silent" and result["text"] == "" + assert task["metadata"]["presence_result_text"] == "" + assert "terminal_origin" not in task # absence is not relabelled as host + + +@pytest.fixture +def native_agent(tmp_path, monkeypatch): + monkeypatch.setattr(agent_module.OuroborosAgent, "_log_worker_boot_once", lambda *_a: None) + monkeypatch.setattr(agent_module, "validate_task_authority_sources", lambda *_a: None) + monkeypatch.setattr(agent_module.OuroborosAgent, "_start_task_heartbeat_loop", lambda *_a: None) + agent = agent_module.OuroborosAgent(agent_module.Env(repo_dir=tmp_path, drive_root=tmp_path)) + ctx = agent.tools._ctx + monkeypatch.setattr(agent, "_prepare_task_context", lambda *_a: (ctx, [], {})) + return agent, ctx + + +@pytest.mark.parametrize("case", ["accepted_empty", "nonstring", "exception", "budget"]) +def test_native_host_replacement_resets_authorship_and_keeps_diagnostic(tmp_path, monkeypatch, native_agent, case): + from ouroboros.usage_accounting import BudgetExceeded + + agent, ctx = native_agent + ctx._presence_completion = {"outcome": "message"} + ctx._presence_completion_accepted = True + + accepted = [] + real_loop = agent_module.run_llm_loop + if case == "accepted_empty": + monkeypatch.setenv("OUROBOROS_TASK_REVIEW_MODE", "off") + ctx.task_contract = {"capability_ceiling": presence_ceiling_payload(_admission().capability_ceiling)} + ctx.is_direct_chat = True + agent.llm = SimpleNamespace(default_model=lambda: "test-model") + replies = iter([_call("message", ""), {"content": ""}]) + monkeypatch.setattr(loop, "call_llm_with_retry", lambda *_a, **_kw: (next(replies), 0.0)) + + def run(**kwargs): + if case == "accepted_empty": + result = real_loop(**kwargs) + accepted.append((ctx._presence_completion_accepted, dict(result[1]))) + return result + if case == "exception": + error = RuntimeError("Synthetic processing failure") + error._ouroboros_loop_usage = {"terminal_origin": "model_final"} + raise error + if case == "budget": + raise BudgetExceeded("Synthetic exhausted budget") + return ("" if case == "accepted_empty" else None), { + "terminal_origin": "model_final", "presence_completion_outcome": "message", + }, {"tool_calls": [], "reasoning_notes": []} + + monkeypatch.setattr(agent_module, "run_llm_loop", run) + events = agent._handle_task_scoped({"id": "replacement", "chat_id": 7, "type": "presence", + "_presence_turn": True, "_is_direct_chat": True, "_skip_post_task_synthesis": True, "text": "Go"}) + result = next(row for row in events if row["type"] == "presence_result") + stored = load_task_result(tmp_path, "replacement") + assert stored["terminal_origin"] == "host_notice" and stored["result"] + assert result["outcome"] == "silent" and result["text"] == "" + assert _cached_result(tmp_path, "replacement").text == "" + if case in {"accepted_empty", "nonstring"}: + assert ctx._presence_completion_accepted is False + assert "empty response" in stored["result"] + assert "presence_completion_outcome" not in stored["loop_outcome"].get("usage", {}) + if case == "accepted_empty": + assert accepted[0][0] is True + assert accepted[0][1]["terminal_origin"] == "model_final" + assert accepted[0][1]["presence_completion_outcome"] == "message" + else: + assert stored["status"] == "failed" + assert stored["reason_code"] == ("task_exception" if case == "exception" else "budget_exhausted") diff --git a/tests/test_presence_terminal_latency.py b/tests/test_presence_terminal_latency.py index 76027c4fa..f75c24bf6 100644 --- a/tests/test_presence_terminal_latency.py +++ b/tests/test_presence_terminal_latency.py @@ -139,5 +139,7 @@ def test_pending_finish_cannot_hide_a_failed_empty_agent_result(tmp_path, monkey events = agent._handle_task_scoped({"id": "failed", "chat_id": 7, "type": "presence", "_presence_turn": True, "_is_direct_chat": True, "_skip_post_task_synthesis": True, "text": "Go"}) result = next(row for row in events if row["type"] == "presence_result") - assert result["outcome"] == "message" and "empty response" in result["text"] - assert load_task_result(tmp_path, "failed")["status"] == "failed" + assert result["outcome"] == "silent" and result["text"] == "" + stored = load_task_result(tmp_path, "failed") + assert stored["status"] == "failed" and "empty response" in stored["result"] + assert stored["terminal_origin"] == "host_notice" diff --git a/tests/test_provider_terminal_notice.py b/tests/test_provider_terminal_notice.py index 1fdd8dabb..df4013e09 100644 --- a/tests/test_provider_terminal_notice.py +++ b/tests/test_provider_terminal_notice.py @@ -78,7 +78,7 @@ def test_pipeline_delivery_and_rebuild_keep_raw_bytes_and_known_wait_custody(tmp @pytest.mark.parametrize("outcome", ["message", "deferred", "silent", "tool_delivered"]) @pytest.mark.parametrize("current", [False, True]) -def test_actual_presence_render_and_cached_read_keep_failure_notice_over_pending_outcome(tmp_path, monkeypatch, outcome, current): +def test_actual_presence_and_cached_read_keep_authored_speech_and_owner_notice_separate(tmp_path, monkeypatch, outcome, current): monkeypatch.setattr(pipeline, "_run_post_task_processing_async", lambda *_a, **_k: None) repo, data = tmp_path / "repo", tmp_path / "data" repo.mkdir() @@ -106,8 +106,7 @@ def test_actual_presence_render_and_cached_read_keep_failure_notice_over_pending stored = load_task_result(data, first.task_id) assert stored["result"] == RAW assert "no terminal provider outcome" in stored["terminal_provider_notice"] - assert first.text.startswith(RAW + "\n\n[Host status]") - assert first.text.count("[Host status]") == 1 + assert first.text == (RAW if current else "") assert stored["metadata"]["presence_result_text"] == first.text assert first.work_ref == "next-task" diff --git a/tests/test_provider_wall_rail_contract.py b/tests/test_provider_wall_rail_contract.py index afeceb119..197e47844 100644 --- a/tests/test_provider_wall_rail_contract.py +++ b/tests/test_provider_wall_rail_contract.py @@ -91,7 +91,7 @@ def test_wall_exhausted_body_429_empty_is_infra_not_a_model_failure(tmp_path, mo def test_scheduled_presence_handoff_survives_the_no_call_rail(tmp_path): - """Presence keeps the admitted child and discloses the current provider outage.""" + """Presence keeps the admitted child without sending host-salvaged text.""" from ouroboros.presence_runner import build_presence_result_event tools_ctx = SimpleNamespace( task_metadata={"presence": {"binding_id": "presence-binding"}}, @@ -113,10 +113,10 @@ def test_scheduled_presence_handoff_survives_the_no_call_rail(tmp_path): assert usage["reason_code"] == "provider_unavailable" terminal = build_presence_result_event( {"id": "presence-turn"}, text, tools_ctx, - provider_notice=usage["terminal_provider_notice"], + terminal_origin=usage["terminal_origin"], retain_scheduled_handoff=True, ) assert terminal["work_ref"] == "t-child-1" assert terminal["outcome"] == "deferred" # admitted work retains its polling custody after the parent failure - assert terminal["text"].startswith("PARTIAL RESULT.") - assert terminal["text"].count(usage["terminal_provider_notice"]) == 1 + assert usage["terminal_origin"] == "host_salvage" + assert terminal["text"] == "" diff --git a/tests/test_terminal_host_notice.py b/tests/test_terminal_host_notice.py index 975842501..63458303d 100644 --- a/tests/test_terminal_host_notice.py +++ b/tests/test_terminal_host_notice.py @@ -611,7 +611,7 @@ def test_open_custody_is_its_own_card_row_live_and_on_history_replay(tmp_path, m @pytest.mark.parametrize("outcome", ["message", "deferred", "silent", "tool_delivered"]) -def test_presence_delivers_host_notice_once_and_preserves_silence(tmp_path, monkeypatch, outcome): +def test_presence_preserves_authored_speech_and_keeps_host_notice_in_task(tmp_path, monkeypatch, outcome): from ouroboros.presence_runner import PresenceTurnGate, run_presence_turn from tests.test_presence_runner import _admission, _event @@ -635,7 +635,8 @@ def test_presence_delivers_host_notice_once_and_preserves_silence(tmp_path, monk assert run_presence_turn(**args) == first assert first.outcome == outcome assert load_task_result(tmp_path, first.task_id)["result"] == ANSWER - assert first.text == (ANSWER + "\n\n[Host status]\n" + NOTICE if outcome in {"message", "deferred"} else "") + assert first.text == (ANSWER if outcome in {"message", "deferred"} else "") + assert load_task_result(tmp_path, first.task_id)["terminal_host_notice"] == NOTICE def test_a_failed_answer_send_stays_owed_and_owes_no_second_row(tmp_path, monkeypatch): From e7d76d335b32fe219eb03a776e4aa6c6468c67e6 Mon Sep 17 00:00:00 2001 From: Ouroboros Date: Wed, 23 Sep 2026 04:56:44 +0300 Subject: [PATCH 07/15] Keep Presence finalization dependency lazy Reuse the established cross-domain call boundary for authored delivery projection. Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> --- ouroboros/presence_runner.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/ouroboros/presence_runner.py b/ouroboros/presence_runner.py index 8984a04cd..264165d1a 100644 --- a/ouroboros/presence_runner.py +++ b/ouroboros/presence_runner.py @@ -15,10 +15,6 @@ from ouroboros.artifacts import stage_task_attachments from ouroboros.contracts.task_contract import attach_task_contract from ouroboros.presence_admission import PresenceAdmission from ouroboros.presence_authority import presence_ceiling_payload -from ouroboros.task_finalization import ( - HOST_AUTHORED_TERMINAL_ORIGINS, TERMINAL_ORIGIN_MODEL_FINAL, - provider_terminal_body, terminal_notice_text, -) from ouroboros.task_results import load_task_result from ouroboros.utils import append_jsonl, read_json_dict, utc_now_iso @@ -65,6 +61,8 @@ class PresenceTurnResult: def _presence_delivery(outcome: str, text: str, terminal_origin: str, *, legacy: bool = False) -> tuple[str, str]: """Project speech from producer facts, never from its wording or task status.""" + from ouroboros.task_finalization import HOST_AUTHORED_TERMINAL_ORIGINS, TERMINAL_ORIGIN_MODEL_FINAL + if outcome not in {"message", "silent", "tool_delivered", "deferred"}: outcome = "message" if outcome not in {"message", "deferred"}: @@ -79,6 +77,8 @@ def _presence_delivery(outcome: str, text: str, terminal_origin: str, *, legacy: def presence_result_from_stored(stored: Mapping[str, Any], task_id: str) -> PresenceTurnResult: """One replay projection for cached turns and completed delegated work.""" + from ouroboros.task_finalization import TERMINAL_ORIGIN_MODEL_FINAL, provider_terminal_body, terminal_notice_text + metadata = stored.get("metadata") if isinstance(stored.get("metadata"), dict) else {} text = metadata["presence_result_text"] if "presence_result_text" in metadata else stored.get("result") origin = str(stored.get("terminal_origin") or "") From d9ba61915f349e3a102402816dda39c4a997e259 Mon Sep 17 00:00:00 2001 From: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> Date: Wed, 23 Sep 2026 04:58:38 +0300 Subject: [PATCH 08/15] fix: keep landing patch history within the release cap --- README.md | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/README.md b/README.md index 782a82cba..c51332cd1 100644 --- a/README.md +++ b/README.md @@ -462,9 +462,8 @@ and the reason. | 7.2.0 | 2026-09-19 | **feat: Light, Dark and System appearance, review that reads the repository instead of a packed snapshot, and one fast local test battery.** The web UI gains a client-local Appearance choice — Light, Dark or System — applied before first paint in the app and the onboarding wizard and carried through in-page charts, diagrams and host-rendered widget charts, while framed skill widgets keep their own colours (#1107). With no saved choice the UI follows the system scheme, so an existing install on a light OS opens in Light until Dark is chosen; to remember the choice the desktop shell now keeps website data, including cookies, and a desktop app that only took in-app updates remembers it once the 7.2.0 installer has been installed. Project questions are one row in Main and become a card only while the task waits (#1106). Scope review retrieves what it needs from the repository rather than receiving one packed snapshot (#1043); plan review keeps paid feedback (#1051), acceptance review keeps its rework controls, final review text and explicit author completion (#1042, #1069, #1089), and system review mail no longer reopens accepted results (#1088). `python scripts/run_tests.py` is the one documented local battery: it runs the node lane first, then every default-lane test in one parallel run (#1101). First-run setup recovers subscription sign-in and model discovery (#1099), and Finish stays usable after a refused Main-reviewer recovery (#1100); delegated requests are stored by reference and keep pending recovery (#1067); terminal outcomes, usage accounting, cancellation provenance, task relay authorship and process environments are preserved end to end (#1054, #1064, #1083, #1098). Also: the managed Claudexor runtime moves to 3.12.4, an experimental Android host with optional release artifacts (#873), bounded Cowork Bench meter reads that preserve interrupted work (#1044), chat composer, Activity row and project-link fixes (#1063, #1095), clearer Presence delivery (#1053), subagent scheduling refusals (#1092), settings and helper-history reporting (#1090), and scope-review context logging from a community contribution (#752). | | 7.1.0 | 2026-09-17 | **feat: memory that understands people, Background Consciousness as an ordinary Main turn, and a chat that leads with narration.** Memory keeps an understanding of people in front of the mind through resident summaries, one memory shared by every room and honest consolidation (#910). Background Consciousness becomes an ordinary Main turn on an alarm clock (#988, closing the class behind #654). Chat blocks lead with narration while tool calls fold into one evidence row (#1007); task cards follow their work and treat acceptance review as advice to Ouroboros (#970); question pointers say whether an answer is wanted and show the recorded answer (#1025); the unified UI control system, source picker and model chooser name models and accounts truthfully (#768, #862, #890). Planning is asynchronous and Swarm tasks are admitted directly (#851); delegated work survives durable cancellation (#850); a task's own reviewer is never swept as its delegation (#1008); semantic duplicate vetoes leave subagent admission (#887); Presence profiles work in an owner-selected folder (#941). Prompt caches extend through an append-only acceptance observation and a shared Codex cache key per install and model (#929, #1015). Update letters include merged branch changes (#1003), the owned Claudexor daemon latches failed starts with a typed diagnosis, and the managed runtime advances to Claudexor 3.12.1 (#891, #962), alongside Windows, platform-installation and UI-smoke repairs. | | 7.0.0 | 2026-09-08 | **v7: a modular runtime with full ordinary-conversation tools, complete delegated inputs and visible owner dialogue.** Main and Project conversations retain their chosen tools and working folders; required owner waits preserve the live browser without occupying pooled execution capacity. Publish admission is visible in its intended chat, and GitHub/image/plan failures retain their actual causes. Planning advice stays optional; source-backed evidence, separate host notices and exact-workload test reuse preserve honest review without new approval machinery. Integrates the subscription model and account-role controls, owned startup/restart custody and cross-platform repairs, with Claudexor pinned to 3.10.1. | -| 6.113.5 | 2026-08-31 | **fix: browser cleanup after a timed-out tool is generation-safe (integrates community PR #429 by @mikemikimike, closes #409).** A stateful-tool timeout now retires the whole browser generation: the shared state slot is replaced with a fresh object, the abandoned worker keeps writing only into its retired one, and the close is queued on the retiring executor so it always runs on the owning worker thread — including the already-settled race — with the cognitive lease closing after that cleanup. A late infrastructure-error retry that observes a replaced generation closes only its own retired session, so it can no longer cross-thread-kill the next command's browser. Closes the follow-up findings of the #440 post-merge audit; the never-settling-worker session leak stays a disclosed residual of in-process Playwright. | | 6.113.0 | 2026-08-29 | **feat: delegation by construction — the nanny charter, typed $0 terminals, truthful executor cards, and honest route health (Claudexor runtime 3.9.0).** An `agent_session` child now IS work on its harness: the host pre-starts the physical leaf through the same configured `delegate_start` wrapper BEFORE the nanny's first LLM round and never waits — her first round arrives with a live `configured_session_started` receipt, waiting is her own `delegate_wait` decision, and children may run beside the leaf. A definite refusal to start (typed pre-POST, dispatch-blocked, engine-rejected — with a custody-handle guard that always prefers a model episode over a false terminal) ends the task typed and unrun at $0; ambiguity always wakes the model, and durable zero-run/unknown-evidence fences outrank blocked terminals. Zero-run receipts narrow to incomplete\|unknown, actor cleanliness requires a SUCCEEDED delegated run (children are evidence, never a completion path), unreadable custody projects typed unknown all the way into the finalization nudge, and only acts of delegation reset the economics baseline (the reminder-storm class is dead). route_health stops refusing on aggregate doctor status — admission belongs to the engine; the owner's enabled toggle stays a typed `route_disabled`. Acceptance sees substrate facts as visibility with zero gates. The executor chip tells the run truth for the whole lifecycle (dispatched → counted `N ok, M failed` → evidence honesty), all-failed can never render clean, actual_substrate reaches the wire, and the terminal evidence frame survives chat-0/A2A routing. The pinned Claudexor runtime moves to 3.9.0: per-vendor quota pacing with typed Retry-After floors (a poll 429 is never a quota fact), honest foreground cooldowns, first-429 short-circuit, cached accounts default, and the cursor delegation belt (live-E2E proven). This tag also carries the untagged 6.111.0 and 6.112.0 (P13 Emergence) below, and heals the branch's latent size-ratchet debt root-cause (settings_integrity extraction, cybergym module splits, regenerated manifest). | -Older releases are preserved in this repository's history. Rows 6.114.0 and 6.113.2 were rolled off in this release. Older 6.x rows (including 6.113.1, 6.110.0, 6.109.0, 6.110.1, 6.108.1, 6.106.0, 6.101.1, 6.97.2, 6.105.0, 6.97.1, 6.97.0, 6.96.1, 6.96.0, 6.95.0, 6.94.0, 6.93.0, 6.92.1, 6.92.0, 6.91.1, 6.90.3, 6.91.0, 6.90.2, 6.90.0, 6.87.5, 6.87.4, 6.87.3, 6.87.2, 6.84.0, 6.87.1, 6.83.0, 6.86.1, 6.81.1, 6.76.0, 6.75.0, 6.74.5, 6.74.4, 6.74.1, 6.74.0, 6.73.2, 6.73.1, 6.73.0, 6.72.0, 6.71.2, 6.71.1, 6.71.0, 6.70.0, 6.69.0, 6.68.0, 6.67.0, 6.66.0, 6.65.4, 6.65.3, 6.65.2, 6.65.1, 6.65.0, 6.64.3, 6.64.2, 6.64.1, 6.64.0, 6.63.0, 6.62.0, 6.61.4, 6.61.3, 6.61.1, 6.61.0, 6.60.0, 6.59.0, 6.58.0, 6.57.0, 6.56.0, 6.55.0, 6.54.4, 6.54.2, 6.54.1, 6.54.0, 6.53.4, 6.53.0, 6.51.0), the 5.2.0 through 5.33.0-rc.6 rows, and former `4.0.0` rows are rolled off to respect the P9 changelog cap; their full bodies remain in this file's git history, with release tags where present for historical versions. +Older releases are preserved in this repository's history. Rows 6.114.0, 6.113.2 and 6.113.5 are retained in Git history. Older 6.x rows (including 6.113.1, 6.110.0, 6.109.0, 6.110.1, 6.108.1, 6.106.0, 6.101.1, 6.97.2, 6.105.0, 6.97.1, 6.97.0, 6.96.1, 6.96.0, 6.95.0, 6.94.0, 6.93.0, 6.92.1, 6.92.0, 6.91.1, 6.90.3, 6.91.0, 6.90.2, 6.90.0, 6.87.5, 6.87.4, 6.87.3, 6.87.2, 6.84.0, 6.87.1, 6.83.0, 6.86.1, 6.81.1, 6.76.0, 6.75.0, 6.74.5, 6.74.4, 6.74.1, 6.74.0, 6.73.2, 6.73.1, 6.73.0, 6.72.0, 6.71.2, 6.71.1, 6.71.0, 6.70.0, 6.69.0, 6.68.0, 6.67.0, 6.66.0, 6.65.4, 6.65.3, 6.65.2, 6.65.1, 6.65.0, 6.64.3, 6.64.2, 6.64.1, 6.64.0, 6.63.0, 6.62.0, 6.61.4, 6.61.3, 6.61.1, 6.61.0, 6.60.0, 6.59.0, 6.58.0, 6.57.0, 6.56.0, 6.55.0, 6.54.4, 6.54.2, 6.54.1, 6.54.0, 6.53.4, 6.53.0, 6.51.0), the 5.2.0 through 5.33.0-rc.6 rows, and former `4.0.0` rows are rolled off to respect the P9 changelog cap; their full bodies remain in this file's git history, with release tags where present for historical versions. --- From a9e65e974fba4c75fdff5bca3f7b817773097c4e Mon Sep 17 00:00:00 2001 From: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> Date: Wed, 23 Sep 2026 05:20:52 +0300 Subject: [PATCH 09/15] fix: retain producer tool payloads beside dispatcher annotations Preserve exact machine-readable sources through task artifact promotion while keeping warning and outcome projections intact. --- docs/architecture/13-external-skills-layer.md | 2 + ouroboros/loop_tool_execution.py | 32 ++++- ouroboros/observability.py | 16 +-- ouroboros/size_ratchet_manifest.py | 1 + ouroboros/tools/extension_dispatch.py | 43 +++--- ouroboros/tools/tool_result.py | 27 +++- tests/test_reference_book_budgets.py | 3 +- tests/test_registry_core.py | 2 + tests/test_tool_producer_promotion.py | 65 +++++++++ tests/test_tool_producer_sources.py | 131 ++++++++++++++++++ 10 files changed, 286 insertions(+), 36 deletions(-) create mode 100644 tests/test_tool_producer_promotion.py create mode 100644 tests/test_tool_producer_sources.py diff --git a/docs/architecture/13-external-skills-layer.md b/docs/architecture/13-external-skills-layer.md index bb2c4441a..b07508ed1 100644 --- a/docs/architecture/13-external-skills-layer.md +++ b/docs/architecture/13-external-skills-layer.md @@ -16,4 +16,6 @@ The optional `model_experience` manifest prose — what the skill adds to the mo Marketplace installs are bounded archives staged privately and landed atomically with per-file hash checks. Inert resource bytes, PDF/PPTX included, need no suffix allowlist; archive validity, extraction paths, checksum and actual loader support remain separate facts, and Cyber policy findings are warnings. Install metadata drives isolated dependencies (`marketplace/install_specs.py`); manual instructions remain guidance, not execution. Fresh review and hash-covered declarations authorize literal build/check argv; verified resources and package caches survive under `state/skills//dependency_cache`, and `deps.json` records resolved package metadata, bytes, outputs and diagnostics. Delivery is distinct from a declared executable check: absent checks remain unknown, and declared-output drift invalidates installed readiness. Large resources stay outside the payload and the Git patch. Marketplace update and adopt rollback (`install.PayloadRollbackSnapshot`, verified `rolled_back`): §6 Skills and extensions; script runtimes and their Go/Deno specifics: `tools/skill_exec.py`, `docs/CREATING_SKILLS.md`. +Tool results retain two views when the host adds route or safety notes. `tools/tool_result.py` captures `producer_text` before composition and keeps `host_annotations` outside the bounded metadata map; the existing `text` projection and typed status remain authoritative for model/review behavior. Builtin, extension and MCP dispatch use this same composer. `loop_tool_execution.py` records both views in observability and stores producer bytes separately through the existing write-once task source store. `PRODUCER_RESULT_SOURCE_JSON` is the parser-readable source, distinct from the complete annotated `FULL_RESULT_SOURCE_JSON` used for review. Notes remain visible after truncation; a failed source write is disclosed rather than presented as a usable file. Unannotated results and historical retained files keep their existing representation. + Extensions import through staged trees (`_stage_extension_import_tree` under `__extension_imports/`), so concurrent workers cannot remove a peer's live import and stale trees stay reclaimable. Per-call child processes may proxy tools/routes/WS/UI/settings/companion descriptors, while persistent subscriptions and supervised tasks require in-process or companion lifecycle — reported through the generic capability matrix, never inferred from a platform name. Isolated children run the same staged loader in a private base with a scrubbed env, so native crashes cannot kill `server.py`. In-process extensions are more powerful, which is why namespacing, declared permissions, per-skill tracking and atomic unload are an executable contract rather than convention. Skill repair enqueues an ordinary managed task with an exact selected resource and revision admission (`skill_repair_admission.py`); a legacy `skill_repair` selector remains readable but selects no reduced profile. Shell, browser and delegation remain normal task capabilities; an installed payload needs no mandatory Git copy, and review never forces unload merely because the caller is repairing it. diff --git a/ouroboros/loop_tool_execution.py b/ouroboros/loop_tool_execution.py index a17ab7497..5bfbcfa8a 100644 --- a/ouroboros/loop_tool_execution.py +++ b/ouroboros/loop_tool_execution.py @@ -397,11 +397,13 @@ def _persist_truncated_tool_source( tool_call_id: str, result: Any, tool_args: Optional[Dict[str, Any]] = None, + *, + force: bool = False, ) -> Dict[str, Any]: """Write a generic over-limit result to this actor's existing artifact root.""" text = str(result) - if ( + if not force and ( _should_skip_tool_result_truncation(tool_name, tool_args) or len(text) <= _tool_result_limit(tool_name) ): @@ -732,6 +734,9 @@ def _execute_single_tool( "round_id": correlation.get("round_id"), "args": args, "result": result, + **({"producer_result": tool_result.producer_text, + "host_annotations": list(tool_result.host_annotations)} + if tool_result.producer_text is not None else {}), "tool_ok": tool_ok, "semantic_ok": not is_error, "result_meta": result_meta, @@ -1311,7 +1316,8 @@ def _maybe_auto_attach_image( typed = exec_result.get("tool_result") if isinstance(typed, ToolResult) and typed.code == "TOOL_REPORTED_FAILURE": return - raw = exec_result.get("result") + raw = (typed.producer_text if isinstance(typed, ToolResult) + and typed.producer_text is not None else exec_result.get("result")) if not isinstance(raw, str) or '"auto_attach_image"' not in raw: return observation = None @@ -1402,6 +1408,25 @@ def process_tool_results( tool_args=exec_result.get("tool_args"), source_ref=result_source_ref, ) + typed = exec_result.get("tool_result") + producer_ref = {} + if isinstance(typed, ToolResult) and typed.producer_text is not None: + if ctx is not None: + producer_ref = _persist_truncated_tool_source( + ctx, fn_name, str(exec_result["tool_call_id"]) + ".producer", + typed.producer_text, force=True, + ) + # The complete annotated source above remains review evidence. This + # separate source is for parsers; neither bytes nor notes live in meta. + if result_partial and typed.host_annotations: + truncated_result += "\n\n" + "\n\n".join(typed.host_annotations) + truncated_result += ( + "\nPRODUCER_RESULT_SOURCE_JSON=" + json.dumps(producer_ref, ensure_ascii=False) + + "\nUnannotated tool data for programmatic reading; host notes and outcome still apply." + if producer_ref else + "\nPRODUCER_RESULT_SOURCE_UNAVAILABLE=true" + "\nHost notes remain in the result; no clean producer file was retained." + ) messages.append({ "role": "tool", @@ -1448,6 +1473,9 @@ def process_tool_results( ), } if result_partial else {}), **(exec_result.get("result_meta") or {}), + **({"producer_source_ref": producer_ref, + "host_annotations": list(typed.host_annotations)} + if isinstance(typed, ToolResult) and typed.producer_text is not None else {}), }) if fn_name == "task_acceptance_review" and not is_error: raw = str(exec_result.get("result") or "") diff --git a/ouroboros/observability.py b/ouroboros/observability.py index 2025e9928..d7bb77d64 100644 --- a/ouroboros/observability.py +++ b/ouroboros/observability.py @@ -572,7 +572,7 @@ _PUBLISHED_CHILD_REF_FIELDS = frozenset( } ) _SOURCE_HANDLES_SUBDIR = "source_handles" -_TASK_SOURCE_MARKER = "FULL_RESULT_SOURCE_JSON=" +_TASK_SOURCE_MARKERS = ("FULL_RESULT_SOURCE_JSON=", "PRODUCER_RESULT_SOURCE_JSON=") _SERVICE_REF_TOOLS = frozenset({"service_logs", "stop_service"}) @@ -857,17 +857,17 @@ def _rewrite_task_source_markers( task_id: str, state: Dict[str, Any], ) -> str: - """Rewrite only Phase3B's explicit actor-source envelope inside tool text.""" + """Promote the host's full-view and clean-producer source envelopes.""" rewritten_lines: List[str] = [] for line in str(text).splitlines(keepends=True): body = line.rstrip("\r\n") - newline = line[len(body):] - if not body.startswith(_TASK_SOURCE_MARKER): + marker = next((prefix for prefix in _TASK_SOURCE_MARKERS if body.startswith(prefix)), None) + if marker is None: rewritten_lines.append(line) continue try: - ref = json.loads(body[len(_TASK_SOURCE_MARKER):]) + ref = json.loads(body[len(marker):]) except (TypeError, ValueError): rewritten_lines.append(line) continue @@ -878,14 +878,14 @@ def _rewrite_task_source_markers( parent_root, child_root, task_id, ref, state ) rewritten_lines.append( - _TASK_SOURCE_MARKER + marker + json.dumps( promoted, ensure_ascii=False, sort_keys=True, separators=(",", ":"), ) - + newline + + line[len(body):] ) return "".join(rewritten_lines) @@ -940,7 +940,7 @@ def _rewrite_child_ref_tree( _rewrite_child_ref_tree(item, parent_root, child_root, task_id, state) for item in value ] - if isinstance(value, str) and _TASK_SOURCE_MARKER in value: + if isinstance(value, str) and any(marker in value for marker in _TASK_SOURCE_MARKERS): return _rewrite_task_source_markers( value, parent_root, child_root, task_id, state ) diff --git a/ouroboros/size_ratchet_manifest.py b/ouroboros/size_ratchet_manifest.py index 06ed35ee9..38a075d3f 100644 --- a/ouroboros/size_ratchet_manifest.py +++ b/ouroboros/size_ratchet_manifest.py @@ -160,6 +160,7 @@ BAND_PATHS = { "ouroboros/tools/skill_preflight.py": "Entered the band by absorbing the upstream classic-script validator, widget entry/grammar findings and the preflight schema description beside the v7 typed _run_check rework they attach to.", "ouroboros/tools/skill_publish.py": "Entered the band from 952 lines: publish now writes the OuroborosHub publication receipt at pr_opened through the shared locked-update seam and maps the receipt from the validated serialized form (hubflow sprint, receipt-as-only-stored-fact design).", "ouroboros/tools/subagent_integration.py": "Existing native result integration owner handles source patches and complete file artifacts under one disposition and target authority.", + "ouroboros/tools/tool_result.py": "Typed result composition owns producer payload and dispatch annotations together while preserving the status and metadata contract.", "ouroboros/usage_compaction.py": "Entered the band from 971 lines: the C6 round-4 fixes homed here \u2014 dir-fd/O_NOFOLLOW anchoring of the archive writer and reader (a link planted after any path check cannot receive or serve monetary history) and the swap's last-instant snapshot re-proof inside the atomic replace \u2014 defenses that belong beside the compaction pass they defend.", "ouroboros/workspace_executor.py": None, "scripts/claudexor_platform_smoke.py": "The managed Claudexor platform smoke owns a multi-platform fixture, lifecycle receipt, and cleanup proof; keeping this runner in the documented band preserves the release gate without moving those checks into product runtime.", diff --git a/ouroboros/tools/extension_dispatch.py b/ouroboros/tools/extension_dispatch.py index 478748fa1..264bbe7b6 100644 --- a/ouroboros/tools/extension_dispatch.py +++ b/ouroboros/tools/extension_dispatch.py @@ -13,7 +13,8 @@ from ouroboros.tools.tool_context import ToolContext from ouroboros.tools.tool_result import ( ToolResult, ToolStatus, - _compose_execute_result, + _compose_execute_result_result, + _replace_tool_result, _structured_failure, ) @@ -98,11 +99,10 @@ def _dispatch_mcp_tool_result( return ToolResult(status="error", code="TOOL_ERROR", text=text) if not safety_msg: return result - text = _compose_execute_result(result.text, "", safety_msg) - meta = {**dict(result.meta), "safety_warning": True} - if result.code == "OK": - return ToolResult(status="ok", code="SAFETY_WARNING", text=text, meta=meta) - return ToolResult(status=result.status, code=result.code, text=text, meta=meta) + return _replace_tool_result( + _compose_execute_result_result(name, result, "", safety_msg), + meta_updates={"safety_warning": True}, + ) def _extension_result( @@ -138,21 +138,18 @@ def _extension_completion(result: str, safety_msg: str) -> ToolResult: structured check is the adapter's, so there is exactly one implementation of what a self-reported failure is.""" reported_failure = _structured_failure(result) + base = _extension_result( + "error" if reported_failure else "ok", + "TOOL_REPORTED_FAILURE" if reported_failure else "OK", + result, + dispatched=True, + ) if safety_msg: - # #447 H1: the warning TRAILS the payload — line 1 belongs to the - # extension, so a structured {"ok": false} answer (and any first-line - # marker) stays readable to every text-only consumer downstream. - text = f"{result}\n\n{safety_msg}" - return _extension_result( - "error" if reported_failure else "ok", - "TOOL_REPORTED_FAILURE" if reported_failure else "SAFETY_WARNING", - text, - safety_warning=True, - dispatched=True, + return _replace_tool_result( + _compose_execute_result_result("", base, "", safety_msg), + meta_updates={"safety_warning": True}, ) - if reported_failure: - return _extension_result("error", "TOOL_REPORTED_FAILURE", result, dispatched=True) - return _extension_result("ok", "OK", result, dispatched=True) + return base def _generation_digest_for(ext_tool: Dict[str, Any]) -> str: @@ -208,11 +205,9 @@ def _dispatch_extension_tool_result( or not digest ): return result - return ToolResult( - status=result.status, - code=result.code, - text=result.text, - meta={**dict(result.meta), "extension_generation": digest, + return _replace_tool_result( + result, + meta_updates={"extension_generation": digest, **({"content_hash": content_hash} if content_hash else {})}, ) diff --git a/ouroboros/tools/tool_result.py b/ouroboros/tools/tool_result.py index 70978da6b..0ed02e943 100644 --- a/ouroboros/tools/tool_result.py +++ b/ouroboros/tools/tool_result.py @@ -496,12 +496,18 @@ TOOL_CODE_SPECS: Mapping[str, ToolCodeSpec] = MappingProxyType( @dataclass(frozen=True) class ToolResult: - """Internal result; ``text`` remains the complete model-facing projection.""" + """Internal result; ``text`` includes notes, ``producer_text`` never does. + + Producer text is captured only when the host adds annotations, before any + composition. It is not bounded metadata and is never reconstructed from text. + """ status: ToolStatus code: str text: str meta: Mapping[str, Any] = field(default_factory=dict) + producer_text: str | None = None + host_annotations: tuple[str, ...] = () def __post_init__(self) -> None: if not isinstance(self.code, str) or not _CODE_RE.fullmatch(self.code): @@ -513,6 +519,12 @@ class ToolResult: raise ValueError(f"status {self.status!r} does not match {self.code} ({spec.status!r})") if not isinstance(self.text, str): raise TypeError("tool result text must be a string") + if self.producer_text is not None and not isinstance(self.producer_text, str): + raise TypeError("tool producer text must be a string or None") + if not isinstance(self.host_annotations, tuple) or any( + not isinstance(note, str) for note in self.host_annotations + ): + raise TypeError("host annotations must be a tuple of strings") raw_meta = dict(self.meta or {}) if any(not isinstance(key, str) for key in raw_meta): raise ValueError("tool result meta keys must be strings") @@ -566,6 +578,8 @@ def _replace_tool_result( code=selected_code, text=result.text if text is None else text, meta=meta, + producer_text=result.producer_text, + host_annotations=result.host_annotations, ) @@ -967,6 +981,14 @@ def _compose_execute_result_result( else LegacyTextResultAdapter.from_text(tool_name, base) ) text = _compose_execute_result(base_result.text, route_note, safety_msg) + notes = tuple(note for note in (route_note, safety_msg) if note) + source = { + "producer_text": ( + base_result.producer_text if base_result.producer_text is not None + else base_result.text if notes else None + ), + "host_annotations": base_result.host_annotations + notes, + } meta = dict(base_result.meta) if route_note: meta["route_note"] = True @@ -983,6 +1005,7 @@ def _compose_execute_result_result( code=base_result.code, text=text, meta=meta, + **source, ) if base_result.code == "OK": return ToolResult( @@ -990,6 +1013,7 @@ def _compose_execute_result_result( code="SAFETY_WARNING", text=text, meta=meta, + **source, ) meta["safety_warning"] = True return ToolResult( @@ -997,4 +1021,5 @@ def _compose_execute_result_result( code=base_result.code, text=text, meta=meta, + **source, ) diff --git a/tests/test_reference_book_budgets.py b/tests/test_reference_book_budgets.py index eb78edf20..12026f41b 100644 --- a/tests/test_reference_book_budgets.py +++ b/tests/test_reference_book_budgets.py @@ -119,7 +119,8 @@ CHAPTER_BYTE_BUDGETS: dict[str, int] = { # 7764 -> 8600 (#1195): the fresh selected-subject + immutable peer projection # execution check (`skill_peer_inventory.py`, `skill_conflicts.py`) replaces # whole-inventory hashing; the chapter had no description of that seam to swap out. - "docs/architecture/13-external-skills-layer.md": 8600, + # Dispatcher producer/annotation separation and its retained-source lifetime. + "docs/architecture/13-external-skills-layer.md": 9500, "docs/development/01-role-and-authority.md": 2437, "docs/development/02-naming-and-boundaries.md": 36372, # 22873 -> 23100: one new invariant (notifications ring for live events diff --git a/tests/test_registry_core.py b/tests/test_registry_core.py index 06728d6a9..710166a4e 100644 --- a/tests/test_registry_core.py +++ b/tests/test_registry_core.py @@ -305,6 +305,8 @@ def test_registry_uses_typed_required_root_not_note_or_tool_name(tmp_path, monke code="OK", text="OK\n\n⚠️ AUTO_ROUTED_TO_ACTIVE_WORKSPACE: benign additive note", meta={"route_note": True}, + producer_text="OK", + host_annotations=("⚠️ AUTO_ROUTED_TO_ACTIVE_WORKSPACE: benign additive note",), ) assert len(calls) == 1 diff --git a/tests/test_tool_producer_promotion.py b/tests/test_tool_producer_promotion.py new file mode 100644 index 000000000..d4b30596a --- /dev/null +++ b/tests/test_tool_producer_promotion.py @@ -0,0 +1,65 @@ +"""Clean tool sources survive when only a model-request closure is published.""" + +from __future__ import annotations + +import json +from pathlib import Path +from types import SimpleNamespace + +import pytest + +from ouroboros.artifacts import read_actor_source_bytes +from ouroboros.headless import copy_child_task_result, prepare_task_drive, prune_headless_task_drives +from ouroboros.loop_tool_execution import process_tool_results +from ouroboros.observability import persist_call, read_blob_ref +from ouroboros.task_results import STATUS_COMPLETED, write_task_result +from ouroboros.tools.core import _read_file +from ouroboros.tools.extension_dispatch import _extension_completion +from ouroboros.tools.tool_context import ToolContext + + +def _marker_ref(text: str, marker: str) -> dict: + return json.loads(next(line[len(marker):] for line in text.splitlines() if line.startswith(marker))) + + +@pytest.mark.parametrize("large", [False, True]) +def test_clean_source_in_model_request_survives_child_copyback_and_pruning(tmp_path, large): + parent = tmp_path / "canonical" + task_id = "producer-copyback" + child = prepare_task_drive(parent, task_id, "empty") + assert child is not None + repo = tmp_path / "repo" + repo.mkdir() + ctx = ToolContext(repo_dir=repo, drive_root=child, task_id=task_id) + payload = json.dumps({"ok": False, "text": "雪" * (20000 if large else 1)}, ensure_ascii=False) + warning = "⚠️ SAFETY_WARNING: the returned content still needs inspection." + typed = _extension_completion(payload, warning) + messages, trace = [], {"tool_calls": []} + process_tool_results( + [{"fn_name": "ext_fixture", "tool_call_id": "fixture-call", "result": typed.text, + "tool_result": typed, "is_error": True, "tool_args": {}, "args_for_log": {}}], + messages, trace, lambda *a, **kw: None, SimpleNamespace(_ctx=ctx), + ) + source = _marker_ref(messages[0]["content"], "PRODUCER_RESULT_SOURCE_JSON=") + request = persist_call(child, task_id=task_id, call_id="producer-request", call_type="llm_request", + payload={"messages": messages}) + # The actual collector retains call refs, not the in-memory tool result row. + # Publish only the model request to prove marker-based dependency custody. + write_task_result(child, task_id, STATUS_COMPLETED, result="done", artifact_status="ready", + trace_refs={"llm_call_refs": [{"request_ref": request["manifest_ref"]}]}) + copied = copy_child_task_result(parent, {"id": task_id, "drive_root": str(child)}) + assert copied is not None and copied["child_ref_promotion"]["status"] == "complete" + request_ref = copied["trace_refs"]["llm_call_refs"][0]["request_ref"] + manifest = json.loads(Path(request_ref["path"]).read_text(encoding="utf-8")) + promoted = read_blob_ref(parent, manifest["full_payload_ref"])["messages"][0]["content"] + assert warning in promoted + assert _marker_ref(promoted, "PRODUCER_RESULT_SOURCE_JSON=") == source + prune_headless_task_drives(parent, retention_days=0, now=4_000_000_000.0) + assert not child.exists() + assert read_actor_source_bytes(parent, task_id, source) == payload.encode("utf-8") + canonical_ctx = ToolContext(repo_dir=repo, drive_root=parent, task_id=task_id) + assert "雪" in _read_file(canonical_ctx, **source["read"]["arguments"]) + assert json.loads(read_actor_source_bytes(parent, task_id, source))["ok"] is False + if large: + full = _marker_ref(promoted, "FULL_RESULT_SOURCE_JSON=") + assert read_actor_source_bytes(parent, task_id, full).decode("utf-8") == typed.text diff --git a/tests/test_tool_producer_sources.py b/tests/test_tool_producer_sources.py new file mode 100644 index 000000000..944e9d87b --- /dev/null +++ b/tests/test_tool_producer_sources.py @@ -0,0 +1,131 @@ +"""Host notes stay visible without corrupting data retained for a program.""" + +from __future__ import annotations + +import json +from types import SimpleNamespace + +import pytest + +from ouroboros import artifacts +from ouroboros import loop_tool_execution as execution +from ouroboros.tools import extension_dispatch +from ouroboros.tools.core import _read_file +from ouroboros.tools.tool_context import ToolContext +from ouroboros.tools.tool_result import ToolResult, _compose_execute_result_result + + +WARNING = "⚠️ SAFETY_WARNING: inspect the returned data before acting." +PAYLOAD = json.dumps({"ok": False, "body": "Unicode: 雪\n\n---\nSAFETY_WARNING is data", "rows": [None, True]}) + + +@pytest.mark.parametrize("route", ["builtin", "extension", "mcp"]) +@pytest.mark.parametrize("warning", ["", WARNING]) +def test_dispatch_preserves_payload_and_failure_without_parsing_notes(monkeypatch, route, warning): + base = ToolResult(status="error", code="TOOL_REPORTED_FAILURE", text=PAYLOAD) + if route == "builtin": + result = _compose_execute_result_result("fixture", base, "", warning) + elif route == "extension": + result = extension_dispatch._extension_completion(PAYLOAD, warning) + else: + monkeypatch.setattr("ouroboros.safety.check_safety", lambda *a, **k: (True, warning)) + monkeypatch.setattr("ouroboros.mcp_client._call_mcp_tool_result", lambda *a, **k: base) + result = extension_dispatch._dispatch_mcp_tool_result(SimpleNamespace(), "mcp_demo", {}) + assert result.status == "error" and result.code == "TOOL_REPORTED_FAILURE" + assert result.text == PAYLOAD + ("\n\n" + warning if warning else "") + assert result.producer_text == (PAYLOAD if warning else None) + assert result.host_annotations == ((warning,) if warning else ()) + assert json.loads(result.producer_text or result.text)["ok"] is False + + +def test_nested_notes_and_generation_stamp_keep_original_payload(monkeypatch): + result = extension_dispatch._extension_completion(PAYLOAD, WARNING) + result = _compose_execute_result_result("fixture", result, "route note", "") + monkeypatch.setattr(extension_dispatch, "_dispatch_extension_tool_untagged", lambda *a: result) + stamped = extension_dispatch._dispatch_extension_tool_result( + SimpleNamespace(), "ext_demo", {"extension_generation": "generation", "content_hash": "hash"}, {}, + ) + assert stamped.producer_text == PAYLOAD + assert stamped.host_annotations == (WARNING, "route note") + assert stamped.meta["extension_generation"] == "generation" + assert stamped.text == PAYLOAD + "\n\n" + WARNING + "\n\nroute note" + + +@pytest.mark.parametrize("actor", ["presence", "project", "local_readonly_subagent"]) +@pytest.mark.parametrize("large", [False, True]) +def test_loop_retains_parseable_source_and_annotated_review_source(tmp_path, monkeypatch, actor, large): + repo = tmp_path / "repo" + repo.mkdir() + drive = tmp_path / actor / "data" + (drive / "logs").mkdir(parents=True) + ctx = ToolContext( + repo_dir=repo, drive_root=drive, task_id="producer-source", + budget_drive_root=tmp_path / "canonical", + workspace_root=tmp_path / "project" if actor != "presence" else None, + workspace_mode="external" if actor != "presence" else "", + task_depth=1 if actor == "local_readonly_subagent" else 0, + ) + ctx.task_metadata = {"task_id": ctx.task_id} + if actor == "presence": + ctx.task_metadata.update(_presence_turn=True, presence={"profile": {}}) + elif actor == "project": + ctx.task_metadata["project_id"] = "project-demo" + else: + ctx.task_constraint = {"mode": "local_readonly_subagent"} + ctx.task_metadata["parent_task_id"] = "parent" + payload = json.dumps({"ok": False, "text": "雪" * (20000 if large else 1)}, ensure_ascii=False) + result = extension_dispatch._extension_completion(payload, WARNING) + registry = SimpleNamespace(_ctx=ctx, CODE_TOOLS=frozenset(), execute_result=lambda *a: result) + captured = [] + monkeypatch.setattr(execution, "persist_call", lambda *a, **k: captured.append(k["payload"]) or {}) + row = execution._execute_single_tool( + registry, {"id": "call-json", "function": {"name": "ext_demo", "arguments": "{}"}}, + drive / "logs", ctx.task_id, + ) + messages, trace = [], {"tool_calls": []} + execution.process_tool_results([row], messages, trace, lambda *a, **k: None, registry) + recorded = trace["tool_calls"][0] + ref = recorded["producer_source_ref"] + exact = artifacts.read_actor_source_bytes(drive, ctx.task_id, ref) + assert exact == payload.encode("utf-8") + assert json.loads(exact)["ok"] is False + assert "雪" in _read_file(ctx, **ref["read"]["arguments"]) + assert WARNING in messages[0]["content"] + assert "PRODUCER_RESULT_SOURCE_JSON=" in messages[0]["content"] + assert recorded["is_error"] is True + assert captured[0]["producer_result"] == payload + assert captured[0]["result"] == result.text + assert captured[0]["host_annotations"] == [WARNING] + if large: + full_ref = recorded["result_source_ref"] + assert artifacts.read_actor_source_bytes(drive, ctx.task_id, full_ref).decode("utf-8") == result.text + assert full_ref != ref + + +def test_clean_small_result_keeps_existing_projection(tmp_path): + ctx = ToolContext(repo_dir=tmp_path, drive_root=tmp_path, task_id="clean-source") + result = extension_dispatch._extension_completion(PAYLOAD, "") + messages, trace = [], {"tool_calls": []} + execution.process_tool_results( + [{"fn_name": "ext_demo", "tool_call_id": "clean", "result": PAYLOAD, + "tool_result": result, "is_error": True, "tool_args": {}, "args_for_log": {}}], + messages, trace, lambda *a, **k: None, SimpleNamespace(_ctx=ctx), + ) + assert messages[0]["content"] == PAYLOAD + assert "producer_source_ref" not in trace["tool_calls"][0] + + +def test_missing_source_is_disclosed_without_hiding_warning_or_failure(tmp_path, monkeypatch): + monkeypatch.setattr(artifacts, "store_actor_source_bytes", lambda *a, **k: (_ for _ in ()).throw(OSError("disk full"))) + ctx = ToolContext(repo_dir=tmp_path, drive_root=tmp_path, task_id="source-failure") + result = extension_dispatch._extension_completion(PAYLOAD, WARNING) + messages, trace = [], {"tool_calls": []} + execution.process_tool_results( + [{"fn_name": "ext_demo", "tool_call_id": "failed", "result": result.text, + "tool_result": result, "is_error": True, "tool_args": {}, "args_for_log": {}}], + messages, trace, lambda *a, **k: None, SimpleNamespace(_ctx=ctx), + ) + assert result.text in messages[0]["content"] + assert "PRODUCER_RESULT_SOURCE_UNAVAILABLE=true" in messages[0]["content"] + assert trace["tool_calls"][0]["is_error"] is True + assert trace["tool_calls"][0]["producer_source_ref"] == {} From ecf9394d8919c69ed698080271919f2b4a648f5d Mon Sep 17 00:00:00 2001 From: Ouroboros Date: Wed, 23 Sep 2026 05:20:59 +0300 Subject: [PATCH 10/15] Retain Presence child custody after an empty final is replaced Keep an admitted child pollable when a host-authored terminal replaces an empty Presence answer, even when the existing lifecycle remains completed. Preserve successful model replies and reject unconfirmed child references. Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> --- ouroboros/presence_runner.py | 5 ++-- tests/test_presence_terminal_authorship.py | 31 ++++++++++++++++++++++ 2 files changed, 34 insertions(+), 2 deletions(-) diff --git a/ouroboros/presence_runner.py b/ouroboros/presence_runner.py index 264165d1a..66a7ae911 100644 --- a/ouroboros/presence_runner.py +++ b/ouroboros/presence_runner.py @@ -101,6 +101,7 @@ def presence_result_from_stored(stored: Mapping[str, Any], task_id: str) -> Pres def build_presence_result_event(task: dict[str, Any], text: str, ctx: Any, *, terminal_origin: str = "", retain_scheduled_handoff: bool = False) -> dict[str, Any]: """Freeze typed delivery metadata before the ordinary durable result write.""" + from ouroboros.task_finalization import HOST_AUTHORED_TERMINAL_ORIGINS completion = getattr(ctx, "_presence_completion", None) completion = completion if ( @@ -116,8 +117,8 @@ def build_presence_result_event(task: dict[str, Any], text: str, ctx: Any, *, te ) if outcome == "deferred" and not work_ref: outcome = "message" - if retain_scheduled_handoff and work_ref: - # A failed/forced parent still owes an already admitted child's result. + if work_ref and (retain_scheduled_handoff or terminal_origin in HOST_AUTHORED_TERMINAL_ORIGINS): + # A failed/forced or host-replaced final still owes an admitted child's result. # Transports poll only deferred outcomes, even when no reply was authored. outcome = "deferred" outcome, result_text = _presence_delivery(outcome, str(text or ""), terminal_origin) diff --git a/tests/test_presence_terminal_authorship.py b/tests/test_presence_terminal_authorship.py index 33309e38b..db0b81986 100644 --- a/tests/test_presence_terminal_authorship.py +++ b/tests/test_presence_terminal_authorship.py @@ -219,3 +219,34 @@ def test_native_host_replacement_resets_authorship_and_keeps_diagnostic(tmp_path else: assert stored["status"] == "failed" assert stored["reason_code"] == ("task_exception" if case == "exception" else "budget_exhausted") + + +@pytest.mark.parametrize("admission", ["scheduled", "unconfirmed", "rejected"]) +def test_empty_deferred_final_keeps_only_confirmed_child_polling(tmp_path, monkeypatch, native_agent, admission): + from ouroboros.tools.control_routing import _finish_swarm_handoff + + agent, ctx = native_agent + monkeypatch.setenv("OUROBOROS_TASK_REVIEW_MODE", "off") + ctx.task_contract = {"capability_ceiling": presence_ceiling_payload(_admission().capability_ceiling)} + ctx.task_metadata = {"presence": {"binding_id": "test-binding"}} + ctx.is_direct_chat = True + _finish_swarm_handoff(ctx, {"task_id": "managed-child"}, "Admission receipt", status=admission) + agent.llm = SimpleNamespace(default_model=lambda: "test-model") + replies, calls = iter([_call("deferred", ""), {"content": ""}]), [] + + def respond(*_a, **_kw): + calls.append(1) + return next(replies), 0.0 + + monkeypatch.setattr(loop, "call_llm_with_retry", respond) + events = agent._handle_task_scoped({"id": "empty-deferred", "chat_id": 7, "type": "presence", + "_presence_turn": True, "_is_direct_chat": True, "_skip_post_task_synthesis": True, "text": "Go", + "metadata": {"presence": {"binding_id": "test-binding"}}}) + result = next(row for row in events if row["type"] == "presence_result") + stored, cached = load_task_result(tmp_path, "empty-deferred"), _cached_result(tmp_path, "empty-deferred") + assert len(calls) == 2 and ctx._presence_completion_accepted is False + assert stored["terminal_origin"] == "host_notice" and "empty response" in stored["result"] + assert stored["status"] == "completed" # the existing lifecycle does not release child custody + assert result["outcome"] == cached.outcome == ("deferred" if admission == "scheduled" else "silent") + assert result["work_ref"] == cached.work_ref == ("managed-child" if admission == "scheduled" else "") + assert result["text"] == cached.text == "" From 28472b761d7d481180ba98e8ae48aaf667115557 Mon Sep 17 00:00:00 2001 From: Ouroboros Date: Wed, 23 Sep 2026 05:23:20 +0300 Subject: [PATCH 11/15] Preserve Presence delivery facts through post-task memory Seal the event-time automatic reply separately from internal terminal diagnostics for normal and recovered synthesis. Retain historical reply evidence without applying current replay policy, clarify correction topic scope, and reconcile the transport expectation and documentation budgets. Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> --- docs/architecture/06-agent-core.md | 4 +- ...12-host-service-companions-and-chat-ids.md | 4 +- ouroboros/room_consolidation.py | 4 +- ouroboros/task_finalization.py | 48 +++++++- tests/test_loop_transport_wait.py | 3 +- tests/test_presence_post_task_truth.py | 107 ++++++++++++++++++ 6 files changed, 157 insertions(+), 13 deletions(-) create mode 100644 tests/test_presence_post_task_truth.py diff --git a/docs/architecture/06-agent-core.md b/docs/architecture/06-agent-core.md index f6c8fe68c..2bd08fb5b 100644 --- a/docs/architecture/06-agent-core.md +++ b/docs/architecture/06-agent-core.md @@ -20,7 +20,7 @@ The delivery-control protocol is resolved here and only here (DEVELOPMENT keeps The child-absorption gate is an action gate: while undispositioned direct children remain, the loop HOLDS the candidate (`child_absorption_or_revision_required`) instead of arming the JSON-only control instruction — the hold-vs-arm split exists so the model never receives two contradictory instructions in one round. A typed keep cannot close the gate; after the one bounded reminder it forces the best-effort `children_unabsorbed` rail with a current `id [status] sha256` listing. The absorption digest's `## child` header carries the child's typed custody debt (`delegated_runs_unreconciled`, bounded, with a `get_task_result` pointer) as visibility only — the parent's authority over that patch is exactly the orphan rule, and a child's debt never relabels the root card (DESIGN §4). Finalizing over an UNDISPOSED OWN delegated patch is deliberately NOT gated: the consequence is disclosed where the decision is made (the `integrate_delegated_patch` schema, the apply receipts) and lands as the additive Done-with-warnings custody overlay rather than a hold; a pre-finalization reminder and propagation of child custody debt into the acceptance-subtree snapshot remain disclosed deferred gaps. -Provider death is the one forced rail that is NOT a best-effort completion: it salvages the best available text but stamps `infra_failed`, so the task terminalizes `failed` with the typed `provider_unavailable` reason and an immediate "provider outage — NOT completed" owner notification; a waited-out transport outage reaches the same rail through its deterministic no-resend branch (`transport_unavailable_no_resend`, or `provider_outcome_unknown_no_resend` when the round still holds a transport-death repeat record). The rail makes its one forced model call only while a call can still land: a transport that spent its same-model retry wall stamps `_llm_retry_wall_exhausted`, a no-call gate beside `context_overflow` and `provider_outcome_unknown`, so the forced rail never re-pays a second retry window over a proven-dead provider (disclosed residual: the marker is a last-invocation bool on the shared usage dict). Terminal delivery preserves producer authorship: `terminal_provider_notice` is host presentation beside the raw result, and receipts and secondary System incidents carry the wait duration, finalization cause and unknown-outcome warning and never recommend a blind rerun. Every forced rail stamps one closed-vocabulary producer word at the single forced-finalization sink: `model_final` for complete model text (host-authored notices stay separate), `host_notice` for host-written terminal text (an owner System row, never automatic external Presence speech), and `host_salvage` only on the provider-death rail, where managed/direct delivery replaces the text with the enriched outage receipt and the full bytes stay in task details. A missing origin remains unknown. Host text replacements in `agent.py` reset origin instead of inheriting the replaced answer's authorship. +Provider death is the one forced rail that is NOT a best-effort completion: it salvages the best available text but stamps `infra_failed`, so the task terminalizes `failed` with the typed `provider_unavailable` reason and an immediate "provider outage — NOT completed" owner notification; a waited-out transport outage reaches the same rail through its deterministic no-resend branch (`transport_unavailable_no_resend`, or `provider_outcome_unknown_no_resend` when the round still holds a transport-death repeat record). The rail makes its one forced model call only while a call can still land: a transport that spent its same-model retry wall stamps `_llm_retry_wall_exhausted`, a no-call gate beside `context_overflow` and `provider_outcome_unknown`, so the forced rail never re-pays a second retry window over a proven-dead provider (disclosed residual: the marker is a last-invocation bool on the shared usage dict). Terminal delivery preserves producer authorship: `terminal_provider_notice` is host presentation beside the raw result, and receipts and secondary System incidents carry the wait duration, finalization cause and unknown-outcome warning and never recommend a blind rerun. Every forced rail stamps one closed-vocabulary producer word at the single forced-finalization sink: `model_final` for complete model text (host-authored notices stay separate), `host_notice` for host-written terminal text (an owner System row, never automatic external Presence speech), and `host_salvage` only on the provider-death rail, where managed/direct delivery replaces the text with the enriched outage receipt and the full bytes stay in task details. Missing origin stays unknown; `agent.py` resets it on host text replacement. Presence synthesis seals the recorded adapter body and internal result separately (chapter 12), never a delivery receipt. Finalization controls are typed owner-mailbox entries rather than injected owner prose. The supervisor may request one bounded tool-less answer, salvage the last persisted assistant text, and retain a full canonical copy when a preview would truncate it. A grace episode has one durable control and can be revoked atomically when the task itself resumes; descendant activity does not count as the task's own progress. A process that cannot be killed remains visibly running, and custody checks prevent another runtime instance from reaping work it does not own. @@ -541,7 +541,7 @@ Every IMPLICIT claim — the UI conversion, that admission, the reaper's retry a `context.py` assembles static governance, semi-stable memory, and dynamic task evidence without treating truncation as forgetting; the recent-activity sections are each task's OWN newest rows (progress 50 rendered; tools 20 selected, 10 rendered and 20 scanned for review markers; events 200 counted by type) through the bounded reader `jsonl_tail.py` (`Memory.read_task_recent`: a doubling live tail plus at most three newest archives), never a global tail filtered afterwards (issue #131), and their header's coverage line names the rows, the window and any unopened archives while `read_file` pages the rest; a subagent child gets the same three windows beside its `## Working sources` block, its tools and events read from its own execution drive (its worker rows; host-side rows such as waits stay in the canonical log, as the header says) and progress from the canonical log; the Development context matrix and `context_layout.py` own which reference form is resident. When the rendered scratchpad exceeds `SCRATCHPAD_SECTION_BUDGET_CHARS`, `context.py` keeps the newest whole blocks that fit and drops the oldest behind an in-band gap marker naming `memory/scratchpad.md` as the live source; no block is retired by a context build, and scratchpad replacement keeps its explicit summary and source-journal provenance. -`consolidator.py` publishes a logical dialogue block only after every part succeeds, preserving raw generations and their captured cursor; an unfindable generation appends `[MEMORY GAP]`, never silently resets the offset. `context_fit` measures full Light requests against fresh route/account capacity and calibrated density (`llm_local.local_context_limits` owns local output reservation; missing/stale evidence remains unknown). `room_consolidation.py` drafts and corrects each room separately, then deterministically assembles the sections. Episodic text is grounded in that room's source; cumulative knowledge replacements require the complete current note plus the episode in BOTH stages. Corrected entries bind to the corrector's complete delivered read, never the draft's revision credit. A narrow or older episode cannot negate prior facts or later receipts; supported corrections and removals remain model judgment. Range reads, authored views, CAS and old/new history remain the publication path; no new stage or store. Failed, empty or truncated correction withholds the chunk, cursor and nominations. +`consolidator.py` publishes a dialogue block only after every part succeeds, retaining raw generations and their cursor; an unfindable generation appends `[MEMORY GAP]`, never silently resets the offset. `context_fit` measures full Light requests against fresh route/account capacity and calibrated density (`llm_local.local_context_limits` owns local output reservation; missing/stale evidence remains unknown). `room_consolidation.py` drafts and corrects each room separately, then deterministically assembles the sections. Episodic text is grounded in that room's source; cumulative knowledge replacements require the complete current note plus the episode in BOTH stages. Corrected entries bind to the corrector's complete delivered read, never the draft's revision credit. A narrow or older episode cannot negate prior facts or later receipts; supported corrections and removals remain model judgment. Range reads, authored views, CAS and old/new history remain the publication path; no new stage or store. Failed, empty or truncated correction withholds the chunk, cursor and nominations. Oversized source splits without clipping, including inside an entry; continuation context stays outside source bytes. A real refusal records its source hash and strictly smaller same-route byte bound in `dialogue_meta.json` (`consolidation_retry`); changed source, route, capacity or output reserve invalidates it. Era compression regroups each recorded room across blocks and reassembles deterministically; legacy untyped blocks remain explicitly unknown provenance. Failed/overflowed eras preserve old blocks. Failures retain `last_consolidation_error`, cleared by an advance without a new failure; incomplete knowledge publication retains `last_unpublished_nominations`. Unknown spend remains nullable and control/resource/unknown model errors retain `propagate_model_error` semantics. diff --git a/docs/architecture/12-host-service-companions-and-chat-ids.md b/docs/architecture/12-host-service-companions-and-chat-ids.md index 5de716e4c..b7a30f05c 100644 --- a/docs/architecture/12-host-service-companions-and-chat-ids.md +++ b/docs/architecture/12-host-service-companions-and-chat-ids.md @@ -12,9 +12,9 @@ Admission is one limiter with two policies (`host_service._RateLimiter`): the WS Operation correlation: a named injected message has `operation_ref=:` on 202, 200, 504 and disconnect responses. `supervisor.message_bus.accept_local_message` serializes check, canonical inbound-row acceptance and enqueue, and the `log_chat` write must succeed before work is queued. Repeated same-id, same-text, same-skill delivery rejoins even before supervisor dequeue; changed content or source is refused with 409. The queue stays in-memory: a crash after acceptance can lose delivery and reads honestly as `lost`, never authorizing a second enqueue. Routing annotations and outbound task ids are discovery hints only — task reads and cancellation require the actual queue/task record's complete `origin_message_ref` to match the authenticated skill's canonical source (`DirectActivityRegistry` carries the same origin), and named response waits poll that exact operation and its retry-aware effective task result, because chat ordering alone never proves a reply. Cancel enters the durable intent and cascade-custody owner (§5) only for work with that origin and the same installation root; a different or unavailable owner root and unaddressable or foreign work are `cancel_unsupported` before any intent is written, and unresolved custody never becomes a false `cancelled`. -Presence flow (`presence_runner.py`): `POST /presence/turn` requires the hash-bound `presence` permission, an owner binding from `state/presence_bindings.json`, an exact provider/account/conversation/thread event and optional files confined to the calling skill's state root. Fresh agents share one autobiography; event-derived task IDs make retries idempotent, and cross-process locks enforce the installation cap and per-conversation serialization. Outcomes are message/silent/tool_delivered/deferred; deferred requires correlated `work_ref`, read through `GET /presence/work/{work_ref}` rather than the general task API. Promotion clears requested Project/workspace/source widening and carries the ceiling, cost and return context by value; an unusable folder refuses typed (`workspace_unusable`, repair in `detail`) without sending text. `presence_cancel_work` requires the current binding and conversation. Owner chat and Background Consciousness at Act or above may `initiate_presence` on an enabled binding. +Presence (`presence_runner.py`): `POST /presence/turn` admits an exact provider/account/conversation/thread event and skill-state-confined files under the hash-bound `presence` permission and an owner binding (`state/presence_bindings.json`). Agents share one autobiography. Event-derived IDs deduplicate retries; cross-process locks cap concurrency and serialize conversations. Outcomes are message/silent/tool_delivered/deferred; deferred needs a correlated `work_ref`, read via `GET /presence/work/{work_ref}`, not the general task API. Promotion discards requested Project/workspace/source widening and copies the ceiling, cost and return context; unusable folders return `workspace_unusable` with repair detail. `presence_cancel_work` requires the current binding and conversation. Owner chat and consciousness at Act or above may `initiate_presence` on an enabled binding. -The ceiling includes knowledge, scratchpad, identity and chat history. Correspondents gain no tools or owner-command authority. Unselected baseline `chat_history` remains unbound, so model judgment can surface owner words, as knowledge recall can; older frozen ceilings retain their digest and acquire no new baseline until recompiled. `presence_context.py` distinguishes `communication.current_reply_route`, the admission filter and the separate proactive endpoint; original event and argument-binding meanings remain. Room/person facts do not establish system ownership. Early speech uses selected transport tools, never automatic Working forwarding. Explicit `presence_finish` enters common completion gates after the tool batch/control/budget tail; omitted message/deferred text retains a subsequent answer round. Holds and revised or failed work cannot reuse stale completion text. `presence_runner.py` projects immediate, cached and deferred results from terminal authorship: current model replies remain deliverable after partial failure, while host diagnostics and notices stay in the owner task. Explicit empty external text never falls back to the raw result; known host origins also suppress diagnostic bodies frozen by older code. A failed/forced parent retains an actually scheduled `work_ref` as deferred even with no authored text, so adapters keep polling without sending the diagnostic. Ordinary implicit replies and accepted silence/tool delivery remain supported; missing stored origin is explicit legacy compatibility, not proof of model authorship. +The ceiling includes knowledge, scratchpad, identity and chat history, not correspondent tool or owner-command authority. Unselected baseline `chat_history` is unbound: the model may recall owner words; frozen older ceilings keep their digest without acquiring new baseline tools. `presence_context.py` names current reply, admission and proactive routes; room/person facts grant no ownership. Early speech uses selected transport tools, never Working forwarding. `presence_finish` enters common completion checks after the tool/control/budget tail; omitted message/deferred text retains an answer round. Stale completion text cannot survive revised/failed work. Current authored replies, including best effort, remain speech; host diagnostics stay owner-side. Host-only terminals retain scheduled child custody as deferred with an empty body. Live/cache/work readers share authorship, preserve empty bodies, suppress known host text and retain unstamped legacy compatibility. Synthesis keeps the recorded adapter body, outcome/origin/work reference and internal result, preserving history rather than applying today's replay policy; preparation is no delivery receipt. Receipt reporting is negotiated through `/identity` (`presence_delivery_version: 1`) and optional `delivery_reporting_version: 1` on a turn; mode survives cached/deferred results. Mode1 writes outgoing history on provider receipts; mode0 retains an authored, delivery-unconfirmed row. `presence_delivery.py` accepts authenticated `POST /presence/delivery` observations through the chat writer. Parts retain target, text/format and provider facts; SMTP `accepted` means provider acceptance only. Speech is typed `presence_delivery`, never a task-finalizing untyped row; failed/uncertain attempts are System facts, queued remains in tool/outbox receipts. Memory, history and consolidation retain destination/state/details. A Host-context projection rebuilds once from retained chat generations and updates after required writes; identical report retries deduplicate, changed facts conflict. Transport outboxes own provider receipts and separate report ACK/backoff: a slow or failed report neither resends nor blocks provider delivery. No second store or scheduler; old queues are not imported. Wire fields and procedure: CREATING_SKILLS, “Reporting actual Presence delivery”. diff --git a/ouroboros/room_consolidation.py b/ouroboros/room_consolidation.py index aeffe578b..0b8173364 100644 --- a/ouroboros/room_consolidation.py +++ b/ouroboros/room_consolidation.py @@ -120,8 +120,8 @@ def correction_prompt( room_label = json.dumps(str(room_label), ensure_ascii=False) knowledge_check = ("If the draft has a `KNOWLEDGE_ENTRIES_JSON:` block, check those cumulative updates " "against each complete current note you read YOURSELF and this episode. The draft is a " - "proposal, not a source. Return corrected nominations for the same topics after the " - "episodic memory, or drop them. Unsupported episode claims must not survive in a note; " + "proposal, not a source. After the episodic memory, return only draft-nominated topics " + "(including proposed new notes), or drop them. Unsupported episode claims must not survive in a note; " "independently established knowledge may remain without becoming an event or approval " "in this episode.\n" + knowledge_instruction if knowledge_instruction else "") return f"""Compare this draft memory of Ouroboros against its complete source and return the corrected memory. diff --git a/ouroboros/task_finalization.py b/ouroboros/task_finalization.py index 76d94b99f..b10272257 100644 --- a/ouroboros/task_finalization.py +++ b/ouroboros/task_finalization.py @@ -490,7 +490,7 @@ def focus_source_projection( def build_sealed_final_package(result_row: Any, final_text: str) -> Dict[str, Any]: - """Host-attested final outcome: delivered text + artifact-store manifest. + """Seal recorded reply preparation and artifact facts, not delivery receipts. The manifest comes from the DURABLE task result the pipeline just stored (whose ``artifacts`` were merged from ``collect_task_artifact_records``, @@ -499,6 +499,7 @@ def build_sealed_final_package(result_row: Any, final_text: str) -> Dict[str, An independent filesystem walk (no second source of truth). """ from ouroboros.outcomes import artifact_bundle_from_result + from ouroboros.dialogue_provenance import is_presence_task row = result_row if isinstance(result_row, dict) else {} manifest = [ @@ -508,13 +509,28 @@ def build_sealed_final_package(result_row: Any, final_text: str) -> Dict[str, An if isinstance(record, dict) and record.get("name") ] omitted = max(0, len(manifest) - _SEALED_MANIFEST_MAX_FILES) - return { + package = { "final_result_text": str(final_text or ""), "artifact_manifest": manifest[:_SEALED_MANIFEST_MAX_FILES], **({"artifact_manifest_omitted": omitted} if omitted else {}), "completion_observations": row.get("completion_observations") or {"status": "unavailable"}, **({"terminal_host_notice": terminal_host_notice_text(row)} if terminal_host_notice_text(row) else {}), } + if is_presence_task(row): + metadata = row.get("metadata") if isinstance(row.get("metadata"), dict) else {} + # Remember the event-time body, not today's replay policy: a historical + # host diagnostic may actually have reached an adapter before this fix. + package["final_result_text"] = str(metadata.get("presence_result_text") or "") + package["presence_delivery"] = { + "reply_recorded": "presence_result_text" in metadata, + "outcome": str(metadata.get("presence_outcome") or "unknown"), + "work_ref": str(metadata.get("presence_work_ref") or ""), + **{key: str(row.get(key) or "unknown") for key in ("terminal_origin", "status", "reason_code")}, + } + package["internal_terminal_text"] = str(row.get("result") or final_text or "") + if notice := terminal_notice_text(row): + package["terminal_host_notice"] = notice + return package def sealed_final_prompt_section(sealed_final: Dict[str, Any] | None) -> str: @@ -540,21 +556,41 @@ def sealed_final_prompt_section(sealed_final: Dict[str, Any] | None) -> str: host_notice = truncate_review_artifact( str(sealed_final.get("terminal_host_notice") or ""), limit=_SEALED_FINAL_TEXT_PROMPT_CHARS, ) - return ( - "## Sealed final outcome (host-attested ground truth)\n" + introduction = ( "Below are the final answer submitted for delivery and a host-built\n" "manifest of this task's durable artifact store (plain filesystem facts).\n" "Outcomes stated here OVERRIDE impressions from the error trace: if the\n" "trace suggests failure but this package shows a delivered result or\n" "artifact, describe the recovery honestly instead of declaring the\n" "deliverable missing.\n" + ) + heading, presence_facts, internal_text = "Final result text (submitted for delivery):\n", "", "" + presence = sealed_final.get("presence_delivery") + if isinstance(presence, dict): + introduction = ( + "The recorded automatic-reply body was prepared for the Presence adapter;\n" + "it is not a provider delivery receipt. Preserve its recorded origin and task status.\n" + "Internal terminal text is diagnostic evidence, not an authored external reply.\n" + "Historical host-origin reply bodies remain historical facts, not proof of delivery.\n" + ) + heading = "Recorded automatic-reply body:\n" + presence_facts = "Recorded Presence facts:\n" + json.dumps(presence, ensure_ascii=False) + "\n" + if not presence.get("reply_recorded"): + final_text = "(automatic reply record unavailable)" + raw = str(sealed_final.get("internal_terminal_text") or "") + if raw and raw != str(sealed_final.get("final_result_text") or ""): + internal_text = "Internal terminal text (separate from the automatic reply):\n" + truncate_review_artifact( + raw, limit=_SEALED_FINAL_TEXT_PROMPT_CHARS, + ) + "\n" + return ( + "## Sealed final outcome (host-attested ground truth)\n" + + introduction + "An empty final answer, manifest, or observation section is not evidence that no action occurred.\n" "Tool success records a submitted/queued action, not a chat receipt or proof the owner received it.\n" "Skill readiness is current task-related state; it does not attribute an owner's click to this task.\n" "Use the inline counts/results and coverage below; source refs are for later agent readers,\n" "not additional evidence you have read. Omitted or unavailable facts remain unknown.\n" - "Final result text (submitted for delivery):\n" - f"{final_text}\n" + + presence_facts + heading + f"{final_text}\n" + internal_text + (f"Host-authored terminal notice (separate from the model answer):\n{host_notice}\n" if host_notice else "") + "Artifact store manifest (task_results/artifacts//):\n" f"{manifest_text}\nTask completion observations:\n{observations}\n\n" diff --git a/tests/test_loop_transport_wait.py b/tests/test_loop_transport_wait.py index 78750778b..0c4519768 100644 --- a/tests/test_loop_transport_wait.py +++ b/tests/test_loop_transport_wait.py @@ -363,10 +363,11 @@ def test_presence_handoff_retains_work_ref_when_transport_terminal_fails(tmp_pat events, task, result, usage, trace, 0.0, tmp_path / "logs", ctx=registry._ctx) delivery = next(row for row in events if row["type"] == "presence_result") assert delivery["outcome"] == "deferred" and delivery["work_ref"] == "presence-work" - assert delivery["text"].count("[Host status]") == 1 + assert delivery["text"] == "" stored = load_task_result(tmp_path, task["id"]) assert stored["status"] == "failed" and stored["reason_code"] == "provider_unavailable" assert stored["metadata"]["presence_work_ref"] == "presence-work" + assert stored["result"] == result and stored["terminal_provider_notice"] @pytest.mark.parametrize("turn_flag", [None, "is_direct_chat"]) diff --git a/tests/test_presence_post_task_truth.py b/tests/test_presence_post_task_truth.py new file mode 100644 index 000000000..6095c90d6 --- /dev/null +++ b/tests/test_presence_post_task_truth.py @@ -0,0 +1,107 @@ +"""Memory synthesis remembers reply preparation separately from diagnostics.""" + +from types import SimpleNamespace + +import pytest + +from ouroboros import agent_task_pipeline as pipeline +from ouroboros.task_finalization import build_sealed_final_package, sealed_final_prompt_section +from ouroboros.task_results import load_task_result, write_task_result + + +@pytest.mark.parametrize("completion, origin, failed, work_ref, expected", [ + (None, "model_final", False, "", "message"), + (None, "host_notice", True, "", "silent"), + (None, "host_salvage", True, "child-work", "deferred"), + ("silent", "model_final", False, "", "silent"), + ("tool_delivered", "model_final", False, "", "tool_delivered"), + ("deferred", "model_final", False, "child-work", "deferred"), +]) +def test_live_and_recovered_synthesis_use_recorded_presence_body( + tmp_path, monkeypatch, completion, origin, failed, work_ref, expected, +): + captured = [] + monkeypatch.setattr(pipeline, "_run_post_task_processing_async", + lambda *_a, **kwargs: captured.append(kwargs["sealed_final"])) + raw = "Internal terminal diagnostic" if failed else "Current authored result" + task = {"id": "presence-synthesis", "root_task_id": "presence-synthesis", "type": "presence", + "chat_id": 7, "text": "Read the request", "_is_direct_chat": True, + "metadata": {"presence": {"binding_id": "binding"}}} + ctx = SimpleNamespace(_presence_completion={"outcome": completion} if completion else None, + _presence_completion_accepted=bool(completion), + _swarm_handoff_attempt={"status": "scheduled", "task_id": work_ref} if work_ref else None) + usage = {"terminal_origin": origin, "terminal_provider_notice": "Internal provider status"} + if failed: + usage.update(execution_status="failed", reason_code="round_limit") + events = [] + pipeline.emit_task_results( + SimpleNamespace(drive_root=tmp_path, repo_dir=tmp_path), None, None, events, task, raw, + usage, {"tool_calls": [], "reasoning_notes": []}, 0.0, tmp_path / "logs", ctx=ctx, + ) + event = next(row for row in events if row["type"] == "presence_result") + stored = load_task_result(tmp_path, task["id"]) + assert event["outcome"] == expected + assert stored["result"] == raw + assert len(captured) == 1 + sealed = captured[0] + assert sealed["final_result_text"] == event["text"] + assert sealed["internal_terminal_text"] == raw + assert sealed["presence_delivery"] == { + "reply_recorded": True, "outcome": expected, "work_ref": work_ref, + "terminal_origin": origin, "status": stored["status"], + "reason_code": stored.get("reason_code") or "unknown", + } + prompt = sealed_final_prompt_section(sealed) + assert "not a provider delivery receipt" in prompt + assert "Final result text (submitted for delivery)" not in prompt + assert "Internal provider status" in prompt + if not event["text"]: + assert "Recorded automatic-reply body:\n(empty final message)\nInternal terminal text" in prompt + assert raw in prompt.split("Internal terminal text (separate from the automatic reply):\n", 1)[1] + else: + assert f"Recorded automatic-reply body:\n{raw}\n" in prompt + + # The ordinary restart path reconstructs the same event-time evidence; + # it must not use the raw diagnostic as a fallback for an empty body. + write_task_result(tmp_path, task["id"], stored["status"], + root_phase_checkpoint={"post_task_synthesis": "pending_once"}) + assert pipeline.recover_pending_root_post_task_synthesis(tmp_path, tmp_path) == 1 + assert len(captured) == 2 and captured[1] == sealed + + +def test_historical_host_reply_is_remembered_without_claiming_delivery(): + row = {"task_id": "old-turn", "status": "failed", "reason_code": "round_limit", + "terminal_origin": "host_notice", "result": "Historic host diagnostic", + "metadata": {"presence": {}, "presence_outcome": "message", + "presence_result_text": "Historic host diagnostic"}} + sealed = build_sealed_final_package(row, row["result"]) + assert sealed["final_result_text"] == "Historic host diagnostic" + assert sealed["presence_delivery"]["terminal_origin"] == "host_notice" + assert sealed["presence_delivery"]["outcome"] == "message" + assert "not a provider delivery receipt" in sealed_final_prompt_section(sealed) + # Today's retry policy prevents a new send; it cannot erase biography. + from ouroboros.presence_runner import presence_result_from_stored + + assert presence_result_from_stored(row, "old-turn").text == "" + + +def test_missing_automatic_reply_record_is_unknown_not_raw_fallback(): + row = {"status": "failed", "result": "Internal-only text", "metadata": { + "presence": {}, "presence_outcome": "deferred", "presence_work_ref": "child"}} + sealed = build_sealed_final_package(row, row["result"]) + assert sealed["final_result_text"] == "" + assert sealed["presence_delivery"]["reply_recorded"] is False + assert sealed["presence_delivery"]["outcome"] == "deferred" + assert sealed["presence_delivery"]["work_ref"] == "child" + assert sealed["presence_delivery"]["terminal_origin"] == "unknown" + prompt = sealed_final_prompt_section(sealed) + assert "(automatic reply record unavailable)" in prompt and "Internal-only text" in prompt + + +def test_ordinary_owner_synthesis_retains_its_existing_package_and_prompt(): + sealed = build_sealed_final_package({"terminal_origin": "model_final"}, "Owner answer") + assert sealed == {"final_result_text": "Owner answer", "artifact_manifest": [], + "completion_observations": {"status": "unavailable"}} + prompt = sealed_final_prompt_section(sealed) + assert "Final result text (submitted for delivery):\nOwner answer\n" in prompt + assert "Presence" not in prompt and "Internal terminal text" not in prompt From 33da7a4f3c0d5385cdbcb65a5b8ee4f84e3d45b3 Mon Sep 17 00:00:00 2001 From: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> Date: Wed, 23 Sep 2026 07:47:19 +0300 Subject: [PATCH 12/15] docs: fit merged release diagnostics prose within chapter budget --- docs/architecture/06-agent-core.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/architecture/06-agent-core.md b/docs/architecture/06-agent-core.md index 6d2ee6828..881373ee7 100644 --- a/docs/architecture/06-agent-core.md +++ b/docs/architecture/06-agent-core.md @@ -357,7 +357,7 @@ The hermetic runner (`preflight_runner.py`) alone mints `ctx._preflight_test_pro #### Commit advisory cycle -`preflight_review(commit_message="...", deterministic_only=True, source="worktree" | "index")` returns a release-only diagnostic report before sync, staging, custody/state access, provider checks, tests or fingerprints. `commit_admission.release_metadata_diagnostics` owns source acquisition and reports all independent applicable `findings` plus `unavailable` sources; `release_sync.release_metadata_findings` reuses the carrier validators, release grammar and P9 history counters. The report identifies its source and status (`clean`, `blocked`, `unavailable` or `not_applicable`), gives no review freshness, and writes no review history. It checks the actual selected VERSION, never an imagined next release. VERSION/README absence and failed reads are unavailable evidence; an absent optional older carrier stays optional, while a readable malformed carrier is a finding. Coverage is release metadata only, not syntax, structural size, tests or critic review; size policy stays warning-only locally. +`preflight_review(commit_message="...", deterministic_only=True, source="worktree" | "index")` returns a release-only diagnostic report before sync, staging, custody/state access, provider checks, tests or fingerprints. `commit_admission.release_metadata_diagnostics` owns source acquisition and reports all independent applicable `findings` plus `unavailable` sources; `release_sync.release_metadata_findings` reuses the carrier validators, release grammar and P9 history counters. The report identifies its source and status (`clean`, `blocked`, `unavailable` or `not_applicable`), gives no review freshness, and writes no review history. It checks the selected VERSION, not a future release. VERSION/README absence and failed reads are unavailable evidence; an absent optional older carrier stays optional, while a readable malformed carrier is a finding. Coverage is release metadata only, not syntax, structural size, tests or critic review; size policy stays warning-only locally. Standalone advisory retains automatic carrier sync and worktree reads. Prepared advisory follows the commit gate's whole-index applicability, regardless of a narrower paths hint, including partial staging and the version-neutral carve; standalone documentation-only scope stays exempt. A present malformed VERSION is now a finding, not a silent skip. Optional[str] wrappers format the complete release findings, separating unavailable evidence from defects. Author continuation uses the shared name-status formatter and unavailable classification. Changelog prose, history trimming and version allocation remain deliberate. From ccaae8870244f7eb32536562eea5633ec24b58bb Mon Sep 17 00:00:00 2001 From: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:26:35 +0300 Subject: [PATCH 13/15] fix: preserve Git text decoding and catalog button ownership --- devtools/e2e_live/scenarios.py | 4 +-- .../03-web-ui-pages-and-buttons.md | 2 +- ouroboros/commit_admission.py | 15 ++++++---- tests/test_release_metadata_diagnostics.py | 22 +++++++++++++-- tests/test_ui_smoke_status_attention.py | 4 ++- web/modules/settings_catalog.js | 6 +++- web/tests/settings_action_row.test.js | 2 +- web/tests/settings_catalog.test.js | 28 +++++++++++++++++++ 8 files changed, 69 insertions(+), 14 deletions(-) diff --git a/devtools/e2e_live/scenarios.py b/devtools/e2e_live/scenarios.py index 486a83a88..99d8db69d 100644 --- a/devtools/e2e_live/scenarios.py +++ b/devtools/e2e_live/scenarios.py @@ -475,8 +475,8 @@ def worktree_after_commit(clone: pathlib.Path) -> tuple[bool, str, list[str]]: def _git_show(clone: pathlib.Path, rev: str, path: str) -> str: """The exact text of ``path`` at ``rev`` ('' when absent there).""" - proc = subprocess.run(["git", "show", f"{rev}:{path}"], cwd=str(clone), check=False, capture_output=True, text=True) - return proc.stdout if proc.returncode == 0 else "" + proc = subprocess.run(["git", "show", f"{rev}:{path}"], cwd=str(clone), check=False, capture_output=True) + return proc.stdout.decode("utf-8") if proc.returncode == 0 else "" def release_carriers_desync_at(clone: pathlib.Path, rev: str) -> str: diff --git a/docs/architecture/03-web-ui-pages-and-buttons.md b/docs/architecture/03-web-ui-pages-and-buttons.md index 4aac0c9e6..5c8a4188f 100644 --- a/docs/architecture/03-web-ui-pages-and-buttons.md +++ b/docs/architecture/03-web-ui-pages-and-buttons.md @@ -281,7 +281,7 @@ The synchronous lock-owning apply executor publishes process-local stage observa Settings has Accounts, Secrets, Models, Agents, Behavior, Appearance, Advanced and About tabs — a sequence from connections to runtime detail. Accounts: managed subscriptions and their shared service banner, API providers, custom compatible endpoints, local runtime entry points, and the optional non-loopback network gate. Secrets: known provider/integration secrets, skill-requested keys and owner-defined custom keys, without returning stored values. Models: compact source/model/account role rows, ordered fallbacks, context assertions and effort lanes. Agents: task actors and review lanes, with delegation permissions, per-root and depth limits and subagent path roots; their accounts are managed in Accounts. Behavior: context, safety-supervisor coverage, task acceptance, self-evolution, prompt-cache posture. Appearance: the client-local theme choice described above, never a runtime-settings value. Advanced: process, timeout, local-model, integration, source-control and cleanup controls (worker count is process capacity, so it lives here). About reports application/runtime identity. Keys, defaults and per-key semantics are the §7 Default settings table, not this chapter. Models, Available subagents and Review lanes share ONE grouped source select owned by `web/modules/route_editor_primitives.js` (`routeChoiceGroups`, `configuredApiProviders`): the owner picks a source and the editor composes the stored id, so the provider prefixes (`provider::model`, `claudexor::source=model`, `harness=model`) are serialization only, never owner input. -`settings_catalog.js` owns model-catalog reads and the Accounts subscription: a changed confirmed account or recovery from a failed read refreshes discovery; identical settled status stays quiet, and `settings.js` arms/disposes the subscription with the page. Catalog updates preserve the owner's current model draft. +`settings_catalog.js` owns model-catalog reads and the Accounts subscription: a changed confirmed account or recovery from a failed read refreshes discovery; identical settled status stays quiet, and `settings.js` arms/disposes the subscription with the page. Catalog updates preserve drafts. Per-button request ownership releases busy state independently of global response freshness; a newer background read cannot strand a manual button. The Settings client validates the whole current draft before Save; a local error keeps every value available for correction and sends no partial save (`settings_controls.js` keeps custom-key collection pure and dirty reads passive). Ordinary refresh and failed writes preserve current edits, leaving or explicitly reloading a dirty draft asks first, the write response distinguishes saved, unsaved and unknown outcomes, and there is no durable cross-page draft store or secret persistence. diff --git a/ouroboros/commit_admission.py b/ouroboros/commit_admission.py index 0f4577d2b..a966ff6f1 100644 --- a/ouroboros/commit_admission.py +++ b/ouroboros/commit_admission.py @@ -39,8 +39,9 @@ def changed_worktree_paths( try: result = subprocess.run( ["git", "--no-optional-locks", "status", "--porcelain"] + path_args, - cwd=str(repo_dir), capture_output=True, text=True, timeout=10, + cwd=str(repo_dir), capture_output=True, timeout=10, ) + stdout = result.stdout.decode("utf-8") except Exception: if strict: raise @@ -49,7 +50,7 @@ def changed_worktree_paths( if strict: raise RuntimeError("git status failed") return [] - return parse_changed_paths_from_porcelain(result.stdout) + return parse_changed_paths_from_porcelain(stdout) def auto_sync_release_metadata_if_needed( @@ -104,9 +105,11 @@ def read_release_file(repo_dir, path: str, *, source: str) -> str | None: present.check_returncode() result = subprocess.run( ["git", "show", f":{path}"], cwd=str(repo_dir), capture_output=True, - encoding="utf-8", timeout=10, check=True, + timeout=10, check=True, ) - return result.stdout + # Decode on the caller thread (Windows pipe-reader errors otherwise disappear), + # retaining the universal-newline semantics of worktree read_text(). + return result.stdout.decode("utf-8").replace("\r\n", "\n").replace("\r", "\n") def release_metadata_diagnostics( @@ -133,9 +136,9 @@ def release_metadata_diagnostics( result = subprocess.run( ["git", "--no-optional-locks", "diff", "--cached", "--name-only", "--diff-filter=d", "-z"], cwd=str(repo_dir), - capture_output=True, encoding="utf-8", timeout=10, check=True, + capture_output=True, timeout=10, check=True, ) - touched.update(filter(None, result.stdout.split("\0"))) + touched.update(filter(None, result.stdout.decode("utf-8").split("\0"))) except Exception as exc: unavailable.append(f"Changed {source} paths could not be read ({type(exc).__name__}).") diff --git a/tests/test_release_metadata_diagnostics.py b/tests/test_release_metadata_diagnostics.py index 9cee659b6..c80e96235 100644 --- a/tests/test_release_metadata_diagnostics.py +++ b/tests/test_release_metadata_diagnostics.py @@ -74,6 +74,24 @@ def candidate(tmp_path, monkeypatch): emit_progress_fn=lambda *_: None) +def test_crlf_carriers_have_same_text_semantics_in_index_and_worktree(candidate): + repo = candidate.repo_dir + _git(repo, "config", "core.autocrlf", "false") + for name, text in _release("1.2.4").items(): + (repo / name).write_bytes(text.replace("\n", "\r\n").encode("utf-8")) + _git(repo, "add", ".") + for source in ("worktree", "index"): + result = admission.release_metadata_diagnostics(repo, ["VERSION"], source=source) + assert result["status"] == "clean", result + + +def test_unicode_worktree_discovery_is_not_locale_decoded(candidate): + repo = candidate.repo_dir + _git(repo, "config", "core.quotepath", "false") + (repo / "А.py").write_text("value = 1\n", encoding="utf-8") + assert "А.py" in admission.changed_worktree_paths(repo, strict=True) + + def _broken(repo): files = _release("1.2.4") files["README.md"] = _release()["README.md"] + "".join( @@ -208,8 +226,8 @@ def test_index_read_failure_and_unmerged_entries_are_unavailable(candidate): _git(repo, "add", ".") blob = _git(repo, "rev-parse", ":pyproject.toml") # Install an unmerged optional carrier in this disposable fixture index. - subprocess.run(["git", "update-index", "--index-info"], cwd=repo, check=True, text=True, - input=f"0 {'0' * 40}\tpyproject.toml\n100644 {blob} 1\tpyproject.toml\n100644 {blob} 2\tpyproject.toml\n") + subprocess.run(["git", "update-index", "--index-info"], cwd=repo, check=True, + input=(f"0 {'0' * 40}\tpyproject.toml\n100644 {blob} 1\tpyproject.toml\n100644 {blob} 2\tpyproject.toml\n").encode("utf-8")) report = _diagnose(candidate, "index") assert report["status"] == "unavailable" assert any("pyproject.toml" in item for item in report["unavailable"]) diff --git a/tests/test_ui_smoke_status_attention.py b/tests/test_ui_smoke_status_attention.py index 9fa645f49..e0e12e662 100644 --- a/tests/test_ui_smoke_status_attention.py +++ b/tests/test_ui_smoke_status_attention.py @@ -324,7 +324,9 @@ def test_task_status_stays_factual_in_main_and_project_chat( assert failed.get_attribute("data-finished") == "1" assert phase_text(failed) == "Failed" assert_phase_accessibility(failed, "Task", "Failed") - assert "provider_route_failed" not in failed.locator( + # This synthetic code has no producer phrase. Unknown reasons remain + # visible verbatim; hiding them would discard the only available cause. + assert "provider_route_failed" in failed.locator( ":scope > [data-live-summary-button]" ).inner_text() diff --git a/web/modules/settings_catalog.js b/web/modules/settings_catalog.js index 7d84020d3..9de2e4f4b 100644 --- a/web/modules/settings_catalog.js +++ b/web/modules/settings_catalog.js @@ -3,6 +3,7 @@ import { apiFetch } from './api_client.js'; import { setInlineStatus } from './ui_helpers.js'; export const MODEL_CATALOG_TIMEOUT_MS = 25000; let catalogRefreshSeq = 0; +const buttonRefreshes = new WeakMap(); // Account login/status is the authority for subscription model discovery. Keep // one small signature of the confirmed account facts so a newly settled login @@ -194,6 +195,7 @@ export async function refreshModelCatalog({ button } = {}) { const statusEl = document.getElementById('settings-model-catalog-status'); setCatalogStatus(statusEl, 'Refreshing model catalog...', 'muted'); if (button) { + buttonRefreshes.set(button, refreshSeq); button.disabled = true; button.setAttribute('aria-busy', 'true'); } @@ -241,7 +243,9 @@ export async function refreshModelCatalog({ button } = {}) { return { items: [], errors: [{ provider_id: 'catalog', error: String(message) }] }; } finally { clearTimeout(timeoutId); - if (button && refreshSeq === catalogRefreshSeq) { + // Global freshness governs data, not a particular button's busy lease. + if (button && buttonRefreshes.get(button) === refreshSeq) { + buttonRefreshes.delete(button); button.disabled = false; button.removeAttribute('aria-busy'); } diff --git a/web/tests/settings_action_row.test.js b/web/tests/settings_action_row.test.js index 01a1418a6..563acf425 100644 --- a/web/tests/settings_action_row.test.js +++ b/web/tests/settings_action_row.test.js @@ -38,5 +38,5 @@ test('async actions expose the same busy and status semantics', () => { assert.match(catalogJs, /import \{ setInlineStatus \} from '\.\/ui_helpers\.js'/); assert.match(catalogJs, /setInlineStatus\(statusEl, text, tone\)/); assert.match(catalogJs, /refreshModelCatalog\(\{ button \} = \{\}\)/); - assert.match(catalogJs, /refreshSeq === catalogRefreshSeq/); + assert.match(catalogJs, /buttonRefreshes\.get\(button\) === refreshSeq/); }); diff --git a/web/tests/settings_catalog.test.js b/web/tests/settings_catalog.test.js index d7eecf856..21c7feed6 100644 --- a/web/tests/settings_catalog.test.js +++ b/web/tests/settings_catalog.test.js @@ -136,6 +136,34 @@ test('an older Refresh completion cannot replace newer success or its read facts assert.deepEqual(events[0].items, first.items); }); +test('background refresh cannot strand a superseded manual button busy', async (t) => { + const previousDocument = globalThis.document; + const previousFetch = globalThis.fetch; + t.after(() => { globalThis.document = previousDocument; globalThis.fetch = previousFetch; }); + const document = new EventTarget(); + document.getElementById = () => null; + globalThis.document = document; + const button = { disabled: false, setAttribute() {}, removeAttribute() {} }; + const pending = []; + globalThis.fetch = () => new Promise((resolve) => pending.push(resolve)); + const manual = refreshModelCatalog({ button }); + const background = refreshModelCatalog(); + pending[1]({ ok: true, json: async () => first }); + await background; + assert.equal(button.disabled, true, 'manual request still owns its busy state'); + pending[0]({ ok: true, json: async () => first }); + assert.equal((await manual).stale, true); + assert.equal(button.disabled, false); + const old = refreshModelCatalog({ button }); + const latest = refreshModelCatalog({ button }); + pending[2]({ ok: true, json: async () => first }); + await old; + assert.equal(button.disabled, true, 'older request cannot release a newer button owner'); + pending[3]({ ok: true, json: async () => first }); + await latest; + assert.equal(button.disabled, false); +}); + test('read errors drop the httpx documentation pointer and rows get the compact form', () => { const data = { items: [{ value: 'openai/gpt-x' }], From 568bf7f420b0f470abb25ae909cf3c86f28fc6a8 Mon Sep 17 00:00:00 2001 From: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:57:22 +0300 Subject: [PATCH 14/15] fix: retain newline semantics through committed carrier export --- devtools/e2e_live/scenarios.py | 2 +- docs/architecture/03-web-ui-pages-and-buttons.md | 2 +- tests/test_release_metadata_diagnostics.py | 14 +++++++++++++- tests/test_ui_smoke_status_attention.py | 5 ++++- 4 files changed, 19 insertions(+), 4 deletions(-) diff --git a/devtools/e2e_live/scenarios.py b/devtools/e2e_live/scenarios.py index 99d8db69d..32a7f5c99 100644 --- a/devtools/e2e_live/scenarios.py +++ b/devtools/e2e_live/scenarios.py @@ -476,7 +476,7 @@ def worktree_after_commit(clone: pathlib.Path) -> tuple[bool, str, list[str]]: def _git_show(clone: pathlib.Path, rev: str, path: str) -> str: """The exact text of ``path`` at ``rev`` ('' when absent there).""" proc = subprocess.run(["git", "show", f"{rev}:{path}"], cwd=str(clone), check=False, capture_output=True) - return proc.stdout.decode("utf-8") if proc.returncode == 0 else "" + return proc.stdout.decode("utf-8").replace("\r\n", "\n").replace("\r", "\n") if proc.returncode == 0 else "" def release_carriers_desync_at(clone: pathlib.Path, rev: str) -> str: diff --git a/docs/architecture/03-web-ui-pages-and-buttons.md b/docs/architecture/03-web-ui-pages-and-buttons.md index 5c8a4188f..5a4f2ba84 100644 --- a/docs/architecture/03-web-ui-pages-and-buttons.md +++ b/docs/architecture/03-web-ui-pages-and-buttons.md @@ -281,7 +281,7 @@ The synchronous lock-owning apply executor publishes process-local stage observa Settings has Accounts, Secrets, Models, Agents, Behavior, Appearance, Advanced and About tabs — a sequence from connections to runtime detail. Accounts: managed subscriptions and their shared service banner, API providers, custom compatible endpoints, local runtime entry points, and the optional non-loopback network gate. Secrets: known provider/integration secrets, skill-requested keys and owner-defined custom keys, without returning stored values. Models: compact source/model/account role rows, ordered fallbacks, context assertions and effort lanes. Agents: task actors and review lanes, with delegation permissions, per-root and depth limits and subagent path roots; their accounts are managed in Accounts. Behavior: context, safety-supervisor coverage, task acceptance, self-evolution, prompt-cache posture. Appearance: the client-local theme choice described above, never a runtime-settings value. Advanced: process, timeout, local-model, integration, source-control and cleanup controls (worker count is process capacity, so it lives here). About reports application/runtime identity. Keys, defaults and per-key semantics are the §7 Default settings table, not this chapter. Models, Available subagents and Review lanes share ONE grouped source select owned by `web/modules/route_editor_primitives.js` (`routeChoiceGroups`, `configuredApiProviders`): the owner picks a source and the editor composes the stored id, so the provider prefixes (`provider::model`, `claudexor::source=model`, `harness=model`) are serialization only, never owner input. -`settings_catalog.js` owns model-catalog reads and the Accounts subscription: a changed confirmed account or recovery from a failed read refreshes discovery; identical settled status stays quiet, and `settings.js` arms/disposes the subscription with the page. Catalog updates preserve drafts. Per-button request ownership releases busy state independently of global response freshness; a newer background read cannot strand a manual button. +`settings_catalog.js` owns model-catalog reads and the Accounts subscription: a changed confirmed account or recovery from a failed read refreshes discovery; identical settled status stays quiet, and `settings.js` arms/disposes the subscription with the page. Catalog updates preserve drafts. Per-button ownership releases busy state independently of data freshness: background reads cannot strand a manual button. The Settings client validates the whole current draft before Save; a local error keeps every value available for correction and sends no partial save (`settings_controls.js` keeps custom-key collection pure and dirty reads passive). Ordinary refresh and failed writes preserve current edits, leaving or explicitly reloading a dirty draft asks first, the write response distinguishes saved, unsaved and unknown outcomes, and there is no durable cross-page draft store or secret persistence. diff --git a/tests/test_release_metadata_diagnostics.py b/tests/test_release_metadata_diagnostics.py index c80e96235..e48d14fc1 100644 --- a/tests/test_release_metadata_diagnostics.py +++ b/tests/test_release_metadata_diagnostics.py @@ -85,6 +85,18 @@ def test_crlf_carriers_have_same_text_semantics_in_index_and_worktree(candidate) assert result["status"] == "clean", result +def test_sm1_committed_crlf_export_preserves_carrier_lines(candidate): + from devtools.e2e_live.scenarios import _git_show, release_carriers_desync_at + repo = candidate.repo_dir + _git(repo, "config", "core.autocrlf", "false") + for name, text in _release().items(): + (repo / name).write_bytes(text.replace("\n", "\r\n").encode("utf-8")) + _git(repo, "add", ".") + _git(repo, "commit", "-qm", "CRLF carriers") + assert _git_show(repo, "HEAD", "uv.lock") == _release()["uv.lock"] + assert release_carriers_desync_at(repo, "HEAD") == "" + + def test_unicode_worktree_discovery_is_not_locale_decoded(candidate): repo = candidate.repo_dir _git(repo, "config", "core.quotepath", "false") @@ -239,7 +251,7 @@ def test_git_discovery_failure_is_never_clean(candidate, monkeypatch, source): def fail_discovery(argv, **kwargs): if "status" in argv or "diff" in argv: - return subprocess.CompletedProcess(argv, 128, stdout="", stderr="fixture failed") + return subprocess.CompletedProcess(argv, 128, stdout=b"", stderr=b"fixture failed") return real(argv, **kwargs) # The real run(check=True) raises for index discovery; emulate that contract. diff --git a/tests/test_ui_smoke_status_attention.py b/tests/test_ui_smoke_status_attention.py index e0e12e662..8a27ad9ca 100644 --- a/tests/test_ui_smoke_status_attention.py +++ b/tests/test_ui_smoke_status_attention.py @@ -327,7 +327,10 @@ def test_task_status_stays_factual_in_main_and_project_chat( # This synthetic code has no producer phrase. Unknown reasons remain # visible verbatim; hiding them would discard the only available cause. assert "provider_route_failed" in failed.locator( - ":scope > [data-live-summary-button]" + ":scope > [data-live-summary-button] [data-live-activity]" + ).inner_text() + assert "provider_route_failed" not in failed.locator( + ":scope > [data-live-summary-button] [data-live-title]" ).inner_text() status = scope.locator(status_selector) From 14a05d7fd51296b0b9a586937c5c52301475bfce Mon Sep 17 00:00:00 2001 From: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> Date: Wed, 23 Sep 2026 17:21:32 +0300 Subject: [PATCH 15/15] test: guarantee CRLF export fixture changes on Windows --- tests/test_release_metadata_diagnostics.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/test_release_metadata_diagnostics.py b/tests/test_release_metadata_diagnostics.py index e48d14fc1..8f69f564d 100644 --- a/tests/test_release_metadata_diagnostics.py +++ b/tests/test_release_metadata_diagnostics.py @@ -89,11 +89,11 @@ def test_sm1_committed_crlf_export_preserves_carrier_lines(candidate): from devtools.e2e_live.scenarios import _git_show, release_carriers_desync_at repo = candidate.repo_dir _git(repo, "config", "core.autocrlf", "false") - for name, text in _release().items(): + for name, text in _release("1.2.4").items(): (repo / name).write_bytes(text.replace("\n", "\r\n").encode("utf-8")) _git(repo, "add", ".") _git(repo, "commit", "-qm", "CRLF carriers") - assert _git_show(repo, "HEAD", "uv.lock") == _release()["uv.lock"] + assert _git_show(repo, "HEAD", "uv.lock") == _release("1.2.4")["uv.lock"] assert release_carriers_desync_at(repo, "HEAD") == ""