diff --git a/docs/development/08-mutation-attribution-rule.md b/docs/development/08-mutation-attribution-rule.md index a88d37008..2007be8c0 100644 --- a/docs/development/08-mutation-attribution-rule.md +++ b/docs/development/08-mutation-attribution-rule.md @@ -1,11 +1,11 @@ # Mutation Attribution Rule -This chapter binds mutation evidence, staging and interpreter resolution so a reviewed commit does not claim another actor’s work. Attribution informs review; it is not exclusive ownership. +Mutation evidence, staging and interpreter resolution keep commits from claiming another actor’s work. Attribution informs review; it is not exclusive ownership. - Capture a `system_repo` baseline when a queued root starts and terminal candidates at outcome derivation. Pre-existing dirt, stale/missing baselines and failed scans are ambiguity for reviewing LLMs, not structural outcome vetoes. Keep this evidence in the task result; add no lease/holder service, second ledger or runtime writer keyword scanner. - The acceptance packet reads mutation evidence from the canonical results root (`budget_drive_root` first), the same root the writer and the outcome consumer use. - Git staging is attribution-based: `paths=None` means the clean-at-baseline candidate set plus explicitly adopted predecessor candidates; an explicit list must be its subset, and empty never means `git add -A`. Other pre-existing user dirt stays excluded evidence. Whole-tree staging belongs only to typed managed update/release transactions and the typed external patch-capture transaction; contexts without a captured baseline keep the legacy staging contract. -- A later independent root may adopt in-place changes from its host-validated predecessor source. Keep original dirt and separate `predecessor_adoption` in the same evidence. Require terminal quiescence, the same repository, exact terminal content fingerprints and unchanged per-path base. Unrelated landed paths/dirt remain separate; missing or size-only fingerprints prove no transfer. Adoption grants attribution, never review approval, an automatic task or a reset of the old wallet. +- A later independent root may adopt in-place changes through its admission/queue-validated predecessor source. Retain original dirt and separate `predecessor_adoption` evidence. Require terminal quiescence, same repo, exact content/Git-mode fingerprints and unchanged per-path base. Read config/index once per surface; honor `core.filemode` and indexed symlinks without index writes. Missing, size-only or unknown regular-file modes prove no transfer. Other landed work/dirt stays separate; absent legacy modes never block unrelated commits. Adoption grants attribution only, no review approval, automatic task or wallet reset. - Resolve unversioned Python only for `run_command`, `run_script`, `start_service` and run-kind `verify_and_record`, once BEFORE the shell guard, so guard and handler receive identical argv. Resolve bare Node for the same four surfaces but once AFTER the dispatch gates: the node health check executes an argv[0]-steered candidate, and probing before the gates would run a planted PATH shim for a request the fences would refuse. Never rewrite explicit paths, versioned interpreters, shell bodies or remote execution; never install a dependency in response to `ModuleNotFoundError`; with no usable runtime the argv runs as written and fails honestly (a rewritten absolute shebang is a disclosed residual). - Skill Review ordinals and provenance stay in `review_job.json` and the append-only `review_history.jsonl`: allocate under the lifecycle lock, consume a round only after actual start, write one terminal row per `job_id`, and compute legacy ordinals at read time without rewriting history. diff --git a/docs/inventories/FACADE_INVENTORY.md b/docs/inventories/FACADE_INVENTORY.md index bbede94c5..5a2e01469 100644 --- a/docs/inventories/FACADE_INVENTORY.md +++ b/docs/inventories/FACADE_INVENTORY.md @@ -2,7 +2,7 @@ AST-derived inventory of compatibility facades, regenerated by `python scripts/regenerate_inventories.py`. Do not edit. A facade row is any runtime module whose top-level `from import ...` statements carry the `noqa: F401` re-export marker — the codebase's declared "this binding exists for its binding, not for this module's own use" convention (reference FACADE_CONSUMERS method). Leaf domains come from `ouroboros/domains.toml`; a leaf outside the facade's domain is marked ✗ (that edge also appears in the manifest's pinned direction matrix). `tests/test_generated_inventories.py` pins byte-identity, so any re-export surface change must regenerate this file. -- facade modules: **58**; marked re-export bindings: **2273**; cross-domain facade→leaf pairs: **131** +- facade modules: **58**; marked re-export bindings: **2272**; cross-domain facade→leaf pairs: **131** | facade | domain | bindings | leaves | |---|---|---:|---| @@ -19,7 +19,7 @@ AST-derived inventory of compatibility facades, regenerated by `python scripts/r | `ouroboros/headless.py` | D17 | 45 | `ouroboros/contracts/task_constraint.py` (1 ✗D19)
`ouroboros/headless_status.py` (11)
`ouroboros/workspace_patch_capture.py` (20)
`ouroboros/workspace_patch_rules.py` (13) | | `ouroboros/llm.py` | D02 | 92 | `ouroboros/anthropic_native_custody.py` (7)
`ouroboros/context_budget.py` (2 ✗D03)
`ouroboros/llm_anthropic.py` (2)
`ouroboros/llm_attempt.py` (22)
`ouroboros/llm_capability_policy.py` (5)
`ouroboros/llm_fallback.py` (1)
`ouroboros/llm_gigachat.py` (1)
`ouroboros/llm_local.py` (7)
`ouroboros/llm_messages.py` (1)
`ouroboros/llm_openai_compatible.py` (5)
`ouroboros/llm_pricing.py` (4)
`ouroboros/llm_routing.py` (3)
`ouroboros/openrouter_attribution.py` (1)
`ouroboros/provider_models.py` (7)
`ouroboros/request_wire_recovery.py` (7)
`ouroboros/transport_custody.py` (1)
`ouroboros/usage_accounting.py` (14 ✗D16)
`ouroboros/utils.py` (2 ✗D18) | | `ouroboros/loop.py` | D01 | 257 | `ouroboros/config.py` (6 ✗D12)
`ouroboros/context.py` (1 ✗D03)
`ouroboros/context_budget.py` (1 ✗D03)
`ouroboros/context_compaction.py` (2 ✗D03)
`ouroboros/context_fit.py` (2 ✗D03)
`ouroboros/deadline_utils.py` (2)
`ouroboros/llm.py` (3 ✗D02)
`ouroboros/loop_acceptance.py` (21)
`ouroboros/loop_acceptance_review.py` (29)
`ouroboros/loop_budget.py` (13)
`ouroboros/loop_delivery.py` (29)
`ouroboros/loop_forced_finalization.py` (22)
`ouroboros/loop_llm_call.py` (6)
`ouroboros/loop_messages.py` (10)
`ouroboros/loop_model_call.py` (19)
`ouroboros/loop_nudges.py` (16)
`ouroboros/loop_round_limits.py` (16)
`ouroboros/loop_tool_execution.py` (5)
`ouroboros/loop_transport.py` (10)
`ouroboros/nanny_pacing.py` (4 ✗D07)
`ouroboros/observability.py` (1 ✗D16)
`ouroboros/outcomes.py` (17)
`ouroboros/pricing.py` (1 ✗D02)
`ouroboros/review_cycles.py` (1 ✗D06)
`ouroboros/task_finalization.py` (3)
`ouroboros/task_pacing.py` (1)
`ouroboros/tool_policy.py` (4 ✗D04)
`ouroboros/usage_accounting.py` (5 ✗D16)
`ouroboros/utils.py` (3 ✗D18)
`supervisor/owner_stop.py` (4 ✗D09) | -| `ouroboros/loop_acceptance.py` | D01 | 5 | `ouroboros/acceptance_settlement.py` (1)
`ouroboros/loop_messages.py` (4) | +| `ouroboros/loop_acceptance.py` | D01 | 4 | `ouroboros/loop_messages.py` (4) | | `ouroboros/loop_llm_call.py` | D01 | 2 | `ouroboros/config.py` (2 ✗D12) | | `ouroboros/loop_nudges.py` | D01 | 1 | `ouroboros/skill_readiness.py` (1 ✗D14) | | `ouroboros/loop_round_limits.py` | D01 | 5 | `supervisor/owner_stop.py` (5 ✗D09) | diff --git a/ouroboros/agent.py b/ouroboros/agent.py index 0f6c78d3f..fb5104c67 100644 --- a/ouroboros/agent.py +++ b/ouroboros/agent.py @@ -473,8 +473,7 @@ class OuroborosAgent: from ouroboros.mutation_attribution import capture_mutation_baseline predecessor = (task.get("predecessor_authority") or {}).get("source") - if (not isinstance(predecessor, dict) - or predecessor.get("task_id") != task.get("predecessor_task_id")): + if not isinstance(predecessor, dict): predecessor = None capture_mutation_baseline( pathlib.Path( diff --git a/ouroboros/loop_acceptance.py b/ouroboros/loop_acceptance.py index 9a55aacc6..1951c5a2f 100644 --- a/ouroboros/loop_acceptance.py +++ b/ouroboros/loop_acceptance.py @@ -597,9 +597,7 @@ ACCEPTANCE_DECISION_REASONS = ( "author_finish", "author_stop", "review_outcome_received", - # The pacing/wallet reason two branches below already STAMP (`pass_reason == - # REASON_REVIEW_CYCLES_EXHAUSTED`); it was missing from the closed set, so a - # spent shared cap shipped a reason no reader could validate. + # An explicit author stop can retain the wallet's exhausted-cycle reason. REASON_REVIEW_CYCLES_EXHAUSTED, # A-material (2026-08-30): the resubmit carried no changed candidate and no new # obligation disposition, so the recorded verdict was replayed for free. @@ -682,13 +680,11 @@ def merge_agent_acceptance_stance(trace: Dict[str, Any], decision: dict, ctx: An trace["acceptance_decision"] = merged -from ouroboros.acceptance_settlement import expose_acceptance_feedback # noqa: E402,F401 - existing import surface - def _collect_acceptance_obligations(llm_trace: Dict[str, Any], result: Any) -> None: """Typed PER-TASK obligations from critical contributing findings (v6.54.4). - Required+blocking path only. Each critical finding WITH a concrete + Blocking path after review eligibility. Each critical finding WITH a concrete recommendation becomes one open obligation in llm_trace (never the durable commit review_state — a separate SSOT). Clean finalization asks for an agent disposition per obligation (v6.54.0); time/pass gates and the diff --git a/ouroboros/loop_acceptance_review.py b/ouroboros/loop_acceptance_review.py index 1dfce4bce..6ada126a0 100644 --- a/ouroboros/loop_acceptance_review.py +++ b/ouroboros/loop_acceptance_review.py @@ -711,7 +711,8 @@ def _finish_advisory_author(ctx: _TaskAcceptanceContext) -> bool: author = build_author_disposition( disposition=disposition, rationale=str(stance.get("agent_rationale") or ""), subject_hash=ctx.review_binding["binding_hash"], - reviewer_signal=str((feedback or {}).get("aggregate_signal") or ""), enforcement=_loop().get_review_enforcement(), + reviewer_signal=str((feedback or {}).get("aggregate_signal") or ""), + enforcement="blocking" if review_enforcement_blocks(_loop().get_review_enforcement()) else "advisory", ) author["action"] = action from ouroboros.task_results import project_task_acceptance_review_capacity @@ -907,7 +908,7 @@ def _apply_task_acceptance_result( if capsule and open_obligations: _loop()._set_acceptance_decision(ctx.llm_trace, { "status": ACCEPTANCE_FINALIZED_UNACCEPTED, - "reason": pass_reason if pass_reason == REASON_REVIEW_CYCLES_EXHAUSTED else "open_obligations", + "reason": "open_obligations", "source": "task_acceptance_review", "rationale": ( f"Improvement gates exhausted ({pass_reason or 'passes spent'}) with " @@ -931,7 +932,6 @@ def _apply_task_acceptance_result( _loop()._set_acceptance_decision(ctx.llm_trace, { "status": ACCEPTANCE_FINALIZED_UNACCEPTED, "reason": ( - pass_reason if pass_reason == REASON_REVIEW_CYCLES_EXHAUSTED else "improvement_window_closed" if (not ctx.passes_done and pass_reason) else "capsule_spent" diff --git a/ouroboros/loop_model_call.py b/ouroboros/loop_model_call.py index df325758e..2ca7e06e5 100644 --- a/ouroboros/loop_model_call.py +++ b/ouroboros/loop_model_call.py @@ -487,7 +487,7 @@ def _dispatch_round_model( binding = (waiter.register_reprepare(role, lambda kwargs: _reprepare_waiting_main(ctx, kwargs)) if waiter is not None else contextlib.nullcontext()) previous_call = ctx.accumulated_usage.get("_last_llm_call_meta") - from ouroboros.loop_acceptance import expose_acceptance_feedback + from ouroboros.acceptance_settlement import expose_acceptance_feedback observe_feedback = lambda sent: expose_acceptance_feedback( getattr(ctx.tools._ctx, "_execution_trace", {}), sent, str(ctx.task_id)) diff --git a/ouroboros/mutation_attribution.py b/ouroboros/mutation_attribution.py index aa047a9fc..5ea52d0dd 100644 --- a/ouroboros/mutation_attribution.py +++ b/ouroboros/mutation_attribution.py @@ -15,8 +15,10 @@ from __future__ import annotations import hashlib import json +import logging import os import pathlib +import stat import subprocess from typing import Any, Iterable, Mapping, Sequence @@ -159,6 +161,62 @@ def _path_fingerprint(path: pathlib.Path) -> dict[str, Any]: return {"kind": "other", "size": int(stat.st_size)} +def _git_path_fingerprints(root: pathlib.Path, paths: Iterable[str]) -> dict[str, dict[str, Any]]: + """Capture content and ordinary Git staging modes with one config/index read. + + Only regular files need an additional mode fact: symlink targets and absence + already describe their Git type. Missing mode evidence remains unknown. + """ + fingerprints = {path: _path_fingerprint(root / path) for path in paths} + files = {path: row for path, row in fingerprints.items() if row.get("kind") == "file"} + if not files: + return fingerprints + for row in files.values(): + row["git_mode"] = None + try: + config = subprocess.run( + ["git", "config", "--type=bool", "--null", "--get-regexp", r"^core\.(filemode|symlinks)$"], + cwd=str(root), capture_output=True, text=True, check=False, + ) + if config.returncode not in (0, 1): + raise RuntimeError("Git mode configuration unavailable") + options = dict(row.split("\n", 1) for row in config.stdout.split("\0") if row) + modes, unmerged = {}, set() + for entry in _run_git(root, "ls-files", "--stage", "-z").split("\0"): + if not entry: + continue + metadata, path = entry.split("\t", 1) + mode, _oid, stage = metadata.split() + if path in files: + if stage == "0": + modes[path] = mode + else: + unmerged.add(path) + for path, row in files.items(): + if path in unmerged: + continue + indexed = modes.get(path) + if options.get("core.symlinks", "true") == "false" and indexed == "120000": + row["git_mode"] = indexed + elif options.get("core.filemode", "true") == "false": + row["git_mode"] = indexed if indexed in {"100644", "100755"} else "100644" + else: + row["git_mode"] = "100755" if (root / path).lstat().st_mode & stat.S_IXUSR else "100644" + except (OSError, RuntimeError, ValueError): + logging.getLogger(__name__).debug("Git fingerprint mode unavailable", exc_info=True) + return fingerprints + + +def _foreign_fingerprint_matches(previous: Any, current: Any) -> bool: + """Unknown legacy mode must not globally block already-excluded foreign WIP.""" + if not isinstance(previous, dict) or not isinstance(current, dict): + return False + if previous.get("git_mode") is not None and current.get("git_mode") is not None: + return previous == current + return ({key: value for key, value in previous.items() if key != "git_mode"} + == {key: value for key, value in current.items() if key != "git_mode"}) + + def _normalize_known_paths(root: pathlib.Path, values: Iterable[Any]) -> list[str]: normalized: list[str] = [] for value in values: @@ -220,9 +278,7 @@ def _capture_surface(surface: Mapping[str, Any]) -> dict[str, Any]: "base_commit": _run_git(root, "rev-parse", "HEAD").strip(), "base_tree": _run_git(root, "rev-parse", "HEAD^{tree}").strip(), "dirty_paths": dirty_paths, - "dirty_fingerprints": { - path: _path_fingerprint(root / path) for path in dirty_paths - }, + "dirty_fingerprints": _git_path_fingerprints(root, dirty_paths), } return row row["known_path_fingerprints"] = { @@ -382,6 +438,8 @@ def _predecessor_git_changes( if (path not in changed_base and path in (retained.get("candidates") or []) and isinstance(fingerprint, dict) and (fingerprint.get("sha256") or fingerprint.get("kind") in {"missing", "symlink"}) + and (fingerprint.get("kind") != "file" + or fingerprint.get("git_mode") in {"100644", "100755", "120000"}) and fingerprint == (git.get("dirty_fingerprints") or {}).get(path)): adoption["paths"].append(path) return adoption @@ -566,9 +624,10 @@ def attributed_git_candidates( excluded = sorted(path for path in changed if path in dirty) candidates = sorted(path for path in changed if path not in dirty) # A pre-existing dirty path is excluded evidence either way; it becomes a - # blocker only when its content actually CHANGED during the observed window + # blocker only when known content/mode CHANGED during the observed window # (unchanged owner WIP merely persisting must not wedge the task's commits). - if any(_path_fingerprint(root / path) != fingerprints.get(path) for path in excluded): + current_fingerprints = _git_path_fingerprints(root, excluded) + if any(not _foreign_fingerprint_matches(fingerprints.get(path), current_fingerprints[path]) for path in excluded): blockers.append("preexisting_dirty_changed") effect_state = str(evidence.get("effect_state") or "") if effect_state not in _OBSERVED_EFFECT_STATES: @@ -683,8 +742,13 @@ def record_terminal_mutation_candidates( dirty = _foreign_dirty_paths(git) fingerprints = git.get("dirty_fingerprints") or {} excluded = sorted(path for path in changed if path in dirty) + try: + current_fingerprints = _git_path_fingerprints(root, changed) + except OSError: + current_fingerprints = {} + blockers.append("candidate_fingerprint_unavailable") if any( - _path_fingerprint(root / path) != fingerprints.get(path) + not _foreign_fingerprint_matches(fingerprints.get(path), current_fingerprints.get(path)) for path in excluded ): blockers.append("preexisting_dirty_changed") @@ -697,11 +761,7 @@ def record_terminal_mutation_candidates( if isinstance(flag_row, dict) and str(flag_row.get("flag") or "") ) candidates = sorted(path for path in changed if path not in dirty) - candidate_fingerprints = {} - try: - candidate_fingerprints = {path: _path_fingerprint(root / path) for path in candidates} - except OSError: - blockers.append("candidate_fingerprint_unavailable") + candidate_fingerprints = {path: current_fingerprints[path] for path in candidates if path in current_fingerprints} row.update({ "candidates": candidates, "candidate_fingerprints": candidate_fingerprints, @@ -845,9 +905,7 @@ def advance_mutation_baseline( "base_commit": _run_git(root, "rev-parse", "HEAD").strip(), "base_tree": _run_git(root, "rev-parse", "HEAD^{tree}").strip(), "dirty_paths": still_dirty, - "dirty_fingerprints": { - path: _path_fingerprint(root / path) for path in still_dirty - }, + "dirty_fingerprints": _git_path_fingerprints(root, still_dirty), **({"predecessor_adoption": { **row["git"]["predecessor_adoption"], "paths": sorted(set(row["git"]["predecessor_adoption"].get("paths") or []) & set(still_dirty)), diff --git a/ouroboros/project_dialogue.py b/ouroboros/project_dialogue.py index 6eaad5aae..f1d23d768 100644 --- a/ouroboros/project_dialogue.py +++ b/ouroboros/project_dialogue.py @@ -722,7 +722,7 @@ TASK_CAUSE_PHRASES = { # accepted decision with a sentence here still states its cause. "previous_revision_accepted": "The reviewers approved the earlier version of this answer; it changed before they finished.", "author_stop": "Main stopped with unfinished work; no review approval was granted.", - "review_outcome_received": "The review could not start; Main received the recorded reason.", + "review_outcome_received": "Main received the review outcome or recorded limitation.", "author_finish": "The answer was delivered on Main's own judgement; the reviewers had not signed it off.", "review_degraded": "No reviewer verdict was established for this answer.", "infra_failure": "A review infrastructure failure prevented a settled verdict.", diff --git a/ouroboros/review_custody.py b/ouroboros/review_custody.py index 2324c9f49..ed5c15af1 100644 --- a/ouroboros/review_custody.py +++ b/ouroboros/review_custody.py @@ -1035,7 +1035,7 @@ def _settle_review_attempt( if entry.released_early: # plan review's event route: progress line + the settled-wave frame from ouroboros.tools.plan_review_collect import announce_released_settlement - announce_released_settlement(usage_ctx, request=request, task_id=task_id, slot=slot, actor=actor, + announce_released_settlement(usage_ctx, request=request, task_id=task_id, actor=actor, settled_wave=dict(released_wave.get("slots") or {}), roster_size=int(released_wave.get("total") or 0)) if getattr(request, "surface", "") == "task_acceptance" and (released_wave or quorum_wave): from ouroboros.acceptance_settlement import announce_acceptance_settlement diff --git a/ouroboros/review_dispatch.py b/ouroboros/review_dispatch.py index 26ed97426..11b6e2c32 100644 --- a/ouroboros/review_dispatch.py +++ b/ouroboros/review_dispatch.py @@ -210,8 +210,8 @@ def reconcile_pending_acceptance_runs( log.warning("acceptance run %s could not be reconciled: %s", str(run.get("panel_id") or "")[:16], exc) continue - # Keep the paid operation's request; only its producer facts advance. - run.update({key: value for key, value in vars(result).items() if key != "request"}) + # Keep the host panel identity and paid request; only producer facts advance. + run.update({key: value for key, value in vars(result).items() if key not in {"request", "panel_id"}}) advanced += not acceptance_run_pending(run) return advanced diff --git a/ouroboros/review_status_projection.py b/ouroboros/review_status_projection.py index 8c631e14d..0c6133004 100644 --- a/ouroboros/review_status_projection.py +++ b/ouroboros/review_status_projection.py @@ -145,7 +145,14 @@ def build_review_status_payload(projection: Dict[str, Any], *, next_step: str, i } if selected_attempt is not None and selected_attempt.phase in {"review_only", "late_wait"}: from ouroboros.config import get_review_enforcement - if get_review_enforcement() == "advisory": + from ouroboros.review_records import review_outcome_received + + received = review_outcome_received( + [*selected_attempt.triad_raw_results, selected_attempt.scope_raw_result], + findings=[*selected_attempt.critical_findings, *selected_attempt.advisory_findings], + terminal=selected_attempt.phase == "review_only" and selected_attempt.status == "reviewed", + ) + if get_review_enforcement() == "advisory" and received: payload["review_reference"] = {"surface": "commit", **{key: getattr(selected_attempt, key) for key in ("repo_key", "task_id", "tool_name", "attempt", "pre_review_fingerprint")}} payload["next_step"] = ("Read the returned findings. You may revise and request another permitted review, stop, or call " diff --git a/ouroboros/settings_defaults.py b/ouroboros/settings_defaults.py index b1d0b971e..a58f8e3a8 100644 --- a/ouroboros/settings_defaults.py +++ b/ouroboros/settings_defaults.py @@ -446,8 +446,8 @@ def retired_setting_keys_notice(dropped: tuple[str, ...], *, reviewer_slots: tup elif state == "invalid": panel = ( "NO reviewer panel: that setting is malformed, so reviews are refused " - "(commit review blocks; under Advisory enforcement it warns and commits " - "unreviewed) until it is repaired on the Settings page — %s" % parse_error) + "(Blocking prevents committing; Advisory returns the failure for an explicit " + "author decision) until it is repaired on the Settings page — %s" % parse_error) else: panel = "the SHIPPED default reviewer panel until that setting is authored (Settings page)" clauses.append( diff --git a/ouroboros/task_pacing.py b/ouroboros/task_pacing.py index ad7115930..50c8afade 100644 --- a/ouroboros/task_pacing.py +++ b/ouroboros/task_pacing.py @@ -1090,7 +1090,7 @@ def _acceptance_rails_line_inner( "(deadline/budget rails bind)" ) else: - passes_part = f"review passes: {int(passes_done)}/{int(cap)}" + passes_part = f"author passes: {int(passes_done)}/{int(cap)}" # v6.74.4 freeze directive (count axis): the pass launched at # cap-1 is the last one improvement_pass_allowed will admit, so # say so. cap==0 never feeds a capsule back; skip the clause, and diff --git a/ouroboros/tools/plan_review_collect.py b/ouroboros/tools/plan_review_collect.py index b407127d3..3c0fef7e4 100644 --- a/ouroboros/tools/plan_review_collect.py +++ b/ouroboros/tools/plan_review_collect.py @@ -24,7 +24,7 @@ log = logging.getLogger(__name__) def announce_released_settlement( - usage_ctx: Any, *, request: Any, task_id: str, slot: Any, actor: Any, + usage_ctx: Any, *, request: Any, task_id: str, actor: Any, settled_wave: Dict[str, str], roster_size: int = 0, ) -> None: """Attach late evidence and announce whole-wave settlement through its mailbox. diff --git a/ouroboros/tools/review.py b/ouroboros/tools/review.py index 46f312b6b..6ae56be67 100644 --- a/ouroboros/tools/review.py +++ b/ouroboros/tools/review.py @@ -1333,7 +1333,7 @@ def _dispatch_unified_review(ctx: ToolContext, commit_message: str, prepared: di ) return _handle_review_block_or_warning( ctx, blocking_review, blocked_msg, - "Review enforcement=Advisory: review infrastructure failure did not block commit. ", + "Review enforcement=Advisory: review infrastructure failed; an explicit author decision is required. ", ) if "error" in result: @@ -1347,7 +1347,7 @@ def _dispatch_unified_review(ctx: ToolContext, commit_message: str, prepared: di ) return _handle_review_block_or_warning( ctx, blocking_review, blocked_msg, - "Review enforcement=Advisory: review service error did not block commit. ", + "Review enforcement=Advisory: review service failed; an explicit author decision is required. ", ) model_results = result.get("results", []) @@ -1359,7 +1359,7 @@ def _dispatch_unified_review(ctx: ToolContext, commit_message: str, prepared: di "model — commit cannot proceed without a successful review.") return _handle_review_block_or_warning( ctx, blocking_review, blocked_msg, - "Review enforcement=Advisory: review returned no model results; commit proceeding anyway. ") + "Review enforcement=Advisory: no model results were received; an explicit author decision is required. ") critical_fails, advisory_warns, errored_models, _triad_raw = _collect_review_findings(ctx, model_results) models_total = len(model_results) @@ -1381,7 +1381,7 @@ def _dispatch_unified_review(ctx: ToolContext, commit_message: str, prepared: di f"{', '.join(pending_models)}. Retry the same commit to reconcile them without a blind paid resend.") pending_block = _handle_review_block_or_warning( ctx, blocking_review, blocked_msg, - "Review enforcement=Advisory: pending review work did not block commit. ", + "Review enforcement=Advisory: review is pending; collect its outcome before choosing an author continuation. ", ) if pending_block is not None: return pending_block @@ -1401,7 +1401,7 @@ def _dispatch_unified_review(ctx: ToolContext, commit_message: str, prepared: di ) return _handle_review_block_or_warning( ctx, blocking_review, blocked_msg, - "Review enforcement=Advisory: review quorum failure did not block commit. ", + "Review enforcement=Advisory: review quorum was not met; an explicit author decision is required. ", ) if models_total < 2: @@ -1440,14 +1440,10 @@ def _dispatch_unified_review(ctx: ToolContext, commit_message: str, prepared: di ctx, ("Cyber Pro: critical review findings do not prohibit action." if not review_enforcement_blocks("blocking") else - "Review enforcement=Advisory: critical review findings did not block commit."), + "Review enforcement=Advisory: critical findings require an explicit author decision before committing."), ) for finding in getattr(ctx, "_last_review_critical_findings", []) or []: _append_review_warning(ctx, finding) - for warning in getattr(ctx, "_last_review_advisory_findings", []) or []: - _append_review_warning(ctx, warning) - if errored_note: - _append_review_warning(ctx, errored_note) if not critical_fails: # All clear: reset iteration state. With critical findings present diff --git a/tests/test_cyber_review_provenance.py b/tests/test_cyber_review_provenance.py index 68a4205b4..d902cc6d0 100644 --- a/tests/test_cyber_review_provenance.py +++ b/tests/test_cyber_review_provenance.py @@ -291,3 +291,41 @@ def test_legacy_settled_review_replays_its_proven_subject_without_key_error(actu assert h.trace["review_runs"][-1]["actors"] == old_actors assert h.trace["acceptance_decision"]["reason"] != "infra_failure" assert "KeyError" not in json.dumps(h.trace.get("acceptance_decision") or {}) + + +@pytest.mark.parametrize("runtime,enforcement", [("pro", "advisory"), ("cyber_pro", "advisory"), ("cyber_pro", "blocking")]) +@pytest.mark.parametrize("explicit", [False, True]) +def test_task_author_record_preserves_configured_and_effective_authority(actual_acceptance, monkeypatch, runtime, enforcement, explicit): + import inspect + from ouroboros import config + from ouroboros.acceptance_settlement import expose_acceptance_feedback + from ouroboros.outcomes import _objective_axis + from ouroboros.project_dialogue import outcome_phase + + h = actual_acceptance + # Obtain a real independent outcome in ordinary Advisory first. + h.run("Initial answer") + original = copy.deepcopy(h.trace["review_runs"][-1]) + expose_acceptance_feedback(h.trace, inspect.getclosurevars(h.run).nonlocals["messages"], "task") + if explicit or runtime == "pro": + h.annotate() + config.reset_runtime_mode_baseline_for_tests() + config.initialize_runtime_mode_baseline(runtime) + monkeypatch.setenv("OUROBOROS_REVIEW_ENFORCEMENT", enforcement) + try: + h.run("Verified revised answer") + decision = h.trace["acceptance_decision"] + assert decision["enforcement"] == enforcement + assert decision["author_disposition"]["enforcement"] == "advisory" + assert decision["reason"] == "author_finish" + assert h.trace["review_runs"][0]["actors"] == original["actors"] + review = {"status": "fail", "acceptance_decision": decision} + objective = _objective_axis(review) + row = {"status": "completed", "outcome_axes": { + "objective": objective, "review": review, "execution": {"status": "ok"}}} + assert objective["status"] == "pass" + assert outcome_phase(row, {}) == "done" + row["outcome_axes"]["execution"]["status"] = "failed" + assert outcome_phase(row, {}) == "error" + finally: + config.reset_runtime_mode_baseline_for_tests() diff --git a/tests/test_git_mode_attribution.py b/tests/test_git_mode_attribution.py new file mode 100644 index 000000000..9c7529262 --- /dev/null +++ b/tests/test_git_mode_attribution.py @@ -0,0 +1,183 @@ +"""Exact predecessor custody follows ordinary Git modes, not raw permissions.""" +import copy +import hashlib +import os + +import pytest + +from ouroboros import mutation_attribution as attribution +from ouroboros.task_results import load_task_result, write_task_result +from tests.test_mutation_attribution import _git, _repo +from tests.test_predecessor_mutation_handoff import _previous, _source, _start + + +@pytest.mark.parametrize("filemode", [True, False]) +def test_foreign_git_mode_is_not_adopted(tmp_path, filemode): + if filemode and os.name == "nt": + pytest.skip("filesystem executable-bit probe; index-mode case covers Windows") + root, data = _previous(tmp_path) + _git(root, "config", "core.filemode", str(filemode).lower()) + if filemode: + (root / "clean.txt").chmod(0o755) + else: + _git(root, "update-index", "--chmod=+x", "clean.txt") + index = (root / ".git/index").read_bytes() + _start(data, root, "successor", _source()) + selected, evidence, error = attribution.resolve_attributed_git_paths(data, "successor", root, None) + assert selected == ["new.txt"] and not error + assert evidence["excluded_preexisting_dirty"] == ["clean.txt", "dirty.txt"] + assert (root / ".git/index").read_bytes() == index, "observation must preserve staged work" + if filemode: + _git(root, "add", "--", *selected) + assert "mode change" not in _git(root, "diff", "--cached", "--summary") + + +@pytest.mark.parametrize("filemode", [True, False]) +def test_predecessor_executable_correction_keeps_its_mode(tmp_path, filemode): + if filemode and os.name == "nt": + pytest.skip("filesystem executable-bit probe; index-mode case covers Windows") + root, data = _previous(tmp_path) + _git(root, "config", "core.filemode", str(filemode).lower()) + if filemode: + (root / "clean.txt").chmod(0o755) + else: + _git(root, "update-index", "--chmod=+x", "clean.txt") + terminal = attribution.record_terminal_mutation_candidates(data, "previous") + assert terminal["terminal_candidate_snapshot"]["surfaces"][0]["candidate_fingerprints"]["clean.txt"]["git_mode"] == "100755" + _start(data, root, "successor", _source()) + selected, _, error = attribution.resolve_attributed_git_paths(data, "successor", root, None) + assert selected == ["clean.txt", "new.txt"] and not error + _git(root, "add", "--", *selected) + assert "mode change 100644 => 100755 clean.txt" in _git(root, "diff", "--cached", "--summary") + + +@pytest.mark.skipif(os.name == "nt", reason="requires meaningful POSIX chmod") +@pytest.mark.parametrize("filemode,mode", [(True, 0o654), (True, 0o645), (False, 0o755)]) +def test_git_ignored_permission_changes_preserve_handoff(tmp_path, filemode, mode): + root, data = _previous(tmp_path) + _git(root, "config", "core.filemode", str(filemode).lower()) + (root / "clean.txt").chmod(mode) + _start(data, root, "successor", _source()) + selected, _, error = attribution.resolve_attributed_git_paths(data, "successor", root, None) + assert selected == ["clean.txt", "new.txt"] and not error + _git(root, "add", "--", *selected) + assert "mode change" not in _git(root, "diff", "--cached", "--summary") + + +def test_git_mode_capture_is_batched_and_read_only(tmp_path, monkeypatch): + root = _repo(tmp_path) + _git(root, "config", "core.filemode", "false") + paths = ["clean.txt", "dirty.txt"] + [f"new-{i}.txt" for i in range(20)] + for path in paths[2:]: + (root / path).write_text("new\n", encoding="utf-8") + index_before = hashlib.sha256((root / ".git/index").read_bytes()).hexdigest() + calls, run = [], attribution.subprocess.run + + def observe(argv, **kwargs): + calls.append(argv) + return run(argv, **kwargs) + + monkeypatch.setattr(attribution.subprocess, "run", observe) + fingerprints = attribution._git_path_fingerprints(root, paths) + assert len(calls) == 2, "one config and one index read, independent of path count" + assert all(row["sha256"] and row["git_mode"] == "100644" for row in fingerprints.values()) + assert hashlib.sha256((root / ".git/index").read_bytes()).hexdigest() == index_before + + +def test_indexed_symlink_mode_matches_git_on_a_regular_file(tmp_path): + root = _repo(tmp_path) + _git(root, "config", "core.symlinks", "false") + link = root / "link" + link.write_text("clean.txt", encoding="utf-8") + blob = _git(root, "hash-object", "-w", "link") + _git(root, "update-index", "--add", "--cacheinfo", f"120000,{blob},link") + link.write_text("dirty.txt", encoding="utf-8") + fingerprint = attribution._git_path_fingerprints(root, ["link"])["link"] + assert fingerprint["kind"] == "file" and fingerprint["git_mode"] == "120000" + _git(root, "add", "--", "link") + assert _git(root, "ls-files", "--stage", "link").startswith("120000 ") + + +@pytest.mark.skipif(os.name == "nt", reason="requires meaningful POSIX chmod") +def test_predecessor_executable_addition_is_retained(tmp_path): + root, data = _previous(tmp_path) + _git(root, "config", "core.filemode", "true") + (root / "new.txt").chmod(0o744) + attribution.record_terminal_mutation_candidates(data, "previous") + _start(data, root, "successor", _source()) + selected, _, error = attribution.resolve_attributed_git_paths(data, "successor", root, None) + assert selected == ["clean.txt", "new.txt"] and not error + _git(root, "add", "--", *selected) + assert "create mode 100755 new.txt" in _git(root, "diff", "--cached", "--summary") + + +def test_new_files_deletions_and_symlinks_keep_exact_handoff(tmp_path): + root, data = _previous(tmp_path) + (root / "clean.txt").unlink() + try: + (root / "link").symlink_to("new.txt") + except OSError: + pytest.skip("native symlink creation unavailable") + attribution.record_terminal_mutation_candidates(data, "previous") + _start(data, root, "successor", _source()) + selected, _, error = attribution.resolve_attributed_git_paths(data, "successor", root, None) + assert selected == ["clean.txt", "link", "new.txt"] and not error + _git(root, "add", "--", *selected) + summary = _git(root, "diff", "--cached", "--summary") + assert "delete mode 100644 clean.txt" in summary + assert "create mode 120000 link" in summary and "create mode 100644 new.txt" in summary + + +def test_legacy_mode_is_not_invented_for_exact_handoff(tmp_path): + root, data = _previous(tmp_path) + old = copy.deepcopy(load_task_result(data, "previous")["mutation_evidence"]) + for row in old["terminal_candidate_snapshot"]["surfaces"][0]["candidate_fingerprints"].values(): + row.pop("git_mode", None) + write_task_result(data, "previous", "completed", mutation_evidence=old) + _start(data, root, "successor", _source()) + assert attribution.attributed_git_candidates(data, "successor", root)["candidates"] == [] + assert load_task_result(data, "previous")["mutation_evidence"] == old + + +def test_legacy_excluded_wip_does_not_block_other_paths_or_epoch_advance(tmp_path): + root, data = _previous(tmp_path) + evidence = _start(data, root, "independent") + for row in evidence["baseline"]["surfaces"][0]["git"]["dirty_fingerprints"].values(): + row.pop("git_mode", None) + write_task_result(data, "independent", "running", mutation_evidence=evidence) + (root / "mine.txt").write_text("independent work\n", encoding="utf-8") + selected, _, error = attribution.resolve_attributed_git_paths(data, "independent", root, None) + assert selected == ["mine.txt"] and not error + _git(root, "add", "--", *selected) + _git(root, "commit", "-qm", "fixture independent commit") + advanced = attribution.advance_mutation_baseline(data, "independent", root) + dirty = advanced["baseline"]["surfaces"][0]["git"]["dirty_fingerprints"] + assert all(row["git_mode"] == "100644" for row in dirty.values()) + + +@pytest.mark.skipif(os.name == "nt", reason="requires meaningful POSIX chmod") +def test_foreign_mode_change_is_visible_at_current_and_terminal_comparison(tmp_path): + root, data = _previous(tmp_path) + _git(root, "config", "core.filemode", "true") + _start(data, root, "independent") + (root / "dirty.txt").chmod(0o755) + assert "preexisting_dirty_changed" in attribution.attributed_git_candidates(data, "independent", root)["blockers"] + terminal = attribution.record_terminal_mutation_candidates(data, "independent") + assert "preexisting_dirty_changed" in terminal["terminal_candidate_snapshot"]["surfaces"][0]["blockers"] + + +def test_unknown_mode_cannot_authorize_a_regular_file_transfer(tmp_path, monkeypatch): + root, data = _previous(tmp_path) + run_git = attribution._run_git + + def unreadable_index(path, *args): + if args[0] == "ls-files": + raise RuntimeError("index unavailable") + return run_git(path, *args) + + monkeypatch.setattr(attribution, "_run_git", unreadable_index) + terminal = attribution.record_terminal_mutation_candidates(data, "previous") + assert terminal["terminal_candidate_snapshot"]["surfaces"][0]["candidate_fingerprints"]["clean.txt"]["git_mode"] is None + monkeypatch.setattr(attribution, "_run_git", run_git) + _start(data, root, "successor", _source()) + assert attribution.attributed_git_candidates(data, "successor", root)["candidates"] == [] diff --git a/tests/test_git_review_enforcement.py b/tests/test_git_review_enforcement.py index 5f94f98ed..2d5ca58c0 100644 --- a/tests/test_git_review_enforcement.py +++ b/tests/test_git_review_enforcement.py @@ -178,7 +178,7 @@ class TestReviewEnforcementModes: result = review._run_unified_review(ctx, "test commit", repo_dir=ctx.repo_dir) assert result is None assert any( - isinstance(w, str) and "critical review findings did not block commit" in w.lower() + isinstance(w, str) and "critical findings require an explicit author decision" in w.lower() for w in ctx._review_advisory ) assert any( @@ -195,13 +195,18 @@ class TestReviewEnforcementModes: self._mock_staged(monkeypatch, review, changed_files="x.py") monkeypatch.setenv("OUROBOROS_REVIEW_ENFORCEMENT", "advisory") ctx._review_advisory = ["prior deterministic/preflight warning"] - monkeypatch.setattr(review, "_handle_multi_model_review", lambda *a, **kw: self._fake_result( + response = json.loads(self._fake_result( '[{"item":"contract","verdict":"FAIL","severity":"critical","reason":"material original finding"}]', '[{"item":"style","verdict":"FAIL","severity":"advisory","reason":"minor original finding"}]')) + response["results"].append({"model": "failed-critic", "error": "Transport unavailable"}) + monkeypatch.setattr(review, "_handle_multi_model_review", lambda *a, **kw: json.dumps(response)) assert review._run_unified_review(ctx, "candidate", repo_dir=ctx.repo_dir) is None saved = json.dumps(ctx._review_advisory) assert "prior deterministic/preflight warning" in saved - assert "material original finding" in saved and "minor original finding" in saved + assert saved.count("material original finding") == 1 + assert saved.count("minor original finding") == 1 + assert saved.count("prior deterministic/preflight warning") == 1 + assert saved.count("Note: 1 of 3 review models") == 1 @pytest.mark.parametrize("failure", ["nonzero_rc", "non_utf8_rc"]) def test_uncapturable_staged_diff_blocks_instead_of_reviewing_a_placeholder( diff --git a/tests/test_llm_provider_golden.py b/tests/test_llm_provider_golden.py index 4420275a0..c550d0771 100644 --- a/tests/test_llm_provider_golden.py +++ b/tests/test_llm_provider_golden.py @@ -81,6 +81,8 @@ _ROUTE_ENV_NAMES = ( "OUROBOROS_OBSERVABILITY_KEEP_RAW", "OUROBOROS_MODEL_ACCOUNTS", "OUROBOROS_MODEL_CONTEXT_WINDOWS", + "OUROBOROS_PROCESSING_PREFERENCE", + "OUROBOROS_MODEL_PROCESSING_PREFERENCES", ) # Class-level caches LLMClient uses as process-global memory. Reset per case. @@ -714,6 +716,23 @@ def test_llm_provider_route_matches_golden(case): ) + +@pytest.mark.parametrize("role", ["", "main"]) +def test_processing_environment_isolated_but_explicit_case_options_apply(monkeypatch, role): + case = next(case for case in _CASES if case["id"] == "openai.dispatch.happy_path") + spec = copy.deepcopy(case["spec"]) + if role: + spec["call"]["kwargs"]["model_role"] = role + monkeypatch.setenv("OUROBOROS_PROCESSING_PREFERENCE", "fast") + monkeypatch.setenv("OUROBOROS_MODEL_PROCESSING_PREFERENCES", '{"main":"economy"}') + assert _observe(spec) == case["expected"] + key = "OUROBOROS_MODEL_PROCESSING_PREFERENCES" if role else "OUROBOROS_PROCESSING_PREFERENCE" + spec.setdefault("env", {})[key] = '{"main":"standard"}' if role else "standard" + explicit = _observe(spec) + assert explicit["sends"][0]["payload"]["service_tier"] == "default" + assert explicit["returned"]["usage"]["processing"]["requested"] == "standard" + + def test_golden_covers_every_declared_provider_lane(): """Coverage floor: dropping a lane's fixtures must fail, not go unnoticed.""" covered = {case["id"].split(".", 1)[0] for case in _CASES} diff --git a/tests/test_loop_acceptance_gate.py b/tests/test_loop_acceptance_gate.py index bc25ec296..edf1ab7c0 100644 --- a/tests/test_loop_acceptance_gate.py +++ b/tests/test_loop_acceptance_gate.py @@ -108,7 +108,8 @@ def test_every_host_acceptance_writer_emits_a_canonical_status_and_typed_reason( # cannot escape the guard by living in (or moving to) a leaf. loop_file = pathlib.Path(loop_mod.__file__) src = [] - for path in [loop_file, *sorted(loop_file.parent.glob("loop_*.py"))]: + for path in [loop_file, *sorted(loop_file.parent.glob("loop_*.py")), + loop_file.parent / "acceptance_settlement.py"]: src.extend(path.read_text(encoding="utf-8").splitlines()) starts = [ i for i, line in enumerate(src) @@ -116,7 +117,7 @@ def test_every_host_acceptance_writer_emits_a_canonical_status_and_typed_reason( ] # Include the separate infrastructure-outcome handback; it requests an # author response without manufacturing a critic capsule or reviewer PASS. - assert len(starts) == 21, f"writer inventory changed: {len(starts)} call sites" + assert len(starts) == 22, f"writer inventory changed: {len(starts)} call sites" allowed_status = { "ACCEPTANCE_ACCEPTED", "ACCEPTANCE_REVISION_REQUESTED", "ACCEPTANCE_FINALIZED_UNACCEPTED", @@ -148,7 +149,7 @@ def test_every_host_acceptance_writer_emits_a_canonical_status_and_typed_reason( seen_expression_reasons += 1 assert reason_names[name] in ACCEPTANCE_DECISION_REASONS, name # The widened regex really does catch expression-valued reasons: the two - # `pass_reason if ... == REASON_REVIEW_CYCLES_EXHAUSTED` branches and the + # explicit author-stop REASON_REVIEW_CYCLES_EXHAUSTED branches and the # A-material `REASON_IDENTICAL_ACCEPTANCE_REFUSED` writer. assert seen_expression_reasons >= 3, seen_expression_reasons diff --git a/tests/test_owner_hurry_s3.py b/tests/test_owner_hurry_s3.py index 053d83bd3..59be72d69 100644 --- a/tests/test_owner_hurry_s3.py +++ b/tests/test_owner_hurry_s3.py @@ -473,7 +473,7 @@ def test_required_blocking_unbounded_loop_collapses_to_zero_under_hurry(tmp_path rails = tp_mod.acceptance_rails_line( snapshot, effective, 0, None, required_blocking=True, ) - assert "review passes: 0/0" in rails + assert "author passes: 0/0" in rails assert "no local count cap" not in rails # An unlatched ctx passes the profile through UNCHANGED (identity). unlatched = _acceptance_ctx(tmp_path, latched=False) diff --git a/tests/test_plan_review_historical_supplements.py b/tests/test_plan_review_historical_supplements.py index fe87d625c..accbc06bb 100644 --- a/tests/test_plan_review_historical_supplements.py +++ b/tests/test_plan_review_historical_supplements.py @@ -94,7 +94,7 @@ def test_a_b_late_a_preserves_authority_and_complete_source(tmp_path): before = state(tmp_path) actor, text = complete(tmp_path, a, req, slots[0]) ctx = SimpleNamespace(drive_root=tmp_path, task_id=TASK, emit_progress_fn=lambda text: None) - collect.announce_released_settlement(ctx, request=req, task_id=TASK, slot=slots[0], + collect.announce_released_settlement(ctx, request=req, task_id=TASK, actor=actor, settled_wave={}) after = state(tmp_path) old = tr.plan_review_wave(after, A) diff --git a/tests/test_predecessor_mutation_handoff.py b/tests/test_predecessor_mutation_handoff.py index d6ebf06ef..33ecc8067 100644 --- a/tests/test_predecessor_mutation_handoff.py +++ b/tests/test_predecessor_mutation_handoff.py @@ -134,15 +134,16 @@ def test_real_startup_passes_validated_predecessor_to_baseline(tmp_path): root, data = _previous(tmp_path) task = {"id": "successor", "root_task_id": "successor", "budget_drive_root": str(data), - "predecessor_task_id": "previous", "predecessor_authority_source": _source()} + "predecessor_authority_source": _source()} assert not validate_task_authority_sources(data, task) write_task_result(data, "successor", "running") agent = SimpleNamespace(env=SimpleNamespace(repo_dir=root, drive_root=data, budget_drive_root=str(data))) OuroborosAgent._capture_mutation_baseline(agent, task, {}) assert attributed_git_candidates(data, "successor", root)["candidates"] == ["clean.txt", "new.txt"] # Merely carrying old context is not an explicit selection by this task. - unselected = {**task, "id": "unselected", "root_task_id": "unselected"} - unselected.pop("predecessor_task_id") + unselected = {"id": "unselected", "root_task_id": "unselected", "budget_drive_root": str(data), + "metadata": {"project_last_task_result": {"task_id": "previous"}}} write_task_result(data, "unselected", "running") + assert not validate_task_authority_sources(data, unselected) OuroborosAgent._capture_mutation_baseline(agent, unselected, {}) assert attributed_git_candidates(data, "unselected", root)["candidates"] == [] diff --git a/tests/test_review_author_finality.py b/tests/test_review_author_finality.py index ddfe257a1..cd82ed2ac 100644 --- a/tests/test_review_author_finality.py +++ b/tests/test_review_author_finality.py @@ -161,7 +161,7 @@ def test_post_review_finish_handles_revised_answer_without_another_panel(monkeyp ) assert run("initial answer") is True critic_hash = trace["review_runs"][-1]["binding_hash"] - from ouroboros.loop_acceptance import expose_acceptance_feedback + from ouroboros.acceptance_settlement import expose_acceptance_feedback expose_acceptance_feedback(trace, messages, "author-root") trace["tool_calls"].append({"tool": "task_acceptance_review", "args": {}}) merge_agent_acceptance_stance(trace, {"disposition": "partial", "explicit_finish": change != "ordinary_evidence", "rationale": "I fixed the material issue."}, tools_ctx) diff --git a/tests/test_review_feedback_continuation.py b/tests/test_review_feedback_continuation.py index 70a75f87f..4f62924fd 100644 --- a/tests/test_review_feedback_continuation.py +++ b/tests/test_review_feedback_continuation.py @@ -7,7 +7,8 @@ from types import SimpleNamespace import pytest from ouroboros import loop, review_substrate, task_pacing -from ouroboros.loop_acceptance import expose_acceptance_feedback, merge_agent_acceptance_stance +from ouroboros.acceptance_settlement import expose_acceptance_feedback +from ouroboros.loop_acceptance import merge_agent_acceptance_stance from ouroboros.loop_acceptance_review import _apply_task_acceptance_result from ouroboros.review_records import ReviewRunResult from ouroboros.task_results import effective_task_acceptance_review_cycles diff --git a/tests/test_review_wave_identity.py b/tests/test_review_wave_identity.py new file mode 100644 index 000000000..d1a10fc0c --- /dev/null +++ b/tests/test_review_wave_identity.py @@ -0,0 +1,94 @@ +"""A collected host review keeps one identity across durable publications.""" +import copy +import json +import threading +import time +from types import SimpleNamespace + + + + +def test_one_operation_through_pending_and_settled_publications(tmp_path, monkeypatch): + from ouroboros import review_custody, review_dispatch, review_projection + from ouroboros.loop_acceptance_review import _record_host_acceptance_run, acceptance_run_pending + from ouroboros.review_records import ReviewRequest, ReviewSlot + from ouroboros.review_substrate import run_review_request + from ouroboros.task_results import load_task_result, merge_review_projection + from tests.test_loop_acceptance_gate import _seed_acceptance_root + + entered, release, settled = threading.Event(), threading.Event(), threading.Event() + calls, stamps = [], [] + real_settle = review_custody._settle_review_attempt + + def settle(*args, **kwargs): + try: + return real_settle(*args, **kwargs) + finally: + settled.set() + + monkeypatch.setattr(review_custody, '_settle_review_attempt', settle) + + class HeldModel: + def chat(self, **kwargs): + # A recording LLM bypasses the real API accounting boundary, so + # explicitly invoke its already-bound dispatch stamp there. + review_dispatch.invoke_bound_api_review_paid_stamp() + calls.append(kwargs) + entered.set() + assert release.wait(10) + return {'content': json.dumps({'verdict': 'FAIL', 'summary': 'Actual retained criticism', + 'findings': []})}, {'prompt_tokens': 5, 'completion_tokens': 2} + + ctx = SimpleNamespace(task_id='panel-audit', task_attempt=1, drive_root=tmp_path, + budget_drive_root=tmp_path, task_metadata={}, pending_events=[], event_queue=None) + _seed_acceptance_root(tmp_path, ctx.task_id, ctx) + ctx._review_paid_stamp = review_dispatch.ReviewPaidStamp(lambda: stamps.append('physical')) + request = ReviewRequest(surface='task_acceptance', task_id=ctx.task_id, task_attempt=1, + goal='Audit publication identity', subject='The same complete answer', + evidence={'requirement': 'same'}, retry_key='panel-audit:paid-one', drain_deadline=time.monotonic()) + slot = ReviewSlot(slot_id='one', model='model/original', effort='high', timeout_sec=20) + trace = {'review_runs': []} + snapshots = [] + try: + first = run_review_request(request, slots=[slot], drive_root=tmp_path, usage_ctx=ctx, llm=HeldModel()) + assert entered.wait(5) and acceptance_run_pending(first) + binding = review_projection.build_review_binding(candidate=request.subject, evidence=request.evidence, + fence_token_or_state='same fence') + binding['paid_identity'] = 'paid-one' + owner = SimpleNamespace(tools=SimpleNamespace(_ctx=ctx), llm_trace=trace, review_binding=binding) + run = _record_host_acceptance_run(owner, first) + + def publish(): + review_projection.publish_acceptance_checkpoint(ctx, trace) + saved = load_task_result(tmp_path, ctx.task_id) + snapshots.append(copy.deepcopy(saved['review_projection'])) + return saved + + publish() + advanced = review_dispatch.reconcile_pending_acceptance_runs(trace, drive_root=tmp_path, usage_ctx=ctx) + assert advanced == 0 and acceptance_run_pending(run) + publish() + release.set() + assert settled.wait(5) + advanced = review_dispatch.reconcile_pending_acceptance_runs(trace, drive_root=tmp_path, usage_ctx=ctx) + assert advanced == 1 and not acceptance_run_pending(run) + saved = publish() + assert len(calls) == len(stamps) == 1 + assert len(trace['review_runs']) == 1 + rows = saved['review_projection']['panels'] + assert len(rows) == 1 + assert [len(p["panels"]) for p in snapshots] == [1, 1, 1] + operations = {actor['operation_id'] for panel in rows for actor in panel['actors']} + assert len(operations) == 1 and run['actors'][0]['operation_id'] in operations + assert run['actors'][0]['parsed']['verdict'] == 'FAIL' + assert rows[-1]['publication_revision'] == 3 + assert rows[0]['panel_id'] == binding['panel_id'] + # Neither another logical record at the same binding (host error, + # rebuttal/new wave) nor another task attempt is collapsed. + second = {**rows[0], 'panel_index': 1, 'publication_revision': 4} + next_attempt = {**rows[0], 'task_attempt': 2, 'publication_revision': 1} + merged = merge_review_projection(saved['review_projection'], {'panels': [second, next_attempt]}) + assert len(merged['panels']) == 3 + finally: + release.set() + assert settled.wait(5) diff --git a/tests/test_v674_acceptance_dialogue.py b/tests/test_v674_acceptance_dialogue.py index 235866926..310e2bd95 100644 --- a/tests/test_v674_acceptance_dialogue.py +++ b/tests/test_v674_acceptance_dialogue.py @@ -393,7 +393,7 @@ def test_capsule_leads_with_verdict_blocker_rails_and_three_moves(): result = _fail_result_with([_finding()]) capsule = build_improvement_capsule( result, - rails_line="money: $1.00 spent; time: 10 min left; review passes: 1 done", + rails_line="money: $1.00 spent; time: 10 min left; author passes: 1 done", open_obligations=[{"id": "ob-1"}, {"id": "ob-2"}], ) # The note the model READS says the assessment in words: a ledger token in @@ -725,7 +725,7 @@ def test_rails_final_pass_freeze_directive_workspace(): # rides EVERY workspace rails line (commit triad r1, sol: a deadline/cost # rail can end the loop between capsules), and never a non-workspace one. line = _rails(5, cap=6, workspace=True) - assert "review passes: 5/6" in line + assert "author passes: 5/6" in line assert "FINAL improvement pass, no further passes will run" in line assert "working tree as it stands" in line and "VERIFIED state" in line # Non-workspace: factual finality only, no tree directive. @@ -738,15 +738,15 @@ def test_rails_freeze_directive_absent_off_final_and_edge_caps(monkeypatch): # Non-final pass: no FINAL marker, but the workspace tree directive is # always present for workspace deliveries. mid = _rails(3, cap=6, workspace=True) - assert "FINAL" not in mid and "review passes: 3/6" in mid + assert "FINAL" not in mid and "author passes: 3/6" in mid assert "working tree as it stands" in mid assert "working tree" not in _rails(3, cap=6, workspace=False) # cap == 0 never feeds a capsule back — no misleading FINAL rail. zero = _rails(0, cap=0, workspace=True) - assert "review passes: 0/0" in zero and "FINAL" not in zero + assert "author passes: 0/0" in zero and "FINAL" not in zero # Passes already exhausted (supersede-reset re-review): not a launch. spent = _rails(6, cap=6, workspace=True) - assert "review passes: 6/6" in spent and "FINAL" not in spent + assert "author passes: 6/6" in spent and "FINAL" not in spent # No local cap (required+blocking with the shared review-cycle cap set to # unlimited — D10/D20: otherwise the shared cap binds) — no count-axis clause. monkeypatch.setenv("OUROBOROS_REVIEW_MAX_CYCLES", "unlimited") diff --git a/tests/test_workflow_blocked_handoff_commit.py b/tests/test_workflow_blocked_handoff_commit.py index 3acefc3ef..f947002ae 100644 --- a/tests/test_workflow_blocked_handoff_commit.py +++ b/tests/test_workflow_blocked_handoff_commit.py @@ -1,5 +1,6 @@ """Real two-task Git continuation: saved Blocking correction earns fresh authority.""" from types import SimpleNamespace +import json import subprocess import sys @@ -14,7 +15,49 @@ from ouroboros.tools import git from ouroboros.tools.registry import ToolContext from ouroboros.tools.scope_review import ScopeReviewResult from tests.test_mutation_attribution import _git, _repo -from tests.test_predecessor_mutation_handoff import _source + + +def _promote_retained_correction(root, data, monkeypatch): + from ouroboros.server_routing_context import _task_result_ground_truth + from ouroboros.tools import control_routing + from supervisor import queue, workers + from supervisor.events import _handle_promote_chat_to_task + + pending, running, pool = [], {}, {0: SimpleNamespace()} + for module in (queue, workers): + monkeypatch.setattr(module, "DRIVE_ROOT", data) + monkeypatch.setattr(module, "PENDING", pending) + monkeypatch.setattr(module, "RUNNING", running) + monkeypatch.setattr(workers, "WORKERS", pool) + monkeypatch.setattr(workers, "_WORKER_POOL_DISABLED_REASON", "") + monkeypatch.setattr(queue, "QUEUE_SNAPSHOT_PATH", data / "state/queue_snapshot.json") + for name in ("ADMISSION_RESERVATIONS", "ACCEPTANCE_FENCES", "BUDGET_ROOT_FENCES"): + monkeypatch.setattr(queue, name, {}) + monkeypatch.setattr(queue, "QUEUE_SEQ_COUNTER_REF", {"value": 0}) + supervisor = SimpleNamespace( + DRIVE_ROOT=data, WORKERS=pool, PENDING=pending, RUNNING=running, bridge=None, + enqueue_task=queue.enqueue_task, persist_queue_snapshot=queue.persist_queue_snapshot, + load_state=lambda: {"owner_chat_id": 1}, append_jsonl=lambda *_a, **_kw: None, + ) + + def deliver(_ctx, event): + assert event["predecessor_task_id"] == "first" + return "test_event_bus", _handle_promote_chat_to_task(event, supervisor) + + monkeypatch.setattr(control_routing, "_emit_and_wait_for_routing", deliver) + router = ToolContext(repo_dir=root, drive_root=data, task_id="decision", + current_chat_id=1, task_metadata={"main_routing_manifest": { + "final_results": [_task_result_ground_truth(load_task_result(data, "first"))]}}) + response = control_routing._promote_chat_to_task( + router, "Finish the retained correction", predecessor_task_id="first", workspace="none") + assert "durably scheduled" in response, response + assert len(pending) == 1 and "predecessor_task_id" not in pending[0] + snapshot = json.loads(queue.QUEUE_SNAPSHOT_PATH.read_text(encoding="utf-8")) + assert snapshot["pending"][0]["task"]["predecessor_authority_source"] == pending[0]["predecessor_authority_source"] + pending.clear() + assert queue.restore_pending_from_snapshot() == 1 + assert "predecessor_task_id" not in pending[0] + return pending[0] def test_second_task_reviews_and_commits_only_explicitly_selected_correction(tmp_path, monkeypatch): @@ -89,14 +132,15 @@ def test_second_task_reviews_and_commits_only_explicitly_selected_correction(tmp # Independent admission explicitly selects the retained predecessor. It does # not copy its paid wallet or critic authority, and does not reset first. - task = {"id": "second", "root_task_id": "second", "budget_drive_root": str(data), - "predecessor_task_id": "first", "predecessor_authority_source": _source("first")} - assert not validate_task_authority_sources(data, task) - write_task_result(data, "second", "running") + task = _promote_retained_correction(root, data, monkeypatch) + second_id = task["id"] + assert second_id != "first" and task["root_task_id"] == second_id agent = SimpleNamespace(env=SimpleNamespace(repo_dir=root, drive_root=data, budget_drive_root=str(data))) + assert not validate_task_authority_sources(agent.env, task) + write_task_result(data, second_id, "running") OuroborosAgent._capture_mutation_baseline(agent, task, {}) - second = context("second") - candidates = attributed_git_candidates(data, "second", root) + second = context(second_id) + candidates = attributed_git_candidates(data, second_id, root) assert candidates["candidates"] == ["clean.txt", "new.txt"] assert candidates["excluded_preexisting_dirty"] == ["dirty.txt"] completed = git._repo_commit_push(second, "Commit retained correction after fresh review", skip_advisory_review=True) @@ -106,11 +150,11 @@ def test_second_task_reviews_and_commits_only_explicitly_selected_correction(tmp assert _git(root, "show", "HEAD:dirty.txt") == "base" assert (root / "dirty.txt").read_text(encoding="utf-8") == "unrelated owner WIP\n" assert _git(root, "diff", "--name-only") == "dirty.txt" - assert [row["task"] for row in reviews] == ["first", "second"] + assert [row["task"] for row in reviews] == ["first", second_id] assert reviews[0]["fingerprint"] != reviews[1]["fingerprint"] attempts = load_state(data).attempts assert {task_id: sum(row.paid for row in attempts if row.root_task_id == task_id) - for task_id in ("first", "second")} == {"first": 1, "second": 1} + for task_id in ("first", second_id)} == {"first": 1, second_id: 1} assert attempts[-1].status == "succeeded" and not attempts[-1].author_disposition assert load_task_result(data, "first") == first_result assert len(publications) == 1 and all(row["exit"] == 0 for row in checks) diff --git a/tests/test_workflow_review_outcomes.py b/tests/test_workflow_review_outcomes.py index 2d017eb2e..afba45e75 100644 --- a/tests/test_workflow_review_outcomes.py +++ b/tests/test_workflow_review_outcomes.py @@ -51,6 +51,10 @@ def test_commit_finish_requires_received_outcome(candidate, monkeypatch, basis): if basis == "explicit_prior": reference = json.loads(prior.split('\n', 1)[1])['review_reference'] row = load_state(ctx.drive_root).attempts[-1] + from ouroboros.tools.claude_advisory_review import _handle_review_status + projected = json.loads(_handle_review_status(ctx)) + assert ("author_disposition" in projected["next_step"]) is (basis == "custody_lost") + assert ("review_reference" in projected) is (basis == "custody_lost") assert row.status == 'reviewing' and row.late_result_pending assert not row.critical_findings second = git._repo_commit_push(ctx, 'Fix amount', review_reference=reference, diff --git a/web/modules/log_events.js b/web/modules/log_events.js index f1b6007cf..966d36473 100644 --- a/web/modules/log_events.js +++ b/web/modules/log_events.js @@ -416,7 +416,7 @@ export function taskStoppedWithSummary(evt) { const TASK_CAUSE_PHRASES = { previous_revision_accepted: "The reviewers approved the earlier version of this answer; it changed before they finished.", author_stop: "Main stopped with unfinished work; no review approval was granted.", - review_outcome_received: "The review could not start; Main received the recorded reason.", + review_outcome_received: "Main received the review outcome or recorded limitation.", author_finish: "The answer was delivered on Main's own judgement; the reviewers had not signed it off.", review_degraded: "No reviewer verdict was established for this answer.", infra_failure: "A review infrastructure failure prevented a settled verdict.",