From 6a6db620cf65a4694649ae90458363dd72abda78 Mon Sep 17 00:00:00 2001 From: Ouroboros Date: Thu, 3 Sep 2026 23:21:08 +0300 Subject: [PATCH] acceptance packet: panels lead the prompt, partial rows never resolve, lifecycle from the canonical root; shared review projects retire once MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Authoritative review round of P3: the acceptance-panel block was appended after large review JSON and head-truncated away at 8 000 chars — it now leads the bounded prompt and truncated reasons carry `reason_omitted_chars` plus a response reference; a partial `tool_trajectory` section is typed `partial` (dispatchable, never resolving for API-only acceptance reviewers); acceptance reads a compact root-task→skill projection (`state/skill_review_root_tasks.jsonl`, appended by terminal skill reviews, enrolled as a hot store) instead of every skill's whole review history; skill-lifecycle identity keys are derived from the real skill tool schemas (incl. toggle/publish and the `user_repo` bucket); lifecycle history is read from the canonical root on split-root topologies. Custody (AP10): a shared review project was never retired when the lowest run id settled first and deferred — the last sibling to settle now attempts regardless of order and one project-level PROJECT_RETIRED row discharges every sharer on replay, so contributor receipts bind a consistent final settlement. delegate_custody.py stays at exactly 1 600 lines. The size-ratchet manifest is deliberately left untouched (one regeneration at synthesis). Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> --- ouroboros/agent_startup_checks.py | 7 ++ ouroboros/context_budget.py | 2 + ouroboros/delegate_custody.py | 26 ++--- ouroboros/review_evidence.py | 30 +++-- ouroboros/review_evidence_refs.py | 18 ++- ouroboros/skill_readiness.py | 54 ++++----- ouroboros/skill_review_history.py | 5 + ouroboros/skill_review_runner.py | 11 ++ ouroboros/tools/skill_publish.py | 64 ++++++----- tests/test_acceptance_fixround.py | 131 ++++++++++++++++++++++ tests/test_acceptance_skill_lifecycle.py | 46 +++++++- tests/test_contributor_review_evidence.py | 2 +- tests/test_delegate_registration_sweep.py | 41 +++++++ 13 files changed, 353 insertions(+), 84 deletions(-) diff --git a/ouroboros/agent_startup_checks.py b/ouroboros/agent_startup_checks.py index fe7ae191b..728b62c72 100644 --- a/ouroboros/agent_startup_checks.py +++ b/ouroboros/agent_startup_checks.py @@ -713,6 +713,7 @@ def _hot_store_thresholds() -> Tuple[Tuple[str, int, str], ...]: BG_OBSERVATIONS_WARN_BYTES, PROGRESS_LOG_WARN_BYTES, SCHEDULED_TASKS_WARN_BYTES, + SKILL_REVIEW_ROOT_TASKS_WARN_BYTES, TOOLS_LOG_WARN_BYTES, USAGE_LEDGER_WARN_BYTES, ) @@ -752,6 +753,12 @@ def _hot_store_thresholds() -> Tuple[Tuple[str, int, str], ...]: "under the queue lock, and consumed one-shot follow-ups are retained " "as durable receipts; pruning old consumed receipts is the remediation.", ), + ( + "state/skill_review_root_tasks.jsonl", + SKILL_REVIEW_ROOT_TASKS_WARN_BYTES, + "Acceptance packet assembly reads this compact skill-review index; " + "archive old root-task rows with their review histories.", + ), ) diff --git a/ouroboros/context_budget.py b/ouroboros/context_budget.py index a079b5c51..c2df7d58d 100644 --- a/ouroboros/context_budget.py +++ b/ouroboros/context_budget.py @@ -245,6 +245,8 @@ SCHEDULED_TASKS_WARN_BYTES = 2_000_000 # owner on each wake. This is a warning, not a retention gate: acknowledged # and unacknowledged rows remain durable until a future owner-approved archive. BG_OBSERVATIONS_WARN_BYTES = 20_000_000 +# Compact root-task -> skill review index used by acceptance packet assembly. +SKILL_REVIEW_ROOT_TASKS_WARN_BYTES = 20_000_000 # ``chat_history`` can deliberately replay the archive chain, while ordinary # context reads only the unconsolidated generation suffix. Warn before an # explicit full-history read becomes seconds-scale; this is observability, not diff --git a/ouroboros/delegate_custody.py b/ouroboros/delegate_custody.py index 9cc185066..87b449c50 100644 --- a/ouroboros/delegate_custody.py +++ b/ouroboros/delegate_custody.py @@ -397,10 +397,12 @@ def _apply(state: Dict[str, RunCustody], row: Dict[str, Any]) -> None: elif kind == SETTLED_UNREAD: custody.unread_disclosed = True elif kind == PROJECT_RETIRED: - # Recorded on SUCCESS too, not only on failure: without it, a retirement that - # landed before a failed ledger write would be replayed as still-owned after a + # Recorded on SUCCESS too: without it, a retirement before a failed ledger write replays as owned after a # restart, and the retry would keep failing on an already-removed project. - custody.project_owned = False + project_id = str(row.get("project_id") or custody.project_id or "") + for sibling in state.values(): + if sibling is custody or (project_id and sibling.project_id == project_id): + sibling.project_owned = False elif kind == OUTPUT_SPILLED: if row.get("staged") and str(row.get("artifact") or ""): custody.output_artifact = str(row.get("artifact") or "") @@ -778,12 +780,8 @@ def is_terminal(detail: Dict[str, Any]) -> bool: def retire_project(drive_root: Any, gateway: Any, custody: RunCustody) -> None: - """Discharge the registration obligation. Absence IS discharge (a 404 on - the project is the asked-for outcome, never a failure). REFCOUNT - DEFERRAL: the daemon refuses removal while runs live, so only the - LOWEST-run_id sharer keeps attempting; the rest defer quietly and - discharge on the daemon's 404 (deterministic tie-break: someone always - attempts).""" + """Discharge registration (a project 404 is the asked-for outcome). The last sibling to settle attempts; + one project-level row discharges every sharer.""" if custody.project_persistent: # #362: stable identity outlives the run — discharge the duty DURABLY # (replay must not resurrect owned=True), keep the project itself. @@ -791,7 +789,7 @@ def retire_project(drive_root: Any, gateway: Any, custody: RunCustody) -> None: emit(drive_root, PROJECT_RETIRED, {"run_id": custody.run_id, "task_id": custody.task_id, "project_id": custody.project_id, "project_kept": True}) return - if not (custody.project_owned and custody.project_id): + if not custody.project_id: return try: # Sharers = EVERY run in the project (only the creator carries @@ -811,6 +809,8 @@ def retire_project(drive_root: Any, gateway: Any, custody: RunCustody) -> None: return rows = [run for run in state.values() if run.project_id == custody.project_id and run.run_id] + if not any(run.project_owned for run in rows): + return if any(run.project_persistent for run in rows): # #362: ANY persistent sharer makes the project a durable user # identity — a non-persistent creator must not delete it either. @@ -820,13 +820,10 @@ def retire_project(drive_root: Any, gateway: Any, custody: RunCustody) -> None: return if any(not run.settled and run.run_id != custody.run_id for run in rows): return - sharers = sorted(run.run_id for run in rows if run.project_owned) except Exception: log.warning("Retirement deferred: replay failed for %s", custody.run_id, exc_info=True) return - if sharers and custody.run_id and custody.run_id != sharers[0]: - return # deferred quietly: the canonical sharer carries the lane try: gateway.remove_project(custody.project_id) except Exception as exc: @@ -839,6 +836,9 @@ def retire_project(drive_root: Any, gateway: Any, custody: RunCustody) -> None: "reason": str(exc)[:500]}) return custody.project_owned = False + for sibling in _CUSTODY.values(): + if sibling.project_id == custody.project_id: + sibling.project_owned = False emit(drive_root, PROJECT_RETIRED, {"run_id": custody.run_id, "task_id": custody.task_id, "project_id": custody.project_id}) diff --git a/ouroboros/review_evidence.py b/ouroboros/review_evidence.py index 004d2b295..55324f79a 100644 --- a/ouroboros/review_evidence.py +++ b/ouroboros/review_evidence.py @@ -1130,7 +1130,8 @@ def build_task_acceptance_evidence( # task touched — the same VISIBILITY-ONLY charter as substrate_execution. from ouroboros.skill_readiness import acceptance_skill_lifecycle - if lifecycle := acceptance_skill_lifecycle(drive_root, llm_trace or {}, root_task_id): + lifecycle_root = getattr(ctx, "budget_drive_root", None) or drive_root + if lifecycle := acceptance_skill_lifecycle(lifecycle_root, llm_trace or {}, root_task_id): ev["skill_lifecycle"] = redact_projection(lifecycle).value prov["skill_lifecycle"] = "host_attested" repo_diff = collect_turn_diff(ctx, include_recent_commit=include_recent_commit) @@ -1365,6 +1366,22 @@ _ACCEPTANCE_PANEL_ROW_KEYS = ( ) +def _acceptance_panel_prompt_row(panel: Dict[str, Any]) -> Dict[str, Any]: + row = {key: panel.get(key) for key in _ACCEPTANCE_PANEL_ROW_KEYS if key in panel} + reason = str(panel.get("reason") or "") + if reason: + limit = 300 + row["reason"] = truncate_review_artifact(reason, limit=limit) + row["reason_omitted_chars"] = max(0, len(reason) - limit) + refs = [ + actor.get("response_ref") for actor in (panel.get("actors") or []) + if isinstance(actor, dict) and actor.get("response_ref") + ] + if refs: + row["response_refs"] = refs + return row + + def format_review_evidence_for_prompt( evidence: Dict[str, Any], *, @@ -1379,23 +1396,22 @@ def format_review_evidence_for_prompt( can pass a positive *max_chars* to get an explicit omission note instead of silent clipping. - ``acceptance_panels`` appends the task's OWN acceptance-panel projection. + ``acceptance_panels`` leads with the task's OWN acceptance-panel projection. The commit/advisory lens knows nothing about it, so its absence statement names the lens it describes rather than claiming the task bought no review. """ - sections: List[str] = [] - if evidence and evidence.get("has_evidence"): - sections.append(json.dumps(evidence, ensure_ascii=False, indent=2)) rows = [ - {key: panel.get(key) for key in _ACCEPTANCE_PANEL_ROW_KEYS if key in panel} - | ({"reason": str(panel.get("reason") or "")[:300]} if panel.get("reason") else {}) + _acceptance_panel_prompt_row(panel) for panel in (acceptance_panels if isinstance(acceptance_panels, list) else []) if isinstance(panel, dict) ] + sections: List[str] = [] if rows: sections.append( "TASK ACCEPTANCE PANELS:\n" + json.dumps(rows, ensure_ascii=False, indent=2) ) + if evidence and evidence.get("has_evidence"): + sections.append(json.dumps(evidence, ensure_ascii=False, indent=2)) if not sections: return "(no commit/advisory review evidence recorded for this task)" full = "\n\n".join(sections) diff --git a/ouroboros/review_evidence_refs.py b/ouroboros/review_evidence_refs.py index e91705d19..4f43863fb 100644 --- a/ouroboros/review_evidence_refs.py +++ b/ouroboros/review_evidence_refs.py @@ -40,12 +40,17 @@ DECLARED_INTENT_SECTION = "declared_intent_section" # built outside the acceptance builder). Fail CLOSED: unknown attestation is not # attestation. UNATTESTED_SECTION = "unattested_section" +# A host-recorded section whose decision-bearing bytes were capped for the +# reviewer. It remains enumerable and dispatchable, but cannot resolve a clean +# acceptance criterion until the packet carries the complete section. +PARTIAL_SECTION = "partial" NON_RESOLVING_BASIS_KINDS = frozenset({ CLAIM_ID_UNSUPPORTED, RECEIPT_NOT_PASSING, AGENT_SUPPLIED_SECTION, DECLARED_INTENT_SECTION, UNATTESTED_SECTION, + PARTIAL_SECTION, }) # Section provenance tags (``__provenance__`` in the built packet) that make a @@ -76,7 +81,7 @@ def acceptance_evidence_ref_vocabulary(evidence: Any) -> Dict[str, str]: Maps each valid reviewer ``evidence_ref`` string to its CLOSED basis kind (claim_id | claim_id_unsupported | obligation_id | artifact | - verification_receipt | verification_receipt_not_passing | packet_section | + verification_receipt | verification_receipt_not_passing | packet_section | partial | agent_supplied_section | declared_intent_section | unattested_section — a closed table per ref kind, like ``IDENTITY_KINDS``). Pure derivation over the packet dict: no filesystem reads, no re-execution (a machine comparison must @@ -152,7 +157,12 @@ def acceptance_evidence_ref_vocabulary(evidence: Any) -> Dict[str, str]: if name.startswith("__"): continue tag = str(provenance.get(name) or "") - if name in DECLARED_INTENT_SECTIONS: + if name == "tool_trajectory" and any( + isinstance(row, dict) and row.get("result_complete") is False + for row in (ev.get(name) if isinstance(ev.get(name), list) else []) + ): + basis = PARTIAL_SECTION + elif name in DECLARED_INTENT_SECTIONS: basis = DECLARED_INTENT_SECTION elif tag in HOST_ATTESTED_SECTION_PROVENANCE: basis = "packet_section" @@ -173,8 +183,8 @@ def resolve_criteria_evidence_refs(criteria: Any, vocabulary: Dict[str, str]) -> every ref (rounds-6/7 rule: whatever decides is what is reported) plus ``supported_evidence_resolves`` — whether at least ONE ref resolved, which is all ``task_acceptance_is_clean`` consumes. A basis in - ``NON_RESOLVING_BASIS_KINDS`` (today: a claim id with no host-attested - supporting receipt) is DISCLOSED by name but does not resolve. Never touches + ``NON_RESOLVING_BASIS_KINDS`` (for example, an unsupported claim id or a + partial section) is DISCLOSED by name but does not resolve. Never touches parse validity, quorum, or verdicts (the v6.71.1 starvation class stays closed).""" rows: list = [] diff --git a/ouroboros/skill_readiness.py b/ouroboros/skill_readiness.py index 408ffc8be..5620f2fa4 100644 --- a/ouroboros/skill_readiness.py +++ b/ouroboros/skill_readiness.py @@ -25,8 +25,19 @@ class SkillReadiness: _SKILL_PAYLOAD_EDIT_TOOLS = frozenset({"write_file", "edit_text"}) -_SKILL_LIFECYCLE_TOOLS = frozenset({"skill_review", "skill_preflight", "skill_exec"}) -_SKILL_NAMING_TOOLS = _SKILL_PAYLOAD_EDIT_TOOLS | _SKILL_LIFECYCLE_TOOLS + + +def _skill_tool_identity_mapping() -> Dict[str, str]: + """Live skill-tool name -> identity argument, derived from their schemas.""" + from ouroboros.tools.skill_exec import _EXEC_SCHEMA, _REVIEW_SCHEMA, _TOGGLE_SCHEMA + from ouroboros.tools.skill_preflight import _PREFLIGHT_SCHEMA + from ouroboros.tools.skill_publish import _PUBLISH_SCHEMA + + schemas = (_REVIEW_SCHEMA, _PREFLIGHT_SCHEMA, _EXEC_SCHEMA, _TOGGLE_SCHEMA, _PUBLISH_SCHEMA) + return { + str(schema["name"]): str(schema["parameters"]["required"][0]) + for schema in schemas + } def skill_names_touched_by_trace(llm_trace: Dict[str, Any]) -> List[str]: @@ -37,21 +48,22 @@ def skill_names_touched_by_trace(llm_trace: Dict[str, Any]) -> List[str]: delegated payload the root never wrote with ``write_file``/``edit_text``. """ names: List[str] = [] + identity_keys = _skill_tool_identity_mapping() for call in llm_trace.get("tool_calls") or []: if not isinstance(call, dict): continue tool = str(call.get("tool") or "") - if tool not in _SKILL_NAMING_TOOLS: + if tool not in _SKILL_PAYLOAD_EDIT_TOOLS and tool not in identity_keys: continue args = call.get("args") if isinstance(call.get("args"), dict) else {} - if tool in _SKILL_LIFECYCLE_TOOLS: - named = str(args.get("skill_name") or args.get("name") or "").strip() + if tool in identity_keys: + named = str(args.get(identity_keys[tool]) or "").strip() if named and named not in names: names.append(named) continue bucket = str(args.get("bucket") or "").strip().lower() skill_name = str(args.get("skill_name") or "").strip() - if bucket in {"external", "clawhub", "ouroboroshub"} and skill_name: + if bucket in {"external", "clawhub", "ouroboroshub", "user_repo"} and skill_name: if skill_name not in names: names.append(skill_name) continue @@ -112,34 +124,24 @@ def acceptance_skill_lifecycle( def _skill_names_from_review_history(drive_root: pathlib.Path, root_task_id: str) -> List[str]: - """Skills whose review history records this root task.""" + """Skills named by the compact root-task projection, never full histories.""" if not root_task_id: return [] - import json + from ouroboros.skill_review_history import root_task_projection_path + from ouroboros.utils import iter_jsonl_objects + path = root_task_projection_path(drive_root) names: List[str] = [] - base = drive_root / "state" / "skills" try: - candidates = sorted(entry for entry in base.iterdir() if entry.is_dir()) + rows = iter_jsonl_objects(path) except OSError: return [] - for entry in candidates: - path = entry / "review_history.jsonl" - try: - raw = path.read_text(encoding="utf-8") - except (OSError, UnicodeError): + for row in rows: + if not isinstance(row, dict) or str(row.get("root_task_id") or "") != root_task_id: continue - for line in raw.splitlines(): - line = line.strip() - if not line.startswith("{") or root_task_id not in line: - continue - try: - row = json.loads(line) - except ValueError: - continue - if isinstance(row, dict) and str(row.get("root_task_id") or "") == root_task_id: - names.append(entry.name) - break + name = str(row.get("skill") or "").strip() + if name and name not in names: + names.append(name) return names diff --git a/ouroboros/skill_review_history.py b/ouroboros/skill_review_history.py index 2266908c4..6427b45c6 100644 --- a/ouroboros/skill_review_history.py +++ b/ouroboros/skill_review_history.py @@ -20,6 +20,7 @@ _MARKER_FACT_KEYS = ( "review_contract_fingerprint", "rebuttal_sha256", "usage_attribution_schema", "group_id", "content_hash", "root_task_id", ) +ROOT_TASK_PROJECTION_RELATIVE_PATH = "state/skill_review_root_tasks.jsonl" def _redact_history_payload(payload: Dict[str, Any]) -> Dict[str, Any]: @@ -33,6 +34,10 @@ def review_history_path(drive_root: pathlib.Path, skill_name: str) -> pathlib.Pa return drive_root / "state" / "skills" / skill_name / "review_history.jsonl" +def root_task_projection_path(drive_root: pathlib.Path) -> pathlib.Path: + return drive_root / ROOT_TASK_PROJECTION_RELATIVE_PATH + + def legacy_dispatch_marker_path(drive_root: pathlib.Path, skill_name: str) -> pathlib.Path: """The retired SINGLE-file marker (pre per-wave storage). Tolerated read-only: ``write_dispatch_marker`` flushes it into the history and removes it.""" diff --git a/ouroboros/skill_review_runner.py b/ouroboros/skill_review_runner.py index a99164370..c81842a22 100644 --- a/ouroboros/skill_review_runner.py +++ b/ouroboros/skill_review_runner.py @@ -322,6 +322,17 @@ def _append_terminal_history( ts=ts, ), ) + root_task_id = str(job_data.get("root_task_id") or "") + if appended and root_task_id: + projected = append_jsonl( + skill_review_history.root_task_projection_path(drive_root), + { + "ts": ts, "root_task_id": root_task_id, "skill": skill_name, + "job_id": str(job_data.get("job_id") or ""), + }, + ) + if not projected: + log.warning("skill review root-task projection did not land for %s", skill_name) if not appended: # LOUD failure (F3): a silently lost terminal row un-counts spent # review money and hides the verdict from the derived ledger. The diff --git a/ouroboros/tools/skill_publish.py b/ouroboros/tools/skill_publish.py index 936e2cbd7..f1937a5d0 100644 --- a/ouroboros/tools/skill_publish.py +++ b/ouroboros/tools/skill_publish.py @@ -980,39 +980,43 @@ def _submit_skill_to_hub( ) +_PUBLISH_SCHEMA = { + "name": "submit_skill_to_hub", + "description": ( + "Publish one immutable reviewed skill snapshot to OuroborosHub by " + "opening a GitHub pull request. A failed result is repair evidence " + "for the next agent turn; success contains a validated PR receipt." + ), + "parameters": { + "type": "object", + "properties": { + "skill": { + "type": "string", + "description": "Installed skill name (slug).", + }, + "note": { + "type": "string", + "default": "", + "description": "Optional public author note for the pull request.", + }, + "confirm_public_submission": { + "type": "boolean", + "description": ( + "Must be true: confirms the human approved this public " + "OuroborosHub submission." + ), + }, + }, + "required": ["skill", "confirm_public_submission"], + }, +} + + def get_tools() -> List[ToolEntry]: return [ ToolEntry( - name="submit_skill_to_hub", - schema={ - "name": "submit_skill_to_hub", - "description": ( - "Publish one immutable reviewed skill snapshot to OuroborosHub by " - "opening a GitHub pull request. A failed result is repair evidence " - "for the next agent turn; success contains a validated PR receipt." - ), - "parameters": { - "type": "object", - "properties": { - "skill": { - "type": "string", - "description": "Installed skill name (slug).", - }, - "note": { - "type": "string", - "default": "", - "description": "Optional public author note for the pull request.", - }, - "confirm_public_submission": { - "type": "boolean", - "description": ( - "Must be true: confirms the human approved this public OuroborosHub submission." - ), - }, - }, - "required": ["skill", "confirm_public_submission"], - }, - }, + name=_PUBLISH_SCHEMA["name"], + schema=_PUBLISH_SCHEMA, handler=_submit_skill_to_hub, is_code_tool=False, timeout_sec=180, diff --git a/tests/test_acceptance_fixround.py b/tests/test_acceptance_fixround.py index c5e1ce610..1a0b24433 100644 --- a/tests/test_acceptance_fixround.py +++ b/tests/test_acceptance_fixround.py @@ -5,6 +5,137 @@ from pathlib import Path from types import SimpleNamespace +def test_prompt_projection_keeps_panels_ahead_of_an_oversized_lens(): + from ouroboros.review_evidence import format_review_evidence_for_prompt + + rendered = format_review_evidence_for_prompt( + {"has_evidence": True, "oversized_lens": "L" * 30_000}, + max_chars=4_000, + acceptance_panels=[{ + "panel_id": "panel-must-survive", + "surface": "task_acceptance", + "aggregate_signal": "PASS", + "transport_status": "success", + "parse_status": "valid", + "reason": "deciding finding " + "R" * 1_000, + "actors": [{ + "slot_id": "slot_1", + "response_ref": {"call_id": "response-must-survive"}, + }], + }], + ) + + assert rendered.startswith("TASK ACCEPTANCE PANELS:") + assert "panel-must-survive" in rendered + assert '"aggregate_signal": "PASS"' in rendered + assert "deciding finding" in rendered + assert '"reason_omitted_chars"' in rendered + assert "response-must-survive" in rendered + assert "OMISSION NOTE" in rendered + + +def test_partial_trajectory_is_non_resolving_but_complete_trajectory_resolves(): + from ouroboros.review_dispatch import task_acceptance_zero_physical_refusal + from ouroboros.review_evidence import annotate_criteria_evidence_resolution + from ouroboros.review_evidence_refs import acceptance_evidence_ref_vocabulary + from ouroboros.review_substrate import task_acceptance_is_clean + + def _actor(): + return { + "signal": "PASS", + "parsed": { + "outcome_tier": "solved", + "criteria_used": [{ + "criterion": "tool outcome", + "status": "supported", + "evidence_refs": ["tool_trajectory"], + }], + }, + } + + def _result(actor): + return SimpleNamespace( + aggregate_signal="PASS", degraded=False, actors=[actor], + ) + + partial = { + "tool_trajectory": [{"tool": "read_file", "result_complete": False}], + "__provenance__": {"tool_trajectory": "tool_result"}, + } + partial_actor = _actor() + annotate_criteria_evidence_resolution([partial_actor], partial) + assert task_acceptance_zero_physical_refusal(partial) == {} + assert acceptance_evidence_ref_vocabulary(partial)["tool_trajectory"] == "partial" + assert task_acceptance_is_clean(_result(partial_actor)) is False + + complete = { + "tool_trajectory": [{"tool": "read_file", "result_complete": True}], + "__provenance__": {"tool_trajectory": "tool_result"}, + } + complete_actor = _actor() + annotate_criteria_evidence_resolution([complete_actor], complete) + assert acceptance_evidence_ref_vocabulary(complete)["tool_trajectory"] == "packet_section" + assert task_acceptance_is_clean(_result(complete_actor)) is True + + +def test_skill_history_root_task_projection_avoids_whole_history_reads(monkeypatch, tmp_path): + from ouroboros.skill_readiness import _skill_names_from_review_history + + skill_dir = tmp_path / "state" / "skills" / "large-skill" + skill_dir.mkdir(parents=True) + history = skill_dir / "review_history.jsonl" + history.write_text( + (json.dumps({"root_task_id": "some-other-root", "padding": "x" * 200}) + "\n") + * 20_000, + encoding="utf-8", + ) + (tmp_path / "state" / "skill_review_root_tasks.jsonl").write_text( + json.dumps({"root_task_id": "root-wanted", "skill": "large-skill"}) + "\n", + encoding="utf-8", + ) + original = Path.read_text + + def _guarded_read(path, *args, **kwargs): + if path.name == "review_history.jsonl": + raise AssertionError("acceptance rebuilt the full skill review history") + return original(path, *args, **kwargs) + + monkeypatch.setattr(Path, "read_text", _guarded_read) + assert _skill_names_from_review_history(tmp_path, "root-wanted") == ["large-skill"] + + +def test_terminal_skill_review_updates_the_root_task_projection(tmp_path): + from ouroboros.skill_review_runner import _append_terminal_history + + assert _append_terminal_history( + tmp_path, + "projected-skill", + {"job_id": "job-1", "root_task_id": "root-1"}, + status="pass", + terminal_reason="review_complete", + ts="2026-09-03T00:00:00+00:00", + ) + rows = [ + json.loads(line) + for line in (tmp_path / "state" / "skill_review_root_tasks.jsonl") + .read_text(encoding="utf-8").splitlines() + ] + assert rows == [{ + "ts": "2026-09-03T00:00:00+00:00", + "root_task_id": "root-1", + "skill": "projected-skill", + "job_id": "job-1", + }] + + +def test_skill_review_projection_is_enrolled_as_a_hot_store(): + from ouroboros.agent_startup_checks import _hot_store_thresholds + + assert "state/skill_review_root_tasks.jsonl" in { + relative for relative, _threshold, _remediation in _hot_store_thresholds() + } + + def test_acceptance_slot_fit_uses_packet_density(): from ouroboros.review_dispatch import acceptance_slot_fit diff --git a/tests/test_acceptance_skill_lifecycle.py b/tests/test_acceptance_skill_lifecycle.py index 19d8d43f9..ef99d76ce 100644 --- a/tests/test_acceptance_skill_lifecycle.py +++ b/tests/test_acceptance_skill_lifecycle.py @@ -41,6 +41,9 @@ def _write_history(drive_root: pathlib.Path, name: str, root_task_id: str) -> No "root_task_id": root_task_id}) + "\n", encoding="utf-8", ) + projection = drive_root / "state" / "skill_review_root_tasks.jsonl" + with projection.open("a", encoding="utf-8") as handle: + handle.write(json.dumps({"root_task_id": root_task_id, "skill": name}) + "\n") def _ctx(drive_root: pathlib.Path, tmp_path: pathlib.Path, task_id: str) -> ToolContext: @@ -95,11 +98,48 @@ def test_a_skill_lifecycle_tool_call_names_the_skill_without_a_payload_edit(tmp_ """A free delegation lane integrates a patch and never calls write_file, so the lifecycle tools are the only carrier of the name in that shape.""" trace = {"tool_calls": [ - {"tool": "skill_review", "args": {"skill_name": "delegated"}}, - {"tool": "skill_preflight", "args": {"name": "probed"}}, + {"tool": "skill_review", "args": {"skill": "delegated"}}, + {"tool": "skill_preflight", "args": {"skill": "probed"}}, + {"tool": "skill_exec", "args": {"skill": "executed", "script": "scripts/run.py"}}, + {"tool": "toggle_skill", "args": {"skill": "toggled", "enabled": True}}, + {"tool": "submit_skill_to_hub", "args": { + "skill": "published", "confirm_public_submission": True, + }}, + {"tool": "edit_text", "args": { + "root": "skill_payload", "bucket": "user_repo", + "skill_name": "user-repo-skill", "path": "SKILL.md", + "old_text": "old", "new_text": "new", + }}, {"tool": "run_command", "args": {"cmd": "ls"}}, ]} - assert skill_names_touched_by_trace(trace) == ["delegated", "probed"] + assert skill_names_touched_by_trace(trace) == [ + "delegated", "probed", "executed", "toggled", "published", + "user-repo-skill", + ] + + +def test_split_root_packet_reads_skill_lifecycle_from_the_canonical_root(tmp_path): + canonical = tmp_path / "canonical" + execution = tmp_path / "execution" + canonical.mkdir() + execution.mkdir() + _write_skill(canonical, "canonical-skill") + ctx = ToolContext( + repo_dir=tmp_path, drive_root=execution, budget_drive_root=canonical, + task_id="task-split-skill", + ) + + packet = build_task_acceptance_evidence( + ctx, + llm_trace={"tool_calls": [{ + "tool": "skill_review", "args": {"skill": "canonical-skill"}, + }]}, + drive_root=execution, + task_id="task-split-skill", + ) + + assert packet["skill_lifecycle"][0]["name"] == "canonical-skill" + assert packet["skill_lifecycle"][0].get("present", True) is True def test_the_packet_carries_the_section_and_the_vocabulary_resolves_it(tmp_path): diff --git a/tests/test_contributor_review_evidence.py b/tests/test_contributor_review_evidence.py index 5ab321b49..2cb2e60fe 100644 --- a/tests/test_contributor_review_evidence.py +++ b/tests/test_contributor_review_evidence.py @@ -83,7 +83,7 @@ def test_shared_project_receipts_bind_final_retirement_after_all_slots_settle(tm })) custody.record_started(tmp_path, custody.RunCustody( run_id=run_id, task_id="review", project_id="shared-project", - project_owned=index == 1, ledger_root=str(tmp_path), + project_owned=True, ledger_root=str(tmp_path), )) for index in (1, 2): diff --git a/tests/test_delegate_registration_sweep.py b/tests/test_delegate_registration_sweep.py index f85a91fb9..c595241d4 100644 --- a/tests/test_delegate_registration_sweep.py +++ b/tests/test_delegate_registration_sweep.py @@ -3,7 +3,48 @@ Split from test_delegated_run_isolation.py (module line cap).""" from __future__ import annotations +import itertools + from ouroboros import delegate_custody as custody + + +def test_last_shared_project_sibling_retires_once_in_every_settlement_order(tmp_path): + class _Gateway: + def __init__(self): + self.removals = [] + + def remove_project(self, project_id): + self.removals.append(project_id) + + run_ids = ("run-a", "run-b", "run-c") + for case, order in enumerate(itertools.permutations(run_ids)): + root = tmp_path / str(case) + gateway = _Gateway() + custody._CUSTODY.clear() + for index, run_id in enumerate(run_ids): + custody.record_started(root, custody.RunCustody( + run_id=run_id, + task_id=f"task-{run_id}", + project_id="shared-project", + project_owned=index == 0, + ledger_root=str(root), + )) + custody.emit(root, custody.LEDGER_RECORDED, {"run_id": run_id}) + + for run_id in order: + row = custody.replay(root)[run_id] + custody.settle_run(root, gateway, row, {"summary": {"state": "succeeded"}}) + + assert gateway.removals == ["shared-project"], order + replayed = custody.replay(root) + assert all(not row.project_owned for row in replayed.values()), order + retired = [ + row for row in custody._iter_rows(custody.event_log_path(root)) + if row.get("type") == custody.PROJECT_RETIRED + ] + assert len(retired) == 1, order + custody._CUSTODY.clear() + def test_registration_sweep_defers_behind_a_live_unowned_sharer(tmp_path): """Sharers are ALL runs in a project, owned or not: only the creator carries the registration, but the daemon refuses removal while any