diff --git a/devtools/e2e_live/scenarios.py b/devtools/e2e_live/scenarios.py index 20eca4788..8d92000a3 100644 --- a/devtools/e2e_live/scenarios.py +++ b/devtools/e2e_live/scenarios.py @@ -770,6 +770,20 @@ def _skill_entry(base_url: str, name: str) -> dict: return next((row for row in rows if isinstance(row, dict) and row.get("name") == name), {}) +def sk1_review_gate(review: dict, entry: dict, findings: list) -> tuple[bool, dict]: + """The SK1 review criterion is the PRODUCT gate (owner decision 2026-09-06): the review ran (HTTP 200 with + recorded findings) and the ``/api/extensions`` row says ``executable_review`` — clean, warnings, or blockers + under advisory enforcement by operator choice (``skill_review_gate``). A clean all-PASS review is a recorded + FACT, not the verdict: the rc.15 SK1 rerun on 560f7d71 authored one clean, one warnings and one blockers + payload with every other lifecycle check green, so all-PASS measured the author model, not the product.""" + gate = entry.get("review_gate") if isinstance(entry.get("review_gate"), dict) else {} + failed = [f.get("item") for f in findings if str(f.get("verdict") or "") != "PASS"] + ok = review["status"] == 200 and entry.get("executable_review") is True and bool(findings) + return ok, {"review_status": review["body"].get("status"), "review_executable": entry.get("executable_review"), + "review_enforcement": gate.get("review_enforcement"), "review_blocking_reason": gate.get("blocking_reason"), + "findings": len(findings), "findings_failed": failed, "review_clean": bool(findings) and not failed} + + def run_sk1(ctx: LaneContext) -> None: from ouroboros.extension_surface_names import extension_surface_name @@ -785,11 +799,8 @@ def run_sk1(ctx: LaneContext) -> None: review = _api_status(ctx.server.base_url, "POST", f"/api/skills/{SK1_SKILL}/review", {}, timeout=900) review_state = ctx.oracle._json(f"state/skills/{SK1_SKILL}/review.json") findings = [f for f in (review_state.get("findings") or []) if isinstance(f, dict)] - ctx.check("review_all_pass", - review["status"] == 200 and review["body"].get("status") == "clean" and bool(findings) - and all(str(f.get("verdict") or "") == "PASS" for f in findings), - review_status=review["body"].get("status"), findings=len(findings), - findings_failed=[f.get("item") for f in findings if str(f.get("verdict") or "") != "PASS"]) + ok, review_facts = sk1_review_gate(review, _skill_entry(ctx.server.base_url, SK1_SKILL), findings) + ctx.check("review_executable", ok, **review_facts) grants = _api_status(ctx.server.base_url, "POST", f"/api/skills/{SK1_SKILL}/grants", {"items": SK1_GRANTS}, timeout=120) granted = ctx.oracle._json(f"state/skills/{SK1_SKILL}/grants.json").get("granted_permissions") ctx.check("grants_exactly_requested", (grants["body"].get("grants") or {}).get("all_granted") is True diff --git a/docs/DEVELOPMENT.md b/docs/DEVELOPMENT.md index 6812aba4f..55417ad0c 100644 --- a/docs/DEVELOPMENT.md +++ b/docs/DEVELOPMENT.md @@ -1274,7 +1274,10 @@ computed style read by a browser from the COMMITTED CSS after a restart), SW1 (the Swarm button arms `force_plan`; at least two children with causal lineage, the `swarm_fanout` receipt, the with-children cost rollup, no orphan process by the `/proc` environ scan), SK1 (the model authors `SKILL.md`+`plugin.py` and runs -`skill_preflight`; the runner reviews, grants exactly the manifest's one +`skill_preflight`; the runner reviews — the verdict is the product gate, the +`/api/extensions` row's `executable_review` (clean, warnings, or blockers under advisory +enforcement), with the review status, non-PASS items and the clean state recorded as +facts — grants exactly the manifest's one privileged permission, enables, dispatches, deletes; the author and dispatch tasks keep separate `author_*`/`dispatch_*` terminal checks, and the dispatch counts only on a tools.jsonl row with typed status `ok` and the extension's exact diff --git a/tests/test_e2e_live_sk1_plugin.py b/tests/test_e2e_live_sk1_plugin.py index 55da78a0f..c5c2d0a26 100644 --- a/tests/test_e2e_live_sk1_plugin.py +++ b/tests/test_e2e_live_sk1_plugin.py @@ -1,7 +1,8 @@ """The live stand's SK1 probe plugin presents the host token only to the loopback Host Service (docs/CHECKLISTS.md skill item 12, host_token_handling): a base URL naming any other host or scheme is refused before a request is built. Pinned after the rc.15 paid stand, where the skill -review blocked the stand's own plugin for reading HOST_SERVICE_URL unvalidated.""" +review blocked the stand's own plugin for reading HOST_SERVICE_URL unvalidated. The SK1 review criterion +is the product gate (owner decision 2026-09-06): executable review, with the clean/all-PASS state a fact.""" from __future__ import annotations import types @@ -47,3 +48,28 @@ def test_the_probe_accepts_the_loopback_base_and_only_then_reaches_the_transport with pytest.raises(Exception) as excinfo: _echo_tool()(None, "x") assert not isinstance(excinfo.value, RuntimeError) or "loopback" not in str(excinfo.value) + + +_FINDINGS = [{"item": "manifest_schema", "verdict": "PASS"}, {"item": "bug_hunting", "verdict": "FAIL"}] + + +@pytest.mark.parametrize("status,http,executable,findings,expected", [ + ("clean", 200, True, [{"item": "a", "verdict": "PASS"}], True), + ("warnings", 200, True, _FINDINGS, True), # rc.15 SK1_a2: warnings are executable + ("blockers", 200, True, _FINDINGS, True), # rc.15 SK1_a3: blockers executable under advisory enforcement + ("blockers", 200, False, _FINDINGS, False), # the same review under blocking enforcement + ("clean", 500, True, [{"item": "a", "verdict": "PASS"}], False), # the review call itself failed + ("clean", 200, True, [], False), # no recorded findings: no review actually ran + ("clean", 200, None, [{"item": "a", "verdict": "PASS"}], False), # the entry carries no gate fact +]) +def test_sk1_review_verdict_is_the_product_gate_and_records_the_clean_state_as_a_fact(status, http, executable, + findings, expected): + review = {"status": http, "body": {"status": status}} + entry = {"executable_review": executable, + "review_gate": {"review_enforcement": "advisory", "blocking_reason": "x"}} + ok, facts = scenarios.sk1_review_gate(review, entry, findings) + assert ok is expected + failed = [f["item"] for f in findings if f["verdict"] != "PASS"] + assert facts == {"review_status": status, "review_executable": executable, "review_enforcement": "advisory", + "review_blocking_reason": "x", "findings": len(findings), "findings_failed": failed, + "review_clean": bool(findings) and not failed}