fix(e2e_live): SK1 verdict is the product review gate, clean state a fact

The rc.15 SK1-only rerun on 560f7d71 (three attempts, 2026-09-05) passed
every lifecycle check three times — preflight, review job, auto-grant equal
to the request, enable to a live-loaded dispatchable skill, a physical
extension call echoed into the owner chat, cleanup — while the stand's own
"all-PASS review" criterion passed once: the model-authored payload came
back clean, then with warnings (a SKILL.md/plugin.py contradiction), then
with blockers (manifest without runtime). The criterion measured the
author model, not the product.

Owner decision (2026-09-06, question 3 = A): SK1 counts when the review is
executable by the PRODUCT gate — the /api/extensions row's
executable_review (clean, warnings, or blockers under advisory enforcement
by operator choice, skill_review_gate) — plus the full lifecycle. The
check is now review_executable via sk1_review_gate(); the review status,
enforcement, blocking reason, non-PASS items and the clean/all-PASS state
travel as recorded facts. DEVELOPMENT.md says so; a parametrized test pins
the rule on the three rerun outcomes and the failure shapes (blocking
enforcement, failed review call, no findings, no gate fact).

Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com>
This commit is contained in:
Ouroboros 2026-09-06 00:42:09 +00:00
parent 560f7d71d2
commit d70f93c3c1
3 changed files with 47 additions and 7 deletions

View file

@ -770,6 +770,20 @@ def _skill_entry(base_url: str, name: str) -> dict:
return next((row for row in rows if isinstance(row, dict) and row.get("name") == name), {})
def sk1_review_gate(review: dict, entry: dict, findings: list) -> tuple[bool, dict]:
"""The SK1 review criterion is the PRODUCT gate (owner decision 2026-09-06): the review ran (HTTP 200 with
recorded findings) and the ``/api/extensions`` row says ``executable_review`` — clean, warnings, or blockers
under advisory enforcement by operator choice (``skill_review_gate``). A clean all-PASS review is a recorded
FACT, not the verdict: the rc.15 SK1 rerun on 560f7d71 authored one clean, one warnings and one blockers
payload with every other lifecycle check green, so all-PASS measured the author model, not the product."""
gate = entry.get("review_gate") if isinstance(entry.get("review_gate"), dict) else {}
failed = [f.get("item") for f in findings if str(f.get("verdict") or "") != "PASS"]
ok = review["status"] == 200 and entry.get("executable_review") is True and bool(findings)
return ok, {"review_status": review["body"].get("status"), "review_executable": entry.get("executable_review"),
"review_enforcement": gate.get("review_enforcement"), "review_blocking_reason": gate.get("blocking_reason"),
"findings": len(findings), "findings_failed": failed, "review_clean": bool(findings) and not failed}
def run_sk1(ctx: LaneContext) -> None:
from ouroboros.extension_surface_names import extension_surface_name
@ -785,11 +799,8 @@ def run_sk1(ctx: LaneContext) -> None:
review = _api_status(ctx.server.base_url, "POST", f"/api/skills/{SK1_SKILL}/review", {}, timeout=900)
review_state = ctx.oracle._json(f"state/skills/{SK1_SKILL}/review.json")
findings = [f for f in (review_state.get("findings") or []) if isinstance(f, dict)]
ctx.check("review_all_pass",
review["status"] == 200 and review["body"].get("status") == "clean" and bool(findings)
and all(str(f.get("verdict") or "") == "PASS" for f in findings),
review_status=review["body"].get("status"), findings=len(findings),
findings_failed=[f.get("item") for f in findings if str(f.get("verdict") or "") != "PASS"])
ok, review_facts = sk1_review_gate(review, _skill_entry(ctx.server.base_url, SK1_SKILL), findings)
ctx.check("review_executable", ok, **review_facts)
grants = _api_status(ctx.server.base_url, "POST", f"/api/skills/{SK1_SKILL}/grants", {"items": SK1_GRANTS}, timeout=120)
granted = ctx.oracle._json(f"state/skills/{SK1_SKILL}/grants.json").get("granted_permissions")
ctx.check("grants_exactly_requested", (grants["body"].get("grants") or {}).get("all_granted") is True

View file

@ -1274,7 +1274,10 @@ computed style read by a browser from the COMMITTED CSS after a restart), SW1
(the Swarm button arms `force_plan`; at least two children with causal lineage,
the `swarm_fanout` receipt, the with-children cost rollup, no orphan process by
the `/proc` environ scan), SK1 (the model authors `SKILL.md`+`plugin.py` and runs
`skill_preflight`; the runner reviews, grants exactly the manifest's one
`skill_preflight`; the runner reviews — the verdict is the product gate, the
`/api/extensions` row's `executable_review` (clean, warnings, or blockers under advisory
enforcement), with the review status, non-PASS items and the clean state recorded as
facts — grants exactly the manifest's one
privileged permission, enables, dispatches, deletes; the author and dispatch
tasks keep separate `author_*`/`dispatch_*` terminal checks, and the dispatch
counts only on a tools.jsonl row with typed status `ok` and the extension's exact

View file

@ -1,7 +1,8 @@
"""The live stand's SK1 probe plugin presents the host token only to the loopback Host Service
(docs/CHECKLISTS.md skill item 12, host_token_handling): a base URL naming any other host or
scheme is refused before a request is built. Pinned after the rc.15 paid stand, where the skill
review blocked the stand's own plugin for reading HOST_SERVICE_URL unvalidated."""
review blocked the stand's own plugin for reading HOST_SERVICE_URL unvalidated. The SK1 review criterion
is the product gate (owner decision 2026-09-06): executable review, with the clean/all-PASS state a fact."""
from __future__ import annotations
import types
@ -47,3 +48,28 @@ def test_the_probe_accepts_the_loopback_base_and_only_then_reaches_the_transport
with pytest.raises(Exception) as excinfo:
_echo_tool()(None, "x")
assert not isinstance(excinfo.value, RuntimeError) or "loopback" not in str(excinfo.value)
_FINDINGS = [{"item": "manifest_schema", "verdict": "PASS"}, {"item": "bug_hunting", "verdict": "FAIL"}]
@pytest.mark.parametrize("status,http,executable,findings,expected", [
("clean", 200, True, [{"item": "a", "verdict": "PASS"}], True),
("warnings", 200, True, _FINDINGS, True), # rc.15 SK1_a2: warnings are executable
("blockers", 200, True, _FINDINGS, True), # rc.15 SK1_a3: blockers executable under advisory enforcement
("blockers", 200, False, _FINDINGS, False), # the same review under blocking enforcement
("clean", 500, True, [{"item": "a", "verdict": "PASS"}], False), # the review call itself failed
("clean", 200, True, [], False), # no recorded findings: no review actually ran
("clean", 200, None, [{"item": "a", "verdict": "PASS"}], False), # the entry carries no gate fact
])
def test_sk1_review_verdict_is_the_product_gate_and_records_the_clean_state_as_a_fact(status, http, executable,
findings, expected):
review = {"status": http, "body": {"status": status}}
entry = {"executable_review": executable,
"review_gate": {"review_enforcement": "advisory", "blocking_reason": "x"}}
ok, facts = scenarios.sk1_review_gate(review, entry, findings)
assert ok is expected
failed = [f["item"] for f in findings if f["verdict"] != "PASS"]
assert facts == {"review_status": status, "review_executable": executable, "review_enforcement": "advisory",
"review_blocking_reason": "x", "findings": len(findings), "findings_failed": failed,
"review_clean": bool(findings) and not failed}