mirror of
https://github.com/razzant/ouroboros.git
synced 2026-10-02 19:58:46 +00:00
Module side. Eleven owner leaves cut from tip bytes, every span transplant-tool proof-green against git show HEAD:<monolith> (ast=tokens=byte-roundtrip on every symbol, leaf_invariants=[], exit 0): - review_state.py ONE CUT (owner decision 5.3=B, GIANT 2172->777): review_state_records (44 rows, _rs handle), review_state_model (AdvisoryReviewState, _rs), and the NEW review_state_custody leaf - nine unrowed post-cutoff symbols of the adaptive-timeout/custody train (checkpoint_pending_review_invocation family), recorded in LEDGER_CORRECTIONS as unrowed F5 adoption rows. The authority-shape deserializers stay with the parent STORE. - review_substrate.py (1600->815): review_records (projection-only, off the LEAVES table), review_verdict and review_projection on the _sub handle. Upstream re-homes honored, not dragged back: reviewer_slots stays in reviewer_slot_config, _render_prompt in review_execution, slot_id_for_row in review_dispatch - all pinned by the re-derived extraction suite. - tools/review_helpers.py (1575->764): review_prompt_text (27 rows) + review_file_pack (25 rows) on the _rh handle. - tools/scope_review.py (1597->963): scope_review_pack (19/20 rows, _sr); _load_canonical_context_docs stays a facade def (f-string read of load_governance_doc, which tests rebind on the parent - D10 precedent). The budget leaf is the F2.3b re-derive (#383), untouched here. - review_evidence.py (1559->886): review_evidence_sections (25 rows, _ev); the two capability-delta rows are superseded by upstream's delegate_evidence home and not replayed. - tools/review.py (1550->1269): review_multi_model (7 rows, _rev); _parse_model_response superseded (tools/review_response is the home), the two review-model timeout rows retired with the adaptive-timeout contract. Declared sets are the tool-derived exact read sets (maximal-declared policy); the few f-string/import-time reads the gate refuses stay import-bound to their canonical owners and are named in each leaf docstring (none of those names is monkeypatched on a parent anywhere in tests/). Facades = tip parent - moved spans + EOF re-export block + noqa discipline; dead stdlib imports trimmed. Drift-probe first per leaf: 182 rowed oracle spans probed against tip bytes (155 byte-true, 27 drifted); all bodies emitted from tip bytes, no oracle semantics replayed over drift. Path-keyed mirrors: review_context_atlas._REVIEW_STACK_PATHS and run_external_review._REVIEW_SUBSTRATE_PATHS extended additively with the new leaves beside their parents (D10 closure precedent); domains.toml gains the eleven D06 leaf rows and clears the resolved split_pending entries; the domain quotient report regenerated (no manifest drift). Test side. The two D06 test giants re-cut as the reference theme split from tip bytes, lossless: - test_review_agent_session_route.py (3399, GIANT) -> shared fixtures (_review_session_route_shared) + delivery/poller/scope_wiring siblings + a 1218-line remainder (102 == 102 test names; thirteen post-cutoff tests placed with the sibling that owns their helpers; three reference-only tests not replayed, recorded). - test_review_substrate_v2.py (2986, GIANT) -> shared FakeLLM + extraction/acceptance/actor_truth/prompts siblings + a NEW test_review_substrate_custody.py sibling holding the eighteen post-cutoff custody-train tests (71 == 71 test names). The falsified ledger row (_render_prompt -> substrate) is re-derived: the prompts suite imports it from review_execution; LEDGER_CORRECTIONS carries the correction. Both giants leave GIANT_PATHS; the two band re-entries carry rationales. Dedup (owner decision 5.2=A), disclosed as a test deletion: ten AST-identical tests + seven byte-identical orphan helpers removed from test_review_cycles_dispatch.py; the owner is test_review_cycles_skill_dispatch.py (D14 family). -510 double-executed lines; the dispatch file stays the commit-gate paid-accounting suite and leaves the ratchet band. Dead-patch class re-pointed per the oracle adaptations: six sites patching leaf-internal names on the facade (test_scope_review TestRunScopeReviewFailClosed x4, test_review_convergence_rule fixture) retargeted to scope_review_pack; every other historical monkeypatch target stays live on the parents through the call-time handles. New identity suites: five re-derived extraction contracts + re-derived test_review_owner_facades (superseded/retired rows dropped with reasons); ten LEAVES rows added to test_module_handle_extraction. Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> (cherry picked from commit 05ec5fd1173c9895c7b0ac11124979eeeb44d9ef)
266 lines
11 KiB
Python
266 lines
11 KiB
Python
"""What the review actor records claim, and to whom.
|
|
|
|
Split by theme out of ``tests/test_review_substrate_v2.py``. This module owns
|
|
the actor truth: transport, parse and semantics reported separately, the
|
|
bounded compact projection with redaction before truncation, mixed-panel
|
|
participation counts and the stable review binding.
|
|
"""
|
|
|
|
import json
|
|
|
|
from ouroboros.review_substrate import ReviewRequest, ReviewSlot, run_review_request
|
|
|
|
|
|
class _MixedReviewTruthLLM:
|
|
def chat(self, **kwargs):
|
|
model = str(kwargs.get("model") or "")
|
|
if "timeout" in model:
|
|
# Some timeout types stringify to an empty message. The transport
|
|
# truth must come from the exception type, not incidental wording.
|
|
raise TimeoutError()
|
|
if "malformed" in model:
|
|
return {"content": "not json"}, {}
|
|
return {
|
|
"content": json.dumps({
|
|
"verdict": "DEGRADED",
|
|
"outcome_tier": "best_effort",
|
|
"summary": "Evidence coverage is incomplete.",
|
|
"findings": [],
|
|
}),
|
|
}, {
|
|
"provider": "openrouter",
|
|
"resolved_model": "google/gemini-3.5-flash",
|
|
}
|
|
|
|
|
|
def test_review_actor_truth_separates_transport_parse_and_semantics(tmp_path):
|
|
from ouroboros.review_substrate import compact_review_projection
|
|
|
|
result = run_review_request(
|
|
ReviewRequest(
|
|
surface="task_acceptance", goal="g", subject="candidate",
|
|
policy={"min_successful_slots": 2}, task_id="truth",
|
|
),
|
|
slots=[
|
|
ReviewSlot("timeout", "anthropic/timeout-model", role_hint="acceptance reviewer"),
|
|
ReviewSlot("malformed", "openai/malformed-model", role_hint="acceptance reviewer"),
|
|
ReviewSlot("degraded", "google/degraded-model", role_hint="acceptance reviewer"),
|
|
],
|
|
drive_root=tmp_path,
|
|
llm=_MixedReviewTruthLLM(),
|
|
)
|
|
actors = {actor["slot_id"]: actor for actor in result.actors}
|
|
assert result.panel_id.startswith("panel_")
|
|
assert len(result.panel_id) == len("panel_") + 16
|
|
assert actors["timeout"]["transport_status"] == "timeout"
|
|
assert actors["timeout"]["parse_status"] == "malformed"
|
|
assert actors["timeout"]["semantic_verdict"] == ""
|
|
assert actors["malformed"]["transport_status"] == "success"
|
|
assert actors["malformed"]["parse_status"] == "malformed"
|
|
assert actors["degraded"]["parse_status"] == "valid"
|
|
assert actors["degraded"]["semantic_verdict"] == "DEGRADED"
|
|
assert actors["degraded"]["parsed"]["outcome_tier"] == "best_effort"
|
|
assert actors["degraded"]["provider"] == "openrouter"
|
|
assert actors["degraded"]["model"] == "google/gemini-3.5-flash"
|
|
assert actors["degraded"]["actor_role"] == "acceptance reviewer"
|
|
|
|
run = dict(result.__dict__)
|
|
run.update({
|
|
"authority": "host_root",
|
|
"candidate_hash": "c" * 64,
|
|
"evidence_revision": "e" * 64,
|
|
"fence_hash": "f" * 64,
|
|
"enforcement_impact": "degrades_completion",
|
|
})
|
|
panel = compact_review_projection([run])["panels"][0]
|
|
assert panel["panel_id"] == result.panel_id
|
|
assert panel["transport_status"] == "partial"
|
|
assert panel["parse_status"] == "malformed"
|
|
assert panel["quorum"] == {"required": 2, "contributed": 0, "configured": 3}
|
|
assert len(panel["actors"]) == 3
|
|
assert next(
|
|
actor for actor in panel["actors"] if actor["slot_id"] == "degraded"
|
|
)["outcome_tier"] == "best_effort"
|
|
assert all("raw_text" not in actor for actor in panel["actors"])
|
|
|
|
|
|
def test_compact_review_projection_redacts_public_reasons_before_truncation():
|
|
from ouroboros.review_substrate import compact_review_projection
|
|
|
|
secret = "sk-or-" + ("ReviewSecret123" * 4)
|
|
benign_reason = "Evidence coverage is incomplete, but the consumer flow is clear."
|
|
actor_prefix = benign_reason + ("x" * 400) + " credential="
|
|
panel_prefix = "Panel retained its benign diagnostic context. " + ("y" * 735)
|
|
run = {
|
|
"request": {"surface": "task_acceptance", "policy": {"min_successful_slots": 1}},
|
|
"aggregate_signal": "DEGRADED",
|
|
"reason": panel_prefix + " " + secret,
|
|
"actors": [
|
|
{
|
|
"slot_id": "benign",
|
|
"model": "model-safe",
|
|
"status": "ok",
|
|
"signal": "PASS",
|
|
"parsed": {"verdict": "PASS", "summary": benign_reason, "findings": []},
|
|
"quorum_contribution": True,
|
|
},
|
|
{
|
|
"slot_id": "secret-bearing",
|
|
"model": "model-secret",
|
|
"status": "ok",
|
|
"signal": "DEGRADED",
|
|
"parsed": {
|
|
"verdict": "DEGRADED",
|
|
"summary": actor_prefix + secret,
|
|
"findings": [],
|
|
},
|
|
},
|
|
],
|
|
}
|
|
|
|
panel = compact_review_projection([run])["panels"][0]
|
|
actors = {actor["slot_id"]: actor for actor in panel["actors"]}
|
|
rendered = json.dumps(panel, ensure_ascii=False)
|
|
|
|
assert actors["benign"]["reason"] == benign_reason
|
|
assert benign_reason in actors["secret-bearing"]["reason"]
|
|
assert "Panel retained its benign diagnostic context." in panel["reason"]
|
|
assert secret not in rendered
|
|
assert secret[:20] not in rendered
|
|
assert "***REDACTED***" in actors["secret-bearing"]["reason"]
|
|
assert "***REDACTED***" in panel["reason"]
|
|
|
|
|
|
class _MixedPassPassFailLLM:
|
|
def chat(self, **kwargs):
|
|
if str(kwargs.get("model") or "").endswith("-2"):
|
|
body = {
|
|
"verdict": "FAIL",
|
|
"outcome_tier": "blocked_with_evidence",
|
|
"completion_coach": "Resolve the verified acceptance gap.",
|
|
"findings": [{
|
|
"severity": "high",
|
|
"item": "acceptance_gap",
|
|
"evidence": "The required behavior is not demonstrated.",
|
|
"recommendation": "Add independent evidence for the missing behavior.",
|
|
}],
|
|
"summary": "The candidate is not ready.",
|
|
}
|
|
else:
|
|
body = {
|
|
"verdict": "PASS",
|
|
"outcome_tier": "solved",
|
|
"completion_coach": "Ship the candidate.",
|
|
# Production panels always required criteria evidence (the knob was
|
|
# constant-true and is deleted): a contributing solved PASS carries
|
|
# supported criteria with refs.
|
|
"criteria_used": [{
|
|
"criterion": "candidate is verified", "status": "supported",
|
|
"evidence_refs": ["verification_summary"],
|
|
}],
|
|
"findings": [],
|
|
"summary": "The candidate is ready.",
|
|
}
|
|
return {"content": json.dumps(body)}, {}
|
|
|
|
|
|
def test_mixed_panel_counts_valid_participation_independently_of_veto(tmp_path):
|
|
from ouroboros.review_substrate import (
|
|
aggregate_outcome_tier,
|
|
compact_review_projection,
|
|
task_acceptance_is_clean,
|
|
)
|
|
|
|
result = run_review_request(
|
|
ReviewRequest(
|
|
surface="task_acceptance",
|
|
goal="g",
|
|
subject="candidate",
|
|
policy={"classify_outcome_tier": True, "min_successful_slots": 2},
|
|
task_id="mixed-panel",
|
|
),
|
|
slots=[ReviewSlot(f"s{i}", f"m-{i}") for i in range(3)],
|
|
drive_root=tmp_path,
|
|
llm=_MixedPassPassFailLLM(),
|
|
)
|
|
|
|
assert result.aggregate_signal == "FAIL"
|
|
assert aggregate_outcome_tier(result) == "blocked_with_evidence"
|
|
assert task_acceptance_is_clean(result) is False
|
|
actors = {actor["slot_id"]: actor for actor in result.actors}
|
|
assert all(actor["quorum_contribution"] is True for actor in actors.values())
|
|
assert actors["s0"]["enforcement_impact"] == "supports_pass"
|
|
assert actors["s1"]["enforcement_impact"] == "supports_pass"
|
|
assert actors["s2"]["enforcement_impact"] == "veto"
|
|
|
|
run = dict(result.__dict__)
|
|
run["authority"] = "host_root"
|
|
panel = compact_review_projection([run])["panels"][0]
|
|
assert panel["aggregate_signal"] == "FAIL"
|
|
assert panel["quorum"] == {"required": 2, "contributed": 3, "configured": 3}
|
|
assert panel["coverage"]["quorum_contributing"] == 3
|
|
assert [actor["enforcement_impact"] for actor in panel["actors"]] == [
|
|
"supports_pass",
|
|
"supports_pass",
|
|
"veto",
|
|
]
|
|
|
|
|
|
class _ArrayReviewTruthLLM:
|
|
def chat(self, **_kwargs):
|
|
return {
|
|
"content": json.dumps([{
|
|
"verdict": "FAIL",
|
|
"item": "missing_visual_evidence",
|
|
"evidence": "No inspected screenshot is attached.",
|
|
"recommendation": "Inspect the captured consumer flow.",
|
|
}]),
|
|
}, {
|
|
"provider": "openrouter",
|
|
"resolved_model": "anthropic/claude-fable-5",
|
|
}
|
|
|
|
|
|
def test_review_actor_truth_preserves_array_coverage_and_physical_route(tmp_path):
|
|
result = run_review_request(
|
|
ReviewRequest(
|
|
surface="multi_model_review",
|
|
goal="g",
|
|
subject="candidate",
|
|
policy={"min_successful_slots": 1},
|
|
task_id="array-truth",
|
|
),
|
|
slots=[ReviewSlot("array", "anthropic/array-model")],
|
|
drive_root=tmp_path,
|
|
llm=_ArrayReviewTruthLLM(),
|
|
)
|
|
|
|
actor = result.actors[0]
|
|
assert actor["parse_status"] == "valid"
|
|
assert actor["semantic_verdict"] == "FAIL"
|
|
assert actor["coverage"]["findings"] == 1
|
|
assert actor["reason"] == "No inspected screenshot is attached."
|
|
assert actor["provider"] == "openrouter"
|
|
assert actor["model"] == "anthropic/claude-fable-5"
|
|
|
|
def test_review_binding_is_stable_and_tracks_each_exact_input():
|
|
from ouroboros.review_substrate import build_review_binding
|
|
|
|
base = build_review_binding(
|
|
candidate="answer", evidence={"claims": ["verified"]}, fence_token_or_state="fence-1",
|
|
)
|
|
assert base == build_review_binding(
|
|
candidate="answer", evidence={"claims": ["verified"]}, fence_token_or_state="fence-1",
|
|
)
|
|
assert base["candidate_hash"] != build_review_binding(
|
|
candidate="changed", evidence={"claims": ["verified"]}, fence_token_or_state="fence-1",
|
|
)["candidate_hash"]
|
|
assert base["evidence_revision"] != build_review_binding(
|
|
candidate="answer", evidence={"claims": ["changed"]}, fence_token_or_state="fence-1",
|
|
)["evidence_revision"]
|
|
assert base["binding_hash"] != build_review_binding(
|
|
candidate="answer", evidence={"claims": ["verified"]}, fence_token_or_state="fence-2",
|
|
)["binding_hash"]
|
|
assert len(base["fence_hash"]) == 64
|
|
assert "fence_token_or_state" not in base
|
|
assert "fence-1" not in json.dumps(base)
|