ouroboros/devtools/e2e_live/stub_lane.py
Ouroboros f856286372 Preserve late work evidence and remove the extinct rehearsal router
Classify complete tool observations before dismissing a retired hidden turn.
Show late proven work through the existing card projection with its recorded
outcome and cost. Preserve an existing terminal card under metadata-less late
starts; reuse the terminal predicate with the metrics lifecycle axis, since
metrics do not carry the authored-message terminal-status field. Keep live
metrics nonterminal and old/incomplete addressing aggregates hidden.

Remove only the dead promoted_task_toolset router branch and SW1 script role
from the keyless rehearsal. Exercise its actual managed-root loopback request
and preserve repeatable child/probe final templates.

This is the root-approved S1/S2 batch after review of8416d944. Native focused
verification caught and resolved D1/D2 within this batch; their red receipts
remain preserved. Protected size_ratchet_manifest.py was regenerated with the
standard script: chat.js205675 to205673 only, with no baseline expansion.

Validation:125sharedJS tests and132independentJS/probes pass;78focusedPython
checks plus2SM1unit controls pass, including docs/BIBLE and both real ratchet
nodes. Final ratchet delta and Ruff F pass. Full SM1 rehearsal was not rerun.
Exact final model reviews and browser delta follow this immutable checkpoint;
root retains composition, joint live cognitive workflow and public delivery.

(cherry picked from commit b52b4c0898258132d3a2656eae6fa6b4d8871918)
2026-09-12 20:36:17 +03:00

93 lines
4.3 KiB
Python

"""``--stub``: the $0 rehearsal of every scenario against the loopback stub model.
Reuses the system-E2E harness (``tests/system_e2e/harness.py``: the loopback model server, the
review-organ classification with its canned parse-clean verdicts, ``keyless_settings``) instead
of a second stub; the runner imports it lazily and only in stub mode. The one thing added here
is per-role sequencing: a swarm scenario interleaves managed-root, child and admission-probe
calls on one wire, so the script is a map of per-role queues rather than one ordered list
(``scenarios.<id>_stub_script``).
"""
from __future__ import annotations
import json
STUB_MODEL_SLUG = "openai-compatible::mock-model" # == harness.MOCK_SLUG (asserted in stub_settings)
STUB_CHILD_SLUG = "openai-compatible::mock-child"
STUB_MODEL_SLOTS = {"OUROBOROS_MODEL": STUB_MODEL_SLUG, "OUROBOROS_MODEL_LIGHT": STUB_MODEL_SLUG}
def routed_stub_model(script: dict):
"""A loopback model serving ``{role: [steps]}``; review-organ calls stay canned."""
from tests.system_e2e import harness
class RoutedStubModel(harness.LoopbackModelServer):
def __init__(self) -> None:
super().__init__()
self.queues = {role: list(steps) for role, steps in script.items()}
self.roles: list[str] = []
def _model_ids(self) -> list[str]:
return ["mock-model", "mock-child"]
def _route(self, body: dict) -> str:
if not body.get("tools"):
return "probe"
if "mock-child" in str(body.get("model") or ""):
return "child"
return "agent"
def _answer(self, body: dict, seq: int) -> tuple[str, dict]:
kind = harness.classify_call(body)
canned = harness.canned_review_answer(kind)
if canned is not None:
return kind, canned
role = self._route(body)
self.roles.append(role)
queue = self.queues.get(role) or []
step = queue[0] if queue else None
if callable(step):
step = step(harness.body_text(body))
if step is None:
return "final", {"role": "assistant", "content": f"Stub: no scripted step left for role {role}."}
if "final" in step:
if len(queue) > 1 or role in ("agent",):
queue.pop(0) # a final closes ONE task; a child/probe final repeats for every caller
return "final", {"role": "assistant", "content": str(step["final"])}
queue.pop(0)
call = {"name": str(step["tool"]), "arguments": json.dumps(step.get("arguments") or {})}
return role, {"role": "assistant", "content": "still working",
"tool_calls": [{"id": f"call_{seq}", "type": "function", "function": call}]}
def consumed(self) -> dict:
return {role: len(queue) for role, queue in self.queues.items()}
return RoutedStubModel()
def stub_settings(stub, template: dict) -> dict:
"""The keyless lane settings: every slot the tree declares pinned (the loop slots to the
loopback stub, the rest empty), the review panel and the advisory row on the stub, then the
run template's knobs (budget, workers, evolution) on top. Refuses a template that would
smuggle a paid slot or a credential."""
from tests.system_e2e import harness
if harness.MOCK_SLUG != STUB_MODEL_SLUG:
raise RuntimeError("stub slug drifted from the harness MOCK_SLUG")
cfg = harness.keyless_settings(
stub,
OUROBOROS_REVIEWER_SLOTS=harness.keyless_reviewer_slots(advisory=True),
OUROBOROS_RUNTIME_MODE="advanced",
# The tree's default context mode, declared explicitly: the suite's keyless Low would
# either be normalized to Max at boot (a persist the strict snapshot pin refuses) or,
# with the owner marker, SKIP whole-repository scope review by design.
OUROBOROS_CONTEXT_MODE="max",
OUROBOROS_CONTEXT_MODE_AUTO_LOW="false",
)
paid = {k: v for k, v in template.items()
if k.startswith("OUROBOROS_MODEL") and v and str(v) != STUB_MODEL_SLUG}
if paid:
raise RuntimeError(f"stub template carries paid model slots: {sorted(paid)}")
cfg.update(template)
cfg["OPENAI_COMPATIBLE_BASE_URL"] = stub.base_url
harness.assert_settings_keyless(cfg)
return cfg