mirror of
https://github.com/razzant/ouroboros.git
synced 2026-10-03 04:07:04 +00:00
Classify complete tool observations before dismissing a retired hidden turn. Show late proven work through the existing card projection with its recorded outcome and cost. Preserve an existing terminal card under metadata-less late starts; reuse the terminal predicate with the metrics lifecycle axis, since metrics do not carry the authored-message terminal-status field. Keep live metrics nonterminal and old/incomplete addressing aggregates hidden. Remove only the dead promoted_task_toolset router branch and SW1 script role from the keyless rehearsal. Exercise its actual managed-root loopback request and preserve repeatable child/probe final templates. This is the root-approved S1/S2 batch after review of8416d944. Native focused verification caught and resolved D1/D2 within this batch; their red receipts remain preserved. Protected size_ratchet_manifest.py was regenerated with the standard script: chat.js205675 to205673 only, with no baseline expansion. Validation:125sharedJS tests and132independentJS/probes pass;78focusedPython checks plus2SM1unit controls pass, including docs/BIBLE and both real ratchet nodes. Final ratchet delta and Ruff F pass. Full SM1 rehearsal was not rerun. Exact final model reviews and browser delta follow this immutable checkpoint; root retains composition, joint live cognitive workflow and public delivery. (cherry picked from commit b52b4c0898258132d3a2656eae6fa6b4d8871918)
93 lines
4.3 KiB
Python
93 lines
4.3 KiB
Python
"""``--stub``: the $0 rehearsal of every scenario against the loopback stub model.
|
|
|
|
Reuses the system-E2E harness (``tests/system_e2e/harness.py``: the loopback model server, the
|
|
review-organ classification with its canned parse-clean verdicts, ``keyless_settings``) instead
|
|
of a second stub; the runner imports it lazily and only in stub mode. The one thing added here
|
|
is per-role sequencing: a swarm scenario interleaves managed-root, child and admission-probe
|
|
calls on one wire, so the script is a map of per-role queues rather than one ordered list
|
|
(``scenarios.<id>_stub_script``).
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
|
|
STUB_MODEL_SLUG = "openai-compatible::mock-model" # == harness.MOCK_SLUG (asserted in stub_settings)
|
|
STUB_CHILD_SLUG = "openai-compatible::mock-child"
|
|
STUB_MODEL_SLOTS = {"OUROBOROS_MODEL": STUB_MODEL_SLUG, "OUROBOROS_MODEL_LIGHT": STUB_MODEL_SLUG}
|
|
|
|
|
|
def routed_stub_model(script: dict):
|
|
"""A loopback model serving ``{role: [steps]}``; review-organ calls stay canned."""
|
|
from tests.system_e2e import harness
|
|
|
|
class RoutedStubModel(harness.LoopbackModelServer):
|
|
def __init__(self) -> None:
|
|
super().__init__()
|
|
self.queues = {role: list(steps) for role, steps in script.items()}
|
|
self.roles: list[str] = []
|
|
|
|
def _model_ids(self) -> list[str]:
|
|
return ["mock-model", "mock-child"]
|
|
|
|
def _route(self, body: dict) -> str:
|
|
if not body.get("tools"):
|
|
return "probe"
|
|
if "mock-child" in str(body.get("model") or ""):
|
|
return "child"
|
|
return "agent"
|
|
|
|
def _answer(self, body: dict, seq: int) -> tuple[str, dict]:
|
|
kind = harness.classify_call(body)
|
|
canned = harness.canned_review_answer(kind)
|
|
if canned is not None:
|
|
return kind, canned
|
|
role = self._route(body)
|
|
self.roles.append(role)
|
|
queue = self.queues.get(role) or []
|
|
step = queue[0] if queue else None
|
|
if callable(step):
|
|
step = step(harness.body_text(body))
|
|
if step is None:
|
|
return "final", {"role": "assistant", "content": f"Stub: no scripted step left for role {role}."}
|
|
if "final" in step:
|
|
if len(queue) > 1 or role in ("agent",):
|
|
queue.pop(0) # a final closes ONE task; a child/probe final repeats for every caller
|
|
return "final", {"role": "assistant", "content": str(step["final"])}
|
|
queue.pop(0)
|
|
call = {"name": str(step["tool"]), "arguments": json.dumps(step.get("arguments") or {})}
|
|
return role, {"role": "assistant", "content": "still working",
|
|
"tool_calls": [{"id": f"call_{seq}", "type": "function", "function": call}]}
|
|
|
|
def consumed(self) -> dict:
|
|
return {role: len(queue) for role, queue in self.queues.items()}
|
|
|
|
return RoutedStubModel()
|
|
|
|
|
|
def stub_settings(stub, template: dict) -> dict:
|
|
"""The keyless lane settings: every slot the tree declares pinned (the loop slots to the
|
|
loopback stub, the rest empty), the review panel and the advisory row on the stub, then the
|
|
run template's knobs (budget, workers, evolution) on top. Refuses a template that would
|
|
smuggle a paid slot or a credential."""
|
|
from tests.system_e2e import harness
|
|
|
|
if harness.MOCK_SLUG != STUB_MODEL_SLUG:
|
|
raise RuntimeError("stub slug drifted from the harness MOCK_SLUG")
|
|
cfg = harness.keyless_settings(
|
|
stub,
|
|
OUROBOROS_REVIEWER_SLOTS=harness.keyless_reviewer_slots(advisory=True),
|
|
OUROBOROS_RUNTIME_MODE="advanced",
|
|
# The tree's default context mode, declared explicitly: the suite's keyless Low would
|
|
# either be normalized to Max at boot (a persist the strict snapshot pin refuses) or,
|
|
# with the owner marker, SKIP whole-repository scope review by design.
|
|
OUROBOROS_CONTEXT_MODE="max",
|
|
OUROBOROS_CONTEXT_MODE_AUTO_LOW="false",
|
|
)
|
|
paid = {k: v for k, v in template.items()
|
|
if k.startswith("OUROBOROS_MODEL") and v and str(v) != STUB_MODEL_SLUG}
|
|
if paid:
|
|
raise RuntimeError(f"stub template carries paid model slots: {sorted(paid)}")
|
|
cfg.update(template)
|
|
cfg["OPENAI_COMPATIBLE_BASE_URL"] = stub.base_url
|
|
harness.assert_settings_keyless(cfg)
|
|
return cfg
|