mirror of
https://github.com/razzant/ouroboros.git
synced 2026-10-03 04:07:04 +00:00
v7next F0: system-E2E harness skeleton - keyless real-server lane with two green smokes
tests/system_e2e/: SystemHarness over IsolatedServer with (1) a genuinely KEYLESS lane - every provider credential family and proxy var is stripped from the child env AND the settings builder pins every model slot to the loopback stub, with a regression test planting live-shaped keys and proving none reach the child, plus a companion test that PINS the upstream pass-through hole itself so its eventual fix collapses our override loudly; (2) ScriptedStubModel - ordered per-scenario tool scripts and a review-organ branch placed BEFORE the finalization check, classification pinned against the tree's own parsers and prompt-marker literals (drift = red); (3) ArtifactOracle durable readers incl. per-task forked drive roots. Smokes on this exact tree: S1 boot/identity/contract 15s; S2 review-organ 62s - a doc-only commit_reviewed LANDS in the isolated clone under BLOCKING enforcement with stub triad ([]+NO_FINDINGS) and scope (8-item matrix validated by normalize_scope_items), advisory-bypass ledger and events asserted. 11 passed re-verified independently; ruff F clean. Operational facts recorded for F4: the stub must advertise a large context window (capability-evidence gates commit_reviewed pre-dispatch), and an observed publish-before-persist race (/api/tasks says completed before the durable result lands) is a candidate upstream finding, not worked around silently - wait_durable_result() names it.
This commit is contained in:
parent
5d3398c11d
commit
b044bac582
3 changed files with 1008 additions and 0 deletions
0
tests/system_e2e/__init__.py
Normal file
0
tests/system_e2e/__init__.py
Normal file
606
tests/system_e2e/harness.py
Normal file
606
tests/system_e2e/harness.py
Normal file
|
|
@ -0,0 +1,606 @@
|
|||
"""SystemHarness — the Ф0 skeleton of the v7next deep-integration suite (plan §8).
|
||||
|
||||
Not a test module (pytest collects ``test_*.py`` only) — this is the machinery the
|
||||
``tests/system_e2e/test_*`` scenario modules drive: a KEYLESS isolated real-server
|
||||
stack (roast F21), a scriptable loopback stub model whose review-organ branch sits
|
||||
BEFORE the finalization-turn check (roast F22 / plan §8), and readers for the durable
|
||||
artifacts every scenario asserts against. The direct precedent is
|
||||
``tests/fixtures_e2e_cancellation.py`` on the ``ouroboros_v7_wip`` reference branch
|
||||
(same split, same stub idiom, same 0600 settings write); this file generalizes it from
|
||||
the cancellation protocol to the whole system surface and hardens the egress story.
|
||||
|
||||
KEYLESS LANE CONTRACT (F21). The mock lane must be structurally unable to spend money
|
||||
or leak an operator credential into a child the scenarios do not control:
|
||||
|
||||
* the isolated ``settings.json`` is built from scratch (never copied from live
|
||||
settings), pins EVERY model-slot key the tree declares
|
||||
(``provider_models.ACTIVE_MODEL_SETTING_KEYS`` + legacy) so a new upstream slot is
|
||||
pinned by construction, and carries exactly one "credential" — the loopback stub's
|
||||
non-secret placeholder pair;
|
||||
* ``KeylessIsolatedServer`` strips every provider credential the tree knows about
|
||||
(``server_runner._PROVIDER_ENV_KEYS`` ∪ ``provider_models.ALL_PROVIDER_CREDENTIAL_KEYS``)
|
||||
plus all proxy variables from the child environment, ON TOP of the base
|
||||
``IsolatedServer`` sanitization — the base ``_is_secret_env_key`` deliberately
|
||||
EXEMPTS provider keys (benchmark servers need them), which for this lane is exactly
|
||||
the ANTHROPIC_API_KEY hole the plan names;
|
||||
* an un-pinned slot therefore routes to a slug whose provider has no credential and
|
||||
fails loudly instead of silently reaching a paid provider.
|
||||
|
||||
Full egress interception (socket-level deny + evidence) is Ф4 scope, not Ф0.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import pathlib
|
||||
import subprocess
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
|
||||
import pytest
|
||||
|
||||
REPO_ROOT = pathlib.Path(__file__).resolve().parents[2]
|
||||
if str(REPO_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(REPO_ROOT))
|
||||
|
||||
from devtools.benchmarks.common.server_runner import ( # noqa: E402
|
||||
IsolatedServer,
|
||||
_PROVIDER_ENV_KEYS,
|
||||
supervisor_state_is_ready, # noqa: F401 (re-export: THE readiness contract)
|
||||
)
|
||||
from ouroboros.provider_models import ( # noqa: E402
|
||||
ACTIVE_MODEL_SETTING_KEYS,
|
||||
ALL_PROVIDER_CREDENTIAL_KEYS,
|
||||
LEGACY_MODEL_SETTING_KEYS,
|
||||
)
|
||||
from ouroboros.tools.scope_review_contract import SCOPE_REQUIRED_ITEMS # noqa: E402
|
||||
|
||||
LANE_MOCK = "mock"
|
||||
|
||||
# The scenario inventory of this suite (plan §8). The scenario test module fails if an
|
||||
# id loses its test — a scenario must be retired deliberately, not by deletion.
|
||||
# Scenarios land WITH their phases (roast F22); Ф0 carries only the two smokes that
|
||||
# prove the skeleton itself.
|
||||
SCENARIOS = {
|
||||
"S1": ("boot / identity / task contract smoke", LANE_MOCK),
|
||||
"S2": ("review-organ smoke: commit_reviewed triad+scope on a doc-only diff", LANE_MOCK),
|
||||
}
|
||||
|
||||
MOCK_SLUG = "openai-compatible::mock-model"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Prompt markers the stub classifies review-organ calls by (roast F22).
|
||||
#
|
||||
# These are VERBATIM literals from the tree under test and WILL drift with upstream:
|
||||
# REVIEWER_SLOT_MARKER — ouroboros/review_execution.py::_render_prompt_parts
|
||||
# ACCEPTANCE_KEYS_MARKER — same function, the task_acceptance criteria_used key list
|
||||
# TRIAD_USER_MARKER — ouroboros/tools/review.py::_dispatch_unified_review
|
||||
# SCOPE_USER_MARKER — ouroboros/tools/scope_review.py::_call_scope_llm
|
||||
# The default-lane marker-pin test greps them out of the source files so drift is a
|
||||
# named test failure, not a silently mute stub.
|
||||
# ---------------------------------------------------------------------------
|
||||
REVIEWER_SLOT_MARKER = "You are an independent Ouroboros reviewer slot."
|
||||
ACCEPTANCE_KEYS_MARKER = "criteria_used (the acceptance criteria you re-derived"
|
||||
TRIAD_USER_MARKER = "Review the staged diff and context provided in the instructions above."
|
||||
SCOPE_USER_MARKER = "Review the staged change and context above. Output ONLY a JSON array."
|
||||
FINALIZATION_MARKERS = ("[OWNER_STOP]", "[FINALIZE_NOW]")
|
||||
|
||||
MARKER_SOURCES = {
|
||||
REVIEWER_SLOT_MARKER: "ouroboros/review_execution.py",
|
||||
ACCEPTANCE_KEYS_MARKER: "ouroboros/review_execution.py",
|
||||
TRIAD_USER_MARKER: "ouroboros/tools/review.py",
|
||||
SCOPE_USER_MARKER: "ouroboros/tools/scope_review.py",
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Opt-in gate
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def lane_enabled(lane: str) -> bool:
|
||||
selected = str(os.environ.get("OUROBOROS_E2E_DEEP") or "").strip().lower()
|
||||
return selected == lane
|
||||
|
||||
|
||||
def require_lane(lane: str) -> None:
|
||||
if not lane_enabled(lane):
|
||||
pytest.skip(
|
||||
f"set OUROBOROS_E2E_DEEP={lane} to run the {lane} deep-integration lane "
|
||||
"(spawns a real isolated server; see tests/system_e2e/)"
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Message flattening: review prompts arrive as block lists (cached_prompt_blocks),
|
||||
# agent-loop prompts as plain strings — marker checks must see both.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def message_text(message) -> str:
|
||||
content = message.get("content") if isinstance(message, dict) else None
|
||||
if isinstance(content, str):
|
||||
return content
|
||||
if isinstance(content, list):
|
||||
return "\n".join(
|
||||
str(block.get("text") or "")
|
||||
for block in content
|
||||
if isinstance(block, dict)
|
||||
)
|
||||
return ""
|
||||
|
||||
|
||||
def body_text(body: dict) -> str:
|
||||
return "\n".join(message_text(m) for m in (body.get("messages") or []))
|
||||
|
||||
|
||||
def classify_call(body: dict) -> str:
|
||||
"""Name the branch a chat-completion body belongs to.
|
||||
|
||||
Returns one of: ``safety``, ``scope_review``, ``triad_review``, ``acceptance``,
|
||||
``reviewer_slot``, ``finalization``, ``agent``. ORDER MATTERS (roast F22): every
|
||||
review-organ branch is checked BEFORE the finalization-turn check, because a
|
||||
review packet may quote a transcript that itself contains a finalization marker —
|
||||
a stub that answered such a packet with a final chat answer would silently break
|
||||
the review organ mid-scenario.
|
||||
"""
|
||||
fmt = body.get("response_format")
|
||||
if isinstance(fmt, dict) and fmt.get("type") == "json_object":
|
||||
return "safety"
|
||||
user_tail = "\n".join(
|
||||
message_text(m) for m in (body.get("messages") or [])
|
||||
if isinstance(m, dict) and m.get("role") == "user"
|
||||
)
|
||||
full = body_text(body)
|
||||
# Scope before triad: both user messages start with "Review the staged".
|
||||
if SCOPE_USER_MARKER in user_tail:
|
||||
return "scope_review"
|
||||
if TRIAD_USER_MARKER in user_tail:
|
||||
return "triad_review"
|
||||
if REVIEWER_SLOT_MARKER in full:
|
||||
return "acceptance" if ACCEPTANCE_KEYS_MARKER in full else "reviewer_slot"
|
||||
if any(marker in full for marker in FINALIZATION_MARKERS):
|
||||
return "finalization"
|
||||
return "agent"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Canned review-organ verdicts (all-clean). Shapes come from the tree's own parsers:
|
||||
# triad — triad_review.REVIEW_JSON_ARRAY_CONTRACT ([] + NO_FINDINGS sentinel);
|
||||
# scope — scope_review_contract.normalize_scope_items (required matrix, PASS reasons
|
||||
# must be non-terse); reviewer slot — review_execution's "Return JSON with keys" list.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
TRIAD_CLEAN_TEXT = "[]\nNO_FINDINGS"
|
||||
|
||||
|
||||
def scope_clean_text() -> str:
|
||||
return json.dumps([
|
||||
{
|
||||
"item": item,
|
||||
"verdict": "PASS",
|
||||
"severity": "advisory",
|
||||
"reason": "Stub scope reviewer: checked and clean for this scripted smoke diff.",
|
||||
}
|
||||
for item in sorted(SCOPE_REQUIRED_ITEMS)
|
||||
])
|
||||
|
||||
|
||||
def reviewer_slot_clean_text(kind: str) -> str:
|
||||
verdict = {"verdict": "PASS", "findings": [], "summary": "stub reviewer slot: clean."}
|
||||
if kind == "acceptance":
|
||||
verdict["outcome_tier"] = "solved"
|
||||
verdict["dialogue_status"] = "continue_actionable"
|
||||
verdict["criteria_used"] = []
|
||||
return json.dumps(verdict)
|
||||
|
||||
|
||||
def scripted_completion(body: dict, seq: int, script_next, final_answer: str) -> tuple[str, dict]:
|
||||
"""The stub's whole decision function, pure so the default lane can pin it.
|
||||
|
||||
``script_next`` is a callable returning the next scripted tool step (or None when
|
||||
the script is exhausted); it is only consulted on plain agent turns. Returns
|
||||
``(kind, message)`` where message is the OpenAI-style assistant message.
|
||||
"""
|
||||
kind = classify_call(body)
|
||||
if kind == "safety":
|
||||
return kind, {"role": "assistant",
|
||||
"content": json.dumps({"status": "SAFE", "reason": "stub"})}
|
||||
if kind == "scope_review":
|
||||
return kind, {"role": "assistant", "content": scope_clean_text()}
|
||||
if kind == "triad_review":
|
||||
return kind, {"role": "assistant", "content": TRIAD_CLEAN_TEXT}
|
||||
if kind in ("acceptance", "reviewer_slot"):
|
||||
return kind, {"role": "assistant", "content": reviewer_slot_clean_text(kind)}
|
||||
if kind == "finalization":
|
||||
return kind, {"role": "assistant", "content": final_answer}
|
||||
step = script_next(body) if body.get("tools") else None
|
||||
if step is None:
|
||||
return "final", {"role": "assistant", "content": final_answer}
|
||||
call = {"name": str(step["tool"]),
|
||||
"arguments": json.dumps(step.get("arguments") or {})}
|
||||
return "agent", {
|
||||
"role": "assistant", "content": "still working",
|
||||
"tool_calls": [{"id": f"call_{seq}", "type": "function", "function": call}],
|
||||
}
|
||||
|
||||
|
||||
class ScriptedStubModel:
|
||||
"""Keep-alive OpenAI-compatible stub model with an ordered per-scenario script.
|
||||
|
||||
Extends the ``StubModelServer`` idiom of the cancellation harness: instead of one
|
||||
fixed keepalive tool, a scenario hands the stub an ORDERED list of tool steps
|
||||
(``{"tool": name, "arguments": {...}}``); each plain agent turn consumes one step,
|
||||
and an exhausted script yields the tool-less final answer. Review-organ calls
|
||||
(triad / scope / reviewer-slot / acceptance) NEVER consume script steps — they are
|
||||
classified by prompt markers and answered with canned all-clean verdicts, and that
|
||||
classification runs BEFORE the finalization-turn check (roast F22). Safety
|
||||
supervisor calls (json_object response_format) always get a SAFE verdict.
|
||||
|
||||
Every call is recorded as ``(kind, body)`` in ``self.calls``; ``self.kinds()``
|
||||
gives the observed branch sequence a scenario asserts against.
|
||||
"""
|
||||
|
||||
def __init__(self, script=None, *, final_answer: str = "Final answer: scripted scenario complete.",
|
||||
latency_sec: float = 0.0) -> None:
|
||||
self.script = list(script or [])
|
||||
self.final_answer = final_answer
|
||||
self.latency_sec = latency_sec
|
||||
self.calls: list = [] # (kind, body) in arrival order
|
||||
self._script_index = 0
|
||||
self._lock = threading.Lock()
|
||||
outer = self
|
||||
|
||||
class _Handler(BaseHTTPRequestHandler):
|
||||
def do_GET(self): # noqa: N802 - stdlib callback name
|
||||
if self.path.rstrip("/").endswith("/models"):
|
||||
# >=1M ON PURPOSE, and it is load-bearing twice: the capability-
|
||||
# evidence /models probe stores this as a CONFIRMED window, which
|
||||
# (a) sizes the triad fit budget (a 400K window under the cold
|
||||
# 1.65 density floor caps input at ~202K — BELOW the ~226K
|
||||
# governance pack, blocking every commit_reviewed before
|
||||
# dispatch), and (b) satisfies the BIBLE P3 >=1M floor that
|
||||
# scope review's BLOCKING authority requires.
|
||||
return self._send({"data": [{"id": "mock-model", "max_model_len": 2_000_000}]})
|
||||
self.send_error(404)
|
||||
|
||||
def do_POST(self): # noqa: N802 - stdlib callback name
|
||||
length = int(self.headers.get("Content-Length") or 0)
|
||||
try:
|
||||
body = json.loads((self.rfile.read(length) or b"{}").decode("utf-8"))
|
||||
except ValueError:
|
||||
body = {}
|
||||
if not isinstance(body, dict):
|
||||
body = {}
|
||||
if outer.latency_sec:
|
||||
time.sleep(outer.latency_sec)
|
||||
return self._send(outer._completion(body))
|
||||
|
||||
def _send(self, payload):
|
||||
data = json.dumps(payload).encode("utf-8")
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "application/json")
|
||||
self.send_header("Content-Length", str(len(data)))
|
||||
self.end_headers()
|
||||
self.wfile.write(data)
|
||||
|
||||
def log_message(self, *_args):
|
||||
return
|
||||
|
||||
self._server = ThreadingHTTPServer(("127.0.0.1", 0), _Handler)
|
||||
self._thread = threading.Thread(target=self._server.serve_forever, daemon=True)
|
||||
|
||||
def _next_step(self, _body) -> dict | None:
|
||||
if self._script_index >= len(self.script):
|
||||
return None
|
||||
step = self.script[self._script_index]
|
||||
self._script_index += 1
|
||||
return step
|
||||
|
||||
def _completion(self, body: dict) -> dict:
|
||||
with self._lock:
|
||||
seq = len(self.calls) + 1
|
||||
kind, message = scripted_completion(body, seq, self._next_step, self.final_answer)
|
||||
self.calls.append((kind, body))
|
||||
return {
|
||||
"id": f"stub-{seq}",
|
||||
"object": "chat.completion",
|
||||
"model": str(body.get("model") or "mock-model"),
|
||||
"choices": [{"index": 0, "message": message, "finish_reason": "stop"}],
|
||||
"usage": {"prompt_tokens": 10, "completion_tokens": 5, "total_tokens": 15},
|
||||
}
|
||||
|
||||
def kinds(self) -> list[str]:
|
||||
with self._lock:
|
||||
return [kind for kind, _ in self.calls]
|
||||
|
||||
def script_consumed(self) -> bool:
|
||||
with self._lock:
|
||||
return self._script_index >= len(self.script)
|
||||
|
||||
@property
|
||||
def base_url(self) -> str:
|
||||
return f"http://127.0.0.1:{self._server.server_address[1]}/v1"
|
||||
|
||||
def __enter__(self) -> "ScriptedStubModel":
|
||||
self._thread.start()
|
||||
return self
|
||||
|
||||
def __exit__(self, *_exc) -> None:
|
||||
self._server.shutdown()
|
||||
self._server.server_close()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Keyless isolated server (roast F21)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Everything the child environment must NOT carry in the keyless lane. The env union
|
||||
# closes the documented hole: IsolatedServer._is_secret_env_key EXEMPTS provider keys,
|
||||
# so an inherited ANTHROPIC_API_KEY survives the base sanitization by design.
|
||||
STRIPPED_PROVIDER_ENV_KEYS = frozenset(_PROVIDER_ENV_KEYS) | frozenset(ALL_PROVIDER_CREDENTIAL_KEYS)
|
||||
PROXY_ENV_KEYS = frozenset({
|
||||
"HTTP_PROXY", "HTTPS_PROXY", "ALL_PROXY", "NO_PROXY",
|
||||
"http_proxy", "https_proxy", "all_proxy", "no_proxy",
|
||||
})
|
||||
|
||||
|
||||
class KeylessIsolatedServer(IsolatedServer):
|
||||
"""``IsolatedServer`` whose child env can never carry a provider credential.
|
||||
|
||||
The base class keeps ``_PROVIDER_ENV_KEYS`` in the child on purpose (benchmark
|
||||
servers authenticate from them). This lane's contract is the opposite: the ONLY
|
||||
provider config a scenario server may see is what the scenario's settings.json
|
||||
says, and that file only ever names the loopback stub.
|
||||
"""
|
||||
|
||||
def _env(self) -> dict:
|
||||
env = super()._env()
|
||||
for key in list(env):
|
||||
if key in STRIPPED_PROVIDER_ENV_KEYS or key in PROXY_ENV_KEYS:
|
||||
env.pop(key, None)
|
||||
return env
|
||||
|
||||
|
||||
def keyless_settings(stub: ScriptedStubModel, **overrides) -> dict:
|
||||
"""The isolated settings.json for a keyless scenario server.
|
||||
|
||||
Every model-slot key the TREE declares is pinned — un-listed keys default to the
|
||||
empty string (slot disabled / no fallback), the live loop + review slots to the
|
||||
stub slug. Deriving the slot list from ``provider_models`` (instead of an
|
||||
enumerated literal, as the cancellation-harness precedent did) means an upstream
|
||||
slot added tomorrow is pinned by construction rather than silently defaulting to
|
||||
a live OpenRouter route. Overrides carrying a real provider credential are a
|
||||
scenario bug and are refused loudly.
|
||||
"""
|
||||
stub_pair = {"OPENAI_COMPATIBLE_API_KEY", "OPENAI_COMPATIBLE_BASE_URL"}
|
||||
forbidden = (set(ALL_PROVIDER_CREDENTIAL_KEYS) - stub_pair) & set(overrides)
|
||||
if forbidden:
|
||||
raise ValueError(
|
||||
f"keyless lane: overrides must not carry provider credentials: {sorted(forbidden)}"
|
||||
)
|
||||
cfg: dict = {key: "" for key in (*ACTIVE_MODEL_SETTING_KEYS, *LEGACY_MODEL_SETTING_KEYS)}
|
||||
cfg.update({
|
||||
# Disk-authored keys: config.apply_settings_to_env cannot author these from
|
||||
# the environment, so they have to be in the file, written fresh.
|
||||
"OUROBOROS_SAFETY_MODE": "off",
|
||||
"OUROBOROS_CONTEXT_MODE": "low",
|
||||
"OUROBOROS_RUNTIME_MODE": "light",
|
||||
"OUROBOROS_TASK_REVIEW_MODE": "off",
|
||||
"OUROBOROS_POST_TASK_EVOLUTION": "false",
|
||||
"OUROBOROS_MAX_WORKERS": 4,
|
||||
"TOTAL_BUDGET": 10.0,
|
||||
"OUROBOROS_PER_TASK_COST_USD": 10.0,
|
||||
"OPENAI_COMPATIBLE_BASE_URL": stub.base_url,
|
||||
"OPENAI_COMPATIBLE_API_KEY": "stub-key-not-a-credential",
|
||||
})
|
||||
for slot in ("OUROBOROS_MODEL", "OUROBOROS_MODEL_LIGHT",
|
||||
"OUROBOROS_REVIEW_MODELS", "OUROBOROS_SCOPE_REVIEW_MODELS",
|
||||
"OUROBOROS_SCOPE_REVIEW_MODEL"):
|
||||
cfg[slot] = MOCK_SLUG
|
||||
cfg.update(overrides)
|
||||
return cfg
|
||||
|
||||
|
||||
def assert_settings_keyless(settings: dict) -> None:
|
||||
"""Fail loudly if a scenario's settings smuggle a provider credential."""
|
||||
stub_pair = {"OPENAI_COMPATIBLE_API_KEY", "OPENAI_COMPATIBLE_BASE_URL"}
|
||||
offending = sorted(
|
||||
key for key in settings
|
||||
if key in ALL_PROVIDER_CREDENTIAL_KEYS and key not in stub_pair and str(settings[key] or "").strip()
|
||||
)
|
||||
assert not offending, f"keyless settings carry provider credentials: {offending}"
|
||||
base = str(settings.get("OPENAI_COMPATIBLE_BASE_URL") or "")
|
||||
assert base.startswith("http://127.0.0.1:"), f"stub base_url is not loopback: {base!r}"
|
||||
|
||||
|
||||
def clone_repo(destination: pathlib.Path) -> pathlib.Path:
|
||||
"""One throwaway clone of the checkout under test.
|
||||
|
||||
A clone (not the working tree) is what the runtime is allowed to run against: the
|
||||
server owns its repo directory, so an E2E server must never be pointed at a live
|
||||
worktree. The commit identity is pinned locally so reviewed-commit scenarios never
|
||||
depend on the operator's global git config.
|
||||
"""
|
||||
clone = pathlib.Path(destination) / "clone"
|
||||
subprocess.run(["git", "clone", "--no-hardlinks", "-q", str(REPO_ROOT), str(clone)],
|
||||
check=True, capture_output=True)
|
||||
subprocess.run(["git", "checkout", "-B", "ouroboros"], cwd=str(clone),
|
||||
check=True, capture_output=True)
|
||||
subprocess.run(["git", "remote", "remove", "origin"], cwd=str(clone),
|
||||
check=False, capture_output=True)
|
||||
subprocess.run(["git", "config", "user.name", "SystemHarness"], cwd=str(clone),
|
||||
check=True, capture_output=True)
|
||||
subprocess.run(["git", "config", "user.email", "system-harness@e2e.invalid"],
|
||||
cwd=str(clone), check=True, capture_output=True)
|
||||
return clone
|
||||
|
||||
|
||||
def write_settings_file(settings_path: pathlib.Path, settings: dict) -> None:
|
||||
"""0600-before-content settings write (carried over from the v7_wip harness: a
|
||||
default-umask write_text once briefly published a live key world-readable; this
|
||||
lane never holds a live key, but the shape must not regress when a paid lane
|
||||
reuses it)."""
|
||||
fd = os.open(settings_path, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
|
||||
if hasattr(os, "fchmod"):
|
||||
os.fchmod(fd, 0o600) # O_CREAT's mode only applies on creation
|
||||
with os.fdopen(fd, "w", encoding="utf-8") as fh:
|
||||
fh.write(json.dumps(settings, indent=2))
|
||||
if not hasattr(os, "fchmod"):
|
||||
os.chmod(settings_path, 0o600)
|
||||
|
||||
|
||||
def start_server(clone, root, settings: dict, *, ready_timeout: float = 300) -> KeylessIsolatedServer:
|
||||
assert_settings_keyless(settings)
|
||||
data_root = pathlib.Path(root) / "data"
|
||||
data_root.mkdir(parents=True, exist_ok=True)
|
||||
settings_path = data_root / "settings.json"
|
||||
write_settings_file(settings_path, settings)
|
||||
server = KeylessIsolatedServer(clone, data_root, settings_path)
|
||||
server.start(ready_timeout=ready_timeout)
|
||||
return server
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# ArtifactOracle: readers of the durable artifacts every scenario asserts against.
|
||||
# Never an HTTP 200 on its own, never a harness exit code (AGENTS.md: the exit code
|
||||
# is not the run status) — scenarios read back what the owner and watchdog read.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class ArtifactOracle:
|
||||
def __init__(self, data_root) -> None:
|
||||
self.data_root = pathlib.Path(data_root)
|
||||
|
||||
def task_drive(self, task_id: str) -> "ArtifactOracle":
|
||||
"""The oracle for a HEADLESS task's forked drive root.
|
||||
|
||||
On this tree a headless task's ToolContext drive root is
|
||||
``state/headless_tasks/<task_id>/data`` under the server's data root, so the
|
||||
durable review evidence (state/advisory_review.json, the
|
||||
advisory_review_bypassed / scope_review_complete events) lands THERE, not in
|
||||
the server-level files. Falls back to the server root when the task has no
|
||||
forked drive (e.g. a direct-chat turn)."""
|
||||
forked = self.data_root / "state" / "headless_tasks" / str(task_id) / "data"
|
||||
return ArtifactOracle(forked) if forked.is_dir() else self
|
||||
|
||||
# -- json state files ---------------------------------------------------
|
||||
|
||||
def _json(self, relpath: str) -> dict:
|
||||
path = self.data_root / relpath
|
||||
if not path.exists():
|
||||
return {}
|
||||
loaded = json.loads(path.read_text(encoding="utf-8"))
|
||||
return loaded if isinstance(loaded, dict) else {}
|
||||
|
||||
def queue_snapshot(self) -> dict:
|
||||
return self._json("state/queue_snapshot.json")
|
||||
|
||||
def state(self) -> dict:
|
||||
return self._json("state/state.json")
|
||||
|
||||
def advisory_review(self) -> dict:
|
||||
return self._json("state/advisory_review.json")
|
||||
|
||||
def cancel_intents(self) -> dict:
|
||||
blob = self._json("state/cancel_intents.json")
|
||||
return blob.get("intents") if isinstance(blob.get("intents"), dict) else {}
|
||||
|
||||
def task_result(self, task_id: str) -> dict:
|
||||
return self._json(f"task_results/{task_id}.json")
|
||||
|
||||
def task_result_bytes(self, task_id: str) -> bytes:
|
||||
return (self.data_root / "task_results" / f"{task_id}.json").read_bytes()
|
||||
|
||||
# -- jsonl logs -----------------------------------------------------------
|
||||
|
||||
def _jsonl(self, relpath: str, *, type_filter: str = "") -> list:
|
||||
path = self.data_root / relpath
|
||||
if not path.exists():
|
||||
return []
|
||||
rows = []
|
||||
for line in path.read_text(encoding="utf-8").splitlines():
|
||||
if not line.strip():
|
||||
continue
|
||||
if type_filter and type_filter not in line:
|
||||
continue # cheap pre-filter, exact check below
|
||||
try:
|
||||
row = json.loads(line)
|
||||
except ValueError:
|
||||
continue
|
||||
if not isinstance(row, dict):
|
||||
continue
|
||||
if type_filter and str(row.get("type") or "") != type_filter:
|
||||
continue
|
||||
rows.append(row)
|
||||
return rows
|
||||
|
||||
def events(self, event_type: str = "") -> list:
|
||||
return self._jsonl("logs/events.jsonl", type_filter=event_type)
|
||||
|
||||
def supervisor_rows(self, row_type: str = "") -> list:
|
||||
return self._jsonl("logs/supervisor.jsonl", type_filter=row_type)
|
||||
|
||||
def tools_rows(self) -> list:
|
||||
return self._jsonl("logs/tools.jsonl")
|
||||
|
||||
def chat_bytes(self) -> bytes:
|
||||
path = self.data_root / "logs" / "chat.jsonl"
|
||||
return path.read_bytes() if path.exists() else b""
|
||||
|
||||
def running_ids(self) -> set:
|
||||
return {
|
||||
str(row.get("id") or "")
|
||||
for row in (self.queue_snapshot().get("running") or [])
|
||||
if isinstance(row, dict)
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Small drivers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def wait_until(predicate, timeout: float, interval: float = 0.5):
|
||||
deadline = time.time() + timeout
|
||||
last = None
|
||||
while time.time() < deadline:
|
||||
last = predicate()
|
||||
if last:
|
||||
return last
|
||||
time.sleep(interval)
|
||||
return last
|
||||
|
||||
|
||||
def submit_running(server: IsolatedServer, description: str, *, timeout: float = 120) -> str:
|
||||
"""Submit a task and wait until the supervisor actually has it RUNNING."""
|
||||
task_id = server.submit(description)
|
||||
assert task_id, "submit returned no task id"
|
||||
oracle = ArtifactOracle(server.data_root)
|
||||
running = wait_until(lambda: task_id in oracle.running_ids(), timeout)
|
||||
assert running, f"task {task_id} never reached the RUNNING set"
|
||||
return task_id
|
||||
|
||||
|
||||
def wait_durable_result(oracle: ArtifactOracle, task_id: str, *, timeout: float = 180) -> dict:
|
||||
"""Wait for ``task_results/<id>.json`` to reach a TERMINAL status and return it.
|
||||
|
||||
The HTTP task view can report ``completed`` while the durable terminal write is
|
||||
still in flight behind post-task processing (observed live on this tree: the
|
||||
stored row said ``scheduled`` seconds after the API said ``completed``). A
|
||||
scenario that asserts the durable record must wait for the record, not for the
|
||||
HTTP answer.
|
||||
"""
|
||||
terminal = {"completed", "failed", "cancelled", "rejected_duplicate"}
|
||||
stored = wait_until(
|
||||
lambda: (
|
||||
oracle.task_result(task_id)
|
||||
if str(oracle.task_result(task_id).get("status") or "") in terminal
|
||||
else None
|
||||
),
|
||||
timeout,
|
||||
)
|
||||
assert stored, (
|
||||
f"task {task_id} durable result never reached a terminal status: "
|
||||
f"{oracle.task_result(task_id)!r}"
|
||||
)
|
||||
return stored
|
||||
402
tests/system_e2e/test_system_scenarios.py
Normal file
402
tests/system_e2e/test_system_scenarios.py
Normal file
|
|
@ -0,0 +1,402 @@
|
|||
"""S1-S2 — the Ф0 smokes of the deep-integration suite (v7next plan §8, roast F22).
|
||||
|
||||
WHAT THIS FILE IS. The harness skeleton (``tests/system_e2e/harness.py``) lands in Ф0;
|
||||
the full scenario matrix (subagents, delegation, update engine, cancellation E-suite,
|
||||
skills, UI truth, …) lands WITH its phases. The two scenarios here exist to prove the
|
||||
skeleton itself on the CURRENT upstream-shaped tree, and must survive the domain
|
||||
transplants unchanged:
|
||||
|
||||
* S1 — boot / identity / task contract: a real ``server.py`` on an isolated clone +
|
||||
data root boots to the frozen readiness contract, attests its identity against its
|
||||
own checkout, runs one scripted stub task to completion, and leaves a sane durable
|
||||
``task_results/<id>.json`` behind.
|
||||
* S2 — review organ: a scripted task drives ``commit_reviewed`` over a doc-only diff
|
||||
with the advisory pre-review explicitly skipped (audited bypass) and BLOCKING
|
||||
enforcement, the stub answers the triad packet and the scope-matrix packet with
|
||||
all-clean verdicts, and the commit lands in the isolated clone. Landing under
|
||||
``blocking`` makes the git log itself the proof that both review organs ran and
|
||||
passed — under advisory a failed review would still commit.
|
||||
|
||||
LANES. Default (always-on) tests pin the harness's own contracts with no server and no
|
||||
sockets: the scenario manifest, the stub's branch classification (review-organ branch
|
||||
BEFORE the finalization check), the prompt-marker literals against the tree's source,
|
||||
and the keyless/egress hardening (roast F21). The ``mock`` lane spawns real isolated
|
||||
servers: opt in with ``OUROBOROS_E2E_DEEP=mock``; both scenarios are ``serial`` (real
|
||||
ports, real process trees). No paid lane exists in Ф0.
|
||||
|
||||
Every scenario asserts durable artifacts — never an HTTP 200 on its own and never a
|
||||
harness exit code (AGENTS.md: the exit code is not the run status).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
|
||||
import pytest
|
||||
|
||||
from tests.system_e2e.harness import (
|
||||
ACCEPTANCE_KEYS_MARKER,
|
||||
LANE_MOCK,
|
||||
MARKER_SOURCES,
|
||||
MOCK_SLUG,
|
||||
PROXY_ENV_KEYS,
|
||||
REPO_ROOT,
|
||||
REVIEWER_SLOT_MARKER,
|
||||
SCENARIOS,
|
||||
SCOPE_USER_MARKER,
|
||||
STRIPPED_PROVIDER_ENV_KEYS,
|
||||
TRIAD_USER_MARKER,
|
||||
ArtifactOracle,
|
||||
KeylessIsolatedServer,
|
||||
ScriptedStubModel,
|
||||
assert_settings_keyless,
|
||||
classify_call,
|
||||
clone_repo,
|
||||
keyless_settings,
|
||||
require_lane,
|
||||
scripted_completion,
|
||||
start_server,
|
||||
submit_running,
|
||||
supervisor_state_is_ready,
|
||||
wait_durable_result,
|
||||
wait_until,
|
||||
write_settings_file,
|
||||
)
|
||||
|
||||
# ===========================================================================
|
||||
# Default lane: harness self-contracts. No server, no model, no egress.
|
||||
# ===========================================================================
|
||||
|
||||
|
||||
def test_system_manifest_is_covered():
|
||||
"""Every S-id in the scenario manifest still has at least one test here."""
|
||||
import sys
|
||||
|
||||
names = [name for name in dir(sys.modules[__name__]) if name.startswith("test_")]
|
||||
for scenario_id, (title, _lane) in SCENARIOS.items():
|
||||
prefix = f"test_{scenario_id.lower()}_"
|
||||
assert any(name.startswith(prefix) for name in names), (
|
||||
f"scenario {scenario_id} ({title}) has no {prefix}* test"
|
||||
)
|
||||
|
||||
|
||||
def test_prompt_markers_still_exist_in_the_tree():
|
||||
"""The stub's review-organ classification is prompt-marker based; a marker that
|
||||
drifts out of the source it was pinned from would leave the stub silently mute on
|
||||
that organ — surface the drift as a NAMED failure instead."""
|
||||
for marker, relpath in MARKER_SOURCES.items():
|
||||
source = (REPO_ROOT / relpath).read_text(encoding="utf-8")
|
||||
assert marker in source, (
|
||||
f"marker {marker!r} no longer appears in {relpath}: upstream prompt drifted, "
|
||||
"re-pin the literal in tests/system_e2e/harness.py"
|
||||
)
|
||||
|
||||
|
||||
def _agent_body(text: str = "keep going", *, tools: bool = True) -> dict:
|
||||
body: dict = {"messages": [{"role": "user", "content": text}]}
|
||||
if tools:
|
||||
body["tools"] = [{"type": "function", "function": {"name": "list_files"}}]
|
||||
return body
|
||||
|
||||
|
||||
def test_stub_classification_review_branch_beats_finalization():
|
||||
"""Roast F22: the review-organ branch sits BEFORE the finalization-turn check.
|
||||
|
||||
A triad/scope/reviewer-slot packet that happens to QUOTE a finalization marker
|
||||
(review of a stopped task's transcript) must still be answered as a review."""
|
||||
scope_body = {"messages": [
|
||||
{"role": "system", "content": [{"type": "text", "text": "scope pack [OWNER_STOP] quoted"}]},
|
||||
{"role": "user", "content": SCOPE_USER_MARKER},
|
||||
], "tools": []}
|
||||
triad_body = {"messages": [
|
||||
{"role": "system", "content": [{"type": "text", "text": "triad pack [FINALIZE_NOW] quoted"}]},
|
||||
{"role": "user", "content": "Review the staged diff and context provided in the instructions above."},
|
||||
]}
|
||||
slot_body = {"messages": [
|
||||
{"role": "system", "content": REVIEWER_SLOT_MARKER + "\nSurface: plan_review\n [OWNER_STOP]"},
|
||||
{"role": "user", "content": "Subject: ..."},
|
||||
]}
|
||||
acceptance_body = {"messages": [
|
||||
{"role": "system", "content": REVIEWER_SLOT_MARKER + "\n" + ACCEPTANCE_KEYS_MARKER},
|
||||
{"role": "user", "content": "Subject: ..."},
|
||||
]}
|
||||
assert classify_call(scope_body) == "scope_review"
|
||||
assert classify_call(triad_body) == "triad_review"
|
||||
assert classify_call(slot_body) == "reviewer_slot"
|
||||
assert classify_call(acceptance_body) == "acceptance"
|
||||
assert classify_call({"messages": [{"role": "user", "content": "[FINALIZE_NOW] wrap up"}]}) == "finalization"
|
||||
assert classify_call({"messages": [{"role": "user", "content": "hi"}],
|
||||
"response_format": {"type": "json_object"}}) == "safety"
|
||||
assert classify_call(_agent_body()) == "agent"
|
||||
|
||||
|
||||
def test_stub_verdicts_satisfy_the_trees_own_parsers():
|
||||
"""The canned all-clean answers must parse under the REAL review contracts of this
|
||||
tree — a stub that emits an unparseable verdict turns every review into a
|
||||
parse_failure and the S2 smoke into a lie."""
|
||||
from ouroboros.tools.scope_review_contract import (
|
||||
SCOPE_REQUIRED_ITEMS,
|
||||
classify_scope_findings,
|
||||
normalize_scope_items,
|
||||
)
|
||||
from ouroboros.triad_review import empty_array_is_verified_clean
|
||||
|
||||
_kind, scope_message = scripted_completion(
|
||||
{"messages": [{"role": "user", "content": SCOPE_USER_MARKER}]}, 1, lambda _b: None, "x")
|
||||
items, errors = normalize_scope_items(json.loads(scope_message["content"]))
|
||||
assert not errors, f"stub scope verdict rejected by normalize_scope_items: {errors}"
|
||||
assert {item["item"] for item in items} == set(SCOPE_REQUIRED_ITEMS)
|
||||
critical, advisory = classify_scope_findings(items)
|
||||
assert critical == [] and advisory == []
|
||||
|
||||
_kind, triad_message = scripted_completion(
|
||||
{"messages": [{"role": "user", "content": TRIAD_USER_MARKER}]}, 1, lambda _b: None, "x")
|
||||
assert empty_array_is_verified_clean(triad_message["content"])
|
||||
|
||||
_kind, slot_message = scripted_completion(
|
||||
{"messages": [{"role": "system", "content": REVIEWER_SLOT_MARKER}]}, 1, lambda _b: None, "x")
|
||||
verdict = json.loads(slot_message["content"])
|
||||
assert verdict["verdict"] == "PASS" and verdict["findings"] == []
|
||||
|
||||
|
||||
def test_stub_consumes_the_script_in_order_then_finalizes():
|
||||
steps = iter([{"tool": "write_file", "arguments": {"path": "a.md"}},
|
||||
{"tool": "commit_reviewed", "arguments": {"commit_message": "m"}}])
|
||||
|
||||
def _next(_body):
|
||||
return next(steps, None)
|
||||
|
||||
kind1, msg1 = scripted_completion(_agent_body(), 1, _next, "done")
|
||||
kind2, msg2 = scripted_completion(_agent_body(), 2, _next, "done")
|
||||
kind3, msg3 = scripted_completion(_agent_body(), 3, _next, "done")
|
||||
assert (kind1, kind2, kind3) == ("agent", "agent", "final")
|
||||
assert msg1["tool_calls"][0]["function"]["name"] == "write_file"
|
||||
assert msg2["tool_calls"][0]["function"]["name"] == "commit_reviewed"
|
||||
assert "tool_calls" not in msg3 and msg3["content"] == "done"
|
||||
# A tool-less prompt (final synthesis turn) never consumes a script step.
|
||||
kind4, _ = scripted_completion(_agent_body(tools=False), 4, _next, "done")
|
||||
assert kind4 == "final"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Egress hardening (roast F21): the regression the plan names — a planted
|
||||
# ANTHROPIC_API_KEY in the CALLER env must never reach the child server env.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_f21_planted_provider_key_never_reaches_child_env(tmp_path, monkeypatch):
|
||||
planted = {
|
||||
"ANTHROPIC_API_KEY": "sk-ant-planted-must-not-leak",
|
||||
"OPENROUTER_API_KEY": "sk-or-planted-must-not-leak",
|
||||
"OPENAI_API_KEY": "sk-planted-must-not-leak",
|
||||
"OPENAI_COMPATIBLE_API_KEY": "planted-must-not-leak",
|
||||
"GIGACHAT_CREDENTIALS": "planted-must-not-leak",
|
||||
"CLOUDRU_FOUNDATION_MODELS_API_KEY": "planted-must-not-leak",
|
||||
"HTTP_PROXY": "http://proxy.invalid:3128",
|
||||
"https_proxy": "http://proxy.invalid:3128",
|
||||
"ALL_PROXY": "socks5://proxy.invalid:1080",
|
||||
"NO_PROXY": "localhost",
|
||||
}
|
||||
for key, value in planted.items():
|
||||
monkeypatch.setenv(key, value)
|
||||
server = KeylessIsolatedServer(
|
||||
tmp_path / "clone", tmp_path / "data", tmp_path / "data" / "settings.json")
|
||||
child_env = server._env()
|
||||
leaked = sorted(set(planted) & set(child_env))
|
||||
assert not leaked, f"planted caller-env values leaked into the child env: {leaked}"
|
||||
# The whole families, not just the planted samples:
|
||||
assert not (STRIPPED_PROVIDER_ENV_KEYS & set(child_env))
|
||||
assert not (PROXY_ENV_KEYS & set(child_env))
|
||||
# The child still gets its 4-var isolation set, pointing INTO the throwaway root.
|
||||
for key in ("OUROBOROS_APP_ROOT", "OUROBOROS_REPO_DIR",
|
||||
"OUROBOROS_DATA_DIR", "OUROBOROS_SETTINGS_PATH"):
|
||||
assert str(tmp_path) in child_env[key], (key, child_env[key])
|
||||
|
||||
|
||||
def test_f21_base_isolated_server_still_leaks_provider_keys(tmp_path, monkeypatch):
|
||||
"""The hole the keyless lane closes, pinned so its future upstream fix is VISIBLE:
|
||||
the base ``IsolatedServer`` deliberately keeps provider keys in the child. When
|
||||
this test starts failing, upstream closed the hole itself — collapse
|
||||
``KeylessIsolatedServer`` accordingly instead of keeping a dead override."""
|
||||
from devtools.benchmarks.common.server_runner import IsolatedServer
|
||||
|
||||
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-ant-planted")
|
||||
server = IsolatedServer(
|
||||
tmp_path / "clone", tmp_path / "data", tmp_path / "data" / "settings.json")
|
||||
assert server._env().get("ANTHROPIC_API_KEY") == "sk-ant-planted"
|
||||
|
||||
|
||||
def test_f21_keyless_settings_pin_every_slot_and_refuse_credentials():
|
||||
from ouroboros.provider_models import (
|
||||
ACTIVE_MODEL_SETTING_KEYS,
|
||||
LEGACY_MODEL_SETTING_KEYS,
|
||||
)
|
||||
|
||||
class _FakeStub:
|
||||
base_url = "http://127.0.0.1:1/v1"
|
||||
|
||||
cfg = keyless_settings(_FakeStub())
|
||||
for slot in (*ACTIVE_MODEL_SETTING_KEYS, *LEGACY_MODEL_SETTING_KEYS):
|
||||
assert slot in cfg, f"model slot {slot} left unpinned"
|
||||
assert cfg[slot] in ("", MOCK_SLUG), (slot, cfg[slot])
|
||||
assert cfg["OUROBOROS_MODEL"] == MOCK_SLUG
|
||||
assert_settings_keyless(cfg)
|
||||
with pytest.raises(ValueError, match="provider credentials"):
|
||||
keyless_settings(_FakeStub(), ANTHROPIC_API_KEY="sk-ant-nope")
|
||||
with pytest.raises(AssertionError):
|
||||
assert_settings_keyless({**cfg, "OPENROUTER_API_KEY": "sk-or-nope"})
|
||||
with pytest.raises(AssertionError):
|
||||
assert_settings_keyless({**cfg, "OPENAI_COMPATIBLE_BASE_URL": "https://api.example.com/v1"})
|
||||
|
||||
|
||||
@pytest.mark.skipif(os.name != "posix", reason="POSIX mode bits are meaningless on Windows")
|
||||
def test_settings_file_is_created_secret_safe(tmp_path):
|
||||
"""0600-before-content (carried over from the v7_wip cancellation harness)."""
|
||||
settings_path = tmp_path / "settings.json"
|
||||
write_settings_file(settings_path, {"OPENAI_COMPATIBLE_API_KEY": "not-a-credential"})
|
||||
assert (settings_path.stat().st_mode & 0o777) == 0o600
|
||||
settings_path.chmod(0o664)
|
||||
write_settings_file(settings_path, {"OPENAI_COMPATIBLE_API_KEY": "not-a-credential"})
|
||||
assert (settings_path.stat().st_mode & 0o777) == 0o600
|
||||
|
||||
|
||||
# ===========================================================================
|
||||
# Mock lane: real isolated servers. Opt in with OUROBOROS_E2E_DEEP=mock.
|
||||
# ===========================================================================
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def e2e_clone(tmp_path_factory):
|
||||
"""One throwaway clone of the checkout under test, shared by every scenario server."""
|
||||
require_lane(LANE_MOCK)
|
||||
return clone_repo(tmp_path_factory.mktemp("system_e2e_clone"))
|
||||
|
||||
|
||||
S1_SCRIPT = [
|
||||
{"tool": "list_files", "arguments": {"path": "."}},
|
||||
]
|
||||
|
||||
S2_COMMIT_MESSAGE = "docs: system_e2e S2 review-organ smoke (doc-only)"
|
||||
S2_DOC_PATH = "docs/notes/system_e2e_smoke.md"
|
||||
S2_SCRIPT = [
|
||||
{"tool": "write_file", "arguments": {
|
||||
"root": "system_repo",
|
||||
"path": S2_DOC_PATH,
|
||||
"content": ("# system_e2e S2 smoke\n\n"
|
||||
"Doc-only change landed through commit_reviewed by the scripted stub.\n"),
|
||||
}},
|
||||
{"tool": "commit_reviewed", "arguments": {
|
||||
"commit_message": S2_COMMIT_MESSAGE,
|
||||
"paths": [S2_DOC_PATH],
|
||||
# Audited advisory-only skip (recorded as `bypassed` in the ledger) — the
|
||||
# scenario's subject is the triad+scope organ, not the advisory pre-review.
|
||||
"skip_advisory_review": True,
|
||||
# The post-commit hermetic pytest is out of scope for a smoke that proves the
|
||||
# review organ; the skip is recorded in the commit attempt.
|
||||
"skip_tests": True,
|
||||
"goal": "Land a doc-only smoke note through the full triad+scope review organ.",
|
||||
"scope": f"{S2_DOC_PATH} only.",
|
||||
}},
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.serial
|
||||
def test_s1_boot_identity_and_task_contract(e2e_clone, tmp_path_factory):
|
||||
require_lane(LANE_MOCK)
|
||||
root = tmp_path_factory.mktemp("s1")
|
||||
with ScriptedStubModel(S1_SCRIPT) as stub:
|
||||
server = start_server(e2e_clone, root, keyless_settings(stub))
|
||||
try:
|
||||
# Boot + identity: the frozen readiness contract, and the attestation the
|
||||
# readiness path took (runtime identity == the clone it booted from).
|
||||
state = server._state()
|
||||
assert supervisor_state_is_ready(state), state
|
||||
attestation = server.attestation
|
||||
assert attestation.get("ok") is True, attestation
|
||||
assert re.fullmatch(r"[0-9a-f]{40}", str(attestation.get("repo_head") or "")), attestation
|
||||
assert attestation.get("runtime_version") == attestation.get("repo_version")
|
||||
|
||||
# Contract: one scripted task to completion over the same HTTP surface the
|
||||
# UI posts to.
|
||||
task_id = submit_running(server, "List the repository root and finish.")
|
||||
result = server.wait_task(task_id, timeout=300)
|
||||
assert result.get("status") == "completed", result
|
||||
|
||||
# Durable truth, not the HTTP answer: task_results/<id>.json.
|
||||
oracle = ArtifactOracle(server.data_root)
|
||||
stored = wait_durable_result(oracle, task_id)
|
||||
assert stored.get("task_id") == task_id, stored
|
||||
assert stored.get("status") == "completed", stored
|
||||
assert str(stored.get("result") or "").strip(), "durable result text is empty"
|
||||
json.loads(oracle.task_result_bytes(task_id)) # bytes on disk are valid JSON
|
||||
|
||||
# The queue drained and the stub actually drove the loop.
|
||||
assert wait_until(lambda: task_id not in oracle.running_ids(), 60)
|
||||
kinds = stub.kinds()
|
||||
assert "agent" in kinds and "final" in kinds, kinds
|
||||
assert stub.script_consumed(), "S1 script was not fully consumed"
|
||||
assert oracle.events(), "events.jsonl is empty after a completed task"
|
||||
finally:
|
||||
server.stop()
|
||||
|
||||
|
||||
@pytest.mark.serial
|
||||
def test_s2_commit_reviewed_triad_and_scope_pass_on_doc_only_diff(e2e_clone, tmp_path_factory):
|
||||
require_lane(LANE_MOCK)
|
||||
root = tmp_path_factory.mktemp("s2")
|
||||
with ScriptedStubModel(S2_SCRIPT) as stub:
|
||||
settings = keyless_settings(
|
||||
stub,
|
||||
# The review organ needs the self-modification surface: advanced runtime
|
||||
# (light restricts repo writes), BLOCKING enforcement so the landed commit
|
||||
# PROVES the organ passed rather than being waved through.
|
||||
OUROBOROS_RUNTIME_MODE="advanced",
|
||||
OUROBOROS_REVIEW_ENFORCEMENT="blocking",
|
||||
)
|
||||
server = start_server(e2e_clone, root, settings)
|
||||
try:
|
||||
task_id = submit_running(
|
||||
server,
|
||||
"Write the smoke note and land it through commit_reviewed, then finish.",
|
||||
)
|
||||
result = server.wait_task(task_id, timeout=600)
|
||||
assert result.get("status") == "completed", result
|
||||
|
||||
oracle = ArtifactOracle(server.data_root)
|
||||
stored = wait_durable_result(oracle, task_id)
|
||||
assert stored.get("status") == "completed", stored
|
||||
|
||||
# The review organ ran: the stub answered a triad packet AND a scope packet.
|
||||
kinds = stub.kinds()
|
||||
assert "triad_review" in kinds, kinds
|
||||
assert "scope_review" in kinds, kinds
|
||||
|
||||
# The commit LANDED in the isolated clone — under blocking enforcement this
|
||||
# is only reachable through PASS verdicts from both organs.
|
||||
log_output = subprocess.run(
|
||||
["git", "log", "-n", "5", "--format=%s"],
|
||||
cwd=str(e2e_clone), check=True, capture_output=True, text=True,
|
||||
).stdout
|
||||
assert S2_COMMIT_MESSAGE in log_output, log_output
|
||||
committed_doc = subprocess.run(
|
||||
["git", "show", f"HEAD:{S2_DOC_PATH}"],
|
||||
cwd=str(e2e_clone), check=False, capture_output=True, text=True,
|
||||
)
|
||||
assert committed_doc.returncode == 0, "smoke doc is not in the committed tree"
|
||||
|
||||
# Durable review evidence lives in the task's FORKED drive root
|
||||
# (state/headless_tasks/<id>/data — headless-task isolation on this tree):
|
||||
# the audited advisory bypass and the scope round.
|
||||
task_oracle = oracle.task_drive(task_id)
|
||||
assert task_oracle.data_root != oracle.data_root, (
|
||||
"task drive root missing — headless drive layout changed?")
|
||||
runs = task_oracle.advisory_review().get("advisory_runs") or []
|
||||
bypassed = [r for r in runs if isinstance(r, dict) and r.get("status") == "bypassed"]
|
||||
assert bypassed, f"no bypassed advisory run in the task ledger: {runs!r}"
|
||||
assert bypassed[0].get("commit_message") == S2_COMMIT_MESSAGE, bypassed[0]
|
||||
assert task_oracle.events("advisory_review_bypassed"), "bypass event missing"
|
||||
assert task_oracle.events("scope_review_complete"), "scope completion event missing"
|
||||
finally:
|
||||
server.stop()
|
||||
Loading…
Add table
Add a link
Reference in a new issue