diff --git a/devtools/e2e_live/run_live_lanes.py b/devtools/e2e_live/run_live_lanes.py index 737ebf337..a351124a9 100644 --- a/devtools/e2e_live/run_live_lanes.py +++ b/devtools/e2e_live/run_live_lanes.py @@ -14,7 +14,8 @@ clean DETACHED clone of ``--seed`` (a commit or ref of the source; never the ope worktree, so concurrent edits cannot dirty it), the effective settings written from the TREE'S DEFAULTS (D-09) with the budget knobs as settings keys (never env), and the lane pool. ``--total-budget`` is a RUN-WIDE cap kept by ``RunBudget``: an attempt is scheduled only while -spent + reserved stays under it and every lane's TOTAL_BUDGET is the headroom left at its start. +spent + reserved stays under it and every lane's TOTAL_BUDGET is its OWN reservation — an immutable +ceiling disjoint from every other lane's, so the ceilings in flight plus the spend never exceed the cap. The run-root template is redacted (the key value lives only in each lane's 0600 settings file and is disclosed by fingerprint). The manifest names the model from the APPLIED settings file, not argv. Every lane leaves ``lanes/_a/result.json`` (checks, digests, grants by @@ -81,7 +82,8 @@ SEED_POLICY = "detached_clone_of_ref" # A lane's TOTAL_BUDGET must stay POSITIVE: the runtime reads a non-positive value as "no finite # global budget" (``settings_setup_contract.resolve_total_budget_usd``), the opposite of a cap. LANE_BUDGET_FLOOR_USD = 0.01 -RESERVATION_RULE = "per_task_usd x root_tasks (children spend under their root task's ceiling)" +RESERVATION_RULE = ("per_task_usd x root_tasks (children spend under their root task's ceiling); " + "the lane's TOTAL_BUDGET is that reservation, so in-flight ceilings + spend <= cap") def _log(msg: str) -> None: @@ -96,7 +98,7 @@ def parse_args(argv: list[str] | None = None) -> argparse.Namespace: ap.add_argument("--attempts", type=int, default=1, help="attempts per scenario (every one runs and is recorded)") ap.add_argument("--pass-of", type=int, default=1, help="passes needed for a scenario verdict") ap.add_argument("--total-budget", type=float, default=100.0, - help="RUN-WIDE USD cap shared by every lane (each lane's TOTAL_BUDGET is the headroom at its start)") + help="RUN-WIDE USD cap shared by every lane (each lane's TOTAL_BUDGET is its own reservation)") ap.add_argument("--per-task-usd", type=float, default=8.0, help="OUROBOROS_PER_TASK_COST_USD in the lane settings; also the per-root-task reservation unit") ap.add_argument("--task-timeout", type=int, default=1500) @@ -154,8 +156,8 @@ def effective_settings(args: argparse.Namespace, key: str) -> dict: """The run template (D-09: the defaults of the tree under test, never the owner's live settings). Every model slot is written explicitly so the manifest can name it from the FILE; in stub mode the slots name the loopback stub, which IS the model of that run. The - template's TOTAL_BUDGET is the RUN cap; each lane rewrites it with its own headroom - (``RunBudget.headroom``) before its server starts.""" + template's TOTAL_BUDGET is the RUN cap; each lane rewrites it with its own ceiling + (``RunBudget.ceiling``: its reservation) before its server starts.""" slots = stub_lane.STUB_MODEL_SLOTS if args.stub else declared_model_settings({}) overrides = { **slots, @@ -247,6 +249,10 @@ class RunBudget: reservation <= cap``; the first refusal halts scheduling for the rest of the run and every later attempt is recorded ``not_run``. ``spent`` is re-read from the lanes' durable usage on every question, so a lane overrunning its reservation is seen by the next admission. + Each lane's TOTAL_BUDGET is its OWN reservation (``ceiling``): the ceilings in flight are + disjoint and, by the admission rule, their sum plus the spend never exceeds the cap — the + first draft handed every lane ``cap - others' reservations``, which two concurrent lanes + could sum above the cap. """ def __init__(self, cap_usd: float, per_task_usd: float, @@ -289,12 +295,12 @@ class RunBudget: self._live[job] = (pathlib.Path(data_root), need) return True, facts - def headroom(self, job: tuple) -> float: - """The lane's TOTAL_BUDGET: the cap minus what is spent minus the OTHER in-flight - reservations — never the whole cap while anyone else is running, never non-positive.""" + def ceiling(self, job: tuple) -> float: + """The lane's TOTAL_BUDGET: its own reservation — immutable, disjoint from every other + lane's, never non-positive (the runtime reads a non-positive budget as NO cap).""" with self._lock: - spent, _unknown = self._spent_locked() - return max(LANE_BUDGET_FLOOR_USD, round(self.cap - spent - self._reserved_locked(except_job=job), 4)) + entry = self._live.get(job) + return max(LANE_BUDGET_FLOOR_USD, round(entry[1] if entry else 0.0, 4)) def settle(self, job: tuple) -> None: with self._lock: @@ -592,8 +598,8 @@ def run_lane(job: tuple[str, int], args: argparse.Namespace, out: pathlib.Path, cfg.update(scenario.overrides(child_model)) if args.profile == "wiring": cfg["OUROBOROS_REVIEW_ENFORCEMENT"] = "advisory" - # The lane's ceiling is the run's REMAINING headroom at this moment, never the whole cap. - cfg["TOTAL_BUDGET"] = budget.headroom(job) + # The lane's ceiling is its own reservation: disjoint from the other lanes', never the whole cap. + cfg["TOTAL_BUDGET"] = budget.ceiling(job) row["budget"] = {"reservation_usd": budget.reservation(scenario.root_tasks), "lane_total_budget_usd": cfg["TOTAL_BUDGET"], "per_task_usd": float(args.per_task_usd)} sha = write_settings(settings_path, cfg) @@ -687,6 +693,7 @@ def run_lane(job: tuple[str, int], args: argparse.Namespace, out: pathlib.Path, lambda: not harness.pids_with_env_value(str(data_root)), 30)) if not row["no_orphans_after_stop"] and row["status"] == "pass": row["status"] = "fail" + row["reason_code"] = "checks_failed" # the index never carries a failed row with an empty reason row["checks"]["no_orphans_after_stop"] = row["no_orphans_after_stop"] row["budget"]["spent_usd"], row["budget"]["unknown_cost_rows"] = lane_spend(data_root) row["ended_at"], row["duration_sec"] = now_iso(), round(time.time() - started, 1) diff --git a/docs/DEVELOPMENT.md b/docs/DEVELOPMENT.md index a896e208a..083b74722 100644 --- a/docs/DEVELOPMENT.md +++ b/docs/DEVELOPMENT.md @@ -1280,8 +1280,9 @@ cap: a ledger sums the lanes' durable `llm_usage` costs, reserves under their root's `OUROBOROS_PER_TASK_COST_USD` fence — SK1 two), schedules a new attempt only while `spent + reserved + reservation ≤ cap`, halts scheduling at the first refusal (later attempts are recorded `not_run` with -`reason_code=budget_cap`), and writes each lane's TOTAL_BUDGET as the headroom -left at its start; the manifest records the cap, the spend, the reservation rule +`reason_code=budget_cap`), and writes each lane's TOTAL_BUDGET as its OWN reservation — an +immutable ceiling disjoint from every other lane's, so the ceilings in flight plus the +spend never exceed the cap; the manifest records the cap, the spend, the reservation rule and the stop reason. `--per-task-usd` (default 8) is the runtime's per-root-task fence: with the tree's default review panel a blocking triad that includes claude-opus-5 plus the scope review exceeded $8 on SM1 in the first paid run diff --git a/tests/test_e2e_live_runner.py b/tests/test_e2e_live_runner.py index cbc1c097c..2ca40483d 100644 --- a/tests/test_e2e_live_runner.py +++ b/tests/test_e2e_live_runner.py @@ -51,11 +51,11 @@ def _git_seed(root: pathlib.Path, *, dirty: bool = False) -> pathlib.Path: def _fake_lane(job, args, out, template, stagger, states, seed, budget=None, *, key="", seed_sha=""): sid, attempt = job lane = out / "lanes" / f"{sid}_a{attempt}" - headroom = budget.headroom(job) if budget is not None else None # the real lane reads it before spending + ceiling = budget.ceiling(job) if budget is not None else None # the real lane reads it before spending lane.mkdir(parents=True) row = {"scenario": sid, "attempt": attempt, "status": "pass", "checks": {"fake": True}, "error": "", "duration_sec": 0.1, "model_slots": {"OUROBOROS_MODEL": template.get("OUROBOROS_MODEL")}, - "lane_total_budget_usd": headroom, + "lane_total_budget_usd": ceiling, "template_has_key": "OPENROUTER_API_KEY" in template, "key_handed": bool(key), "seed_sha": seed_sha} (lane / "result.json").write_text(json.dumps(row), encoding="utf-8") return row @@ -233,18 +233,21 @@ def test_lane_spend_sums_durable_llm_usage_rows_and_counts_unknown_costs(tmp_pat def test_run_budget_reservation_rule_halts_new_attempts_at_the_cap(tmp_path): """spent (durable, re-read) + reserved (in flight) + this attempt's reservation must fit the - cap; the first refusal halts the rest of the run; a lane's TOTAL_BUDGET is the headroom - minus the OTHER reservations at its start.""" + cap; the first refusal halts the rest of the run; a lane's TOTAL_BUDGET is its OWN reservation, + so the ceilings of the lanes in flight are disjoint and their sum plus the spend never + exceeds the cap (the first draft handed each lane cap - others' reservations: two + concurrent lanes could sum above the cap).""" spend = {} budget = run_live_lanes.RunBudget(20.0, 8.0, reader=lambda root: (spend.get(root.name, 0.0), 0)) assert budget.reservation(1) == 8.0 and budget.reservation(2) == 16.0 and budget.reservation(0) == 8.0 ok, facts = budget.admit(("SM1", 1), 1, tmp_path / "a") assert ok and facts == {"cap_usd": 20.0, "spent_usd": 0.0, "reserved_usd": 0.0, "reservation_usd": 8.0, "unknown_cost_rows": 0} - assert budget.headroom(("SM1", 1)) == 20.0 # nobody else in flight, nothing spent + assert budget.ceiling(("SM1", 1)) == 8.0 # its own reservation, never the whole cap ok, facts = budget.admit(("SW1", 1), 1, tmp_path / "b") assert ok and facts["reserved_usd"] == 8.0 - assert budget.headroom(("SW1", 1)) == 12.0 # the other lane's reservation is excluded + assert budget.ceiling(("SW1", 1)) == 8.0 # disjoint from lane a: 8 + 8 + spent 0 <= cap 20 + assert budget.ceiling(("SM1", 1)) + budget.ceiling(("SW1", 1)) <= 20.0 spend["a"] = 5.0 # lane a spends while in flight: visible now ok, facts = budget.admit(("SK1", 1), 2, tmp_path / "c") # 5 + 16 + 16 > 20 assert not ok and facts["spent_usd"] == 5.0 and facts["halt"]["first_refused"] == "SK1_a1" @@ -256,11 +259,15 @@ def test_run_budget_reservation_rule_halts_new_attempts_at_the_cap(tmp_path): assert snap["spent_usd"] == 5.0 and snap["reserved_usd"] == 0.0 and snap["lanes_settled"] == 2 assert snap["halted"] and snap["halt"]["reason"] == "budget_cap" and snap["attempts_not_run"] == ["SK1_a1", "SM1_a2"] assert snap["reservation_rule"] == run_live_lanes.RESERVATION_RULE - # The floor: a lane's TOTAL_BUDGET is never non-positive (the runtime reads that as NO cap), - # even when an in-flight lane has already overrun the whole cap. + # The ceiling ignores what OTHER lanes spend (it is this lane's reservation), and the floor + # keeps it positive (the runtime reads a non-positive TOTAL_BUDGET as NO cap). tiny = run_live_lanes.RunBudget(10.0, 8.0, reader=lambda root: (20.0, 0)) assert tiny.admit(("SM1", 1), 1, tmp_path / "x")[0] - assert tiny.headroom(("SM1", 1)) == run_live_lanes.LANE_BUDGET_FLOOR_USD + assert tiny.ceiling(("SM1", 1)) == 8.0 + assert tiny.ceiling(("never", 9)) == run_live_lanes.LANE_BUDGET_FLOOR_USD # not admitted: the floor, not the cap + micro = run_live_lanes.RunBudget(10.0, 0.001, reader=lambda root: (0.0, 0)) + assert micro.admit(("SM1", 1), 1, tmp_path / "y")[0] + assert micro.ceiling(("SM1", 1)) == run_live_lanes.LANE_BUDGET_FLOOR_USD # --------------------------------------------------------------------------- # @@ -446,8 +453,8 @@ def test_dispatch_verdict_requires_ok_status_and_the_exact_echo(): assert verdict == {"row_present": True, "status": "ok", "generation": gen, "generation_ok": True, "physical_dispatch": True, "echo_ok": True} assert scenarios.dispatch_verdict([], scenarios.SK1_ECHO_EXPECTED)["row_present"] is False - assert scenarios.SK1_ECHO_EXPECTED == "echo: ping-e2e-live" and scenarios.SK1_ECHO_MESSAGE in scenarios.SK1_PLUGIN or True - assert f"'{scenarios.SK1_ECHO_MESSAGE}'" in scenarios.SCENARIOS["SK1"].stub_script(REPO_ROOT)["agent"][4]["arguments"]["message"].join(["'", "'"]) + assert scenarios.SK1_ECHO_EXPECTED == f"echo: {scenarios.SK1_ECHO_MESSAGE}" + assert scenarios.SCENARIOS["SK1"].stub_script(REPO_ROOT)["agent"][4]["arguments"]["message"] == scenarios.SK1_ECHO_MESSAGE def test_commit_refusal_facts_name_every_typed_refusal(): @@ -704,7 +711,7 @@ def test_run_wide_cap_stops_new_attempts_and_records_not_run_rows(tmp_path, monk "verdict": "fail"} rows = {json.loads(p.read_text(encoding="utf-8"))["attempt"]: json.loads(p.read_text(encoding="utf-8")) for p in out.glob("lanes/SM1_*/result.json")} - assert rows[1]["lane_total_budget_usd"] == 20.0 and rows[2]["lane_total_budget_usd"] == 15.0 + assert rows[1]["lane_total_budget_usd"] == 8.0 and rows[2]["lane_total_budget_usd"] == 8.0 # each: its reservation refused = json.loads((out / "lanes" / "SK1_a1" / "result.json").read_text(encoding="utf-8")) assert refused["status"] == "not_run" and refused["reason_code"] == "budget_cap" assert refused["refusal"]["code"] == "budget_cap" and refused["budget"]["halt"]["first_refused"] == "SK1_a1" @@ -800,7 +807,8 @@ def test_ui_client_prefers_the_suite_interface_when_it_has_this_surface(monkeypa @pytest.mark.serial def test_stub_sm1_end_to_end_on_a_real_isolated_server(tmp_path): """Real server, loopback stub model, no key: the commit lands through the review organ - (both stylesheets, tests preflight included), the durable rows and receipts exist, the + (both stylesheets; the stub rehearsal skips the tests preflight — the disclosed residual), + the durable rows and receipts exist, the seed is a clean detached clone of this tree's HEAD and the manifest names the stub as the model.""" if str(os.environ.get("OUROBOROS_E2E_DEEP") or "").strip().lower() != "mock":