From 90f02b0228fa7ee32682e1fca219e23383fe17e1 Mon Sep 17 00:00:00 2001 From: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> Date: Sat, 5 Sep 2026 14:39:34 +0000 Subject: [PATCH] e2e_live: reserve 2x per-task per root task (inverse of the product's hard stop) The rc.14 paid run (per-task $20) failed lane SM1_a3 with budget_exhausted at $10.21 accounted while its cap was $20: the product's in-task ceiling is min(cost_hard_stop_pct of the GLOBAL remaining at task start, per-task cap - planning margin), and in a lane the global remaining IS the lane budget = the 1x reservation, so every root task's ceiling collapsed to 50% of its cap. On a real install the global budget is far larger and the root-cap axis binds. Second, --self-mod runs the post-task evolution cycle as another root task under the same per-task fence (SM1_a3: task $10.2 + cycle ~$8 of $20), so a 1x reservation under-reserves the lane's legitimate spend. The reservation is now max(LANE_BUDGET_FLOOR_USD, HARD_STOP_INVERSE x per_task_usd x root_tasks), where HARD_STOP_INVERSE = 100 / the product's _DEFAULT_COST_HARD_STOP_PCT imported from ouroboros.task_pacing (2 today; it follows the product's default rather than copying the number). The lane's TOTAL_BUDGET stays equal to its reservation, so the ceilings in flight remain disjoint and settled spend + in-flight ceilings <= cap. RESERVATION_RULE (the manifest's reservation_rule), the docstrings, the --per-task-usd help and the handbook paragraph state the rule and both reasons. Tests: the ledger pins use per-task $4 so the $8 unit and every admission number are unchanged; a new EQUALITY pin drives run_lane up to the written settings file and asserts that per-task $20 with one root reserves $40 and that exactly $40 reaches the lane's settings.json as TOTAL_BUDGET (never the template's $100 run cap), and that the factor equals 100 / the product's default cost_hard_stop_pct. Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> --- devtools/e2e_live/run_live_lanes.py | 49 +++++++++++++++++------ docs/DEVELOPMENT.md | 14 +++++-- tests/test_e2e_live_runner.py | 61 +++++++++++++++++++++++++---- 3 files changed, 100 insertions(+), 24 deletions(-) diff --git a/devtools/e2e_live/run_live_lanes.py b/devtools/e2e_live/run_live_lanes.py index 5fa131969..0cf824869 100644 --- a/devtools/e2e_live/run_live_lanes.py +++ b/devtools/e2e_live/run_live_lanes.py @@ -17,7 +17,11 @@ DEFAULTS (D-09) with the budget knobs as settings keys (never env), and the lane fit next to the settled spend is refused (a ``not_run`` row, PER attempt — no run-wide halt); one that fits the cap but not the reservations in flight WAITS for a lane to settle and asks again; every lane's TOTAL_BUDGET is its OWN reservation — an immutable ceiling disjoint from every other lane's, -so the settled spend plus the ceilings in flight never exceed the cap. +so the settled spend plus the ceilings in flight never exceed the cap. The reservation is +``HARD_STOP_INVERSE x --per-task-usd x root tasks`` (``RESERVATION_RULE``): the product stops a task +at ``cost_hard_stop_pct`` (default 50%) of the GLOBAL remaining, which in a lane IS the lane budget, +so a 1x reservation halved every root task's in-task ceiling, and ``--self-mod`` adds the post-task +evolution cycle as a second root task under the same per-task fence. The run-root template is redacted (the key value lives only in each lane's 0600 settings file and is disclosed by fingerprint). The manifest names the model from the APPLIED settings file, not argv. Every lane leaves ``lanes/_a/result.json`` (checks, digests, grants by @@ -70,6 +74,7 @@ from devtools.e2e_live import stub_lane from devtools.e2e_live.scenarios import SCENARIOS, LaneContext, diff_sha256, head_sha, now_iso from devtools.e2e_live.ui_probe import resolve_ui_client from ouroboros.provider_models import ALL_PROVIDER_CREDENTIAL_KEYS, declared_model_settings +from ouroboros.task_pacing import _DEFAULT_COST_HARD_STOP_PCT MAX_LANES = 6 STAGGER_BOUNDS = (2.0, 3.0) @@ -95,8 +100,21 @@ PROCFS_AVAILABLE = os.path.isdir("/proc") # the orphan scan reads /proc enviro # A lane's TOTAL_BUDGET must stay POSITIVE: the runtime reads a non-positive value as "no finite # global budget" (``settings_setup_contract.resolve_total_budget_usd``), the opposite of a cap. LANE_BUDGET_FLOOR_USD = 0.01 -RESERVATION_RULE = ("per_task_usd x root_tasks (children spend under their root task's ceiling); " - "the lane's TOTAL_BUDGET is that reservation, so settled spend + in-flight ceilings <= cap") +# The reservation unit is per_task_usd x the INVERSE of the product's default in-task hard stop +# (``task_pacing.resolve_cost_ceiling``: min(cost_hard_stop_pct of the GLOBAL remaining at task +# start, root cap - planning margin)). In a lane the global remaining IS its TOTAL_BUDGET = its +# reservation, so a 1x reservation collapsed every root task's ceiling to 50% of the per-task cap +# (rc.14 paid run: SM1_a3 'budget_exhausted' at $10.21 under a $20 cap, two review rounds); on a +# real install the global budget is far larger and the root-cap axis binds. Second, --self-mod +# runs the post-task evolution cycle as ANOTHER root task under the same per-task fence (SM1_a3: +# task $10.2 + cycle ~$8 of a $20 reservation). Imported, not copied: the factor follows the product. +HARD_STOP_INVERSE = 100.0 / _DEFAULT_COST_HARD_STOP_PCT +RESERVATION_RULE = (f"max({LANE_BUDGET_FLOOR_USD:g}, {HARD_STOP_INVERSE:g} x per_task_usd x root_tasks) — the factor " + f"is 100 / the product's default cost_hard_stop_pct ({_DEFAULT_COST_HARD_STOP_PCT}%): in a lane the " + "global remaining IS the lane budget, so a 1x reservation would halve every root task's in-task " + "ceiling, and --self-mod runs the post-task evolution cycle as a second root task under the same " + "per-task fence. The lane's TOTAL_BUDGET is that reservation (children spend under their root " + "task's ceiling), so settled spend + in-flight ceilings <= cap") def _log(msg: str) -> None: @@ -113,7 +131,9 @@ def parse_args(argv: list[str] | None = None) -> argparse.Namespace: ap.add_argument("--total-budget", type=float, default=100.0, help="RUN-WIDE USD cap shared by every lane (each lane's TOTAL_BUDGET is its own reservation)") ap.add_argument("--per-task-usd", type=float, default=8.0, - help="OUROBOROS_PER_TASK_COST_USD in the lane settings; also the per-root-task reservation unit") + help="OUROBOROS_PER_TASK_COST_USD in the lane settings; each root task reserves " + f"{HARD_STOP_INVERSE:g} x this (100 / the product's default cost_hard_stop_pct: in a lane " + "the global remaining IS the lane budget; --self-mod adds the evolution cycle as a root task)") ap.add_argument("--task-timeout", type=int, default=1500) ap.add_argument("--ready-timeout", type=int, default=300) ap.add_argument("--profile", choices=("full", "wiring"), default="full", @@ -259,10 +279,15 @@ class RunBudget: """The RUN-WIDE ledger behind ``--total-budget`` (the first paid run copied the whole cap into every lane: nine attempts, nine $100 ceilings, no run cap at all). - Reservation rule, per attempt: ``per_task_usd x root_tasks`` — the runtime fences each - ROOT task tree at ``OUROBOROS_PER_TASK_COST_USD`` and children spend under their root's - ceiling, so SW1 (one root, two scouts) reserves for one root and SK1 (author + dispatch) - for two. Admission asks two questions, PER attempt: ``spent + reservation > cap`` — it can + Reservation rule, per attempt: ``HARD_STOP_INVERSE x per_task_usd x root_tasks`` — the runtime + fences each ROOT task tree at ``OUROBOROS_PER_TASK_COST_USD`` and children spend under their + root's ceiling, so SW1 (one root, two scouts) reserves for one root and SK1 (author + dispatch) + for two. The factor is the inverse of the product's default ``cost_hard_stop_pct``: the in-task + ceiling is min(that pct of the GLOBAL remaining at task start, per-task cap - planning margin), + and in a lane the global remaining IS the lane budget, so a 1x reservation halved every root + task's ceiling (rc.14: SM1_a3 'budget_exhausted' at $10.21 of a $20 cap); with ``--self-mod`` + the post-task evolution cycle is a second root task under the same per-task fence (SM1_a3: + $10.2 + ~$8 of $20). Admission asks two questions, PER attempt: ``spent + reservation > cap`` — it can NEVER fit, refused and recorded ``not_run`` (a later attempt with a smaller reservation is asked on its own; nothing halts the run); otherwise ``spent + reserved(in flight) + reservation > cap`` — it cannot fit YET, so it waits on the ledger and re-asks after every @@ -290,10 +315,10 @@ class RunBudget: self.not_run: list[str] = [] def reservation(self, root_tasks: int) -> float: - """The ONE effective ceiling of an attempt: ``per_task_usd x root_tasks``, floored at - ``LANE_BUDGET_FLOOR_USD`` and never rounded upward — admission, the stored reservation, - the lane's TOTAL_BUDGET and the reports all carry this same number.""" - return max(LANE_BUDGET_FLOOR_USD, self.per_task * max(1, int(root_tasks or 1))) + """The ONE effective ceiling of an attempt: ``HARD_STOP_INVERSE x per_task_usd x root_tasks``, + floored at ``LANE_BUDGET_FLOOR_USD`` and never rounded upward — admission, the stored + reservation, the lane's TOTAL_BUDGET and the reports all carry this same number.""" + return max(LANE_BUDGET_FLOOR_USD, HARD_STOP_INVERSE * self.per_task * max(1, int(root_tasks or 1))) def _spent_locked(self) -> tuple[float, int]: spent, unknown = 0.0, 0 diff --git a/docs/DEVELOPMENT.md b/docs/DEVELOPMENT.md index 355698724..dd66023e6 100644 --- a/docs/DEVELOPMENT.md +++ b/docs/DEVELOPMENT.md @@ -1277,8 +1277,14 @@ only in each lane's 0600 settings file, disclosed by fingerprint as the runtime grant); the preflight takes `min(key limit remaining, account credits)` and refuses below `--min-credit-usd`. `--total-budget` (default 100) is the RUN-WIDE cap: a ledger sums the lanes' durable `llm_usage` costs, reserves -`--per-task-usd × root tasks` per attempt (SM1 and SW1 one root — scouts spend -under their root's `OUROBOROS_PER_TASK_COST_USD` fence — SK1 two), admits an +`max(0.01, HARD_STOP_INVERSE × --per-task-usd × root tasks)` per attempt (SM1 and SW1 one root — scouts spend +under their root's `OUROBOROS_PER_TASK_COST_USD` fence — SK1 two; the factor is +100 / the product's default `cost_hard_stop_pct`, imported from `task_pacing`, 2 today: +the in-task ceiling is min(that pct of the GLOBAL remaining at task start, per-task cap − +planning margin) and in a lane the global remaining IS the lane budget, so a 1× reservation +halved every root task's ceiling — rc.14 paid run, SM1_a3 `budget_exhausted` at $10.21 of a +$20 cap; and `--self-mod` runs the post-task evolution cycle as a second root task under the +same per-task fence, $10.2 + ~$8 of that $20), admits an attempt only while `spent + reserved(in flight) + reservation ≤ cap` — an attempt that cannot fit YET (blocked only by reservations in flight) waits for a settle and asks again, one that can NEVER fit (`spent + reservation > cap`) is refused and recorded `not_run` with @@ -1286,8 +1292,8 @@ one that can NEVER fit (`spent + reservation > cap`) is refused and recorded `no is asked on its own; the manifest carries `refusals` and `first_refused`), and writes each lane's TOTAL_BUDGET as its OWN reservation — an immutable ceiling disjoint from every other lane's, so the settled spend plus the ceilings in flight never exceed the cap; the manifest records the cap, the spend, the reservation rule -and the stop reason. `--per-task-usd` (default 8) is the runtime's per-root-task -fence: with the tree's default review panel a blocking triad that includes +and the stop reason. `--per-task-usd` (default 8; each root task reserves 2× it) is the +runtime's per-root-task fence: with the tree's default review panel a blocking triad that includes claude-opus-5 plus the scope review exceeded $8 on SM1 in the first paid run (lanes spent $2–6 and still hit the fence, which counts reserved upper bounds), so size it to the review panel (16–20 for a blocking SM1) and let the run-wide diff --git a/tests/test_e2e_live_runner.py b/tests/test_e2e_live_runner.py index b6ec7e1b6..36bbfe3ef 100644 --- a/tests/test_e2e_live_runner.py +++ b/tests/test_e2e_live_runner.py @@ -294,9 +294,11 @@ def test_run_budget_waits_on_in_flight_reservations_and_refuses_only_what_can_ne paid run wrote SW1/SK1 off at t=+21 min behind two SM1 reservations still in flight); spent only grows, so a waiter can end refused with its wait recorded. A lane's TOTAL_BUDGET is its OWN reservation, so the ceilings in flight are disjoint and settled spend + in-flight ceilings - never exceeds the cap (the first draft handed each lane cap - others' reservations).""" + never exceeds the cap (the first draft handed each lane cap - others' reservations). The + reservation unit is HARD_STOP_INVERSE (2) x per-task: $4 per task reserves $8 per root.""" spend = {} - budget = run_live_lanes.RunBudget(20.0, 8.0, reader=lambda root: (spend.get(root.name, 0.0), 0)) + budget = run_live_lanes.RunBudget(20.0, 4.0, reader=lambda root: (spend.get(root.name, 0.0), 0)) + assert run_live_lanes.HARD_STOP_INVERSE * 4.0 == 8.0 assert budget.reservation(1) == 8.0 and budget.reservation(2) == 16.0 and budget.reservation(0) == 8.0 ok, facts = budget.admit(("SM1", 1), 1, tmp_path / "a") assert ok and facts == {"cap_usd": 20.0, "spent_usd": 0.0, "reserved_usd": 0.0, "reservation_usd": 8.0, @@ -334,7 +336,7 @@ def test_run_budget_waits_on_in_flight_reservations_and_refuses_only_what_can_ne assert snap["reservation_rule"] == run_live_lanes.RESERVATION_RULE # The ceiling ignores what OTHER lanes spend (it is this lane's reservation), and the floor # keeps it positive (the runtime reads a non-positive TOTAL_BUDGET as NO cap). - tiny = run_live_lanes.RunBudget(10.0, 8.0, reader=lambda root: (20.0, 0)) + tiny = run_live_lanes.RunBudget(10.0, 4.0, reader=lambda root: (20.0, 0)) assert tiny.admit(("SM1", 1), 1, tmp_path / "x")[0] assert tiny.ceiling(("SM1", 1)) == 8.0 assert tiny.ceiling(("never", 9)) == run_live_lanes.LANE_BUDGET_FLOOR_USD # not admitted: the floor, not the cap @@ -355,8 +357,8 @@ def test_run_budget_waits_on_in_flight_reservations_and_refuses_only_what_can_ne below = run_live_lanes.RunBudget(0.005, 0.001, reader=lambda root: (0.0, 0)) assert not below.admit(("SM1", 1), 1, tmp_path / "z")[0] # the floored reservation exceeds the cap: refused # Fractional reservations are never rounded upward (round(0.01006, 4) would hand out 0.0101): - # two exact 0.01006 reservations fill a 0.02012 cap and each lane receives exactly 0.01006. - frac = run_live_lanes.RunBudget(0.02012, 0.01006, reader=lambda root: (0.0, 0)) + # two exact 2 x 0.00503 reservations fill a 0.02012 cap and each lane receives exactly 0.01006. + frac = run_live_lanes.RunBudget(0.02012, 0.00503, reader=lambda root: (0.0, 0)) assert frac.admit(("SM1", 1), 1, tmp_path / "f1")[0] and frac.admit(("SM1", 2), 1, tmp_path / "f2")[0] assert frac.ceiling(("SM1", 1)) == 0.01006 and frac.ceiling(("SM1", 2)) == 0.01006 assert frac.ceiling(("SM1", 1)) + frac.ceiling(("SM1", 2)) <= 0.02012 @@ -368,6 +370,49 @@ def test_run_budget_waits_on_in_flight_reservations_and_refuses_only_what_can_ne assert not third.is_alive() and box["r"][0] +def test_reservation_is_the_inverse_hard_stop_times_per_task_and_is_the_lane_total_budget(tmp_path, monkeypatch): + """EQUALITY pins of the rc.14 finding: the product's in-task ceiling is min(cost_hard_stop_pct + of the GLOBAL remaining at task start, per-task cap - margin) and in a lane the global + remaining IS the lane budget, so a 1x reservation halved every root task's ceiling (SM1_a3 + 'budget_exhausted' at $10.21 of $20); with --self-mod the evolution cycle is a second root task + under the same fence. The factor is 100 / the product's default (imported, not copied); for + per-task $20 and one root the reservation is $40 and that exact number reaches the lane's + settings file as TOTAL_BUDGET through ``run_lane`` (never the template's run-wide cap).""" + from ouroboros import task_pacing + _short_tmp(monkeypatch) + assert run_live_lanes.HARD_STOP_INVERSE == 100 / task_pacing._DEFAULT_COST_HARD_STOP_PCT == 2.0 + rule = run_live_lanes.RESERVATION_RULE + assert rule.startswith("max(0.01, 2 x per_task_usd x root_tasks)") and "cost_hard_stop_pct (50%)" in rule + assert "global remaining IS the lane budget" in rule and "--self-mod" in rule and "second root task" in rule + budget = run_live_lanes.RunBudget(100.0, 20.0, reader=lambda root: (0.0, 0)) + assert budget.reservation(1) == 40.0 and budget.reservation(2) == 80.0 + seed = _git_seed(tmp_path) + out, job = tmp_path / "out", ("SM1", 1) + ok, facts = budget.admit(job, 1, out / "lanes" / "SM1_a1" / "data") + assert ok and facts["reservation_usd"] == 40.0 and budget.ceiling(job) == 40.0 + + class _NoServer: # the real path up to the written settings, then stop + def __init__(self, *_a, **_k) -> None: + self.base_url = "http://127.0.0.1:0" + + def start(self, **_k) -> None: + raise RuntimeError("no server in this pin: the settings file on disk is the evidence") + + monkeypatch.setattr(run_live_lanes, "IsolatedServer", _NoServer) + args = run_live_lanes.parse_args(["--per-task-usd", "20", "--total-budget", "100", "--scenarios", "SM1", + "--out", str(out), "--watch-interval", "600"]) + template = run_live_lanes.effective_settings(args, FAKE_KEY) + assert template["TOTAL_BUDGET"] == 100.0 # the run cap; every lane rewrites it with its ceiling + row = run_live_lanes.run_lane(job, args, out, template, run_live_lanes.Stagger(2.0), {}, seed, budget, + key=FAKE_KEY, seed_sha=run_live_lanes.head_sha(seed)) + applied = json.loads((out / "lanes" / "SM1_a1" / "data" / "settings.json").read_text(encoding="utf-8")) + assert applied["TOTAL_BUDGET"] == 40.0 == budget.ceiling(job) == budget.reservation(1) + assert applied["OUROBOROS_PER_TASK_COST_USD"] == 20.0 and applied["OPENROUTER_API_KEY"] == FAKE_KEY + assert row["budget"] == {"reservation_usd": 40.0, "lane_total_budget_usd": 40.0, "per_task_usd": 20.0, + "spent_usd": 0.0, "unknown_cost_rows": 0} + assert row["status"] == "infra_error" and row["refusal"]["type"] == "RuntimeError" + + # --------------------------------------------------------------------------- # # The watcher's key probe: informational, bounded, backing off, never on the tick's path # --------------------------------------------------------------------------- # @@ -426,7 +471,7 @@ def test_watcher_tick_never_waits_on_the_key_probe(capsys): release.set() thread.join(timeout=5) line = next(ln for ln in seen.splitlines() if "[watch]" in ln) - assert "spent $2.50/$50.00 reserved $8.00" in line and "SM1_a1=running scenario" in line + assert "spent $2.50/$50.00 reserved $16.00" in line and "SM1_a1=running scenario" in line # 2 x $8, one root assert "key probe pending" in line and "ALERT" not in line @@ -931,14 +976,14 @@ def test_run_root_template_is_redacted_and_the_key_reaches_only_the_lanes(tmp_pa def test_run_wide_cap_refuses_per_attempt_and_records_not_run_rows(tmp_path, monkeypatch): - """cap $20, reservation unit $8, every settled lane read back at $5, one lane (nothing in + """cap $20, per-task $4 = reservation unit $8, every settled lane read back at $5, one lane (nothing in flight): SM1_a1 (0+8), SM1_a2 (5+8) run; SK1_a1 (10+16 > 20) and SK1_a2 are refused; SW1_a1 (10+8) still RUNS after them — a refusal is per attempt, not a halt; SW1_a2 (15+8 > 20) is refused. Every refusal is a recorded row and the manifest's stop_reason.""" monkeypatch.setattr(run_live_lanes, "lane_spend", lambda root: (5.0, 0) if pathlib.Path(root).parent.exists() else (0.0, 0)) out, manifest = _fake_run(tmp_path, monkeypatch, ["--scenarios", "SM1,SK1,SW1", "--attempts", "2", "--lanes", "1", - "--total-budget", "20", "--per-task-usd", "8"], expect_rc=1) + "--total-budget", "20", "--per-task-usd", "4"], expect_rc=1) budget = manifest["extra"]["budget"] assert budget["first_refused"] == "SK1_a1" and "halted" not in budget assert [(r["attempt"], r["spent_usd"], r["reservation_usd"], r["waited_sec"]) for r in budget["refusals"]] == [