diff --git a/ouroboros/context_budget.py b/ouroboros/context_budget.py index f90acb3c9..5f9ae9fbc 100644 --- a/ouroboros/context_budget.py +++ b/ouroboros/context_budget.py @@ -36,6 +36,24 @@ OWNER_LOW_TARGET_TOKENS = 200_000 OWNER_NANO_TARGET_TOKENS = 81_920 NANO_MIN_HEADROOM_TOKENS = 8_192 +# Low-water sizing of the automatic context-reclaim pass. The TRIGGER is +# unchanged: a positive deficit against the binding boundary (the smaller known +# of the owner target T and the route capacity W), one pass per route+round. +# Only the SIZE of the requested pass changes: goal = deficit + +# ceil(boundary / RECLAIM_LOW_WATER_DIVISOR), and 0 without a deficit. A pass +# sized to the deficit alone lands exactly AT the boundary, so the next round's +# ordinary growth re-arms it (a summarizer pass nearly every round). Sized this +# way it lands about an eighth of the boundary below (~125K tokens on a 1M +# route, ~25K under the 200K Low target), so the next pass needs that much real +# growth. Structural constant, not a setting: 8 (12.5 % of the boundary) is a +# disclosed design choice, not a measured optimum; change it here and only here +# (tests/test_context_budget_ssot.py pins it). Cost: older history is condensed +# sooner and each summarizer pass is larger. The materializer, its receipts and +# the route+round latch are unchanged; the checkpoint event records requested +# margin versus achieved headroom (context_fit.measure_main_fit, +# loop_model_call._run_main_reclaim). +RECLAIM_LOW_WATER_DIVISOR = 8 + # One overflow vocabulary for every seam that must recognize a CONTEXT-WINDOW # overflow (Main provider-code precedence, the local transport, and the # summarizer split path). A provider code or message shape added here reaches diff --git a/ouroboros/context_fit.py b/ouroboros/context_fit.py index e1e4af40a..468df2d42 100644 --- a/ouroboros/context_fit.py +++ b/ouroboros/context_fit.py @@ -200,6 +200,9 @@ class MainFitMeasurement: target_deficit_tokens: Optional[int] capacity_deficit_tokens: Optional[int] reclaim_goal_tokens: int + # Low-water margin the goal carries ABOVE the deficit (0 without a deficit): + # the pass is deficit-triggered but sized to land below the boundary. + low_water_margin_tokens: int = 0 @dataclass(frozen=True) @@ -559,6 +562,22 @@ def _route_calibration_ratio( return 1.0 +def reclaim_low_water_margin( + target_total_tokens: Optional[int], capacity_total_tokens: Optional[int], +) -> int: + """Tokens a reclaim pass lands BELOW the binding boundary: ceil(boundary / divisor). + + The boundary is the smaller known positive one of owner target T and route + capacity W; 0 when neither is known. The divisor is read at call time so + the SSOT constant stays the one place to change it. + """ + from ouroboros.context_budget import RECLAIM_LOW_WATER_DIVISOR + + known = [int(value) for value in (target_total_tokens, capacity_total_tokens) + if value is not None and int(value) > 0] + return int(math.ceil(min(known) / RECLAIM_LOW_WATER_DIVISOR)) if known else 0 + + def measure_main_fit( plan: ContextFitPlan, messages: List[Dict[str, Any]], @@ -575,6 +594,9 @@ def measure_main_fit( ``drive_root=None`` reads density from the canonical host evidence root (one observation store) — a child task's own drive must not be consulted. + A positive deficit triggers at most one reclaim pass per route+round; the + requested goal is deficit + ``reclaim_low_water_margin`` so the pass lands + below the boundary instead of exactly at it (``RECLAIM_LOW_WATER_DIVISOR``). """ from ouroboros.capability_evidence import ( canonical_evidence_root, is_known, resolve_main_token_density, @@ -598,10 +620,12 @@ def measure_main_fit( capacity = int(plan.window_tokens or 0) if is_known(plan, require_fresh=True) else None target_deficit = max(0, total - target) if target is not None else None capacity_deficit = max(0, total - capacity) if capacity is not None else None - goal = max( + deficit = max( [value for value in (target_deficit, capacity_deficit) if value is not None] or [0] ) + margin = reclaim_low_water_margin(target, capacity) if deficit > 0 else 0 + goal = deficit + margin measurement = MainFitMeasurement( route_fp=str(plan.route_fp or ""), round_id=str(round_id or ""), @@ -616,6 +640,7 @@ def measure_main_fit( target_deficit_tokens=target_deficit, capacity_deficit_tokens=capacity_deficit, reclaim_goal_tokens=goal, + low_water_margin_tokens=margin, ) if goal > 0 and not automatic_pass_used: action: Literal["send", "reclaim_once", "send_target_miss"] = "reclaim_once" diff --git a/ouroboros/loop_model_call.py b/ouroboros/loop_model_call.py index b0ca03f77..9eaf669da 100644 --- a/ouroboros/loop_model_call.py +++ b/ouroboros/loop_model_call.py @@ -867,6 +867,29 @@ def _run_main_reclaim( ctx.tools._ctx.messages = ctx.messages _loop().seal_task_transcript(ctx.messages) prune_reclaim_trace_refs(ctx.tools._ctx, ctx.messages) + # Low-water facts: the pass is deficit-triggered but sized to land below the + # boundary, so the landing is re-measured on the SAME fit basis as the trigger + # and "reached the boundary" stays distinct from "achieved the margin" + # (reclaimed == deficit is AT the boundary, not below it). + deficit = max(int(measurement.target_deficit_tokens or 0), + int(measurement.capacity_deficit_tokens or 0)) + requested_margin = int(request.reclaim_goal_tokens) - deficit + landed = measurement + if receipt.status == "applied": + try: + remeasured = _loop()._measure_round_main_fit(ctx, automatic_pass_used=True) + except Exception: # telemetry only: an unmeasurable landing must not fail the pass + log.debug("Post-reclaim fit measurement unavailable", exc_info=True) + remeasured = None + landed = remeasured.measurement if remeasured is not None else None + boundary = [value for value in ( + landed.target_total_tokens, landed.capacity_total_tokens) if value is not None] if landed else [] + # None = the landing could not be measured (unknown), never "not reached". + headroom = (min(boundary) - (landed.estimated_input_tokens + landed.response_reserve_tokens) + if boundary else None) + tool_ctx = ctx.tools._ctx + previous_round = getattr(tool_ctx, "_context_reclaim_last_pass_round", None) + tool_ctx._context_reclaim_last_pass_round = int(ctx.round_idx) _loop()._emit_checkpoint_event(ctx.event_queue, ctx.task_id, ctx.drive_logs, { "type": "context_reclaim", "checkpoint_kind": "context_reclaim_automatic", @@ -878,6 +901,13 @@ def _run_main_reclaim( "reclaimed_tokens": receipt.reclaimed_tokens, "goal_reached": receipt.goal_reached, "checkpoint_ref": receipt.checkpoint_ref, + "deficit_tokens": deficit, + "requested_margin_tokens": requested_margin, + "achieved_headroom_tokens": headroom, + "boundary_reached": None if headroom is None else headroom >= 0, + "below_boundary": None if headroom is None else headroom >= max(1, requested_margin), + "rounds_since_previous_pass": ( + int(ctx.round_idx) - int(previous_round) if previous_round is not None else None), }) return receipt @@ -997,7 +1027,15 @@ def _call_round_model(ctx: _RoundModelCallContext) -> Tuple[Any, float, str]: return msg, cost, ctx.active_context_mode key = _fit_key(overflow_fit) if key not in _loop()._context_reclaim_passes(ctx.tools._ctx): - _loop()._run_main_reclaim(ctx, overflow_fit, minimum_goal_tokens=1) + # The provider proved the prediction short by an unknown amount: request a + # low-water-sized pass, never a token-sized one, so the single strict-shrink + # retry has real headroom (the goal already carries the margin when the + # measurement itself found a deficit). + from ouroboros.context_fit import reclaim_low_water_margin + + landed = overflow_fit.measurement + _loop()._run_main_reclaim(ctx, overflow_fit, minimum_goal_tokens=max( + 1, reclaim_low_water_margin(landed.target_total_tokens, landed.capacity_total_tokens))) overflow_fit = _measure_after_reclaim(ctx) if overflow_fit is None: return msg, cost, ctx.active_context_mode diff --git a/tests/test_context_budget_ssot.py b/tests/test_context_budget_ssot.py index 392b16a7a..c19e87ec1 100644 --- a/tests/test_context_budget_ssot.py +++ b/tests/test_context_budget_ssot.py @@ -24,6 +24,25 @@ def test_agent_context_budget_values_pinned(): assert cb.MAX_RECENT_CHAT_TAIL == 1000 assert cb.CHAT_ARCHIVE_SCAN_WARN_BYTES == 100_000_000 assert not hasattr(cb, "CONTEXT_SOFT_CAP_TOKENS") + # Structural low-water divisor of the automatic reclaim pass (12.5 % of the + # binding boundary): a disclosed design choice, not a setting. + assert cb.RECLAIM_LOW_WATER_DIVISOR == 8 + + +def test_reclaim_low_water_divisor_is_one_constant_read_at_call_time(monkeypatch): + """CHECKLISTS item 20: the fit consumes the SSOT name (no bare literal), reads it + at call time so changing the one constant changes every pass, and the margin + is the LAST measurement field (appended; older readers stay positional-safe).""" + from ouroboros import context_fit + + assert "RECLAIM_LOW_WATER_DIVISOR" in _src("ouroboros/context_fit.py") + assert "/ 8" not in inspect.getsource(context_fit.measure_main_fit) + assert dataclasses.fields(context_fit.MainFitMeasurement)[-1].name == "low_water_margin_tokens" + assert context_fit.reclaim_low_water_margin(200_000, 500_000) == 25_000 # target binds + assert context_fit.reclaim_low_water_margin(None, 70_000) == 8_750 # capacity alone + assert context_fit.reclaim_low_water_margin(None, None) == 0 # nothing known + monkeypatch.setattr(cb, "RECLAIM_LOW_WATER_DIVISOR", 4) + assert context_fit.reclaim_low_water_margin(200_000, 500_000) == 50_000 def test_reclaim_request_and_receipt_are_exact_frozen_records(): diff --git a/tests/test_context_fit_integration.py b/tests/test_context_fit_integration.py index d663dcc77..105d569ed 100644 --- a/tests/test_context_fit_integration.py +++ b/tests/test_context_fit_integration.py @@ -123,6 +123,138 @@ def test_owner_low_deficit_reclaim_remeasures_on_one_basis(monkeypatch, tmp_path assert after.measurement.reclaim_goal_tokens == 0 +def _growth_unit(tag: str, chars: int): + return [{ + "role": "assistant", + "content": "investigating", + "tool_calls": [{ + "id": f"call-{tag}", + "type": "function", + "function": {"name": "read_file", "arguments": "x" * chars}, + }], + }, {"role": "tool", "tool_call_id": f"call-{tag}", "content": "y" * chars}] + + +def _install_full_budget_materializer(monkeypatch): + """A13 worst case: every summary comes back at the FULL summary budget of its + source (never a tiny summary), the private checkpoint is stubbed, density 1.0.""" + from ouroboros import capability_evidence, context_compaction as cc + + monkeypatch.setattr( + capability_evidence, "resolve_main_token_density", + lambda *_a, **_kw: (1.0, "fresh_route_usage"), + ) + monkeypatch.setattr(cc, "_summarizer_spec", lambda: { + "model": "summary-model", "resolved_model": "summary-model", "provider": "test", + "route_fp": "summary-route", "effort": "low", "output_budget": 32_768, "use_local": False, + }) + monkeypatch.setattr( + cc, "_persist_reclaim_checkpoint", + lambda *_a, **_kw: {"path": "checkpoint", "sha256": "c" * 64}, + ) + monkeypatch.setattr(cc, "_call_summarizer", lambda parts, *, summary_budgets, **_kw: { + part.source_id: "s" * (4 * int(summary_budgets[part.root_id])) for part in parts + }) + + +def _simulate_growing_transcript(tmp_path, *, window: int, rounds: int, growth_chars: int): + """Owner Max on a known ``window``: fill to just under the capacity boundary, then + grow one completed tool unit per round, running the REAL fit and the REAL + materializer the way the loop does (at most one automatic pass per route+round, + the landing re-measured on the same basis). Returns (unit tokens, per-pass rows + of (round, requested margin, achieved headroom, receipt)).""" + from ouroboros import context_compaction as cc + from ouroboros.context_budget import ContextReclaimRequest + from ouroboros.context_fit import estimate_context_prompt_tokens, measure_main_fit + + plan = _plan(preferred="max", window=window) + messages = plan.messages_for("max") + unit_tokens = estimate_context_prompt_tokens(_growth_unit("probe", growth_chars)) + boundary_input = window - plan.output_reserve_tokens + filler = (boundary_input - estimate_context_prompt_tokens(messages) - unit_tokens // 2) // unit_tokens + for index in range(filler): + messages = messages + _growth_unit(f"f{index}", growth_chars) + + def fit(current, round_idx, *, used): + return measure_main_fit( + plan, current, [], drive_root=tmp_path, profile="owner_max", rendered_mode="max", + round_id=f"exec:round:{round_idx}", automatic_pass_used=used, + ) + + memo: set = set() + rows = [] + for round_idx in range(1, rounds + 1): + messages = messages + _growth_unit(f"g{round_idx}", growth_chars) + before = fit(messages, round_idx, used=False) + if before.action != "reclaim_once": + assert before.action == "send" + continue + measurement = before.measurement + request = ContextReclaimRequest( + route_fp=measurement.route_fp, round_id=measurement.round_id, + transcript_sha256=cc.context_reclaim_transcript_sha256(messages), + measurement_basis=measurement.measurement_basis, + measurement_density=measurement.measurement_density, + reclaim_goal_tokens=measurement.reclaim_goal_tokens, + ) + messages, receipt, _usage = cc.compact_tool_history_llm( + messages, request=request, drive_root=tmp_path, negative_memo=memo, + ) + after = fit(messages, round_idx, used=True).measurement + rows.append(( + round_idx, measurement.low_water_margin_tokens, + window - (after.estimated_input_tokens + after.response_reserve_tokens), receipt, + )) + return unit_tokens, rows + + +def test_low_water_margin_bounds_automatic_passes_under_a_full_budget_summarizer(monkeypatch, tmp_path): + """Anti-thrash class. A pass sized to the deficit alone lands AT the boundary, so the + next round's ordinary growth re-arms it: one summarizer pass nearly every round. + Sized deficit + boundary/RECLAIM_LOW_WATER_DIVISOR it lands below the boundary and + the next pass needs real growth. Under the A13 worst-case stub every selected unit + halves, so a pass lands about HALF the requested margin below (the receipt says + goal_reached=False): the bound proven with that stub is ceil(N*g / (margin/2)) + 1; + the ideal-summarizer bound ceil(N*g / margin) + 1 holds for the ACHIEVED headroom and + is asserted in that form (requested margin is not achieved headroom). With the + margin removed the same assertions fail.""" + import math + + from ouroboros import context_budget as cb + + _install_full_budget_materializer(monkeypatch) + window, rounds, chars = 400_000, 40, 4_000 + margin = math.ceil(window / cb.RECLAIM_LOW_WATER_DIVISOR) + + unit_tokens, passes = _simulate_growing_transcript( + tmp_path, window=window, rounds=rounds, growth_chars=chars, + ) + growth = rounds * unit_tokens + bound = math.ceil(growth / (margin // 2)) + 1 + assert 2 <= len(passes) <= bound + assert [row[1] for row in passes] == [margin] * len(passes) + assert all(row[3].status == "applied" for row in passes) + # Every pass reached the boundary; none achieved the full margin (partial shrink + # under full-budget summaries), and the receipt discloses that underlanding. + assert all(0 <= row[2] < margin for row in passes) + assert not any(row[3].goal_reached for row in passes) + achieved = min(row[2] for row in passes) + assert len(passes) <= math.ceil(growth / achieved) + 1 + gaps = [later[0] - earlier[0] for earlier, later in zip(passes, passes[1:])] + assert min(gaps) > achieved // unit_tokens + + monkeypatch.setattr(cb, "RECLAIM_LOW_WATER_DIVISOR", 10 ** 9) # margin 1: deficit-sized passes + _unit, thrash = _simulate_growing_transcript( + tmp_path, window=window, rounds=rounds, growth_chars=chars, + ) + assert len(thrash) >= rounds - 2 + assert len(thrash) > bound + assert [row[1] for row in thrash] == [1] * len(thrash) + # A deficit-sized pass lands at (here: still above) the boundary, which is exactly + # what re-arms it on the next round. + assert sum(1 for row in thrash if row[2] < 0) >= len(thrash) - 2 + + def test_target_miss_is_non_terminal_fit_evidence(monkeypatch, tmp_path): from ouroboros import capability_evidence, loop from ouroboros.context_fit import measure_main_fit diff --git a/tests/test_context_fit_v664.py b/tests/test_context_fit_v664.py index c03c3d60d..a11557a4f 100644 --- a/tests/test_context_fit_v664.py +++ b/tests/test_context_fit_v664.py @@ -4,10 +4,13 @@ from __future__ import annotations import inspect import json +import math from types import SimpleNamespace import pytest +from ouroboros.context_budget import RECLAIM_LOW_WATER_DIVISOR + def _projection(mode: str): from ouroboros.context_fit import ContextFitProjection @@ -126,10 +129,89 @@ def test_confirmed_window_below_target_wins_even_when_target_fits(monkeypatch, t measurement = disposition.measurement assert measurement.target_deficit_tokens == 0 assert measurement.capacity_deficit_tokens > 0 - assert measurement.reclaim_goal_tokens == measurement.capacity_deficit_tokens + # Deficit-triggered, low-water-sized: the goal carries an eighth of the BINDING + # boundary (the 65,550 window, not the 200K target) above the deficit. + assert measurement.low_water_margin_tokens == math.ceil(65_550 / RECLAIM_LOW_WATER_DIVISOR) + assert measurement.reclaim_goal_tokens == ( + measurement.capacity_deficit_tokens + measurement.low_water_margin_tokens + ) assert disposition.action == "reclaim_once" +@pytest.mark.parametrize("profile,preferred,window,known,boundary", [ + ("owner_max", "max", 70_000, True, 70_000), # capacity binds a Max route + ("owner_low", "low", 500_000, True, 200_000), # the economy target binds Low + ("owner_low", "low", 0, False, 200_000), # unknown capacity: the target alone binds +]) +def test_low_water_margin_follows_the_binding_boundary( + monkeypatch, tmp_path, profile, preferred, window, known, boundary, +): + plan = _plan(preferred=preferred, window=window, known=known) + messages = plan.messages_for(preferred) + [{"role": "user", "content": "x" * 600_000}] + disposition = _measure( + monkeypatch, tmp_path, plan=plan, profile=profile, mode=preferred, messages=messages, + ) + measurement = disposition.measurement + deficit = max(value for value in ( + measurement.target_deficit_tokens, measurement.capacity_deficit_tokens, + ) if value is not None) + assert deficit > 0 + assert measurement.low_water_margin_tokens == math.ceil(boundary / RECLAIM_LOW_WATER_DIVISOR) + assert measurement.reclaim_goal_tokens == deficit + measurement.low_water_margin_tokens + assert disposition.action == "reclaim_once" + # The margin sizes the pass; it never re-arms the route+round latch. + after = _measure( + monkeypatch, tmp_path, plan=plan, profile=profile, mode=preferred, messages=messages, + used=True, + ) + assert after.action != "reclaim_once" + assert after.measurement.low_water_margin_tokens == measurement.low_water_margin_tokens + + +@pytest.mark.parametrize("profile,preferred,window,known", [ + ("owner_max", "max", 0, False), + ("owner_max", "max", 500_000, True), + ("owner_low", "low", 500_000, True), + ("task_local_low", "low", 500_000, True), +]) +def test_no_deficit_means_no_margin_and_no_goal(monkeypatch, tmp_path, profile, preferred, window, known): + plan = _plan(preferred=preferred, window=window, known=known) + disposition = _measure( + monkeypatch, tmp_path, plan=plan, profile=profile, mode=preferred, + messages=plan.messages_for(preferred), + ) + assert disposition.measurement.low_water_margin_tokens == 0 + assert disposition.measurement.reclaim_goal_tokens == 0 + assert disposition.action == "send" + + +def test_low_water_divisor_is_the_one_knob_in_both_directions(monkeypatch, tmp_path): + from ouroboros import context_budget as cb + + plan = _plan(window=70_000, known=True) + messages = plan.messages_for("max") + [{"role": "user", "content": "x" * 40_000}] + base = _measure( + monkeypatch, tmp_path, plan=plan, profile="owner_max", mode="max", messages=messages, + ).measurement + deficit = base.capacity_deficit_tokens + assert deficit > 0 and base.low_water_margin_tokens == math.ceil(70_000 / 8) + + monkeypatch.setattr(cb, "RECLAIM_LOW_WATER_DIVISOR", 4) + wider = _measure( + monkeypatch, tmp_path, plan=plan, profile="owner_max", mode="max", messages=messages, + ).measurement + assert wider.low_water_margin_tokens == math.ceil(70_000 / 4) + assert wider.reclaim_goal_tokens == deficit + wider.low_water_margin_tokens + + monkeypatch.setattr(cb, "RECLAIM_LOW_WATER_DIVISOR", 10 ** 9) # the margin removed + removed = _measure( + monkeypatch, tmp_path, plan=plan, profile="owner_max", mode="max", messages=messages, + ).measurement + assert removed.low_water_margin_tokens == 1 + assert removed.reclaim_goal_tokens == deficit + 1 + assert removed.capacity_deficit_tokens == deficit # the trigger never moved + + def test_task_local_low_does_not_inherit_owner_economy_target(monkeypatch, tmp_path): plan = _plan(window=500_000, known=True) disposition = _measure( diff --git a/tests/test_loop_compaction.py b/tests/test_loop_compaction.py index 611f8fd73..a25bd5973 100644 --- a/tests/test_loop_compaction.py +++ b/tests/test_loop_compaction.py @@ -2,9 +2,15 @@ from __future__ import annotations +import math from dataclasses import replace from types import SimpleNamespace +import pytest + +from ouroboros.context_budget import RECLAIM_LOW_WATER_DIVISOR +from ouroboros.context_fit import reclaim_low_water_margin + def _fit( *, @@ -15,6 +21,8 @@ def _fit( target_deficit=None, capacity_deficit=None, used=False, + estimated_input=120_000, + low_water_margin=0, ): from ouroboros.context_fit import MainFitDisposition, MainFitMeasurement @@ -23,7 +31,7 @@ def _fit( round_id="exec:round:1", profile=profile, rendered_mode=mode, - estimated_input_tokens=120_000, + estimated_input_tokens=estimated_input, response_reserve_tokens=65_536, target_total_tokens=200_000 if profile == "owner_low" else None, capacity_total_tokens=500_000, @@ -32,6 +40,7 @@ def _fit( target_deficit_tokens=target_deficit, capacity_deficit_tokens=capacity_deficit, reclaim_goal_tokens=goal, + low_water_margin_tokens=low_water_margin, ) return MainFitDisposition( measurement=measurement, @@ -220,6 +229,175 @@ def test_checkpoint_proves_automatic_materializer_attempt_even_on_binding_mismat assert ("route-a", "exec:round:1") in context.tools._ctx._context_reclaim_materializations +def _applied_receipt(*, reclaimed: int, goal_reached: bool): + from ouroboros.context_budget import ContextReclaimReceipt + + return ContextReclaimReceipt( + status="applied", + before_transcript_sha256="a" * 64, + after_transcript_sha256="b" * 64, + selection_fingerprint="f" * 64, + selected_unit_ids=("unit",), + reclaimed_tokens=reclaimed, + goal_reached=goal_reached, + checkpoint_ref={"path": "checkpoint", "sha256": "c" * 64}, + capsule_refs=(), + ) + + +# Owner Low: the 200,000 target binds (capacity 500,000); its input boundary is +# 200,000 - 65,536 = 134,464 estimated tokens; margin = ceil(200,000 / 8) = 25,000. +_LOW_BOUNDARY_INPUT = 200_000 - 65_536 + + +@pytest.mark.parametrize("landed_input,headroom,reached,below", [ + (_LOW_BOUNDARY_INPUT, 0, True, False), # reclaimed == deficit: AT the boundary, not below + (_LOW_BOUNDARY_INPUT - 25_000, 25_000, True, True), # the full margin achieved + (_LOW_BOUNDARY_INPUT - 12_000, 12_000, True, False), # under-landed: reached, margin missed + (_LOW_BOUNDARY_INPUT + 2_000, -2_000, False, False), # still above the boundary +]) +def test_reclaim_checkpoint_separates_boundary_from_low_water( + tmp_path, monkeypatch, landed_input, headroom, reached, below, +): + from ouroboros import loop + + context = _ctx(tmp_path, preferred="low", mode="low") + disposition = _fit( + action="reclaim_once", profile="owner_low", mode="low", goal=10_000 + 25_000, + target_deficit=10_000, capacity_deficit=0, estimated_input=_LOW_BOUNDARY_INPUT + 10_000, + low_water_margin=25_000, + ) + landed = _fit( + action="send", profile="owner_low", mode="low", used=True, estimated_input=landed_input, + ) + measured, events = [], [] + monkeypatch.setattr( + loop, "compact_tool_history_llm", + lambda *_a, **_kw: (context.messages, _applied_receipt(reclaimed=1, goal_reached=False), None), + ) + monkeypatch.setattr( + loop, "_measure_round_main_fit", lambda _ctx, **kwargs: measured.append(kwargs) or landed, + ) + monkeypatch.setattr(loop, "_emit_checkpoint_event", lambda _q, _t, _d, data: events.append(data)) + + loop._run_main_reclaim(context, disposition) + + # The landing is re-measured on the SAME basis as the trigger, once, as a used pass. + assert measured == [{"automatic_pass_used": True}] + event = events[-1] + assert event["checkpoint_kind"] == "context_reclaim_automatic" + assert event["deficit_tokens"] == 10_000 + assert event["requested_margin_tokens"] == 25_000 + assert event["reclaim_goal_tokens"] == 35_000 + assert event["achieved_headroom_tokens"] == headroom + assert event["boundary_reached"] is reached + assert event["below_boundary"] is below + assert event["rounds_since_previous_pass"] is None + assert context.tools._ctx._context_reclaim_last_pass_round == 1 + + +def test_reclaim_checkpoint_counts_rounds_since_the_previous_pass_without_remeasuring_a_no_op( + tmp_path, monkeypatch, +): + from ouroboros import loop + + context = replace(_ctx(tmp_path, preferred="low", mode="low"), round_idx=9) + context.tools._ctx._context_reclaim_last_pass_round = 3 + disposition = _fit( + action="reclaim_once", profile="owner_low", mode="low", goal=10_000 + 25_000, + target_deficit=10_000, capacity_deficit=0, estimated_input=_LOW_BOUNDARY_INPUT + 10_000, + low_water_margin=25_000, + ) + disposition = replace( + disposition, measurement=replace(disposition.measurement, round_id="exec:round:9"), + ) + events = [] + monkeypatch.setattr( + loop, "compact_tool_history_llm", + lambda *_a, **_kw: (context.messages, replace( + _applied_receipt(reclaimed=0, goal_reached=False), status="no_eligible", + ), None), + ) + monkeypatch.setattr( + loop, "_measure_round_main_fit", + lambda *_a, **_kw: pytest.fail("a pass that changed nothing has no new landing to measure"), + ) + monkeypatch.setattr(loop, "_emit_checkpoint_event", lambda _q, _t, _d, data: events.append(data)) + + loop._run_main_reclaim(context, disposition) + + event = events[-1] + assert event["status"] == "no_eligible" + assert event["rounds_since_previous_pass"] == 6 + # Unchanged transcript: the pre-pass fit IS the landing, 10,000 above the boundary. + assert event["achieved_headroom_tokens"] == -10_000 + assert event["boundary_reached"] is False + assert event["below_boundary"] is False + assert event["requested_margin_tokens"] == 25_000 + assert context.tools._ctx._context_reclaim_last_pass_round == 9 + + +def test_reclaim_checkpoint_reports_an_unmeasurable_landing_as_unknown(tmp_path, monkeypatch): + """A route/plan rebind can leave the post-pass fit unmeasurable (the fit seam + returns None by contract): the pass still applies and the event says UNKNOWN, + never "boundary not reached".""" + from ouroboros import loop + + context = _ctx(tmp_path, preferred="low", mode="low") + disposition = _fit( + action="reclaim_once", profile="owner_low", mode="low", goal=10_000 + 25_000, + target_deficit=10_000, capacity_deficit=0, estimated_input=_LOW_BOUNDARY_INPUT + 10_000, + low_water_margin=25_000, + ) + events = [] + monkeypatch.setattr( + loop, "compact_tool_history_llm", + lambda *_a, **_kw: (context.messages, _applied_receipt(reclaimed=30_000, goal_reached=False), None), + ) + monkeypatch.setattr(loop, "_measure_round_main_fit", lambda *_a, **_kw: None) + monkeypatch.setattr(loop, "_emit_checkpoint_event", lambda _q, _t, _d, data: events.append(data)) + + receipt = loop._run_main_reclaim(context, disposition) + + assert receipt.status == "applied" + assert events[-1]["reclaimed_tokens"] == 30_000 + assert events[-1]["achieved_headroom_tokens"] is None + assert events[-1]["boundary_reached"] is None + assert events[-1]["below_boundary"] is None + assert events[-1]["requested_margin_tokens"] == 25_000 + + +def test_overflow_minimum_goal_is_low_water_sized_even_without_a_predicted_deficit( + tmp_path, monkeypatch, +): + """Both directions: the measurement found no deficit (goal 0), the provider still + overflowed; the pass requests the low-water minimum, and its telemetry reports + that minimum as the requested margin over a zero deficit.""" + from ouroboros import loop + + context = _ctx(tmp_path, preferred="low", mode="low") + disposition = _fit(action="send", profile="owner_low", mode="low", goal=0, target_deficit=0, + capacity_deficit=0, estimated_input=_LOW_BOUNDARY_INPUT - 500) + requests, events = [], [] + + def compact(messages, *, request, **_kwargs): + requests.append(request.reclaim_goal_tokens) + return messages, replace(_applied_receipt(reclaimed=0, goal_reached=False), + status="no_eligible"), None + + monkeypatch.setattr(loop, "compact_tool_history_llm", compact) + monkeypatch.setattr(loop, "_emit_checkpoint_event", lambda _q, _t, _d, data: events.append(data)) + minimum = max(1, reclaim_low_water_margin(200_000, 500_000)) # what the overflow path passes + loop._run_main_reclaim(context, disposition, minimum_goal_tokens=minimum) + + assert requests == [25_000] + assert events[-1]["deficit_tokens"] == 0 + assert events[-1]["requested_margin_tokens"] == 25_000 + assert events[-1]["achieved_headroom_tokens"] == 500 + assert events[-1]["boundary_reached"] is True + assert events[-1]["below_boundary"] is False + + def test_predicted_reclaim_runs_once_then_sends_target_miss(tmp_path, monkeypatch): from ouroboros import loop @@ -276,7 +454,9 @@ def test_actual_max_overflow_reprojects_and_retries_only_smaller_context(tmp_pat return disposition def reclaim(ctx, _disposition, **kwargs): - assert kwargs["minimum_goal_tokens"] == 1 + # An actual overflow asks for a low-water-sized pass (an eighth of the 500K + # route), never a token-sized one, before its single strict-shrink retry. + assert kwargs["minimum_goal_tokens"] == math.ceil(500_000 / RECLAIM_LOW_WATER_DIVISOR) ctx.tools._ctx._context_reclaim_passes.add(("route-a", "exec:round:1")) sends = [] diff --git a/tests/test_owner_wait_cold_loop.py b/tests/test_owner_wait_cold_loop.py index b02260114..9082f0c4b 100644 --- a/tests/test_owner_wait_cold_loop.py +++ b/tests/test_owner_wait_cold_loop.py @@ -3,6 +3,7 @@ import asyncio from dataclasses import replace import json +import math import socket import threading from types import SimpleNamespace @@ -10,6 +11,7 @@ from types import SimpleNamespace import pytest from ouroboros import context, loop, task_pacing, usage_accounting as accounting +from ouroboros.context_budget import RECLAIM_LOW_WATER_DIVISOR from ouroboros.contracts.task_contract import normalize_budget_profile from ouroboros.owner_wait import checkpoint_owner_wait, set_owner_wait from ouroboros.task_results import write_task_result @@ -78,7 +80,9 @@ def test_cold_switched_model_keeps_actual_overflow_reclaim_and_retry(tmp_path, m sends, reclaimed = [], [] def reclaim(call, disposition, **kwargs): - assert kwargs["minimum_goal_tokens"] == 1 + # An actual overflow requests a low-water-sized pass (an eighth of the + # 500K route), never a token-sized one, before its single strict-shrink retry. + assert kwargs["minimum_goal_tokens"] == math.ceil(500_000 / RECLAIM_LOW_WATER_DIVISOR) reclaimed.append(call.active_model) loop._context_reclaim_passes(call.tools._ctx).add( (disposition.measurement.route_fp, disposition.measurement.round_id)) diff --git a/tests/test_transcript_prefix.py b/tests/test_transcript_prefix.py index 36d612304..3e5691d99 100644 --- a/tests/test_transcript_prefix.py +++ b/tests/test_transcript_prefix.py @@ -315,8 +315,17 @@ def test_automatic_reclaim_inside_the_model_call_is_a_sanctioned_break_on_its_ro if ctx.round_idx != 3 or fired or automatic_pass_used: return None fired.append(True) - measurement = SimpleNamespace(route_fp="fp", round_id="r3", measurement_basis="cold_estimate", - measurement_density=1.0, reclaim_goal_tokens=100) + from ouroboros.context_fit import MainFitMeasurement + + # A real measurement: only a positive deficit decides "reclaim_once", and the + # reclaim's low-water telemetry reads the deficit and boundary fields. + measurement = MainFitMeasurement( + route_fp="fp", round_id="r3", profile="owner_low", rendered_mode="low", + estimated_input_tokens=200_000 - 65_536 + 100, response_reserve_tokens=65_536, + target_total_tokens=200_000, capacity_total_tokens=None, + measurement_basis="cold_estimate", measurement_density=1.0, + target_deficit_tokens=100, capacity_deficit_tokens=None, reclaim_goal_tokens=100, + ) return SimpleNamespace(action="reclaim_once", measurement=measurement, automatic_pass_used=False) monkeypatch.setattr(loop, "_measure_round_main_fit", measure)