mirror of
https://github.com/razzant/ouroboros.git
synced 2026-10-02 19:58:46 +00:00
Size the automatic context-reclaim pass to a low-water margin
The trigger is unchanged: a positive deficit against the binding boundary (the smaller known of owner target and route capacity), one pass per route+round. The requested goal becomes deficit + ceil(boundary / RECLAIM_LOW_WATER_DIVISOR), so a pass lands about an eighth of the boundary below it instead of exactly at it, where the next round's ordinary growth re-armed it nearly every round. RECLAIM_LOW_WATER_DIVISOR = 8 is a documented structural constant in context_budget (not a setting; pinned by test_context_budget_ssot so it stays a one-constant change). MainFitMeasurement gains low_water_margin_tokens, appended last. The context_reclaim checkpoint event records deficit_tokens, requested_margin_tokens, achieved_headroom_tokens (the landing re-measured on the same fit basis as the trigger), boundary_reached, below_boundary (reclaiming exactly the deficit reaches the boundary, it does not get below it) and rounds_since_previous_pass; an unmeasurable landing is reported as unknown, never as "not reached". The actual-overflow path requests the same low-water minimum instead of a token-sized pass before its one strict-shrink retry. Materializer, receipts and the route+round latch are untouched. Tests: margin arithmetic in both directions (binding boundary, no deficit, divisor as the one knob), the anti-thrash class test with the worst-case full-budget summarizer stub (4 passes in 40 rounds against 39 with the margin removed; the stub halves every selected unit, so the asserted bound is ceil(N*g / (margin/2)) + 1 plus the achieved-headroom form), the checkpoint telemetry, the unmeasured landing, and the overflow-minimum pins. Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com>
This commit is contained in:
parent
90ce019b8e
commit
98ffcf2a62
9 changed files with 515 additions and 8 deletions
|
|
@ -36,6 +36,24 @@ OWNER_LOW_TARGET_TOKENS = 200_000
|
|||
OWNER_NANO_TARGET_TOKENS = 81_920
|
||||
NANO_MIN_HEADROOM_TOKENS = 8_192
|
||||
|
||||
# Low-water sizing of the automatic context-reclaim pass. The TRIGGER is
|
||||
# unchanged: a positive deficit against the binding boundary (the smaller known
|
||||
# of the owner target T and the route capacity W), one pass per route+round.
|
||||
# Only the SIZE of the requested pass changes: goal = deficit +
|
||||
# ceil(boundary / RECLAIM_LOW_WATER_DIVISOR), and 0 without a deficit. A pass
|
||||
# sized to the deficit alone lands exactly AT the boundary, so the next round's
|
||||
# ordinary growth re-arms it (a summarizer pass nearly every round). Sized this
|
||||
# way it lands about an eighth of the boundary below (~125K tokens on a 1M
|
||||
# route, ~25K under the 200K Low target), so the next pass needs that much real
|
||||
# growth. Structural constant, not a setting: 8 (12.5 % of the boundary) is a
|
||||
# disclosed design choice, not a measured optimum; change it here and only here
|
||||
# (tests/test_context_budget_ssot.py pins it). Cost: older history is condensed
|
||||
# sooner and each summarizer pass is larger. The materializer, its receipts and
|
||||
# the route+round latch are unchanged; the checkpoint event records requested
|
||||
# margin versus achieved headroom (context_fit.measure_main_fit,
|
||||
# loop_model_call._run_main_reclaim).
|
||||
RECLAIM_LOW_WATER_DIVISOR = 8
|
||||
|
||||
# One overflow vocabulary for every seam that must recognize a CONTEXT-WINDOW
|
||||
# overflow (Main provider-code precedence, the local transport, and the
|
||||
# summarizer split path). A provider code or message shape added here reaches
|
||||
|
|
|
|||
|
|
@ -200,6 +200,9 @@ class MainFitMeasurement:
|
|||
target_deficit_tokens: Optional[int]
|
||||
capacity_deficit_tokens: Optional[int]
|
||||
reclaim_goal_tokens: int
|
||||
# Low-water margin the goal carries ABOVE the deficit (0 without a deficit):
|
||||
# the pass is deficit-triggered but sized to land below the boundary.
|
||||
low_water_margin_tokens: int = 0
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
|
|
@ -559,6 +562,22 @@ def _route_calibration_ratio(
|
|||
return 1.0
|
||||
|
||||
|
||||
def reclaim_low_water_margin(
|
||||
target_total_tokens: Optional[int], capacity_total_tokens: Optional[int],
|
||||
) -> int:
|
||||
"""Tokens a reclaim pass lands BELOW the binding boundary: ceil(boundary / divisor).
|
||||
|
||||
The boundary is the smaller known positive one of owner target T and route
|
||||
capacity W; 0 when neither is known. The divisor is read at call time so
|
||||
the SSOT constant stays the one place to change it.
|
||||
"""
|
||||
from ouroboros.context_budget import RECLAIM_LOW_WATER_DIVISOR
|
||||
|
||||
known = [int(value) for value in (target_total_tokens, capacity_total_tokens)
|
||||
if value is not None and int(value) > 0]
|
||||
return int(math.ceil(min(known) / RECLAIM_LOW_WATER_DIVISOR)) if known else 0
|
||||
|
||||
|
||||
def measure_main_fit(
|
||||
plan: ContextFitPlan,
|
||||
messages: List[Dict[str, Any]],
|
||||
|
|
@ -575,6 +594,9 @@ def measure_main_fit(
|
|||
|
||||
``drive_root=None`` reads density from the canonical host evidence root
|
||||
(one observation store) — a child task's own drive must not be consulted.
|
||||
A positive deficit triggers at most one reclaim pass per route+round; the
|
||||
requested goal is deficit + ``reclaim_low_water_margin`` so the pass lands
|
||||
below the boundary instead of exactly at it (``RECLAIM_LOW_WATER_DIVISOR``).
|
||||
"""
|
||||
from ouroboros.capability_evidence import (
|
||||
canonical_evidence_root, is_known, resolve_main_token_density,
|
||||
|
|
@ -598,10 +620,12 @@ def measure_main_fit(
|
|||
capacity = int(plan.window_tokens or 0) if is_known(plan, require_fresh=True) else None
|
||||
target_deficit = max(0, total - target) if target is not None else None
|
||||
capacity_deficit = max(0, total - capacity) if capacity is not None else None
|
||||
goal = max(
|
||||
deficit = max(
|
||||
[value for value in (target_deficit, capacity_deficit) if value is not None]
|
||||
or [0]
|
||||
)
|
||||
margin = reclaim_low_water_margin(target, capacity) if deficit > 0 else 0
|
||||
goal = deficit + margin
|
||||
measurement = MainFitMeasurement(
|
||||
route_fp=str(plan.route_fp or ""),
|
||||
round_id=str(round_id or ""),
|
||||
|
|
@ -616,6 +640,7 @@ def measure_main_fit(
|
|||
target_deficit_tokens=target_deficit,
|
||||
capacity_deficit_tokens=capacity_deficit,
|
||||
reclaim_goal_tokens=goal,
|
||||
low_water_margin_tokens=margin,
|
||||
)
|
||||
if goal > 0 and not automatic_pass_used:
|
||||
action: Literal["send", "reclaim_once", "send_target_miss"] = "reclaim_once"
|
||||
|
|
|
|||
|
|
@ -867,6 +867,29 @@ def _run_main_reclaim(
|
|||
ctx.tools._ctx.messages = ctx.messages
|
||||
_loop().seal_task_transcript(ctx.messages)
|
||||
prune_reclaim_trace_refs(ctx.tools._ctx, ctx.messages)
|
||||
# Low-water facts: the pass is deficit-triggered but sized to land below the
|
||||
# boundary, so the landing is re-measured on the SAME fit basis as the trigger
|
||||
# and "reached the boundary" stays distinct from "achieved the margin"
|
||||
# (reclaimed == deficit is AT the boundary, not below it).
|
||||
deficit = max(int(measurement.target_deficit_tokens or 0),
|
||||
int(measurement.capacity_deficit_tokens or 0))
|
||||
requested_margin = int(request.reclaim_goal_tokens) - deficit
|
||||
landed = measurement
|
||||
if receipt.status == "applied":
|
||||
try:
|
||||
remeasured = _loop()._measure_round_main_fit(ctx, automatic_pass_used=True)
|
||||
except Exception: # telemetry only: an unmeasurable landing must not fail the pass
|
||||
log.debug("Post-reclaim fit measurement unavailable", exc_info=True)
|
||||
remeasured = None
|
||||
landed = remeasured.measurement if remeasured is not None else None
|
||||
boundary = [value for value in (
|
||||
landed.target_total_tokens, landed.capacity_total_tokens) if value is not None] if landed else []
|
||||
# None = the landing could not be measured (unknown), never "not reached".
|
||||
headroom = (min(boundary) - (landed.estimated_input_tokens + landed.response_reserve_tokens)
|
||||
if boundary else None)
|
||||
tool_ctx = ctx.tools._ctx
|
||||
previous_round = getattr(tool_ctx, "_context_reclaim_last_pass_round", None)
|
||||
tool_ctx._context_reclaim_last_pass_round = int(ctx.round_idx)
|
||||
_loop()._emit_checkpoint_event(ctx.event_queue, ctx.task_id, ctx.drive_logs, {
|
||||
"type": "context_reclaim",
|
||||
"checkpoint_kind": "context_reclaim_automatic",
|
||||
|
|
@ -878,6 +901,13 @@ def _run_main_reclaim(
|
|||
"reclaimed_tokens": receipt.reclaimed_tokens,
|
||||
"goal_reached": receipt.goal_reached,
|
||||
"checkpoint_ref": receipt.checkpoint_ref,
|
||||
"deficit_tokens": deficit,
|
||||
"requested_margin_tokens": requested_margin,
|
||||
"achieved_headroom_tokens": headroom,
|
||||
"boundary_reached": None if headroom is None else headroom >= 0,
|
||||
"below_boundary": None if headroom is None else headroom >= max(1, requested_margin),
|
||||
"rounds_since_previous_pass": (
|
||||
int(ctx.round_idx) - int(previous_round) if previous_round is not None else None),
|
||||
})
|
||||
return receipt
|
||||
|
||||
|
|
@ -997,7 +1027,15 @@ def _call_round_model(ctx: _RoundModelCallContext) -> Tuple[Any, float, str]:
|
|||
return msg, cost, ctx.active_context_mode
|
||||
key = _fit_key(overflow_fit)
|
||||
if key not in _loop()._context_reclaim_passes(ctx.tools._ctx):
|
||||
_loop()._run_main_reclaim(ctx, overflow_fit, minimum_goal_tokens=1)
|
||||
# The provider proved the prediction short by an unknown amount: request a
|
||||
# low-water-sized pass, never a token-sized one, so the single strict-shrink
|
||||
# retry has real headroom (the goal already carries the margin when the
|
||||
# measurement itself found a deficit).
|
||||
from ouroboros.context_fit import reclaim_low_water_margin
|
||||
|
||||
landed = overflow_fit.measurement
|
||||
_loop()._run_main_reclaim(ctx, overflow_fit, minimum_goal_tokens=max(
|
||||
1, reclaim_low_water_margin(landed.target_total_tokens, landed.capacity_total_tokens)))
|
||||
overflow_fit = _measure_after_reclaim(ctx)
|
||||
if overflow_fit is None:
|
||||
return msg, cost, ctx.active_context_mode
|
||||
|
|
|
|||
|
|
@ -24,6 +24,25 @@ def test_agent_context_budget_values_pinned():
|
|||
assert cb.MAX_RECENT_CHAT_TAIL == 1000
|
||||
assert cb.CHAT_ARCHIVE_SCAN_WARN_BYTES == 100_000_000
|
||||
assert not hasattr(cb, "CONTEXT_SOFT_CAP_TOKENS")
|
||||
# Structural low-water divisor of the automatic reclaim pass (12.5 % of the
|
||||
# binding boundary): a disclosed design choice, not a setting.
|
||||
assert cb.RECLAIM_LOW_WATER_DIVISOR == 8
|
||||
|
||||
|
||||
def test_reclaim_low_water_divisor_is_one_constant_read_at_call_time(monkeypatch):
|
||||
"""CHECKLISTS item 20: the fit consumes the SSOT name (no bare literal), reads it
|
||||
at call time so changing the one constant changes every pass, and the margin
|
||||
is the LAST measurement field (appended; older readers stay positional-safe)."""
|
||||
from ouroboros import context_fit
|
||||
|
||||
assert "RECLAIM_LOW_WATER_DIVISOR" in _src("ouroboros/context_fit.py")
|
||||
assert "/ 8" not in inspect.getsource(context_fit.measure_main_fit)
|
||||
assert dataclasses.fields(context_fit.MainFitMeasurement)[-1].name == "low_water_margin_tokens"
|
||||
assert context_fit.reclaim_low_water_margin(200_000, 500_000) == 25_000 # target binds
|
||||
assert context_fit.reclaim_low_water_margin(None, 70_000) == 8_750 # capacity alone
|
||||
assert context_fit.reclaim_low_water_margin(None, None) == 0 # nothing known
|
||||
monkeypatch.setattr(cb, "RECLAIM_LOW_WATER_DIVISOR", 4)
|
||||
assert context_fit.reclaim_low_water_margin(200_000, 500_000) == 50_000
|
||||
|
||||
|
||||
def test_reclaim_request_and_receipt_are_exact_frozen_records():
|
||||
|
|
|
|||
|
|
@ -123,6 +123,138 @@ def test_owner_low_deficit_reclaim_remeasures_on_one_basis(monkeypatch, tmp_path
|
|||
assert after.measurement.reclaim_goal_tokens == 0
|
||||
|
||||
|
||||
def _growth_unit(tag: str, chars: int):
|
||||
return [{
|
||||
"role": "assistant",
|
||||
"content": "investigating",
|
||||
"tool_calls": [{
|
||||
"id": f"call-{tag}",
|
||||
"type": "function",
|
||||
"function": {"name": "read_file", "arguments": "x" * chars},
|
||||
}],
|
||||
}, {"role": "tool", "tool_call_id": f"call-{tag}", "content": "y" * chars}]
|
||||
|
||||
|
||||
def _install_full_budget_materializer(monkeypatch):
|
||||
"""A13 worst case: every summary comes back at the FULL summary budget of its
|
||||
source (never a tiny summary), the private checkpoint is stubbed, density 1.0."""
|
||||
from ouroboros import capability_evidence, context_compaction as cc
|
||||
|
||||
monkeypatch.setattr(
|
||||
capability_evidence, "resolve_main_token_density",
|
||||
lambda *_a, **_kw: (1.0, "fresh_route_usage"),
|
||||
)
|
||||
monkeypatch.setattr(cc, "_summarizer_spec", lambda: {
|
||||
"model": "summary-model", "resolved_model": "summary-model", "provider": "test",
|
||||
"route_fp": "summary-route", "effort": "low", "output_budget": 32_768, "use_local": False,
|
||||
})
|
||||
monkeypatch.setattr(
|
||||
cc, "_persist_reclaim_checkpoint",
|
||||
lambda *_a, **_kw: {"path": "checkpoint", "sha256": "c" * 64},
|
||||
)
|
||||
monkeypatch.setattr(cc, "_call_summarizer", lambda parts, *, summary_budgets, **_kw: {
|
||||
part.source_id: "s" * (4 * int(summary_budgets[part.root_id])) for part in parts
|
||||
})
|
||||
|
||||
|
||||
def _simulate_growing_transcript(tmp_path, *, window: int, rounds: int, growth_chars: int):
|
||||
"""Owner Max on a known ``window``: fill to just under the capacity boundary, then
|
||||
grow one completed tool unit per round, running the REAL fit and the REAL
|
||||
materializer the way the loop does (at most one automatic pass per route+round,
|
||||
the landing re-measured on the same basis). Returns (unit tokens, per-pass rows
|
||||
of (round, requested margin, achieved headroom, receipt))."""
|
||||
from ouroboros import context_compaction as cc
|
||||
from ouroboros.context_budget import ContextReclaimRequest
|
||||
from ouroboros.context_fit import estimate_context_prompt_tokens, measure_main_fit
|
||||
|
||||
plan = _plan(preferred="max", window=window)
|
||||
messages = plan.messages_for("max")
|
||||
unit_tokens = estimate_context_prompt_tokens(_growth_unit("probe", growth_chars))
|
||||
boundary_input = window - plan.output_reserve_tokens
|
||||
filler = (boundary_input - estimate_context_prompt_tokens(messages) - unit_tokens // 2) // unit_tokens
|
||||
for index in range(filler):
|
||||
messages = messages + _growth_unit(f"f{index}", growth_chars)
|
||||
|
||||
def fit(current, round_idx, *, used):
|
||||
return measure_main_fit(
|
||||
plan, current, [], drive_root=tmp_path, profile="owner_max", rendered_mode="max",
|
||||
round_id=f"exec:round:{round_idx}", automatic_pass_used=used,
|
||||
)
|
||||
|
||||
memo: set = set()
|
||||
rows = []
|
||||
for round_idx in range(1, rounds + 1):
|
||||
messages = messages + _growth_unit(f"g{round_idx}", growth_chars)
|
||||
before = fit(messages, round_idx, used=False)
|
||||
if before.action != "reclaim_once":
|
||||
assert before.action == "send"
|
||||
continue
|
||||
measurement = before.measurement
|
||||
request = ContextReclaimRequest(
|
||||
route_fp=measurement.route_fp, round_id=measurement.round_id,
|
||||
transcript_sha256=cc.context_reclaim_transcript_sha256(messages),
|
||||
measurement_basis=measurement.measurement_basis,
|
||||
measurement_density=measurement.measurement_density,
|
||||
reclaim_goal_tokens=measurement.reclaim_goal_tokens,
|
||||
)
|
||||
messages, receipt, _usage = cc.compact_tool_history_llm(
|
||||
messages, request=request, drive_root=tmp_path, negative_memo=memo,
|
||||
)
|
||||
after = fit(messages, round_idx, used=True).measurement
|
||||
rows.append((
|
||||
round_idx, measurement.low_water_margin_tokens,
|
||||
window - (after.estimated_input_tokens + after.response_reserve_tokens), receipt,
|
||||
))
|
||||
return unit_tokens, rows
|
||||
|
||||
|
||||
def test_low_water_margin_bounds_automatic_passes_under_a_full_budget_summarizer(monkeypatch, tmp_path):
|
||||
"""Anti-thrash class. A pass sized to the deficit alone lands AT the boundary, so the
|
||||
next round's ordinary growth re-arms it: one summarizer pass nearly every round.
|
||||
Sized deficit + boundary/RECLAIM_LOW_WATER_DIVISOR it lands below the boundary and
|
||||
the next pass needs real growth. Under the A13 worst-case stub every selected unit
|
||||
halves, so a pass lands about HALF the requested margin below (the receipt says
|
||||
goal_reached=False): the bound proven with that stub is ceil(N*g / (margin/2)) + 1;
|
||||
the ideal-summarizer bound ceil(N*g / margin) + 1 holds for the ACHIEVED headroom and
|
||||
is asserted in that form (requested margin is not achieved headroom). With the
|
||||
margin removed the same assertions fail."""
|
||||
import math
|
||||
|
||||
from ouroboros import context_budget as cb
|
||||
|
||||
_install_full_budget_materializer(monkeypatch)
|
||||
window, rounds, chars = 400_000, 40, 4_000
|
||||
margin = math.ceil(window / cb.RECLAIM_LOW_WATER_DIVISOR)
|
||||
|
||||
unit_tokens, passes = _simulate_growing_transcript(
|
||||
tmp_path, window=window, rounds=rounds, growth_chars=chars,
|
||||
)
|
||||
growth = rounds * unit_tokens
|
||||
bound = math.ceil(growth / (margin // 2)) + 1
|
||||
assert 2 <= len(passes) <= bound
|
||||
assert [row[1] for row in passes] == [margin] * len(passes)
|
||||
assert all(row[3].status == "applied" for row in passes)
|
||||
# Every pass reached the boundary; none achieved the full margin (partial shrink
|
||||
# under full-budget summaries), and the receipt discloses that underlanding.
|
||||
assert all(0 <= row[2] < margin for row in passes)
|
||||
assert not any(row[3].goal_reached for row in passes)
|
||||
achieved = min(row[2] for row in passes)
|
||||
assert len(passes) <= math.ceil(growth / achieved) + 1
|
||||
gaps = [later[0] - earlier[0] for earlier, later in zip(passes, passes[1:])]
|
||||
assert min(gaps) > achieved // unit_tokens
|
||||
|
||||
monkeypatch.setattr(cb, "RECLAIM_LOW_WATER_DIVISOR", 10 ** 9) # margin 1: deficit-sized passes
|
||||
_unit, thrash = _simulate_growing_transcript(
|
||||
tmp_path, window=window, rounds=rounds, growth_chars=chars,
|
||||
)
|
||||
assert len(thrash) >= rounds - 2
|
||||
assert len(thrash) > bound
|
||||
assert [row[1] for row in thrash] == [1] * len(thrash)
|
||||
# A deficit-sized pass lands at (here: still above) the boundary, which is exactly
|
||||
# what re-arms it on the next round.
|
||||
assert sum(1 for row in thrash if row[2] < 0) >= len(thrash) - 2
|
||||
|
||||
|
||||
def test_target_miss_is_non_terminal_fit_evidence(monkeypatch, tmp_path):
|
||||
from ouroboros import capability_evidence, loop
|
||||
from ouroboros.context_fit import measure_main_fit
|
||||
|
|
|
|||
|
|
@ -4,10 +4,13 @@ from __future__ import annotations
|
|||
|
||||
import inspect
|
||||
import json
|
||||
import math
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
|
||||
from ouroboros.context_budget import RECLAIM_LOW_WATER_DIVISOR
|
||||
|
||||
|
||||
def _projection(mode: str):
|
||||
from ouroboros.context_fit import ContextFitProjection
|
||||
|
|
@ -126,10 +129,89 @@ def test_confirmed_window_below_target_wins_even_when_target_fits(monkeypatch, t
|
|||
measurement = disposition.measurement
|
||||
assert measurement.target_deficit_tokens == 0
|
||||
assert measurement.capacity_deficit_tokens > 0
|
||||
assert measurement.reclaim_goal_tokens == measurement.capacity_deficit_tokens
|
||||
# Deficit-triggered, low-water-sized: the goal carries an eighth of the BINDING
|
||||
# boundary (the 65,550 window, not the 200K target) above the deficit.
|
||||
assert measurement.low_water_margin_tokens == math.ceil(65_550 / RECLAIM_LOW_WATER_DIVISOR)
|
||||
assert measurement.reclaim_goal_tokens == (
|
||||
measurement.capacity_deficit_tokens + measurement.low_water_margin_tokens
|
||||
)
|
||||
assert disposition.action == "reclaim_once"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("profile,preferred,window,known,boundary", [
|
||||
("owner_max", "max", 70_000, True, 70_000), # capacity binds a Max route
|
||||
("owner_low", "low", 500_000, True, 200_000), # the economy target binds Low
|
||||
("owner_low", "low", 0, False, 200_000), # unknown capacity: the target alone binds
|
||||
])
|
||||
def test_low_water_margin_follows_the_binding_boundary(
|
||||
monkeypatch, tmp_path, profile, preferred, window, known, boundary,
|
||||
):
|
||||
plan = _plan(preferred=preferred, window=window, known=known)
|
||||
messages = plan.messages_for(preferred) + [{"role": "user", "content": "x" * 600_000}]
|
||||
disposition = _measure(
|
||||
monkeypatch, tmp_path, plan=plan, profile=profile, mode=preferred, messages=messages,
|
||||
)
|
||||
measurement = disposition.measurement
|
||||
deficit = max(value for value in (
|
||||
measurement.target_deficit_tokens, measurement.capacity_deficit_tokens,
|
||||
) if value is not None)
|
||||
assert deficit > 0
|
||||
assert measurement.low_water_margin_tokens == math.ceil(boundary / RECLAIM_LOW_WATER_DIVISOR)
|
||||
assert measurement.reclaim_goal_tokens == deficit + measurement.low_water_margin_tokens
|
||||
assert disposition.action == "reclaim_once"
|
||||
# The margin sizes the pass; it never re-arms the route+round latch.
|
||||
after = _measure(
|
||||
monkeypatch, tmp_path, plan=plan, profile=profile, mode=preferred, messages=messages,
|
||||
used=True,
|
||||
)
|
||||
assert after.action != "reclaim_once"
|
||||
assert after.measurement.low_water_margin_tokens == measurement.low_water_margin_tokens
|
||||
|
||||
|
||||
@pytest.mark.parametrize("profile,preferred,window,known", [
|
||||
("owner_max", "max", 0, False),
|
||||
("owner_max", "max", 500_000, True),
|
||||
("owner_low", "low", 500_000, True),
|
||||
("task_local_low", "low", 500_000, True),
|
||||
])
|
||||
def test_no_deficit_means_no_margin_and_no_goal(monkeypatch, tmp_path, profile, preferred, window, known):
|
||||
plan = _plan(preferred=preferred, window=window, known=known)
|
||||
disposition = _measure(
|
||||
monkeypatch, tmp_path, plan=plan, profile=profile, mode=preferred,
|
||||
messages=plan.messages_for(preferred),
|
||||
)
|
||||
assert disposition.measurement.low_water_margin_tokens == 0
|
||||
assert disposition.measurement.reclaim_goal_tokens == 0
|
||||
assert disposition.action == "send"
|
||||
|
||||
|
||||
def test_low_water_divisor_is_the_one_knob_in_both_directions(monkeypatch, tmp_path):
|
||||
from ouroboros import context_budget as cb
|
||||
|
||||
plan = _plan(window=70_000, known=True)
|
||||
messages = plan.messages_for("max") + [{"role": "user", "content": "x" * 40_000}]
|
||||
base = _measure(
|
||||
monkeypatch, tmp_path, plan=plan, profile="owner_max", mode="max", messages=messages,
|
||||
).measurement
|
||||
deficit = base.capacity_deficit_tokens
|
||||
assert deficit > 0 and base.low_water_margin_tokens == math.ceil(70_000 / 8)
|
||||
|
||||
monkeypatch.setattr(cb, "RECLAIM_LOW_WATER_DIVISOR", 4)
|
||||
wider = _measure(
|
||||
monkeypatch, tmp_path, plan=plan, profile="owner_max", mode="max", messages=messages,
|
||||
).measurement
|
||||
assert wider.low_water_margin_tokens == math.ceil(70_000 / 4)
|
||||
assert wider.reclaim_goal_tokens == deficit + wider.low_water_margin_tokens
|
||||
|
||||
monkeypatch.setattr(cb, "RECLAIM_LOW_WATER_DIVISOR", 10 ** 9) # the margin removed
|
||||
removed = _measure(
|
||||
monkeypatch, tmp_path, plan=plan, profile="owner_max", mode="max", messages=messages,
|
||||
).measurement
|
||||
assert removed.low_water_margin_tokens == 1
|
||||
assert removed.reclaim_goal_tokens == deficit + 1
|
||||
assert removed.capacity_deficit_tokens == deficit # the trigger never moved
|
||||
|
||||
|
||||
def test_task_local_low_does_not_inherit_owner_economy_target(monkeypatch, tmp_path):
|
||||
plan = _plan(window=500_000, known=True)
|
||||
disposition = _measure(
|
||||
|
|
|
|||
|
|
@ -2,9 +2,15 @@
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from dataclasses import replace
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
|
||||
from ouroboros.context_budget import RECLAIM_LOW_WATER_DIVISOR
|
||||
from ouroboros.context_fit import reclaim_low_water_margin
|
||||
|
||||
|
||||
def _fit(
|
||||
*,
|
||||
|
|
@ -15,6 +21,8 @@ def _fit(
|
|||
target_deficit=None,
|
||||
capacity_deficit=None,
|
||||
used=False,
|
||||
estimated_input=120_000,
|
||||
low_water_margin=0,
|
||||
):
|
||||
from ouroboros.context_fit import MainFitDisposition, MainFitMeasurement
|
||||
|
||||
|
|
@ -23,7 +31,7 @@ def _fit(
|
|||
round_id="exec:round:1",
|
||||
profile=profile,
|
||||
rendered_mode=mode,
|
||||
estimated_input_tokens=120_000,
|
||||
estimated_input_tokens=estimated_input,
|
||||
response_reserve_tokens=65_536,
|
||||
target_total_tokens=200_000 if profile == "owner_low" else None,
|
||||
capacity_total_tokens=500_000,
|
||||
|
|
@ -32,6 +40,7 @@ def _fit(
|
|||
target_deficit_tokens=target_deficit,
|
||||
capacity_deficit_tokens=capacity_deficit,
|
||||
reclaim_goal_tokens=goal,
|
||||
low_water_margin_tokens=low_water_margin,
|
||||
)
|
||||
return MainFitDisposition(
|
||||
measurement=measurement,
|
||||
|
|
@ -220,6 +229,175 @@ def test_checkpoint_proves_automatic_materializer_attempt_even_on_binding_mismat
|
|||
assert ("route-a", "exec:round:1") in context.tools._ctx._context_reclaim_materializations
|
||||
|
||||
|
||||
def _applied_receipt(*, reclaimed: int, goal_reached: bool):
|
||||
from ouroboros.context_budget import ContextReclaimReceipt
|
||||
|
||||
return ContextReclaimReceipt(
|
||||
status="applied",
|
||||
before_transcript_sha256="a" * 64,
|
||||
after_transcript_sha256="b" * 64,
|
||||
selection_fingerprint="f" * 64,
|
||||
selected_unit_ids=("unit",),
|
||||
reclaimed_tokens=reclaimed,
|
||||
goal_reached=goal_reached,
|
||||
checkpoint_ref={"path": "checkpoint", "sha256": "c" * 64},
|
||||
capsule_refs=(),
|
||||
)
|
||||
|
||||
|
||||
# Owner Low: the 200,000 target binds (capacity 500,000); its input boundary is
|
||||
# 200,000 - 65,536 = 134,464 estimated tokens; margin = ceil(200,000 / 8) = 25,000.
|
||||
_LOW_BOUNDARY_INPUT = 200_000 - 65_536
|
||||
|
||||
|
||||
@pytest.mark.parametrize("landed_input,headroom,reached,below", [
|
||||
(_LOW_BOUNDARY_INPUT, 0, True, False), # reclaimed == deficit: AT the boundary, not below
|
||||
(_LOW_BOUNDARY_INPUT - 25_000, 25_000, True, True), # the full margin achieved
|
||||
(_LOW_BOUNDARY_INPUT - 12_000, 12_000, True, False), # under-landed: reached, margin missed
|
||||
(_LOW_BOUNDARY_INPUT + 2_000, -2_000, False, False), # still above the boundary
|
||||
])
|
||||
def test_reclaim_checkpoint_separates_boundary_from_low_water(
|
||||
tmp_path, monkeypatch, landed_input, headroom, reached, below,
|
||||
):
|
||||
from ouroboros import loop
|
||||
|
||||
context = _ctx(tmp_path, preferred="low", mode="low")
|
||||
disposition = _fit(
|
||||
action="reclaim_once", profile="owner_low", mode="low", goal=10_000 + 25_000,
|
||||
target_deficit=10_000, capacity_deficit=0, estimated_input=_LOW_BOUNDARY_INPUT + 10_000,
|
||||
low_water_margin=25_000,
|
||||
)
|
||||
landed = _fit(
|
||||
action="send", profile="owner_low", mode="low", used=True, estimated_input=landed_input,
|
||||
)
|
||||
measured, events = [], []
|
||||
monkeypatch.setattr(
|
||||
loop, "compact_tool_history_llm",
|
||||
lambda *_a, **_kw: (context.messages, _applied_receipt(reclaimed=1, goal_reached=False), None),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
loop, "_measure_round_main_fit", lambda _ctx, **kwargs: measured.append(kwargs) or landed,
|
||||
)
|
||||
monkeypatch.setattr(loop, "_emit_checkpoint_event", lambda _q, _t, _d, data: events.append(data))
|
||||
|
||||
loop._run_main_reclaim(context, disposition)
|
||||
|
||||
# The landing is re-measured on the SAME basis as the trigger, once, as a used pass.
|
||||
assert measured == [{"automatic_pass_used": True}]
|
||||
event = events[-1]
|
||||
assert event["checkpoint_kind"] == "context_reclaim_automatic"
|
||||
assert event["deficit_tokens"] == 10_000
|
||||
assert event["requested_margin_tokens"] == 25_000
|
||||
assert event["reclaim_goal_tokens"] == 35_000
|
||||
assert event["achieved_headroom_tokens"] == headroom
|
||||
assert event["boundary_reached"] is reached
|
||||
assert event["below_boundary"] is below
|
||||
assert event["rounds_since_previous_pass"] is None
|
||||
assert context.tools._ctx._context_reclaim_last_pass_round == 1
|
||||
|
||||
|
||||
def test_reclaim_checkpoint_counts_rounds_since_the_previous_pass_without_remeasuring_a_no_op(
|
||||
tmp_path, monkeypatch,
|
||||
):
|
||||
from ouroboros import loop
|
||||
|
||||
context = replace(_ctx(tmp_path, preferred="low", mode="low"), round_idx=9)
|
||||
context.tools._ctx._context_reclaim_last_pass_round = 3
|
||||
disposition = _fit(
|
||||
action="reclaim_once", profile="owner_low", mode="low", goal=10_000 + 25_000,
|
||||
target_deficit=10_000, capacity_deficit=0, estimated_input=_LOW_BOUNDARY_INPUT + 10_000,
|
||||
low_water_margin=25_000,
|
||||
)
|
||||
disposition = replace(
|
||||
disposition, measurement=replace(disposition.measurement, round_id="exec:round:9"),
|
||||
)
|
||||
events = []
|
||||
monkeypatch.setattr(
|
||||
loop, "compact_tool_history_llm",
|
||||
lambda *_a, **_kw: (context.messages, replace(
|
||||
_applied_receipt(reclaimed=0, goal_reached=False), status="no_eligible",
|
||||
), None),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
loop, "_measure_round_main_fit",
|
||||
lambda *_a, **_kw: pytest.fail("a pass that changed nothing has no new landing to measure"),
|
||||
)
|
||||
monkeypatch.setattr(loop, "_emit_checkpoint_event", lambda _q, _t, _d, data: events.append(data))
|
||||
|
||||
loop._run_main_reclaim(context, disposition)
|
||||
|
||||
event = events[-1]
|
||||
assert event["status"] == "no_eligible"
|
||||
assert event["rounds_since_previous_pass"] == 6
|
||||
# Unchanged transcript: the pre-pass fit IS the landing, 10,000 above the boundary.
|
||||
assert event["achieved_headroom_tokens"] == -10_000
|
||||
assert event["boundary_reached"] is False
|
||||
assert event["below_boundary"] is False
|
||||
assert event["requested_margin_tokens"] == 25_000
|
||||
assert context.tools._ctx._context_reclaim_last_pass_round == 9
|
||||
|
||||
|
||||
def test_reclaim_checkpoint_reports_an_unmeasurable_landing_as_unknown(tmp_path, monkeypatch):
|
||||
"""A route/plan rebind can leave the post-pass fit unmeasurable (the fit seam
|
||||
returns None by contract): the pass still applies and the event says UNKNOWN,
|
||||
never "boundary not reached"."""
|
||||
from ouroboros import loop
|
||||
|
||||
context = _ctx(tmp_path, preferred="low", mode="low")
|
||||
disposition = _fit(
|
||||
action="reclaim_once", profile="owner_low", mode="low", goal=10_000 + 25_000,
|
||||
target_deficit=10_000, capacity_deficit=0, estimated_input=_LOW_BOUNDARY_INPUT + 10_000,
|
||||
low_water_margin=25_000,
|
||||
)
|
||||
events = []
|
||||
monkeypatch.setattr(
|
||||
loop, "compact_tool_history_llm",
|
||||
lambda *_a, **_kw: (context.messages, _applied_receipt(reclaimed=30_000, goal_reached=False), None),
|
||||
)
|
||||
monkeypatch.setattr(loop, "_measure_round_main_fit", lambda *_a, **_kw: None)
|
||||
monkeypatch.setattr(loop, "_emit_checkpoint_event", lambda _q, _t, _d, data: events.append(data))
|
||||
|
||||
receipt = loop._run_main_reclaim(context, disposition)
|
||||
|
||||
assert receipt.status == "applied"
|
||||
assert events[-1]["reclaimed_tokens"] == 30_000
|
||||
assert events[-1]["achieved_headroom_tokens"] is None
|
||||
assert events[-1]["boundary_reached"] is None
|
||||
assert events[-1]["below_boundary"] is None
|
||||
assert events[-1]["requested_margin_tokens"] == 25_000
|
||||
|
||||
|
||||
def test_overflow_minimum_goal_is_low_water_sized_even_without_a_predicted_deficit(
|
||||
tmp_path, monkeypatch,
|
||||
):
|
||||
"""Both directions: the measurement found no deficit (goal 0), the provider still
|
||||
overflowed; the pass requests the low-water minimum, and its telemetry reports
|
||||
that minimum as the requested margin over a zero deficit."""
|
||||
from ouroboros import loop
|
||||
|
||||
context = _ctx(tmp_path, preferred="low", mode="low")
|
||||
disposition = _fit(action="send", profile="owner_low", mode="low", goal=0, target_deficit=0,
|
||||
capacity_deficit=0, estimated_input=_LOW_BOUNDARY_INPUT - 500)
|
||||
requests, events = [], []
|
||||
|
||||
def compact(messages, *, request, **_kwargs):
|
||||
requests.append(request.reclaim_goal_tokens)
|
||||
return messages, replace(_applied_receipt(reclaimed=0, goal_reached=False),
|
||||
status="no_eligible"), None
|
||||
|
||||
monkeypatch.setattr(loop, "compact_tool_history_llm", compact)
|
||||
monkeypatch.setattr(loop, "_emit_checkpoint_event", lambda _q, _t, _d, data: events.append(data))
|
||||
minimum = max(1, reclaim_low_water_margin(200_000, 500_000)) # what the overflow path passes
|
||||
loop._run_main_reclaim(context, disposition, minimum_goal_tokens=minimum)
|
||||
|
||||
assert requests == [25_000]
|
||||
assert events[-1]["deficit_tokens"] == 0
|
||||
assert events[-1]["requested_margin_tokens"] == 25_000
|
||||
assert events[-1]["achieved_headroom_tokens"] == 500
|
||||
assert events[-1]["boundary_reached"] is True
|
||||
assert events[-1]["below_boundary"] is False
|
||||
|
||||
|
||||
def test_predicted_reclaim_runs_once_then_sends_target_miss(tmp_path, monkeypatch):
|
||||
from ouroboros import loop
|
||||
|
||||
|
|
@ -276,7 +454,9 @@ def test_actual_max_overflow_reprojects_and_retries_only_smaller_context(tmp_pat
|
|||
return disposition
|
||||
|
||||
def reclaim(ctx, _disposition, **kwargs):
|
||||
assert kwargs["minimum_goal_tokens"] == 1
|
||||
# An actual overflow asks for a low-water-sized pass (an eighth of the 500K
|
||||
# route), never a token-sized one, before its single strict-shrink retry.
|
||||
assert kwargs["minimum_goal_tokens"] == math.ceil(500_000 / RECLAIM_LOW_WATER_DIVISOR)
|
||||
ctx.tools._ctx._context_reclaim_passes.add(("route-a", "exec:round:1"))
|
||||
|
||||
sends = []
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@
|
|||
import asyncio
|
||||
from dataclasses import replace
|
||||
import json
|
||||
import math
|
||||
import socket
|
||||
import threading
|
||||
from types import SimpleNamespace
|
||||
|
|
@ -10,6 +11,7 @@ from types import SimpleNamespace
|
|||
import pytest
|
||||
|
||||
from ouroboros import context, loop, task_pacing, usage_accounting as accounting
|
||||
from ouroboros.context_budget import RECLAIM_LOW_WATER_DIVISOR
|
||||
from ouroboros.contracts.task_contract import normalize_budget_profile
|
||||
from ouroboros.owner_wait import checkpoint_owner_wait, set_owner_wait
|
||||
from ouroboros.task_results import write_task_result
|
||||
|
|
@ -78,7 +80,9 @@ def test_cold_switched_model_keeps_actual_overflow_reclaim_and_retry(tmp_path, m
|
|||
sends, reclaimed = [], []
|
||||
|
||||
def reclaim(call, disposition, **kwargs):
|
||||
assert kwargs["minimum_goal_tokens"] == 1
|
||||
# An actual overflow requests a low-water-sized pass (an eighth of the
|
||||
# 500K route), never a token-sized one, before its single strict-shrink retry.
|
||||
assert kwargs["minimum_goal_tokens"] == math.ceil(500_000 / RECLAIM_LOW_WATER_DIVISOR)
|
||||
reclaimed.append(call.active_model)
|
||||
loop._context_reclaim_passes(call.tools._ctx).add(
|
||||
(disposition.measurement.route_fp, disposition.measurement.round_id))
|
||||
|
|
|
|||
|
|
@ -315,8 +315,17 @@ def test_automatic_reclaim_inside_the_model_call_is_a_sanctioned_break_on_its_ro
|
|||
if ctx.round_idx != 3 or fired or automatic_pass_used:
|
||||
return None
|
||||
fired.append(True)
|
||||
measurement = SimpleNamespace(route_fp="fp", round_id="r3", measurement_basis="cold_estimate",
|
||||
measurement_density=1.0, reclaim_goal_tokens=100)
|
||||
from ouroboros.context_fit import MainFitMeasurement
|
||||
|
||||
# A real measurement: only a positive deficit decides "reclaim_once", and the
|
||||
# reclaim's low-water telemetry reads the deficit and boundary fields.
|
||||
measurement = MainFitMeasurement(
|
||||
route_fp="fp", round_id="r3", profile="owner_low", rendered_mode="low",
|
||||
estimated_input_tokens=200_000 - 65_536 + 100, response_reserve_tokens=65_536,
|
||||
target_total_tokens=200_000, capacity_total_tokens=None,
|
||||
measurement_basis="cold_estimate", measurement_density=1.0,
|
||||
target_deficit_tokens=100, capacity_deficit_tokens=None, reclaim_goal_tokens=100,
|
||||
)
|
||||
return SimpleNamespace(action="reclaim_once", measurement=measurement, automatic_pass_used=False)
|
||||
|
||||
monkeypatch.setattr(loop, "_measure_round_main_fit", measure)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue