mirror of
https://github.com/razzant/ouroboros.git
synced 2026-10-03 12:18:39 +00:00
BREAKING (ABI 7.0 window, owner Q10=A verbatim: 'FLOOR/fail_tasks/ deadline-алиасы — удалить'). Both knobs were deprecated one-minor aliases whose own event announced 'removal: next_major' - this is that major. - contracts/task_contract.py: VALID_IMPROVEMENT_POLICIES = (fixed, adaptive); an unknown policy (incl. the retired spelling) normalizes to 'fixed'; stall_rounds_threshold leaves the normalized profile shape (it was normalized but consumed by nothing). - task_pacing.py: the deprecated_task_pacing_alias event machinery in resolve_budget_profile is gone; the until_deadline count-axis lift in effective_max_improvement_passes is gone, and with it the has_deadline parameter that existed solely for that branch (callers in task_results and the rails line updated; the deadline/reserve TIME rail is untouched). - bench adapters (sanctioned explicitly): programbench schemas.py + README and swe_bench_pro entrypoint_pro.sh + METHODOLOGY switch to improvement_policy=fixed - behavior-identical for those runs because their explicit max_improvement_passes=6 was ALWAYS the binding count axis under every policy; stall_rounds_threshold=12 dropped (never consumed). - docs: ARCHITECTURE task_pacing module line and the acceptance passage now state the removal. - tests: the aliases' own tests removed (review_cycles alias test + rails A3 clause, two v6544 until_deadline tests); vehicle tests adapted to the surviving semantics (wallet-authority test now derives its uncapped lane from the unlimited shared cap; v664 deprecation-noise test now pins that NO deprecation events are emitted at all; headless CLI forwards 'adaptive'; contract-shape and PB/SWE-Pro expectation pins updated). NEW removal pins in tests/test_abi5_q10_removals.py: policy tuple, normalize-away shape, signature no longer takes has_deadline, functional-remnant sweep. Disclosed consequence: a PRE-7.0 stored root contract whose normalized profile says until_deadline is judged malformed by the acceptance-wallet authority (pre-existing unknown-policy behavior); pre-7.0 task-result history is quarantined wholesale by ABI-2 (Q8=B) in this same release, and the ABI-7 RC auditor names the migration. Gates: ruff F clean; size_ratchet 5 passed; affected suites 478+90 passed.
116 lines
4.9 KiB
Python
116 lines
4.9 KiB
Python
"""ProgramBench adapter schemas."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import Any
|
|
|
|
# Official ProgramBench per-task budget (6h). Flows through the task body's
|
|
# ``timeout_sec`` (the gateway turns it into ``deadline_at``), not the contract.
|
|
PROGRAMBENCH_TIMEOUT_SEC = 21600.0
|
|
|
|
|
|
def programbench_budget_profile() -> dict[str, Any]:
|
|
"""Improvement-pacing block mapped onto ``task_contract.budget_profile``.
|
|
|
|
Exactly the five keys ``normalize_budget_profile`` accepts. The original
|
|
prototype's per-round footer keys (max_llm_rounds, show_every_round,
|
|
urgency_show_every_round_below_pct, ...) were rejected — per-round user-turn
|
|
churn breaks prompt caching — so they are deliberately absent: round caps
|
|
come from settings (OUROBOROS_MAX_ROUNDS in settings_base.json) and the 6h
|
|
wall clock from ``timeout_sec``.
|
|
"""
|
|
return {
|
|
# No in-task cost stop for ProgramBench runs (owner decision, v6.56.0):
|
|
# the 6h deadline and the run's own key/budget caps own the bounds;
|
|
# cost milestones stay informational against the start snapshot.
|
|
"cost_hard_stop_pct": 0,
|
|
# ABI 7.0 (Q10=A): the legacy until_deadline alias is removed. The
|
|
# explicit task-local cap below was ALWAYS authoritative over the count
|
|
# axis (an explicit cap binds under every policy), so "fixed" + cap 6
|
|
# reproduces the run behavior exactly; the deadline/reserve rails still
|
|
# stop the loop earlier when time runs out.
|
|
"improvement_policy": "fixed",
|
|
# ProgramBench allows up to six acceptance/improvement passes.
|
|
"max_improvement_passes": 6,
|
|
# 0-100 percentage of the total budget kept for finalization
|
|
# (15% of 6h ≈ the last ~54 minutes).
|
|
"reserve_finalization_pct": 15,
|
|
# stall_rounds_threshold was removed with the 7.0 ABI window (Q10=A):
|
|
# it was normalized into the contract but never consumed by the runtime.
|
|
}
|
|
|
|
|
|
def protected_reference_policy(paths: list[str]) -> dict[str, Any]:
|
|
clean = [str(path) for path in paths if str(path or "").strip()]
|
|
return {
|
|
"protected_artifacts": [
|
|
{
|
|
"id": "programbench_reference",
|
|
"role": "black_box_reference",
|
|
"paths": clean,
|
|
"allow": ["execute"],
|
|
"deny": [
|
|
"read_bytes",
|
|
"copy",
|
|
"hash",
|
|
"static_introspection",
|
|
"dynamic_trace",
|
|
"debug",
|
|
],
|
|
}
|
|
]
|
|
}
|
|
|
|
|
|
def task_body(
|
|
*,
|
|
description: str,
|
|
workspace_root: str,
|
|
executor_ref: dict[str, Any],
|
|
protected_paths: list[str],
|
|
task_id: str = "",
|
|
) -> dict[str, Any]:
|
|
body: dict[str, Any] = {
|
|
"description": description,
|
|
"workspace_root": workspace_root,
|
|
"workspace_mode": "external",
|
|
"memory_mode": "empty",
|
|
"allowed_resources": {"web": False, "network": False, "internet": False},
|
|
"resource_policy": protected_reference_policy(protected_paths),
|
|
"executor_ref": executor_ref,
|
|
# House rule: benches measure the single-model Ouroboros harness, so the
|
|
# external coding-agent gateway is withheld from the solve task.
|
|
"disabled_tools": ["claude_code_edit", "schedule_subagent"], # operator 2026-07-23: subagents=0
|
|
"actor_id": "programbench",
|
|
"source": "programbench",
|
|
# Advisory Observable Acceptance Claims (task-general vocabulary; the
|
|
# deliberately GENERAL wording steers verification toward broadening
|
|
# behavioral coverage without naming any benchmark-specific oracle).
|
|
"acceptance_claims": [
|
|
{
|
|
"id": "behavioral_equivalence",
|
|
"claim": (
|
|
"The built deliverable reproduces the provided reference executable's "
|
|
"observable behavior; a diff against that reference is an independent "
|
|
"oracle available in this environment."
|
|
),
|
|
"surface": "differential runs of the deliverable vs the provided reference executable",
|
|
"support": (
|
|
"verification receipts from differential probe passes; aim verification at "
|
|
"EXPANDING behavioral coverage (flags, boundaries, error paths), not at "
|
|
"repeating already-green probes"
|
|
),
|
|
"priority": "must",
|
|
},
|
|
],
|
|
"metadata": {
|
|
"source": "programbench",
|
|
# POST /api/tasks accepts no top-level task_contract field;
|
|
# metadata.budget_profile is the supported wiring — build_task_contract()
|
|
# normalizes it additively into task_contract.budget_profile.
|
|
"budget_profile": programbench_budget_profile(),
|
|
},
|
|
}
|
|
if task_id:
|
|
body["task_id"] = task_id
|
|
return body
|