From b60b8fe98a15454a12da65e8350f4de8f312e209 Mon Sep 17 00:00:00 2001 From: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> Date: Fri, 25 Sep 2026 17:18:15 +0300 Subject: [PATCH] Narrow the reasoning-effort change to its landed core Revert the route-descriptor registry, the descriptor-driven effort pickers, the Anthropic builder indirection and the effort_not_carried plumbing back to the target's state, and restore DeepSeek's documented projection (xhigh -> high per api-docs.deepseek.com Thinking Mode). What the maintainers found on the combined tree: nine target-green tests went red (five llm_golden compatible/cloudru/minimax cases, three host-message-metadata cases, the provider_models/config leaf guard), the 07-configuration chapter left its byte budget, the picker hid saved and working tiers (xhigh/max/none) and matched the catalog by bare model id, the reviewer-slot wiring passed a fourth argument that the local wrapper dropped, and the new usage fact never reached events.jsonl. The z.ai GLM effort carriage itself is kept and lands as a first-class provider in the next commit. Kept from this change: Z.ai's HTTP 429 code 1113 ("Insufficient balance") is a billing fact, so the provider Test maps it to "No credits" instead of "Rate limited" (typed code only; the status copy stays provider-neutral). --- docs/architecture/07-configuration.md | 12 +- ouroboros/gateway/models.py | 13 - ouroboros/llm_anthropic.py | 10 +- ouroboros/llm_capability_policy.py | 15 -- ouroboros/llm_openai_compatible.py | 71 ++---- ouroboros/llm_probe.py | 10 +- ouroboros/provider_models.py | 261 +------------------- tests/test_deepseek_provider.py | 12 +- tests/test_effort_route_descriptor.py | 282 ---------------------- web/modules/reviewer_slots.js | 10 +- web/modules/route_editor_primitives.js | 63 +---- web/modules/settings.js | 9 +- web/modules/settings_catalog.js | 18 -- web/modules/settings_ui.js | 34 --- web/modules/subagents_settings.js | 4 +- web/tests/effort_route_descriptor.test.js | 66 ----- 16 files changed, 48 insertions(+), 842 deletions(-) delete mode 100644 tests/test_effort_route_descriptor.py delete mode 100644 web/tests/effort_route_descriptor.test.js diff --git a/docs/architecture/07-configuration.md b/docs/architecture/07-configuration.md index 288c93e00..7c8845311 100644 --- a/docs/architecture/07-configuration.md +++ b/docs/architecture/07-configuration.md @@ -221,17 +221,7 @@ A registry of `config.SETTINGS_DEFAULTS` (exact defaults canonical in `settings_ Direct-provider review fallback (legacy name: OpenAI-only review fallback): with exactly one official direct provider configured, `config.get_review_models()` compiles that provider's declarative reviewer-role sequence from provider-prefixed model IDs. Scope covers official OpenAI, Anthropic, MiniMax, DeepSeek, Cloud.ru, and GigaChat; OpenRouter, legacy-base, OpenAI-compatible and mixed configurations stay outside. Per-provider role coverage differs — three independent Main slots down to one role model for every slot (`provider_models.compute_direct_review_models_fallback`). `_exclusive_direct_remote_provider_env` returns empty when OpenRouter, legacy `OPENAI_BASE_URL`, OpenAI-compatible keys or several direct providers are present, and the fallback requires `provider_models.migrate_model_value` to make the main model already start with the exclusive provider prefix, so free text cannot silently enter a single-provider route (DEVELOPMENT "Provider Independence"). -DeepSeek (`deepseek::`): the OpenAI-compatible endpoint is the fixed constant `provider_models.DEEPSEEK_BASE_URL`; a proxy or mirror belongs to `openai-compatible::`, and slash-form `deepseek/...` stays OpenRouter. The canonical effort scale is projected onto the provider's wire dialect at the send boundary, a forced tool choice is served with thinking disabled (thinking accepts only `auto`/`none`), and every tier change is disclosed as `reasoning_effort_clamped` (projection table: `provider_models.EFFORT_ROUTE_ALIASES_LMH`, shared with the GLM/z.ai route). `reasoning_content` stays on CANONICAL assistant turns for strict v4 replay (an explicit empty string marks a turn produced without provider reasoning); other lanes strip the field from their physical send copy and cross-family switches scrub it. On the send copy only, system/assistant/tool content arrays are flattened to strings — the API accepts arrays on user turns alone (`llm_openai_compatible.py`). Caching is automatic, cost stays nullable without an exact catalog, and the 1M context claim needs route-fingerprinted evidence or owner acknowledgement. - -#### Reasoning-effort carriage (descriptor SSOT) - -`provider_models.EFFORT_ROUTE_DESCRIPTOR` is the ONE table declaring, per registered route: the wire `carrier` (`reasoning_effort` | `extra_body.reasoning` | `anthropic.adaptive` | `reasoningEffort` | `none`), the accepted provider `tiers`, the canonical→provider `project`ion, what an ABSENT tier means (`absent_meaning`: `max` | `provider_default` | `off`), and whether a forced tool choice requires thinking suppression. All payload builders (`llm_openai_compatible._build_remote_kwargs`, `llm_anthropic._build_remote_candidate`, the OpenRouter lane) read the table — no per-provider builder branches. Consequences every surface must state: - -- A route with carrier `none` DROPS the tier, disclosed on usage as `effort_not_carried` (never a silent vanish, never a guessed carrier for an unmeasured wire). qwen/kimi/zai placeholder rows carry `none` until measured with a key. -- GLM via `openai-compatible::` (z.ai PAYG and Coding Plan endpoints) resolves to the measured `zai-glm` row by model identity: enum `low/high/max`, shared alias projection `minimal→low, medium→high, xhigh/ultra→max`, and an ABSENT tier bills at MAX — the loudest absent_meaning in the set. Thinking cannot be disabled on this route (HTTP 400 code 1210); `tool_choice: required` and named-function forcing WORK with thinking on (measured on glm-5.3; an earlier draft of this page claimed `auto`-only — corrected by probe), so DeepSeek's forced-tool thinking suppression does NOT apply. -- A descriptor↔capability_evidence disagreement (an observed effort ceiling above the declared top tier, or a floor below the declared bottom) is a loud fact via `provider_models._effort_descriptor_mismatches_observation`, never a silent table override. -- The model catalog exposes the descriptor per entry (`gateway/models.py::_build_model_catalog_entry` → `effort_descriptor{carrier,tiers,canonical_tiers,absent_meaning}`) so the settings UI offers exactly the route's tiers (3 for GLM, not 8) with projection annotations, and names the inheritance when no explicit tier is chosen. -- Z.ai plan exhaustion arrives as HTTP 429 code 1113 "Insufficient balance": billing, not rate limiting — `llm_probe.controlled_probe_error` maps it to the `No credits` reason and the settings test-status line carries the provider message plus a plan hint, never a bare "Rate limited". +DeepSeek (`deepseek::`): the OpenAI-compatible endpoint is the fixed constant `provider_models.DEEPSEEK_BASE_URL`; a proxy or mirror belongs to `openai-compatible::`, and slash-form `deepseek/...` stays OpenRouter. The canonical effort scale is projected onto the provider's wire dialect at the send boundary, a forced tool choice is served with thinking disabled (thinking accepts only `auto`/`none`), and every tier change is disclosed as `reasoning_effort_clamped` (projection table: `provider_models.DEEPSEEK_REASONING_EFFORT_ALIASES`). `reasoning_content` stays on CANONICAL assistant turns for strict v4 replay (an explicit empty string marks a turn produced without provider reasoning); other lanes strip the field from their physical send copy and cross-family switches scrub it. On the send copy only, system/assistant/tool content arrays are flattened to strings — the API accepts arrays on user turns alone (`llm_openai_compatible.py`). Caching is automatic, cost stays nullable without an exact catalog, and the 1M context claim needs route-fingerprinted evidence or owner acknowledgement. GigaChat (`gigachat::`): the native `gigachat` library, not OpenAI-compatible — OpenAI `tools` map to GigaChat `functions`, one `function_call` per turn (parallel `tool_calls` collapse to the first), `tool` results become role `function` and must be valid JSON (plain text is wrapped as `{"result": ...}`), and `system` must come first, so later system-reminders demote to `user` (`llm.py::_chat_gigachat`). `reasoning_effort` is deliberately omitted: hidden reasoning can consume the whole output budget and return empty content. No live cost source exists, so cost stays nullable/unknown, never a hand-maintained tariff. A GigaChat scope row runs native retrieval on its own window, with at most one function call per turn. Reading gaps are diagnostic and never remove its response from quorum; the agent decides whether more reading is needed. Missing inspection tools still produce `native_inspection_unavailable`, not a completed review, and the blocking triad continues to review the full staged diff (§6 Review stack). diff --git a/ouroboros/gateway/models.py b/ouroboros/gateway/models.py index 2072ffbb1..ee0572e3c 100644 --- a/ouroboros/gateway/models.py +++ b/ouroboros/gateway/models.py @@ -59,16 +59,9 @@ def _build_model_catalog_entry( display_name: str, source: str | None = None, ) -> dict[str, str]: - from ouroboros.provider_models import canonical_tiers_for_route, effort_descriptor_for_route - raw_id = str(model_id or "").strip() name = str(display_name or "").strip() or raw_id source_label = source or provider_label - # The reasoning-effort descriptor rides every catalog entry so all UI - # surfaces (Behavior effort cards, reviewer slots, subagent roster, model - # pickers) offer exactly the tiers the route's wire accepts — 3 for a GLM - # route, "no carrier" for unmeasured ones — from ONE SSOT table. - descriptor = effort_descriptor_for_route(provider_id, raw_id) return { "provider_id": provider_id, "provider": provider_label, @@ -77,12 +70,6 @@ def _build_model_catalog_entry( "name": name, "value": _tagged_model_value(provider_id, raw_id), "label": f"{source_label} · {name}", - "effort_descriptor": { - "carrier": descriptor.get("carrier"), - "tiers": list(descriptor.get("tiers") or []), - "canonical_tiers": canonical_tiers_for_route(descriptor), - "absent_meaning": descriptor.get("absent_meaning"), - }, } diff --git a/ouroboros/llm_anthropic.py b/ouroboros/llm_anthropic.py index 603e07e5f..2dddd1b97 100644 --- a/ouroboros/llm_anthropic.py +++ b/ouroboros/llm_anthropic.py @@ -453,18 +453,12 @@ class _AnthropicLaneMixin: "model": str(target.get("resolved_model") or ""), "messages": messages, "max_tokens": max_tokens, } - # Descriptor-driven carriage (provider_models SSOT): the anthropic row - # declares carrier "anthropic.adaptive" (output_config.effort + thinking) - # with minimal→low projection; "none" switches thinking off. - from ouroboros.provider_models import effort_descriptor_for_route, project_effort_for_route - descriptor = effort_descriptor_for_route( - str(target.get("provider") or ""), str(target.get("resolved_model") or "")) effort = normalize_reasoning_effort(reasoning_effort) if effort == "none": payload["thinking"] = {"type": "disabled"} - elif effort and descriptor.get("carrier") == "anthropic.adaptive": + elif effort: payload["thinking"] = {"type": "adaptive"} - payload["output_config"] = {"effort": project_effort_for_route(descriptor, effort)} + payload["output_config"] = {"effort": "low" if effort == "minimal" else effort} if system: payload["system"] = system if temperature is not None: diff --git a/ouroboros/llm_capability_policy.py b/ouroboros/llm_capability_policy.py index a2a4db35d..c764346f1 100644 --- a/ouroboros/llm_capability_policy.py +++ b/ouroboros/llm_capability_policy.py @@ -31,11 +31,6 @@ log = logging.getLogger("ouroboros.llm") # Effort clamp/projection disclosure slot: a ContextVar isolates threads AND # concurrent asyncio tasks (same isolation contract as the reasoning pin note). _EFFORT_CLAMP_CVAR = contextvars.ContextVar("ouroboros_effort_clamp_note", default=None) -# A tier the caller requested but the route's descriptor carries no carrier for: -# the tier is DROPPED from the physical request (never guessed onto a wire that -# was not measured), and this note discloses the drop on usage as -# ``effort_not_carried`` — the loud fact that replaces the old silent vanish. -_EFFORT_NOT_CARRIED_CVAR = contextvars.ContextVar("ouroboros_effort_not_carried_note", default=None) _OPTIONAL_SAMPLING_PARAMS = ("temperature", "top_p", "top_k") @@ -414,16 +409,6 @@ class _CapabilityPolicyMixin: _EFFORT_CLAMP_CVAR.set(None) return pending if isinstance(pending, dict) else None - def _pop_effort_not_carried_disclosure(self) -> Optional[Dict[str, Any]]: - """The pending effort_not_carried record for THIS call's context. - - Mirrors the clamp disclosure: staged by the payload builder when the - route descriptor carries no effort carrier, consumed once on the - response's usage custody so the drop is a visible usage fact.""" - pending = _EFFORT_NOT_CARRIED_CVAR.get() - _EFFORT_NOT_CARRIED_CVAR.set(None) - return pending if isinstance(pending, dict) else None - @classmethod def _record_effort_ceiling(cls, model_id: str, current_effort: str) -> None: """Record a legacy model-global ceiling for diagnostics.""" diff --git a/ouroboros/llm_openai_compatible.py b/ouroboros/llm_openai_compatible.py index 2dad0f738..5a8ee463a 100644 --- a/ouroboros/llm_openai_compatible.py +++ b/ouroboros/llm_openai_compatible.py @@ -22,16 +22,12 @@ from ouroboros.usage_accounting import UsageScope, usage_scope from ouroboros.openrouter_attribution import OPENROUTER_APP_HEADERS from ouroboros.llm_capability_policy import ( _EFFORT_CLAMP_CVAR, - _EFFORT_NOT_CARRIED_CVAR, _OPTIONAL_DROPPABLE_PARAMS, normalize_reasoning_effort, ) from ouroboros.reasoning_artifacts import transcript_has_sealed_reasoning from ouroboros.llm_routing import _resolve_or_provider -from ouroboros.provider_models import ( - effort_descriptor_for_route, - project_effort_for_route, -) +from ouroboros.provider_models import normalize_deepseek_reasoning_effort from ouroboros.request_wire_recovery import ( finalize_wire_response, note_provider_metadata_drop_fields, @@ -182,48 +178,28 @@ class _OpenAICompatibleLaneMixin: # stable governance prefix on the same cache bucket. kwargs["prompt_cache_key"] = cache_identity requested_effort = normalize_reasoning_effort(reasoning_effort) - # Descriptor-driven carriage (one SSOT in provider_models): the - # carrier, tier projection, forced-tool suppression and absent-tier - # meaning all come from EFFORT_ROUTE_DESCRIPTOR — no per-provider - # builder branch. A no-carrier route DROPS the tier but never - # silently: the drop is disclosed as a usage fact (effort_not_carried). - descriptor = effort_descriptor_for_route(provider, resolved_model) - carrier = descriptor.get("carrier") - _EFFORT_CLAMP_CVAR.set(None) # never inherit a stale note - _EFFORT_NOT_CARRIED_CVAR.set(None) - if carrier == "none": - if requested_effort and requested_effort not in ("", "none"): - _EFFORT_NOT_CARRIED_CVAR.set({ - "requested": requested_effort, - "provider": provider, - "model": resolved_model, - "absent_meaning": descriptor.get("absent_meaning") or "provider_default", - }) - elif carrier == "reasoning_effort": - forced_tool = ( - bool(descriptor.get("forced_tool_suppression")) - and bool(prepared_tools) - and tool_choice not in (None, "", "auto", "none") - ) - if requested_effort == "none" and descriptor.get("absent_meaning") == "max": - # GLM/z.ai dialect: thinking cannot be disabled (attempting - # it is HTTP 400 code 1210), and an absent tier bills at - # MAX — the lowest real tier is the honest projection of - # "none" here, disclosed through the clamp note. - applied = "low" - kwargs["reasoning_effort"] = applied - elif requested_effort == "none" and descriptor.get("forced_tool_suppression"): - applied = "none" - kwargs.setdefault("extra_body", {})["thinking"] = {"type": "disabled"} - elif forced_tool: - # Thinking mode accepts only tool_choice auto/none (probed - # 2026-09-03: required and named 400 on both v4 models), so - # a forced tool call is served with thinking disabled. - applied = "none" + if direct_openai: + # Effort-carrying routes honor the OUROBOROS_EFFORT_* lanes + # instead of dropping them like generic compatible lanes. + # Keyed on the PROVIDER id, not a target capability field, so + # a hand-built target (fixtures, probes) cannot silently drop + # the carriage; request-wire recovery adapts on a provider 400. + kwargs["reasoning_effort"] = requested_effort + elif provider == "deepseek": + # Same carriage, projected onto DeepSeek's wire dialect + # (low/high/max; thinking is switched off by a toggle, not an + # effort value). Thinking mode accepts only tool_choice + # auto/none (probed 2026-09-03: required and named 400 on both + # v4 models), so a forced tool call is served with thinking + # disabled. Any tier change is disclosed on usage as + # ``reasoning_effort_clamped``. + forced_tool = bool(prepared_tools) and tool_choice not in (None, "", "auto", "none") + applied = "none" if forced_tool else normalize_deepseek_reasoning_effort(requested_effort) + if applied == "none": kwargs.setdefault("extra_body", {})["thinking"] = {"type": "disabled"} else: - applied = project_effort_for_route(descriptor, requested_effort) kwargs["reasoning_effort"] = applied + _EFFORT_CLAMP_CVAR.set(None) # never inherit a stale note if applied != requested_effort: _EFFORT_CLAMP_CVAR.set({ "requested": requested_effort, "applied": applied, @@ -405,7 +381,6 @@ class _OpenAICompatibleLaneMixin: usage.pop("response_provider", None) usage.pop("reasoning_pin", None) usage.pop("reasoning_effort_clamped", None) - usage.pop("effort_not_carried", None) usage.pop("provider_error", None) # An HTTP-200 that carried a provider body-error (OpenRouter passes # 429/5xx through the body) reaches here only when a same-model reroute @@ -566,12 +541,6 @@ class _OpenAICompatibleLaneMixin: _clamp_note = self._pop_effort_clamp_disclosure() if _clamp_note: usage["reasoning_effort_clamped"] = _clamp_note - # Same disclosure norm for a dropped tier on a no-carrier route: the - # requested tier never reached the wire — that fact must be visible, - # not silently absorbed into "provider default behavior". - _not_carried = self._pop_effort_not_carried_disclosure() - if _not_carried: - usage["effort_not_carried"] = _not_carried # Same disclosure norm for a ≤4-cap cache-marker reduction (v6.77.0): never silent. _cache_note = self._pop_thread_disclosure("_cache_breakpoint_tls") if _cache_note: diff --git a/ouroboros/llm_probe.py b/ouroboros/llm_probe.py index 4910f54fb..86a969ac3 100644 --- a/ouroboros/llm_probe.py +++ b/ouroboros/llm_probe.py @@ -208,6 +208,10 @@ def controlled_probe_error(exc: BaseException) -> dict[str, Any]: """Map typed transport facts to one bounded, provider-neutral reason.""" status, code, error_type = _error_facts(exc) credit_codes = { + # Z.ai answers plan exhaustion as HTTP 429 code 1113 "Insufficient + # balance" (billing, not rate limiting; a Coding Plan key on the + # pay-as-you-go endpoint lands here too). + "1113", "billing_hard_limit_reached", "credit_balance_too_low", "credits_exhausted", @@ -216,11 +220,7 @@ def controlled_probe_error(exc: BaseException) -> dict[str, Any]: } model_codes = {"model_not_found", "unknown_model"} - # Z.ai serves plan exhaustion as 429 code 1113 "Insufficient balance": - # billing, not rate limiting. Mapping it to the bare "Rate limited" reason - # hides the actionable fact (top up / switch plan) behind a retry hint. - insufficient_balance = code == "1113" or "insufficient balance" in error_type - if insufficient_balance or status == 402 or code in credit_codes or error_type in credit_codes: + if status == 402 or code in credit_codes or error_type in credit_codes: reason = "No credits" elif status == 401: reason = "Invalid key" diff --git a/ouroboros/provider_models.py b/ouroboros/provider_models.py index 3428f8e22..9540fc185 100644 --- a/ouroboros/provider_models.py +++ b/ouroboros/provider_models.py @@ -35,13 +35,17 @@ def resolve_minimax_base_url(region: str = "") -> str: DEEPSEEK_BASE_URL = "https://api.deepseek.com/v1" # DeepSeek's Chat Completions ``reasoning_effort`` enum is low/high/max -# (medium/xhigh are documented aliases of high; ultra aliases max) and thinking -# is switched off by ``thinking.type=disabled``, not by an effort value. This -# is the wire dialect of one provider, projected at the physical-send boundary; -# the canonical Ouroboros effort scale stays the SSOT everywhere else. The -# mapping lives in EFFORT_ROUTE_ALIASES_LMH below (the ONE shared -# low/high/max-dialect table: DeepSeek's row and the GLM-family z.ai row of -# EFFORT_ROUTE_DESCRIPTOR both read it); this name is the historical import. +# (medium/xhigh are documented aliases of high) and thinking is switched off by +# ``thinking.type=disabled``, not by an effort value. This is the wire dialect +# of one provider, projected at the physical-send boundary; the canonical +# Ouroboros effort scale stays the SSOT everywhere else. Not a model or pricing +# table and never an admission gate. +DEEPSEEK_REASONING_EFFORT_ALIASES = { + "minimal": "low", + "medium": "high", + "xhigh": "high", + "ultra": "max", +} def normalize_deepseek_reasoning_effort(value: str) -> str: @@ -50,249 +54,6 @@ def normalize_deepseek_reasoning_effort(value: str) -> str: return DEEPSEEK_REASONING_EFFORT_ALIASES.get(normalized, normalized) -# --- Reasoning-effort route descriptor (SSOT for effort carriage) --------------- -# One structural table answers, per registered provider route: WHERE the effort -# tier rides on the wire (the carrier), WHICH provider tier values the route -# accepts, HOW a canonical Ouroboros tier projects onto those values, and WHAT -# an ABSENT parameter means on that route. Before this table each builder -# hard-coded its own branch (openai carries, deepseek projects, everything else -# silently drops), so a tier requested against a no-carrier route vanished with -# no record — the failure class this descriptor makes structurally impossible. -# -# Measurement discipline (deliberately conservative): -# * deepseek / GLM-family low-high-max dialects: measured enums. -# * openai / anthropic / openrouter / claudexor: measured wire contracts. -# * cloudru / minimax / gigachat / generic openai-compatible / local: NOT -# measured for effort carriage → carrier "none" with an honest absent_meaning. -# "zai", "qwen" (DashScope) and "kimi" (Moonshot) are NOT registered routes -# in this baseline (PR #1194 closed unmerged); per the DO-NOT rule their -# carriers are declared "none" — no analogy to GLM without a key/measurement. -# When a measured route lands, it becomes one table row, not a builder edit. -# -# Fields per route: -# carrier: "reasoning_effort" | "extra_body.reasoning" | "anthropic.adaptive" -# | "reasoningEffort" | "none" -# tiers: the provider-side enum values the route accepts (labels for UI); -# empty when carrier is "none". -# project: canonical tier -> provider tier. Identity rows are implicit; only -# re-mapping rows are stored. Absent canonical tier → identity. -# absent_meaning (REQUIRED even for carrier "none"): what the route does when -# the parameter is not sent — "max" (provider silently reasons at its -# top tier: z.ai's behavior, a real cost trap), "provider_default" -# (the provider's own default applies), or "off" (no reasoning -# parameter exists; sending one is an error). -# forced_tool_suppression: whether the route requires thinking disabled when a -# forced tool_choice is used (DeepSeek: thinking accepts only -# auto/none tool_choice, so a forced call ships thinking disabled — -# reason "provider_forced_tool_choice" on the clamp disclosure). -# Do NOT copy this exception to GLM-family routes: measured -# tool_choice required/named work WITH thinking on glm-5.3. -EFFORT_ROUTE_ALIASES_LMH = { - # Shared alias projection for the low/high/max wire dialects (DeepSeek and - # the measured GLM-family z.ai route — Egor's measurement 2026-09-21): - # none/minimal/low→low, medium/high→high, xhigh/ultra→max. ONE mapping; - # DeepSeek's normalize helper reads the same table. (``none`` on DeepSeek - # is intercepted by the builder as the thinking-disable toggle BEFORE this - # projection — see forced_tool_suppression.) - "none": "low", - "minimal": "low", - "medium": "high", - "xhigh": "max", - "ultra": "max", -} -DEEPSEEK_REASONING_EFFORT_ALIASES = EFFORT_ROUTE_ALIASES_LMH - -EFFORT_ROUTE_DESCRIPTOR: dict[str, dict] = { - "openai": { - "carrier": "reasoning_effort", - "tiers": ["minimal", "low", "medium", "high"], - "project": {}, - "absent_meaning": "provider_default", - "forced_tool_suppression": False, - }, - "deepseek": { - "carrier": "reasoning_effort", - "tiers": ["low", "high", "max"], - "project": EFFORT_ROUTE_ALIASES_LMH, - "absent_meaning": "provider_default", - "forced_tool_suppression": True, - }, - # GLM through the generic openai-compatible lane (z.ai PAYG and Coding Plan - # endpoints both speak the OpenAI-compatible shape). The enum and alias - # projection are byte-identical to DeepSeek's dialect (measured on a live - # Coding Plan key, PR #1194 close comment), so this row shares ONE table. - # Absent tier means MAX billing — the loudest absent_meaning in the set. - "zai-glm": { - "carrier": "reasoning_effort", - "tiers": ["low", "high", "max"], - "project": EFFORT_ROUTE_ALIASES_LMH, - "absent_meaning": "max", - "forced_tool_suppression": False, - }, - "anthropic": { - "carrier": "anthropic.adaptive", - "tiers": ["low", "medium", "high"], - "project": {"minimal": "low"}, - "absent_meaning": "provider_default", - "forced_tool_suppression": False, - }, - "openrouter": { - "carrier": "extra_body.reasoning", - "tiers": ["minimal", "low", "medium", "high", "xhigh", "max", "ultra"], - "project": {}, - "absent_meaning": "provider_default", - "forced_tool_suppression": False, - }, - "claudexor": { - "carrier": "reasoningEffort", - "tiers": ["minimal", "low", "medium", "high", "xhigh", "max", "ultra"], - "project": {}, - "absent_meaning": "provider_default", - "forced_tool_suppression": False, - }, - # Unmeasured routes: honest "none". The tier is dropped (and disclosed as - # effort_not_carried); absent_meaning records what the route does without it. - "openai-compatible": { - "carrier": "none", "tiers": [], "project": {}, - "absent_meaning": "provider_default", "forced_tool_suppression": False, - }, - "cloudru": { - "carrier": "none", "tiers": [], "project": {}, - "absent_meaning": "provider_default", "forced_tool_suppression": False, - }, - "minimax": { - "carrier": "none", "tiers": [], "project": {}, - "absent_meaning": "provider_default", "forced_tool_suppression": False, - }, - "gigachat": { - "carrier": "none", "tiers": [], "project": {}, - "absent_meaning": "off", "forced_tool_suppression": False, - }, - "local": { - "carrier": "none", "tiers": [], "project": {}, - "absent_meaning": "off", "forced_tool_suppression": False, - }, - "zai": { - # Placeholder row for the unmerged PR #1194 route: no key, no - # measurement in THIS tree → carrier none by the DO-NOT rule. - "carrier": "none", "tiers": [], "project": {}, - "absent_meaning": "provider_default", "forced_tool_suppression": False, - }, - "qwen": { - "carrier": "none", "tiers": [], "project": {}, - "absent_meaning": "provider_default", "forced_tool_suppression": False, - }, - "kimi": { - "carrier": "none", "tiers": [], "project": {}, - "absent_meaning": "provider_default", "forced_tool_suppression": False, - }, -} - - -# Which canonical tiers each UI surface may OFFER for a route: the descriptor's -# provider tiers with their canonical pre-images, computed — never hand-listed. -def canonical_tiers_for_route(descriptor: dict) -> list[str]: - """Canonical effort choices a route's descriptor supports (empty = none). - - Each provider tier is offered under its IDENTITY canonical name when one - exists (low/high/max are identity rows), and otherwise under a canonical - pre-image — so the GLM low/high/max dialect offers exactly ``low, high, - max``, never ``minimal/medium/xhigh`` aliases of the same wire values.""" - tiers = list(descriptor.get("tiers") or []) - if not tiers: - return [] - project = descriptor.get("project") or {} - inverse: dict[str, list[str]] = {} - for canonical, provider in project.items(): - inverse.setdefault(provider, []).append(canonical) - out: list[str] = [] - for tier in tiers: - if project.get(tier, tier) == tier: - # Identity row (implicit tier→tier): the provider value IS the - # canonical spelling — offer it directly. - out.append(tier) - else: - candidates = inverse.get(tier) or [] - out.append(candidates[0] if candidates else tier) - return out - - -# GLM-family detection for the generic openai-compatible lane. The z.ai hosts -# serve GLM models behind an OpenAI-compatible API whose effort dialect is the -# measured low/high/max enum; a model id naming glm flags the route. Detection -# is by MODEL IDENTITY ONLY — never by a generic base_url heuristic: any vLLM -# server could host anything. -_GLM_MODEL_TOKEN = "glm" - - -def _is_glm_model(resolved_model: str) -> bool: - return _GLM_MODEL_TOKEN in str(resolved_model or "").strip().lower() - - -def effort_descriptor_for_route(provider: str, resolved_model: str = "") -> dict: - """Resolve the effort descriptor for one physical route. - - The z.ai GLM dialect rides the generic ``openai-compatible`` provider lane, - so a GLM model id on that lane resolves to the measured "zai-glm" row; every - other openai-compatible target honestly resolves to carrier "none" (no - measurement, no guessed carriage). Unknown providers fail to the "none" - row — never to a guessed carrier. - """ - provider_id = str(provider or "").strip() - if provider_id == "openai-compatible" and _is_glm_model(resolved_model): - return EFFORT_ROUTE_DESCRIPTOR["zai-glm"] - return EFFORT_ROUTE_DESCRIPTOR.get( - provider_id, - {"carrier": "none", "tiers": [], "project": {}, - "absent_meaning": "provider_default", "forced_tool_suppression": False}, - ) - - -def project_effort_for_route(descriptor: dict, canonical_effort: str) -> str: - """Project a canonical tier onto a route's provider enum via its table.""" - value = str(canonical_effort or "").strip().lower() - if not value: - return value - return str((descriptor.get("project") or {}).get(value, value)) - - -def _effort_descriptor_mismatches_observation( - descriptor: dict, - *, - observed_ceiling: str = "", - observed_floor: str = "", -) -> bool: - """DECLARED vs OBSERVED loud-fact predicate (spec point 3). - - An observation recorded by capability_evidence (an effort ceiling a - provider actually rejected above, or a floor it required below) disagrees - with the descriptor when the observation names a canonical tier that the - descriptor's declared provider tiers cannot express — e.g. an observed - ``ultra`` ceiling on a route whose declared enum tops out at ``max``. - Callers surface a true verdict as a loud fact (test + record), never as a - silent table override: the descriptor changes only through a measured - table edit. - """ - from ouroboros.config import effort_rank - - if descriptor.get("carrier") == "none": - return False - tiers = sorted( - set(descriptor.get("tiers") or []), - key=lambda t: effort_rank(t), - ) - if not tiers: - return False - bottom, top = tiers[0], tiers[-1] - ceiling = str(observed_ceiling or "").strip().lower() - floor = str(observed_floor or "").strip().lower() - if ceiling and effort_rank(ceiling) > effort_rank(top): - return True - if floor and effort_rank(floor) < effort_rank(bottom): - return True - return False - - # Direct-provider prefix → canonical provider name. Un-prefixed models route # through OpenRouter. Order matters only for readability; prefixes are disjoint. PROVIDER_PREFIXES: tuple[tuple[str, str], ...] = ( diff --git a/tests/test_deepseek_provider.py b/tests/test_deepseek_provider.py index f4ff01871..9b4166dbd 100644 --- a/tests/test_deepseek_provider.py +++ b/tests/test_deepseek_provider.py @@ -162,7 +162,7 @@ class TestWireProjection: ("low", "low"), ("medium", "high"), ("high", "high"), - ("xhigh", "max"), + ("xhigh", "high"), ("max", "max"), ("ultra", "max"), ], @@ -170,13 +170,9 @@ class TestWireProjection: def test_effort_projection_matches_deepseek_chat_contract( self, monkeypatch, requested, wire, ): - # Official contract (api-docs.deepseek.com, Thinking Mode) + the shared - # low/high/max-dialect projection table (Egor's z.ai measurement - # 2026-09-21, PR #1194 close comment): the enum is low/high/max, - # minimal→low, medium→high, xhigh/ultra→max (the canonical step ABOVE - # high belongs with the TOP wire tier, not aliased down), and ``none`` - # is the separate ``thinking.type=disabled`` toggle, never a - # reasoning_effort value. + # Official contract (api-docs.deepseek.com, Thinking Mode): the enum is + # low/high/max, medium/xhigh alias high, and ``none`` is the separate + # ``thinking.type=disabled`` toggle, never a reasoning_effort value. client = LLMClient() target = self._target(monkeypatch) kwargs = client._build_remote_kwargs( diff --git a/tests/test_effort_route_descriptor.py b/tests/test_effort_route_descriptor.py deleted file mode 100644 index 30567d5f5..000000000 --- a/tests/test_effort_route_descriptor.py +++ /dev/null @@ -1,282 +0,0 @@ -"""Reasoning-effort route descriptor (SSOT) — projection, carriage and -disclosure tests. - -The descriptor in ``ouroboros/provider_models.py`` is the single table that -answers, per registered provider route: which wire carrier (if any) carries -the effort tier, which provider tier values the route accepts, how canonical -tiers project onto them, and what an ABSENT parameter means. These tests pin: - -* the z.ai/GLM alias projection measured on a live Coding Plan key - (PR #1194 close comment): none/minimal/low→low, medium/high→high, - xhigh/ultra→max; -* the effort_not_carried disclosure when a tier is requested against a - no-carrier route (the tier is dropped, never guessed onto an unmeasured - wire, and the drop is a visible usage fact); -* the descriptor↔capability_evidence declared-vs-observed check; -* the UI-facing tier list (3 choices for GLM, not 8; identity spellings). -""" - -import pytest - -from ouroboros.provider_models import ( - EFFORT_ROUTE_DESCRIPTOR, - canonical_tiers_for_route, - effort_descriptor_for_route, - project_effort_for_route, -) - - -class TestDescriptorShape: - def test_every_row_declares_required_fields(self): - for route, row in EFFORT_ROUTE_DESCRIPTOR.items(): - assert "carrier" in row, route - assert "absent_meaning" in row, route - assert row["absent_meaning"] in ("max", "provider_default", "off"), route - if row["carrier"] == "none": - assert row["tiers"] == [], route - else: - assert row["tiers"], route - for tier in row["tiers"]: - assert row["project"].get(tier, tier) in row["tiers"] or tier in ( - "low", "high", "max", "minimal", "medium", "xhigh", "ultra" - ), (route, tier) - - def test_unknown_provider_fails_to_carrier_none(self): - assert effort_descriptor_for_route("never-measured")["carrier"] == "none" - - def test_unmeasured_chinese_routes_declare_none(self): - # DO-NOT rule: qwen/kimi/zai carriers are NOT declared by analogy to - # GLM — no key, no measurement in this tree. - for route in ("zai", "qwen", "kimi"): - assert EFFORT_ROUTE_DESCRIPTOR[route]["carrier"] == "none", route - - def test_deepseek_and_glm_share_one_alias_table(self): - assert ( - EFFORT_ROUTE_DESCRIPTOR["deepseek"]["project"] - is EFFORT_ROUTE_DESCRIPTOR["zai-glm"]["project"] - ) - - -class TestZaiInsufficientBalance: - """429 code 1113 "Insufficient balance" is billing, not rate limiting.""" - - class _Exc(Exception): - status_code = 429 - code = "1113" - type = "" - - def test_probe_maps_to_no_credits(self): - from ouroboros.llm_probe import controlled_probe_error - - result = controlled_probe_error(self._Exc("Insufficient balance")) - assert result["error"] == "No credits" - assert result["status_code"] == 429 - - def test_plain_429_stays_rate_limited(self): - from ouroboros.llm_probe import controlled_probe_error - - class Plain(Exception): - status_code = 429 - code = "" - type = "" - - assert controlled_probe_error(Plain("too many requests"))["error"] == "Rate limited" - - -class TestZaiProjection: - """Measured on a live Z.ai Coding Plan key (glm-5.3), PR #1194 close.""" - - @pytest.mark.parametrize( - ("canonical", "wire"), - [ - ("none", "low"), # thinking cannot be disabled (400 code 1210) - ("minimal", "low"), - ("low", "low"), - ("medium", "high"), - ("high", "high"), - ("xhigh", "max"), - ("max", "max"), - ("ultra", "max"), - ], - ) - def test_projection(self, canonical, wire): - d = effort_descriptor_for_route("openai-compatible", "glm-5.3") - assert d["carrier"] == "reasoning_effort" - assert project_effort_for_route(d, canonical) == wire - - def test_absent_tier_means_max(self): - d = effort_descriptor_for_route("openai-compatible", "glm-5.3") - assert d["absent_meaning"] == "max" - - def test_glm_detection_is_model_identity_only(self): - assert effort_descriptor_for_route("openai-compatible", "GLM-5.3")["carrier"] == "reasoning_effort" - assert effort_descriptor_for_route("openai-compatible", "zai-org/GLM-4.7")["carrier"] == "reasoning_effort" - # A non-GLM model on the same lane stays honestly unmeasured. - assert effort_descriptor_for_route("openai-compatible", "Qwen/Qwen3-235B")["carrier"] == "none" - - def test_ui_offers_three_identity_tiers_not_eight_aliases(self): - d = effort_descriptor_for_route("openai-compatible", "glm-5.3") - assert canonical_tiers_for_route(d) == ["low", "high", "max"] - - def test_no_forced_tool_suppression_on_glm(self): - # Measured: tool_choice required/named work WITH thinking on glm-5.3 — - # DeepSeek's forced-tool suppression exception must NOT carry over. - d = effort_descriptor_for_route("openai-compatible", "glm-5.3") - assert d["forced_tool_suppression"] is False - assert EFFORT_ROUTE_DESCRIPTOR["deepseek"]["forced_tool_suppression"] is True - - -class TestBuilderCarriage: - """_build_remote_kwargs reads the descriptor for every direct lane.""" - - def _target(self, provider, resolved_model): - return { - "provider": provider, - "resolved_model": resolved_model, - "usage_model": f"{provider}::{resolved_model}", - "supports_openrouter_extensions": False, - } - - def _build(self, provider, resolved_model, effort, tools=None, tool_choice="auto"): - from ouroboros.llm import LLMClient - - client = LLMClient() - return client._build_remote_kwargs( - self._target(provider, resolved_model), - [{"role": "user", "content": "hi"}], - effort, 256, tool_choice, None, tools, - ) - - def test_glm_effort_reaches_the_wire(self): - kwargs = self._build("openai-compatible", "glm-5.3", "low") - assert kwargs["reasoning_effort"] == "low" - - def test_glm_xhigh_projects_to_max(self): - kwargs = self._build("openai-compatible", "glm-5.3", "xhigh") - assert kwargs["reasoning_effort"] == "max" - - def test_glm_medium_projects_to_high(self): - kwargs = self._build("openai-compatible", "glm-5.3", "medium") - assert kwargs["reasoning_effort"] == "high" - - def test_glm_minimal_projects_to_low(self): - kwargs = self._build("openai-compatible", "glm-5.3", "minimal") - assert kwargs["reasoning_effort"] == "low" - - def test_openai_still_carries_identity(self): - kwargs = self._build("openai", "gpt-5.6-terra", "medium") - assert kwargs["reasoning_effort"] == "medium" - - def test_no_carrier_route_drops_with_disclosure(self): - from ouroboros.llm_capability_policy import _EFFORT_NOT_CARRIED_CVAR - - kwargs = self._build("openai-compatible", "some-vllm-model", "high") - assert "reasoning_effort" not in kwargs - note = _EFFORT_NOT_CARRIED_CVAR.get() - assert note == { - "requested": "high", - "provider": "openai-compatible", - "model": "some-vllm-model", - "absent_meaning": "provider_default", - } - - def test_no_disclosure_when_no_tier_requested(self): - from ouroboros.llm_capability_policy import _EFFORT_NOT_CARRIED_CVAR - - self._build("openai-compatible", "some-vllm-model", "none") - assert _EFFORT_NOT_CARRIED_CVAR.get() is None - - def test_usage_surfaces_effort_not_carried(self): - from ouroboros.llm import LLMClient - - client = LLMClient() - self._build("openai-compatible", "some-vllm-model", "high") - msg, usage = client._normalize_remote_response( - {"choices": [{"message": {"role": "assistant", "content": "ok"}}], - "usage": {"prompt_tokens": 1, "completion_tokens": 1}}, - self._target("openai-compatible", "some-vllm-model"), - ) - assert usage["effort_not_carried"]["requested"] == "high" - # consumed once, never sticks to the context - assert "effort_not_carried" not in ( - client._normalize_remote_response( - {"choices": [{"message": {"role": "assistant", "content": "ok"}}], - "usage": {}}, - self._target("openai-compatible", "some-vllm-model"), - )[1] - ) - - def test_glm_forced_tool_keeps_thinking_on(self): - tools = [{"type": "function", "function": {"name": "f", "parameters": {"type": "object"}}}] - kwargs = self._build("openai-compatible", "glm-5.3", "high", tools=tools, tool_choice="required") - assert kwargs["reasoning_effort"] == "high" - assert "extra_body" not in kwargs or "thinking" not in kwargs.get("extra_body", {}) - - def test_deepseek_forced_tool_still_suppresses(self): - from tests.test_deepseek_provider import TestWireProjection # noqa: F401 - tools = [{"type": "function", "function": {"name": "f", "parameters": {"type": "object"}}}] - kwargs = self._build("deepseek", "deepseek-v4-flash", "high", tools=tools, tool_choice="required") - assert "reasoning_effort" not in kwargs - assert kwargs["extra_body"]["thinking"] == {"type": "disabled"} - - -class TestDescriptorCapabilityEvidenceAgreement: - """DECLARED (descriptor) vs OBSERVED (capability_evidence) must never - silently disagree: a mismatch is a loud fact, and the descriptor's - declared ceilings/floors stay within the provider tiers it declares.""" - - def _observed_facts(self): - import inspect - import ouroboros.capability_evidence as ce - - # The observable API surface of the evidence store: recorded - # ceilings/floors/rejected_params are diagnostic namespaces keyed by - # normalized model identity. A descriptor row DISAGREES with the - # observed namespace only when an observed fact names an effort tier - # outside the row's declared provider tiers for a matching route. - return ce - - def test_descriptor_tiers_are_canonical_scale_members(self): - from ouroboros.config import effort_rank - - valid = {"none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra"} - for route, row in EFFORT_ROUTE_DESCRIPTOR.items(): - for tier in row["tiers"]: - assert tier in valid, (route, tier) - for canonical, provider in (row.get("project") or {}).items(): - assert canonical in valid, (route, canonical) - assert provider in valid, (route, provider) - - def test_projection_lands_inside_declared_tiers(self): - # Pinned for the measured low/high/max dialects (DeepSeek, GLM/z.ai): - # every canonical tier projects onto a declared wire value. Identity - # routes (openai/openrouter/claudexor) and anthropic pass unlisted - # tiers through unchanged — request-wire recovery owns those at send - # time, so the projection-table invariant does not apply there. - from ouroboros.provider_models import EFFORT_ROUTE_ALIASES_LMH - - for route, row in EFFORT_ROUTE_DESCRIPTOR.items(): - if row.get("project") is not EFFORT_ROUTE_ALIASES_LMH: - continue - tiers = set(row["tiers"]) - for canonical in ("minimal", "low", "medium", "high", "xhigh", "max", "ultra"): - assert project_effort_for_route(row, canonical) in tiers, (route, canonical) - - def test_mismatch_is_detectable(self): - # The loud-fact check: for each declared carried route, any observed - # effort ceiling ABOVE the descriptor's top tier (or floor BELOW its - # bottom tier) is a declared-vs-observed mismatch. This test pins the - # predicate itself so a future table edit that contradicts a recorded - # observation fails loudly here, not silently in production. - from ouroboros.provider_models import _effort_descriptor_mismatches_observation - - row = dict(EFFORT_ROUTE_DESCRIPTOR["zai-glm"]) - assert not _effort_descriptor_mismatches_observation( - row, observed_ceiling="max", observed_floor="low" - ) - assert _effort_descriptor_mismatches_observation( - row, observed_ceiling="ultra", observed_floor="" - ), "ceiling above declared top tier is a mismatch" - assert _effort_descriptor_mismatches_observation( - row, observed_ceiling="", observed_floor="minimal" - ), "floor below declared bottom tier is a mismatch" diff --git a/web/modules/reviewer_slots.js b/web/modules/reviewer_slots.js index 6b4f9c1a7..b0f683f4f 100644 --- a/web/modules/reviewer_slots.js +++ b/web/modules/reviewer_slots.js @@ -956,13 +956,6 @@ function rowHtml(row, group) { const modelsGap = session ? modelsGapNote(harness, catalogKnown) : ''; if (modelsGap) metaParts.push(modelsGap); const surfaceDefault = CATEGORIES[group]?.surfaceDefault || 'review effort'; - // The route's model identity, in either branch (session split or parsed - // API target), matched against the catalog to find the effort descriptor. - const effortDescriptor = routeEditor.effortDescriptorForModel( - state.catalogModels, split.model || '', - ) || routeEditor.effortDescriptorForModel( - state.catalogModels, row.route?.target_id || '', - ); return `
${reviewerRouteIdentityMarkup(row.route, harnessesById(), identityContext())} @@ -971,7 +964,7 @@ function rowHtml(row, group) { ${session ? modelChooserHtml(`data-slot-model aria-label="${label} agent model"`, split.model, `reviewer-${row.slot_id}-models`, modelOptions, { placeholder: 'Engine default model' }) : routeEditor.routeModelInputHtml(`data-slot-custom-api aria-label="${label} model"`, row.route, state.catalogModels, `reviewer-${row.slot_id}-models`)} ${routeEditor.routeSupportsAccount(row.route) ? selectHtml(`data-slot-profile aria-label="${label} account"`, [{ label: '', options: profileOptions }], row.route.profile_id || '') : ''} - ${effortSelectHtml(`data-slot-effort aria-label="${label} reasoning effort"`, row.effort, surfaceDefault, effortDescriptor)} + ${effortSelectHtml(`data-slot-effort aria-label="${label} reasoning effort"`, row.effort, surfaceDefault)}
${routeEditor.processingDetailsHtml(`data-slot-processing aria-label="${label} processing"`, row.processing_preference, state.processingPreference)} @@ -1042,7 +1035,6 @@ function singletonHtml(spec) { `data-${a}-effort aria-label="${spec.ariaName} effort"`, session ? row.effort : (row.effort === spec.apiEffortDefault ? '' : row.effort), session ? 'route default' : spec.apiEffortLabel, - routeEditor.effortDescriptorForModel(state.catalogModels, split.model || row.route?.target_id || ''), )} ${routeEditor.processingDetailsHtml(`data-${a}-processing aria-label="${spec.ariaName} processing"`, row.processing_preference, processingInheritance(row))} diff --git a/web/modules/route_editor_primitives.js b/web/modules/route_editor_primitives.js index 77843f7ae..5945485a5 100644 --- a/web/modules/route_editor_primitives.js +++ b/web/modules/route_editor_primitives.js @@ -572,8 +572,7 @@ export function selectHtml(attrs, groups, selected) { return ``; } -export function effortSelectHtml(attrs, selected, surfaceDefault = 'route default', descriptor = null) { - if (descriptor) return effortFieldForRoute(attrs, selected, descriptor, { surfaceDefault }); +export function effortSelectHtml(attrs, selected, surfaceDefault = 'route default') { const options = [ { value: '', label: 'Default effort' }, ...EFFORT_CHOICES.map((effort) => ({ value: effort, label: effort })), @@ -585,66 +584,6 @@ export function effortSelectHtml(attrs, selected, surfaceDefault = 'route defaul ); } -/** - * Find a route's effort descriptor in the model catalog by its saved model - * identity. Returns null when the catalog has no matching entry (unknown or - * out-of-catalog model): the caller then falls back to the generic select. - */ -export function effortDescriptorForModel(catalogItems, model) { - const wanted = String(model || '').trim(); - if (!wanted || !Array.isArray(catalogItems)) return null; - for (const item of catalogItems) { - if (!item || typeof item !== 'object') continue; - if (String(item.value || '') === wanted || String(item.id || '') === wanted) { - return item.effort_descriptor || null; - } - } - return null; -} - -/** - * Descriptor-aware effort select: the ONE component every surface (Behavior - * effort cards, reviewer slots, subagent roster) uses when a route's effort - * descriptor is known. Items come from the route's declared canonical tiers — - * 3 for a GLM route, not 8 — each explicit choice annotated with its wire - * projection when it differs ("xhigh → max on this route"), and the empty - * default always shows WHERE it inherits from, never a bare "Default effort". - * - * @param {string} attrs HTML attributes for the select - * @param {string} selected saved canonical tier ('' = inherit) - * @param {{carrier?: string, tiers?: string[], canonical_tiers?: string[], absent_meaning?: string}} descriptor - * @param {{surfaceDefault?: string, routeLabel?: string}} context - */ -export function effortFieldForRoute(attrs, selected, descriptor, { surfaceDefault = 'route default', routeLabel = 'this route' } = {}) { - const d = descriptor || {}; - if (!d.canonical_tiers || !d.canonical_tiers.length || d.carrier === 'none') { - // No measured effort carrier: the route does not accept a tier. Keep - // the value (it still applies to fallback routes) but say so plainly - // instead of offering choices that would be dropped on the wire. - return `
` - + `Route does not accept a reasoning-effort tier${routeLabel ? ` (${escapeHtml(routeLabel)})` : ''}` - + (selected ? ` — saved tier ${escapeHtml(selected)} is not carried on this route` : '') - + (d.absent_meaning === 'max' ? '; an absent tier bills at the provider maximum' : '') - + '
'; - } - const tiers = d.tiers || []; - const canonical = d.canonical_tiers || []; - const options = canonical.map((tier) => { - const wireIndex = canonical.indexOf(tier); - const wire = tiers[wireIndex]; - const label = tier === wire ? tier : `${tier} → ${wire} on this route`; - return { value: tier, label }; - }); - const inheritLabel = surfaceDefault && surfaceDefault !== 'route default' - ? `inherit · ${surfaceDefault}` - : 'inherit from surface default'; - return selectHtml( - `${attrs} title="Reasoning effort — default: ${escapeHtml(surfaceDefault)}"`, - [{ label: '', options: [{ value: '', label: inheritLabel }, ...options] }], - selected || '', - ); -} - export function describeExecutionEvidence(entry) { if (!entry || typeof entry !== 'object') return ''; if ('requested_model' in entry || 'applied_model' in entry) { diff --git a/web/modules/settings.js b/web/modules/settings.js index 162b4d979..55aee3b16 100644 --- a/web/modules/settings.js +++ b/web/modules/settings.js @@ -414,14 +414,7 @@ export function moreProvidersCredentialConfigured({ export function providerTestStatusText(result = {}) { if (result?.ok === true) return 'Works'; const reason = String(result?.error || '').trim(); - if (!reason) return 'Not ready'; - // Z.ai plan exhaustion arrives as 429 code 1113 "Insufficient balance" and - // is mapped to the billing reason server-side; the actionable text is the - // provider message plus a plan hint, never a bare "Rate limited" retry cue. - if (reason === 'No credits') { - return `Not ready — provider reports insufficient balance. Top up the account or switch the plan, then re-test.`; - } - return `Not ready — ${reason}`; + return reason ? `Not ready — ${reason}` : 'Not ready'; } export function providerTestNetworkErrorStatus() { diff --git a/web/modules/settings_catalog.js b/web/modules/settings_catalog.js index 539123d09..9de2e4f4b 100644 --- a/web/modules/settings_catalog.js +++ b/web/modules/settings_catalog.js @@ -1,7 +1,6 @@ import { accountRows, claudexorStatus, READ_OK } from './claudexor_status_store.js'; import { apiFetch } from './api_client.js'; import { setInlineStatus } from './ui_helpers.js'; -import { annotateEffortCardsWithRouteDescriptor } from './settings_ui.js'; export const MODEL_CATALOG_TIMEOUT_MS = 25000; let catalogRefreshSeq = 0; const buttonRefreshes = new WeakMap(); @@ -51,22 +50,6 @@ export function watchAccountModelCatalog() { }; } -/** - * The Behavior effort cards' route annotation follows each successful catalog - * read: the MAIN model setting names the route whose descriptor is shown. - * Best-effort and DOM-optional (non-Settings pages never render the cards). - */ -function annotateEffortCardsFromCatalog(data = {}) { - try { - const mainInput = document.getElementById('s-model-main'); - const mainModel = String(mainInput?.value || '').trim() - || String(data?.settings?.OUROBOROS_MODEL || '').trim(); - annotateEffortCardsWithRouteDescriptor({ catalogItems: data.items || [], mainModel }); - } catch { - // Presentation-only; never fail the catalog read on it. - } -} - /** * Read provenance belongs to discovery, never to the owner's saved assignment. * One unreachable source among several is `partial`: the catalogs that did @@ -235,7 +218,6 @@ export async function refreshModelCatalog({ button } = {}) { } const read_state = catalogReadState(data); fillCatalogDatalist({ ...data, read_state }); - annotateEffortCardsFromCatalog(data); if (read_state !== 'ok') { setCatalogStatus(statusEl, catalogReadNote({ ...data, read_state }), 'warn'); diff --git a/web/modules/settings_ui.js b/web/modules/settings_ui.js index 0fa21fed5..cc028947a 100644 --- a/web/modules/settings_ui.js +++ b/web/modules/settings_ui.js @@ -4,7 +4,6 @@ import { renderAgentAccountsSection, renderAgentsServiceBanner } from './harness import { renderReviewerSlotsSection } from './reviewer_slots.js'; import { renderSubagentsSection } from './subagents_settings.js'; import { modelRolesHost } from './model_roles.js'; -import { effortDescriptorForModel } from './route_editor_primitives.js'; // Reads as a sequence: keys → secrets → which API models → who among the agents // does what → behavior → technical. "Agents", not "Coding agents" (D-10): the @@ -217,43 +216,10 @@ function effortField({ id, label, defaultValue }) { ${renderSegmentedField({ target: id, options })} - `; } -/** - * Annotate the Behavior effort cards with the MAIN route's descriptor facts - * (spec: the value is always visible and the projection is named). The cards - * stay canonical — a surface effort applies across the fallback chain, so it - * is never restricted to one route's enum — but the route's real tiers and - * what an absent tier means are stated instead of left to guess. - */ -export function annotateEffortCardsWithRouteDescriptor({ catalogItems = [], mainModel = '' } = {}) { - const descriptor = effortDescriptorForModel(catalogItems, mainModel); - document.querySelectorAll('.effort-route-note').forEach((note) => { - if (!descriptor || !descriptor.canonical_tiers?.length) { - note.hidden = true; - note.textContent = ''; - return; - } - const carrier = String(descriptor.carrier || ''); - if (carrier === 'none') { - note.textContent = mainModel - ? `Main route (${mainModel}) does not accept a reasoning-effort tier; the tier applies to other routes only.` - : 'Main route does not accept a reasoning-effort tier.'; - note.hidden = false; - return; - } - const tiers = (descriptor.canonical_tiers || []).join(' / '); - const absent = descriptor.absent_meaning === 'max' - ? ' An unsent tier bills at the provider maximum.' - : ''; - note.textContent = `Main route accepts ${tiers}${absent}`; - note.hidden = false; - }); -} - export const SECRET_KEYS = [ ['OPENROUTER_API_KEY', 'OpenRouter API Key', 'sk-or-...'], ['OPENAI_API_KEY', 'OpenAI API Key', 'sk-...'], diff --git a/web/modules/subagents_settings.js b/web/modules/subagents_settings.js index 1bc40cf56..5172d6664 100644 --- a/web/modules/subagents_settings.js +++ b/web/modules/subagents_settings.js @@ -9,7 +9,7 @@ import { harnessIdentityMarkup } from './harness_presentation.js'; import { EFFORT_CHOICES, ROUTE_KIND_AGENT_SESSION, ROUTE_KIND_API_MODEL, compoundSessionEffortConflict, configuredApiProviders, changeRouteChoice, routeModelFields, - routeModelInputHtml, routeTargetFromModel, routeSupportsAccount, effortSelectHtml, effortDescriptorForModel, + routeModelInputHtml, routeTargetFromModel, routeSupportsAccount, effortSelectHtml, encodeRouteChoice, indexProfilesByHarness, mintStableId, profileOptionsFor, routeChoiceGroups, sameEngineAs, selectHtml, serializeRouteSpec, sessionModelOptions, updateRouteControlOptions, PROCESSING_CHOICES, PROCESSING_PREFERENCE_KEY, processingDetailsHtml, processingIntentLabel, accountScopedModelCatalog, @@ -347,7 +347,7 @@ export function availableSubagentRowMarkup(row, state, index = 0) { ${routeSupportsAccount(row.route) ? selectHtml(`data-subagent-field="account" aria-label="Account for Subagent ${ordinal}"`, [{ label: '', options: profileOptions }], row.route.credential_profile_id || '') : ''} - ${effortSelectHtml(`data-subagent-field="effort" aria-label="Reasoning effort for Subagent ${ordinal}"`, row.effort || '', 'route default', effortDescriptorForModel(state.apiModels, split.model || row.route?.target_id || ''))} + ${effortSelectHtml(`data-subagent-field="effort" aria-label="Reasoning effort for Subagent ${ordinal}"`, row.effort || '', 'route default')} ${session ? selectHtml(`id="actor-${escapeHtml(rowKey)}-access" data-subagent-field="access" aria-label="Access for Subagent ${ordinal}"`, [{ label: '', options: [ { value: 'full', label: 'Full system access (default)' }, { value: 'workspace_write', label: 'Working files' }, diff --git a/web/tests/effort_route_descriptor.test.js b/web/tests/effort_route_descriptor.test.js deleted file mode 100644 index 9be947883..000000000 --- a/web/tests/effort_route_descriptor.test.js +++ /dev/null @@ -1,66 +0,0 @@ -import assert from 'node:assert/strict'; -import test from 'node:test'; - -import { providerTestStatusText } from '../modules/settings.js'; -import { - effortFieldForRoute, - effortDescriptorForModel, - effortSelectHtml, -} from '../modules/route_editor_primitives.js'; - -test('z.ai insufficient balance (429 code 1113) carries a plan hint, never a bare rate-limit cue', () => { - const text = providerTestStatusText({ ok: false, error: 'No credits' }); - assert.match(text, /insufficient balance/i); - assert.match(text, /top up the account or switch the plan/i); - assert.doesNotMatch(text, /Rate limited/); -}); - -test('a GLM route descriptor renders exactly three identity tiers with named inheritance', () => { - const glm = { - carrier: 'reasoning_effort', - tiers: ['low', 'high', 'max'], - canonical_tiers: ['low', 'high', 'max'], - absent_meaning: 'max', - }; - const html = effortFieldForRoute('data-x', '', glm, { surfaceDefault: 'low · inherited from review effort' }); - const values = [...html.matchAll(/