mirror of
https://github.com/razzant/ouroboros.git
synced 2026-10-03 20:27:56 +00:00
Merge pull request #563 from razzant/landing/pr458-deepseek-direct-20260903
Some checks are pending
CI / ui-smoke (push) Waiting to run
CI / docker-ui-smoke (push) Waiting to run
CI / docker-portable-test (push) Waiting to run
CI / quick-test (push) Waiting to run
CI / betterleaks-platform-smoke (ubuntu-latest) (push) Waiting to run
CI / full-test (macos-latest) (push) Waiting to run
CI / full-test (ubuntu-latest) (push) Waiting to run
CI / full-test (windows-latest) (push) Waiting to run
CI / betterleaks-platform-smoke (macos-latest) (push) Waiting to run
CI / betterleaks-platform-smoke (windows-latest) (push) Waiting to run
CI / integration-test (push) Waiting to run
CI / skill-smoke (macos-latest) (push) Waiting to run
CI / skill-smoke (ubuntu-latest) (push) Waiting to run
CI / skill-smoke (windows-latest) (push) Waiting to run
CI / marker-guards (push) Waiting to run
CI / release-preflight (push) Blocked by required conditions
CI / build (dmg, macos-latest, macos-arm64, syft_1.50.0_darwin_arm64.tar.gz, syft, e32fdb9d47823fa633748a1efca2528fd77c37469ea93c9e40ab835da44e4cce) (push) Blocked by required conditions
CI / build (tar.gz, ubuntu-latest, linux-x86_64, syft_1.50.0_linux_amd64.tar.gz, syft, bf7b29ff57f06da30918266a0e1c2885a8f99784798d1bdb1628886aa015d788) (push) Blocked by required conditions
CI / build (zip, windows-latest, windows-x64, syft_1.50.0_windows_amd64.zip, syft.exe, 815ee6973ec5dff6a671d7f41b0e78835a8c45b91d5a39f4743ea1cee833d3be) (push) Blocked by required conditions
CI / vendor-package-smoke (push) Blocked by required conditions
CI / release (push) Blocked by required conditions
Claudexor platform gate (API keys — subscription auth NOT covered) / fixture · macos-latest · exact managed runtime, fake harness, no model (push) Waiting to run
Claudexor platform gate (API keys — subscription auth NOT covered) / fixture · ubuntu-latest · exact managed runtime, fake harness, no model (push) Waiting to run
Claudexor platform gate (API keys — subscription auth NOT covered) / fixture · windows-latest · exact managed runtime, fake harness, no model (push) Waiting to run
Claudexor platform gate (API keys — subscription auth NOT covered) / live · macos-latest · claude · API key only, subscription NOT covered (push) Waiting to run
Claudexor platform gate (API keys — subscription auth NOT covered) / live · ubuntu-latest · claude · API key only, subscription NOT covered (push) Waiting to run
Claudexor platform gate (API keys — subscription auth NOT covered) / live · windows-latest · claude · API key only, subscription NOT covered (push) Waiting to run
Claudexor platform gate (API keys — subscription auth NOT covered) / live · macos-latest · codex · API key only, subscription NOT covered (push) Waiting to run
Some checks are pending
CI / ui-smoke (push) Waiting to run
CI / docker-ui-smoke (push) Waiting to run
CI / docker-portable-test (push) Waiting to run
CI / quick-test (push) Waiting to run
CI / betterleaks-platform-smoke (ubuntu-latest) (push) Waiting to run
CI / full-test (macos-latest) (push) Waiting to run
CI / full-test (ubuntu-latest) (push) Waiting to run
CI / full-test (windows-latest) (push) Waiting to run
CI / betterleaks-platform-smoke (macos-latest) (push) Waiting to run
CI / betterleaks-platform-smoke (windows-latest) (push) Waiting to run
CI / integration-test (push) Waiting to run
CI / skill-smoke (macos-latest) (push) Waiting to run
CI / skill-smoke (ubuntu-latest) (push) Waiting to run
CI / skill-smoke (windows-latest) (push) Waiting to run
CI / marker-guards (push) Waiting to run
CI / release-preflight (push) Blocked by required conditions
CI / build (dmg, macos-latest, macos-arm64, syft_1.50.0_darwin_arm64.tar.gz, syft, e32fdb9d47823fa633748a1efca2528fd77c37469ea93c9e40ab835da44e4cce) (push) Blocked by required conditions
CI / build (tar.gz, ubuntu-latest, linux-x86_64, syft_1.50.0_linux_amd64.tar.gz, syft, bf7b29ff57f06da30918266a0e1c2885a8f99784798d1bdb1628886aa015d788) (push) Blocked by required conditions
CI / build (zip, windows-latest, windows-x64, syft_1.50.0_windows_amd64.zip, syft.exe, 815ee6973ec5dff6a671d7f41b0e78835a8c45b91d5a39f4743ea1cee833d3be) (push) Blocked by required conditions
CI / vendor-package-smoke (push) Blocked by required conditions
CI / release (push) Blocked by required conditions
Claudexor platform gate (API keys — subscription auth NOT covered) / fixture · macos-latest · exact managed runtime, fake harness, no model (push) Waiting to run
Claudexor platform gate (API keys — subscription auth NOT covered) / fixture · ubuntu-latest · exact managed runtime, fake harness, no model (push) Waiting to run
Claudexor platform gate (API keys — subscription auth NOT covered) / fixture · windows-latest · exact managed runtime, fake harness, no model (push) Waiting to run
Claudexor platform gate (API keys — subscription auth NOT covered) / live · macos-latest · claude · API key only, subscription NOT covered (push) Waiting to run
Claudexor platform gate (API keys — subscription auth NOT covered) / live · ubuntu-latest · claude · API key only, subscription NOT covered (push) Waiting to run
Claudexor platform gate (API keys — subscription auth NOT covered) / live · windows-latest · claude · API key only, subscription NOT covered (push) Waiting to run
Claudexor platform gate (API keys — subscription auth NOT covered) / live · macos-latest · codex · API key only, subscription NOT covered (push) Waiting to run
Land #458: DeepSeek as a first-class direct provider (kazzand), rebased onto the current tip through a maintainer landing branch, with the wire-contract fixes from the review waves: effort projection onto DeepSeek's low/high/max enum, thinking toggle for none and for forced tool choices, string-only non-user content in the send copy, a two-turn live canary, and a per-call trustworthy effort disclosure. Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com>
This commit is contained in:
commit
85c1e386a1
50 changed files with 1537 additions and 301 deletions
9
.github/workflows/ci.yml
vendored
9
.github/workflows/ci.yml
vendored
|
|
@ -7,8 +7,8 @@
|
|||
# Tier 3 (Build+Release): Tag v* → PyInstaller + GitHub Release (~15 min)
|
||||
#
|
||||
# Tier 2.5 requires OPENROUTER_API_KEY / OPENAI_API_KEY / ANTHROPIC_API_KEY.
|
||||
# Optional MiniMax / Cloud.ru / GigaChat rows run when their repository secrets
|
||||
# exist. The job runs the `integration` pytest marker; locally these tests are
|
||||
# Optional MiniMax / DeepSeek / Cloud.ru / GigaChat rows run when their
|
||||
# repository secrets exist. The job runs the `integration` pytest marker; locally these tests are
|
||||
# excluded by `addopts = -m 'not integration'` in pyproject.toml.
|
||||
|
||||
name: CI
|
||||
|
|
@ -226,8 +226,8 @@ jobs:
|
|||
# Tier 2.5: Integration tests against real provider APIs
|
||||
# Triggered on push to main / ouroboros / ouroboros-stable, manual,
|
||||
# or tag v*. OPENROUTER_API_KEY / OPENAI_API_KEY / ANTHROPIC_API_KEY
|
||||
# are required in the official trusted job; MiniMax / Cloud.ru / GigaChat
|
||||
# credentials are optional and their absent rows remain visible as skips.
|
||||
# are required in the official trusted job; MiniMax / DeepSeek / Cloud.ru /
|
||||
# GigaChat credentials are optional and their absent rows remain visible as skips.
|
||||
# The `integration` pytest marker controls inclusion via `-m integration`.
|
||||
# Confirmed provider-contract failures block release-preflight; the test
|
||||
# classifier keeps quota/rate-limit/5xx/timeout outcomes inconclusive.
|
||||
|
|
@ -249,6 +249,7 @@ jobs:
|
|||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
MINIMAX_API_KEY: ${{ secrets.MINIMAX_API_KEY }}
|
||||
DEEPSEEK_API_KEY: ${{ secrets.DEEPSEEK_API_KEY }}
|
||||
CLOUDRU_FOUNDATION_MODELS_API_KEY: ${{ secrets.CLOUDRU_FOUNDATION_MODELS_API_KEY }}
|
||||
CLOUDRU_FOUNDATION_MODELS_BASE_URL: ${{ secrets.CLOUDRU_FOUNDATION_MODELS_BASE_URL }}
|
||||
GIGACHAT_CREDENTIALS: ${{ secrets.GIGACHAT_CREDENTIALS }}
|
||||
|
|
|
|||
|
|
@ -14,6 +14,7 @@ SECRET_KEYS = (
|
|||
"OPENAI_API_KEY",
|
||||
"ANTHROPIC_API_KEY",
|
||||
"MINIMAX_API_KEY",
|
||||
"DEEPSEEK_API_KEY",
|
||||
"GITHUB_TOKEN",
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -111,6 +111,7 @@ _ISO_SETTINGS_ALLOW_EXACT = frozenset({
|
|||
_PROVIDER_ENV_KEYS = frozenset({
|
||||
"OPENROUTER_API_KEY", "OPENAI_API_KEY", "OPENAI_COMPATIBLE_API_KEY",
|
||||
"CLOUDRU_FOUNDATION_MODELS_API_KEY", "ANTHROPIC_API_KEY", "MINIMAX_API_KEY",
|
||||
"DEEPSEEK_API_KEY",
|
||||
"GIGACHAT_CREDENTIALS", "GIGACHAT_PASSWORD",
|
||||
})
|
||||
|
||||
|
|
@ -136,6 +137,7 @@ _AUTHORITATIVE_ENV_PREFIXES = (
|
|||
"OPENAI_",
|
||||
"ANTHROPIC_",
|
||||
"MINIMAX_",
|
||||
"DEEPSEEK_",
|
||||
"CLOUDRU_",
|
||||
"GIGACHAT_",
|
||||
"CLAUDE_",
|
||||
|
|
|
|||
|
|
@ -56,6 +56,7 @@ _PROVIDER_ENV_KEYS = {
|
|||
"CLOUDRU_FOUNDATION_MODELS_BASE_URL",
|
||||
"MINIMAX_API_KEY",
|
||||
"MINIMAX_REGION",
|
||||
"DEEPSEEK_API_KEY",
|
||||
"GIGACHAT_CREDENTIALS",
|
||||
"GIGACHAT_USER",
|
||||
"GIGACHAT_PASSWORD",
|
||||
|
|
@ -101,6 +102,8 @@ def _credential_keys_for_model(model: str) -> set[str]:
|
|||
return {"GIGACHAT_CREDENTIALS", "GIGACHAT_USER", "GIGACHAT_PASSWORD"}
|
||||
if text.startswith("minimax::"):
|
||||
return {"MINIMAX_API_KEY", "MINIMAX_REGION"}
|
||||
if text.startswith("deepseek::"):
|
||||
return {"DEEPSEEK_API_KEY"}
|
||||
if text.startswith("openai-compatible::"):
|
||||
return {"OPENAI_COMPATIBLE_API_KEY", "OPENAI_COMPATIBLE_BASE_URL"}
|
||||
return {"OPENROUTER_API_KEY"}
|
||||
|
|
|
|||
|
|
@ -110,7 +110,9 @@ def _active_direct_provider(settings: dict[str, Any]) -> str:
|
|||
for provider, key in (
|
||||
("openai", "OPENAI_API_KEY"),
|
||||
("anthropic", "ANTHROPIC_API_KEY"),
|
||||
("minimax", "MINIMAX_API_KEY"),
|
||||
("cloudru", "CLOUDRU_FOUNDATION_MODELS_API_KEY"),
|
||||
("deepseek", "DEEPSEEK_API_KEY"),
|
||||
)
|
||||
if _setting_or_env(settings, key)
|
||||
]
|
||||
|
|
|
|||
|
|
@ -55,6 +55,7 @@ _SECRET_ENV_KEYS = frozenset({
|
|||
"GIGACHAT_CREDENTIALS",
|
||||
"GIGACHAT_PASSWORD",
|
||||
"GIGACHAT_USER",
|
||||
"DEEPSEEK_API_KEY",
|
||||
"MINIMAX_API_KEY",
|
||||
"OPENAI_API_KEY",
|
||||
"OPENAI_COMPATIBLE_API_KEY",
|
||||
|
|
@ -372,6 +373,7 @@ class OuroborosTerminalBenchAgent(BaseInstalledAgent):
|
|||
"GIGACHAT_BASE_URL",
|
||||
"GIGACHAT_VERIFY_SSL_CERTS",
|
||||
"GIGACHAT_PROFANITY_CHECK",
|
||||
"MINIMAX_REGION",
|
||||
"OUROBOROS_MODEL",
|
||||
"OUROBOROS_MODEL_LIGHT",
|
||||
"OUROBOROS_SUBAGENTS",
|
||||
|
|
@ -704,6 +706,14 @@ PY
|
|||
elif env.get("ANTHROPIC_API_KEY"):
|
||||
provider_url = "https://api.anthropic.com/v1/models"
|
||||
provider_name = "anthropic"
|
||||
elif env.get("MINIMAX_API_KEY"):
|
||||
from ouroboros.provider_models import resolve_minimax_base_url
|
||||
provider_url = resolve_minimax_base_url(str(env.get("MINIMAX_REGION") or "")).rstrip("/") + "/models"
|
||||
provider_name = "minimax"
|
||||
elif env.get("DEEPSEEK_API_KEY"):
|
||||
from ouroboros.provider_models import DEEPSEEK_BASE_URL
|
||||
provider_url = DEEPSEEK_BASE_URL.rstrip("/") + "/models"
|
||||
provider_name = "deepseek"
|
||||
elif env.get("CLOUDRU_FOUNDATION_MODELS_API_KEY"):
|
||||
provider_url = (env.get("CLOUDRU_FOUNDATION_MODELS_BASE_URL") or "https://foundation-models.api.cloud.ru/v1").rstrip("/") + "/models"
|
||||
provider_name = "cloudru"
|
||||
|
|
|
|||
|
|
@ -61,6 +61,7 @@ EXTRA_SECRET_FIELDS = (
|
|||
"OPENAI_COMPATIBLE_API_KEY",
|
||||
"CLOUDRU_FOUNDATION_MODELS_API_KEY",
|
||||
"MINIMAX_API_KEY",
|
||||
"DEEPSEEK_API_KEY",
|
||||
"GIGACHAT_PASSWORD",
|
||||
"OUROBOROS_NETWORK_PASSWORD",
|
||||
"TELEGRAM_BOT_TOKEN",
|
||||
|
|
|
|||
|
|
@ -104,6 +104,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de
|
|||
├── loop_tool_execution.py ← Tool dispatch and tool-result handling
|
||||
├── deadline_utils.py ← Shared deadline parsing/remaining-time helpers + the transport-vs-logical wait seam for loop milestones and process-tool/review timeouts
|
||||
├── observability.py ← Private forensic execution ledger: redaction, gzip CAS blobs, call manifests, trace refs
|
||||
├── provider_catalogs.py ← Live pricing-catalog fetchers (OpenRouter + cloud.ru) extracted from `llm.py`; `llm.py` re-exports the historical names
|
||||
├── cancel_intents.py ← Durable cancel-intent projection: compact locked `state/cancel_intents.json` of ACTIVE intents (request id, claim owner/pid + claim GENERATION fencing every mutation, `scope` recording single-vs-cascade so a watchdog replay re-runs the right shape) + forensic `cancel_intent` ledger rows; the ONE ingress `request_cancel` for the agent tool, HTTP single/cascade, and boot migration of legacy latch files — intent never rides the canonical task status; reads are strict and fail closed per §10 (typed `CancelIntentProjectionCorrupt`; enforcement degradation is owner-visible, never a silent "no intent"); a quarantined malformed row discloses once per row content — a ~20 s watchdog must not append the same disclosure forever, and a restart re-announcing once is honest; owns `claim_is_abandoned` and the `allow_settled_target` live-ownership exception (§10, cancellation custody)
|
||||
├── owner_hurry.py ← Owner "hurry": a typed TASK-LOCAL acceleration latch, never a chat message; the durable `owner_hurry` projection is written by `update_json_locked` touching only its own keys — never `write_task_result`, whose status-regression guard could drop concurrent terminal fields — keyed by the real attempt identity `task["_attempt"]`; while latched, the next acceptance panel is skipped with zero reviewer calls (`acceptance_skip_applied`), remaining improvement passes overlay to 0 through `effective_budget_profile` (the immutable task_contract is never rewritten), and force-plan becomes task-locally advisory; the effect DIES WITH THE ATTEMPT (`retry_reset` on every same-id requeue producer), a never-applied request is marked `not_applied_before_terminal`, and the non-chat `owner_hurry` events are hidden from chat by `log_events.js`
|
||||
├── owner_quiz.py ← Owner-quiz lifecycle projection: worker-side `record_asked`, request-id-idempotent first-answer-wins `record_answered` (option index validated against the STORED labels), structural-only `reconcile_terminal` (open → expired_terminal at task done; no host TTL), `quiz_states` replay; same locked-writer idiom as owner_hurry, touching only the `owner_quiz` key
|
||||
|
|
@ -129,7 +130,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de
|
|||
├── delegate_source_coverage.py ← Oversized-work-order source custody: canonical interval union/completeness, strict durable receipt bounds, replay-safe start binding, durable delivery confirmation for safe receipt retry, terminal cannot-verify projection, apply refusal — incomplete source cannot authorize a terminal PASS/apply; reuses `get_task_result` + the existing interaction seam; no alternate store
|
||||
├── delegate_evidence.py ← Read-side execution-evidence projection over the custody rows (`task_execution_evidence`: started/settled/succeeded/failed counts, terminal-state axis, `evidence_read_failed`, disclosed subscription spend, `nanny_nudge_recorded`, and `delegate_start_attempted` counting blocked and uncustodied attempts too, so a refused-but-obedient nanny is never disclosed as nudge-ignoring); owns the stamp writers `record_nanny_nudge_stamp`/`record_start_blocked`; projects `applied_access_profiles` — the access the engine actually served, read off SETTLED rows only (empty = no receipt disclosed it, never "no access") — and `acceptance_patch_dispositions`, the bounded section over `delegate_run_patch_verdict` rows (cap 20 with the exact omitted count, `unreviewed_delegated_apply` headline); absence of the section means NO disposition was recorded, never "reviewed clean", and an unreadable custody log is the typed `evidence_read_failed` marker, never an empty-therefore-clean section
|
||||
├── synthesis_cost_text.py ← Synthesis-prompt renderers for the pre-synthesis cost/outcome snapshot over the SSOT `cost_display`; re-exported by agent_task_pipeline.py
|
||||
├── llm.py ← Multi-provider LLM routing (OpenRouter/OpenAI/compatible/Cloud.ru/MiniMax/GigaChat/Anthropic); canonical conversations stay function-shaped while the physical-send seam delegates exact-route request adaptation to the request-wire leaves below
|
||||
├── llm.py ← Multi-provider LLM routing (OpenRouter/OpenAI/compatible/Cloud.ru/MiniMax/DeepSeek/GigaChat/Anthropic); canonical conversations stay function-shaped while the physical-send seam delegates exact-route request adaptation to the request-wire leaves below
|
||||
├── net_transport.py ← Shared httpx transport construction for remote LLM clients; TCP-keepalive socket options
|
||||
├── transport_custody.py ← Typed transport facts for the physical-attempt custody seam
|
||||
├── openrouter_attribution.py ← Canonical OpenRouter application attribution, centralized so forks do not compete under the same external application identity
|
||||
|
|
@ -159,7 +160,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de
|
|||
├── main_context_authority.py ← Deep-copies the context authority; replaces only oversized raw result strings with source-resolvable narrative or a typed gap
|
||||
├── client_surface.py ← Closed-key bounded client-surface normalizer; surface identity excludes viewport/narrow_layout; mailbox surface-change note
|
||||
├── context_fit.py ← Deterministic Max/Low context projections from one immutable core with labelled measurement + typed reclaim deficit; no routing/retry/global-mode authority
|
||||
├── context_budget.py ← Context budget vocabulary + typed reclaim SSOT (owner-Low 200K economy target); owns `estimate_message_chars` (images counted at `IMAGE_BLOCK_CHAR_EQUIVALENT`) as the one shared bounded basis
|
||||
├── context_budget.py ← Context budget vocabulary + typed reclaim SSOT (owner-Low 200K economy target); owns `estimate_message_chars` (images counted at `IMAGE_BLOCK_CHAR_EQUIVALENT`, replayed `reasoning_content` counted on the DeepSeek echo lane), the bounded basis of the local compaction proxy; remote fit and the density witness measure on `context_fit.estimate_context_prompt_tokens`
|
||||
├── context_mode_compat.py ← One-window compatibility shim for the retired persistent context auto-Low state
|
||||
├── capability_evidence.py ← Sourced capability evidence in `data/state/capability_evidence.json` (confirmed/asserted/unprobeable/failed): authorizing readers require fresh evidence, and unknown fails the ≥1M gates closed; owns `observe_token_density` — the density witness calibrates on the bounded-proxy basis the fit estimator measures, while budget reservation keeps RAW, because over-counting money is the safe direction and the two consumers split on purpose; exact-route dispatch authority lives in `request_wire_compatibility.json`
|
||||
├── context_layout.py ← Doc-layout SSOT: tier-0 always full; ARCHITECTURE full in Max and a lossless fence-aware navigation map in Low; DEVELOPMENT full-or-pointer by task binding; reduction is by relocation with a visible pointer, never silent truncation
|
||||
|
|
@ -557,7 +558,7 @@ Packaged startup is an ordered ownership transaction. The launcher prepares the
|
|||
|
||||
Onboarding runs after the gateway because connecting an agent subscription is a live `/api/*` conversation, not a form field — and a gateway without a supervisor is exactly what the readiness predicate produces, so no second server, mode, or onboarding state machine exists. `ouroboros/launcher_onboarding.py` owns the presentation (readiness decision, setup window, window-lifecycle bridge); when completion reports a boot-pinned value changed, the launcher recycles the managed server rather than counting the exit as a crash. Neither launcher nor server boot normalization may CREATE `settings.json` (the install-time latches below are gated on its absence).
|
||||
|
||||
`has_startup_ready_provider()` is a structural gate, not a network, credential, entitlement, model, or local-process probe: any non-empty recognized remote configuration or active task-capable local routing flag passes (key list: `server_runtime.has_startup_ready_provider`; `USE_LOCAL_HEAVY` is legacy migration input only, `LOCAL_MODEL_SOURCE` alone insufficient). When the gate is false the server marks startup complete without workers so the web UI serves the blocking onboarding overlay; a later successful settings save hot-starts the supervisor.
|
||||
`has_startup_ready_provider()` is a structural gate, not a network, credential, entitlement, model, or local-process probe: any non-empty recognized remote configuration (including DeepSeek) or active task-capable local routing flag passes (key list: `server_runtime.has_startup_ready_provider`; `USE_LOCAL_HEAVY` is legacy migration input only, `LOCAL_MODEL_SOURCE` alone insufficient). When the gate is false the server marks startup complete without workers so the web UI serves the blocking onboarding overlay; a later successful settings save hot-starts the supervisor.
|
||||
|
||||
Every host renders one served page: `GET /onboarding` returns `onboarding_template.html` with the `settings_setup_contract` bootstrap injected, linking wizard CSS and `web/modules/onboarding_wizard.js` as static assets so steps import the same modules as the rest of the UI — an inlined `srcdoc` string cannot. The desktop setup window opens that URL, the blocking overlay frames it, a browser owner opens it directly; `GET /api/onboarding` is the readiness probe (204 once the gate passes, otherwise the page). Steps: providers/access, agents, model slots, review enforcement plus initial runtime mode, budget, summary; context mode is not configured here. The agents step is skippable, owns no input, and mounts the shared login cards in `full` mode because `compact` omits the paste-code entry a Claude login needs when its localhost callback cannot complete; its account facts come from the shared Claudexor status store and become the completion payload's `subscriptionsConnected` declaration — a request to look at the daemon, never an authority.
|
||||
|
||||
|
|
@ -569,7 +570,7 @@ Completion is one HTTP conversation on every host — `POST /api/onboarding/comp
|
|||
|
||||
Validation is structural: at least one exposed remote configuration or a local model source; local-only setup routes at least one active lane locally; Main required, while Light, Vision, Consciousness, and Fallback keep inheritance/empty semantics and Heavy is readable only for bounded migration into an explicit API actor; enforcement and runtime mode are closed enums, budgets finite and positive, the MiniMax region closed, a Hugging Face local source needs a filename. Credential length is checked only on fields changed in the payload: rejecting an unchanged short legacy value would discard the whole form, including its own repair.
|
||||
|
||||
Provider readiness and provider defaulting are separate. With no OpenRouter, legacy OpenAI base, or OpenAI-compatible endpoint, exactly one registered direct provider receives provider-prefixed defaults and migration of untouched shipped/legacy slot values; multiple direct providers stay owner-editable, and OpenRouter keeps router-style routing. An arbitrary OpenAI-compatible endpoint gets no guessed model ids — compatible servers have no universal safe name; the owner selects explicit `openai-compatible::...` routes. A local-source install with no remote provider clears only untouched shipped remote Light/Fallback values that would be unreachable; owner-authored values and explicitly local slots are preserved. This is migration of defaults, not a model allowlist, and never proof the local server is running.
|
||||
Provider readiness and provider defaulting are separate. With no OpenRouter, legacy OpenAI base, or OpenAI-compatible endpoint, exactly one registered direct provider receives provider-prefixed defaults and migration of untouched shipped/legacy slot values; OpenAI, Anthropic, Cloud.ru, GigaChat, MiniMax, and DeepSeek each use their own registered defaults. Multiple direct providers stay owner-editable, and OpenRouter keeps router-style routing. An arbitrary OpenAI-compatible endpoint gets no guessed model ids — compatible servers have no universal safe name; the owner selects explicit `openai-compatible::...` routes. A local-source install with no remote provider clears only untouched shipped remote Light/Fallback values that would be unreachable; owner-authored values and explicitly local slots are preserved. This is migration of defaults, not a model allowlist, and never proof the local server is running.
|
||||
|
||||
`scripts/build_repo_bundle.py` creates the packaged seed only from a clean named checkout, writes a git bundle of that commit/tags, and records schema, version, source SHA, release tag, bundle hash, and managed branch/remote metadata (the release-tag check itself: §8). The launcher validates the manifest fields and the bundle SHA-256 but no per-file member set: clone-time Git verification proves the manifest source object exists and checked-out HEAD equals it.
|
||||
|
||||
|
|
@ -1310,6 +1311,7 @@ A registry of `config.SETTINGS_DEFAULTS` (exact defaults stay canonical in `conf
|
|||
| ANTHROPIC_API_KEY | "" | Official direct-Anthropic credential |
|
||||
| MINIMAX_API_KEY | "" | MiniMax credential |
|
||||
| MINIMAX_REGION | "" | MiniMax region (empty resolves `global_en`) |
|
||||
| DEEPSEEK_API_KEY | "" | Optional. DeepSeek direct provider key (`deepseek::...` model values, OpenAI-compatible API at the fixed official endpoint) |
|
||||
| OUROBOROS_NETWORK_PASSWORD | "" | Non-localhost HTTP gate password (`server_auth.py`; unset only warns — see §8 packaging note) |
|
||||
| OUROBOROS_SERVER_HOST | 127.0.0.1 | HTTP bind host (`0.0.0.0` for Docker/non-loopback) |
|
||||
| OUROBOROS_UPDATE_CHANNEL | `stable` | Update channel: stable/qa/development (§8) |
|
||||
|
|
@ -1449,7 +1451,9 @@ A registry of `config.SETTINGS_DEFAULTS` (exact defaults stay canonical in `conf
|
|||
| GITHUB_REPO | "" | Personal `origin` repository |
|
||||
| OUROBOROS_FILE_BROWSER_DEFAULT | "" | File Browser default root (explicit root required for Docker/non-localhost) |
|
||||
|
||||
Direct-provider review fallback (legacy name: OpenAI-only review fallback): when exactly one official direct provider is configured, `config.get_review_models()` compiles that provider's declarative reviewer-role sequence using provider-prefixed model IDs. Current scope covers official OpenAI, Anthropic, MiniMax, Cloud.ru, and GigaChat; OpenRouter, legacy-base, OpenAI-compatible, and mixed-provider configurations stay outside it. OpenAI and Anthropic run three independent Main-model slots; MiniMax mixes Main/Light; Cloud.ru and GigaChat use their one role model for every slot. `_exclusive_direct_remote_provider_env` returns empty when OpenRouter, legacy `OPENAI_BASE_URL`, OpenAI-compatible keys, or multiple official direct providers are present, and the fallback requires `provider_models.migrate_model_value` to make the main model already start with the exclusive provider prefix — exact prefix checking prevents an arbitrary free-text model from silently entering a single-provider route. This is part of the single-provider independence invariant (docs/DEVELOPMENT.md "Provider Independence").
|
||||
Direct-provider review fallback (legacy name: OpenAI-only review fallback): when exactly one official direct provider is configured, `config.get_review_models()` compiles that provider's declarative reviewer-role sequence using provider-prefixed model IDs. Current scope covers official OpenAI, Anthropic, MiniMax, DeepSeek, Cloud.ru, and GigaChat; OpenRouter, legacy-base, OpenAI-compatible, and mixed-provider configurations stay outside it. OpenAI, Anthropic, and DeepSeek run three independent Main-model slots; MiniMax mixes Main/Light; Cloud.ru and GigaChat use their one role model for every slot. `_exclusive_direct_remote_provider_env` returns empty when OpenRouter, legacy `OPENAI_BASE_URL`, OpenAI-compatible keys, or multiple official direct providers are present, and the fallback requires `provider_models.migrate_model_value` to make the main model already start with the exclusive provider prefix — exact prefix checking prevents an arbitrary free-text model from silently entering a single-provider route. This is part of the single-provider independence invariant (docs/DEVELOPMENT.md "Provider Independence").
|
||||
|
||||
DeepSeek provider specifics (`deepseek::`): the official OpenAI-compatible endpoint is a fixed module constant (`provider_models.DEEPSEEK_BASE_URL`); a proxy or mirror belongs to the generic `openai-compatible::` route. The canonical reasoning scale is projected onto the provider's wire dialect at the send boundary (`minimal`→`low`, `medium`/`xhigh`→`high`, `ultra`→`max`, `none`→`extra_body.thinking.type=disabled`; native tiers pass through), a forced tool choice (`required` or a named tool) is served with thinking disabled because thinking mode accepts only `auto`/`none` (probed 2026-09-03), and every tier-changing projection is disclosed on usage; `reasoning_content` stays on canonical assistant turns for strict v4 tool replay with an explicit empty string for turns produced without provider reasoning, while other lanes strip the field and cross-family switches scrub it. System/assistant/tool content arrays are flattened to strings in the send copy only (the API accepts arrays on user turns alone); canonical block history is untouched. Prompt caching is automatic and cost remains nullable when no exact provider catalog is available. The 1M context claim is admitted only through route-fingerprinted capability evidence or owner acknowledgement. Slash-form `deepseek/...` remains OpenRouter; only `deepseek::...` selects the direct route.
|
||||
|
||||
GigaChat provider specifics (`gigachat::`): routed through the native `gigachat` library, not OpenAI-compatible (`llm.py::_chat_gigachat`). OpenAI `tools` map to GigaChat `functions`; at most ONE `function_call` returns per turn, so parallel `tool_calls` collapse to the first; role `tool` results become role `function` and must be valid JSON (plain text wrapped as `{"result": ...}`); the `system` message must be first, so later system-reminders demote to `user`. `reasoning_effort` is deliberately omitted — hidden reasoning can consume the whole output budget and return empty content. GigaChat exposes no automatic live cost source, so cost stays nullable/unknown rather than a hand-maintained tariff. GigaChat models sit below the 1M scope-review floor; when no ≥1M reviewer is configured, the declared alternatives are the owner-selected `low` context mode (whole-repo scope review declaredly not performed; each commit records a typed `skipped_low_context_mode` evidence row) or an owner-selected retrieving scope slot at ≥200K sourced evidence (BIBLE P3); the blocking triad still reviews the full staged diff in both modes.
|
||||
|
||||
|
|
@ -1474,7 +1478,7 @@ The local `ouroboros-stable` ref is also a recovery fallback maintained by expli
|
|||
|
||||
Quick and full jobs each run a dedicated blocking `size_ratchet` pytest step — the ONLY enforcing surface for the repository size gates (local runs exclude the marker and warn): manifest exactness on the tip plus the pairwise shrink-only transition against the event base in `OURO_SIZE_RATCHET_BASE_REF`; an unresolvable base degrades to the tip's parent manifest verified against the parent's own tree — never a skip — while a resolvable base without a manifest fails closed. Both jobs also run the browser-module suite (`cd web && node --test tests/*.test.js`), the same node lane the hermetic commit gate executes through `ouroboros/preflight_node.py`. Secret-bearing skill review runs before any step that imports downloaded plugin code — untrusted payload code must never share a process with provider credentials — and a missing required key is red rather than skipped.
|
||||
|
||||
Tool-schema compatibility has two layers. Fork-safe PR tests build the complete shipped catalog without MCP or extensions, validate every schema as general JSON Schema plus the cross-provider subset (no empty enum, no root `anyOf`/`oneOf`/`allOf`), and require the OpenRouter/function, direct-Anthropic, GigaChat, and direct-OpenAI projections to preserve the complete tool-name set. The trusted `integration-test` lane sends that same full registry in one bounded `delegate_start` canary per physical route without executing the returned call — OpenRouter Gemini/Opus/GPT/Grok/DeepSeek, the three shipped direct-OpenAI defaults (Main alone keeps a second-turn nonce-bearing continuation), direct Anthropic, and optional MiniMax/Cloud.ru/GigaChat. Each canary requires positive usage, exact provider/model identity, and a normalized schema-valid tool call; quota/billing, 429, 5xx, and timeout outcomes stay typed inconclusive while contract/auth/model/tool/reasoning 4xx are red; one same-route resend is allowed only for a runtime-classified semantic-empty response (bypassing response caches where supported). `response_finish_reason` is retained in host usage for bounded diagnostics only; malformed raw arguments are reported by type/position and hash without copying the provider payload.
|
||||
Tool-schema compatibility has two layers. Fork-safe PR tests build the complete shipped catalog without MCP or extensions, validate every schema as general JSON Schema plus the cross-provider subset (no empty enum, no root `anyOf`/`oneOf`/`allOf`), and require the OpenRouter/function, direct-Anthropic, GigaChat, and direct-OpenAI projections to preserve the complete tool-name set. The trusted `integration-test` lane sends that same full registry in one bounded `delegate_start` canary per physical route without executing the returned call — OpenRouter Gemini/Opus/GPT/Grok/DeepSeek, the three shipped direct-OpenAI defaults (Main alone among them keeps a second-turn nonce-bearing continuation), direct Anthropic, and optional MiniMax/DeepSeek/Cloud.ru/GigaChat (DeepSeek also keeps the continuation because its tool contract is the reasoning_content replay). Each canary requires positive usage, exact provider/model identity, and a normalized schema-valid tool call; quota/billing, 429, 5xx, and timeout outcomes stay typed inconclusive while contract/auth/model/tool/reasoning 4xx are red; one same-route resend is allowed only for a runtime-classified semantic-empty response (bypassing response caches where supported). `response_finish_reason` is retained in host usage for bounded diagnostics only; malformed raw arguments are reported by type/position and hash without copying the provider payload.
|
||||
|
||||
`claudexor-platform-gate.yml` proves the managed Claudexor runtime on three OSes: a fixture lane (fake harness, offline, $0) always, and a live lane only on explicit API keys — subscription auth stays deliberately out of CI, because an interactive machine-bound token must not enter CI secrets. `dependency-graph.yml` reads `ouroboros/claudexor_runtime_pin.json` and submits that direct runtime relationship to GitHub's dependency graph (runs only on pin/workflow changes on `main`/`ouroboros`, plus manual dispatch; `contents: write` only) without presenting Claudexor as a Python or Node package dependency. The Scorecard workflow (`.github/workflows/scorecard.yml`) runs on `main` pushes and weekly, pins every action by full commit SHA, defaults permissions to read-only, and adds only `security-events: write` + `id-token: write` for SARIF upload and OpenSSF publication. `CODE_OF_CONDUCT.md` owns community rules; `CITATION.cff` owns the software and preferred technical-report citations; `site/paper/index.html` owns the canonical paper landing page; `docs/benchmarks/evidence.json` holds the release-bound public benchmark projection; README remains the claim SSOT (guarded by `tests/test_trust_metadata.py` and `tests/test_public_site_metadata.py`).
|
||||
|
||||
|
|
|
|||
|
|
@ -157,7 +157,7 @@ Used by `commit_reviewed` for all changes to the Ouroboros repository.
|
|||
| # | item | what to check | severity when FAIL |
|
||||
|---|------|---------------|--------------------|
|
||||
| 1 | bible_compliance | Does the diff violate any BIBLE.md principle? | critical |
|
||||
| 2 | development_compliance | Does it follow DEVELOPMENT.md patterns? Check explicitly: (a) naming conventions (snake_case modules/vars, PascalCase classes, UPPER_SNAKE_CASE constants); (b) entity type rules — Gateway classes contain ONLY transport, no business logic; Tool functions are thin wrappers; (c) Python everywhere (including `tests/`/`devtools/`) and first-party `web/**/*.js` (including `web/tests/`) target ~1000 lines; exact repo-relative module debt above the 1600-line hard gate, exact `(path, qualname)` Python-function debt above 300 lines, the exact-current 1001-1500 band (new/re-entered paths need a nonblank rationale), and exact byte debt above 200,000 canonical UTF-8/LF bytes are checked in to `ouroboros/size_ratchet_manifest.py`; the enforcing surface for all of these (and for `MAX_TOTAL_FUNCTIONS`) is the official repository CI's `size_ratchet` pytest lane — manifest exactness on the tip tree plus the pairwise base-vs-tip shrink-only transition — while local runs surface the same `validate_size_ratchet` findings as warnings (a stale or growing entry is therefore review debt to flag, not a local commit block); methods above 150 lines are a decomposition signal, runtime-code total Python function/method count stays under `ouroboros/review.py::MAX_TOTAL_FUNCTIONS`, and more than eight parameters is a decomposition signal, not a hard gate; (d) no gratuitous abstract layers, and any SOLID/minimalism finding names an exact symbol/authority, concrete duplication or coupling, and a smaller contract-preserving alternative rather than citing diff size (P7 Minimalism) — and when the diff ADDS a surface (a new module, state file, ledger, resolver, cache, retry path, tool, endpoint, or background loop), the reviewer consults the docs/ARCHITECTURE.md map and NAMES the existing mechanism that already covers the need when one exists (name it exactly — the reuse-first duty this checklist carries for a CHANGE; the plan-review checklist judges an intention and has no such generative duty); absence of a covering mechanism may be stated in one line; (e) new LLM calls go through the shared `LLMClient`/`llm.py` layer, not ad-hoc HTTP clients; (f) cognitive artifacts (identity.md, scratchpad, task reflections, review outputs) must NOT use hardcoded `[:N]` truncation — explicit omission notes required; (g) new `get_tools()` exports follow the ToolEntry pattern in registry.py; (h) provider independence — no change may make a core capability (agent loop, multi-model commit review, scope review, or memory/context flows) silently require a second provider or OpenRouter specifically, and every supported single direct provider (local, OpenAI, Anthropic, MiniMax, Cloud.ru, GigaChat) must keep its model AND review/scope slots self-fillable (see DEVELOPMENT.md "Provider Independence"); (i) a claimed-complete visible UI change includes vision-inspected evidence from at least one relevant real consumer flow. A screenshot file without inspection is insufficient; states/viewports/additional engines are risk-selected, mobile/WebKit are not universal, and an unavailable optional engine alone is not degradation. | critical |
|
||||
| 2 | development_compliance | Does it follow DEVELOPMENT.md patterns? Check explicitly: (a) naming conventions (snake_case modules/vars, PascalCase classes, UPPER_SNAKE_CASE constants); (b) entity type rules — Gateway classes contain ONLY transport, no business logic; Tool functions are thin wrappers; (c) Python everywhere (including `tests/`/`devtools/`) and first-party `web/**/*.js` (including `web/tests/`) target ~1000 lines; exact repo-relative module debt above the 1600-line hard gate, exact `(path, qualname)` Python-function debt above 300 lines, the exact-current 1001-1500 band (new/re-entered paths need a nonblank rationale), and exact byte debt above 200,000 canonical UTF-8/LF bytes are checked in to `ouroboros/size_ratchet_manifest.py`; the enforcing surface for all of these (and for `MAX_TOTAL_FUNCTIONS`) is the official repository CI's `size_ratchet` pytest lane — manifest exactness on the tip tree plus the pairwise base-vs-tip shrink-only transition — while local runs surface the same `validate_size_ratchet` findings as warnings (a stale or growing entry is therefore review debt to flag, not a local commit block); methods above 150 lines are a decomposition signal, runtime-code total Python function/method count stays under `ouroboros/review.py::MAX_TOTAL_FUNCTIONS`, and more than eight parameters is a decomposition signal, not a hard gate; (d) no gratuitous abstract layers, and any SOLID/minimalism finding names an exact symbol/authority, concrete duplication or coupling, and a smaller contract-preserving alternative rather than citing diff size (P7 Minimalism) — and when the diff ADDS a surface (a new module, state file, ledger, resolver, cache, retry path, tool, endpoint, or background loop), the reviewer consults the docs/ARCHITECTURE.md map and NAMES the existing mechanism that already covers the need when one exists (name it exactly — the reuse-first duty this checklist carries for a CHANGE; the plan-review checklist judges an intention and has no such generative duty); absence of a covering mechanism may be stated in one line; (e) new LLM calls go through the shared `LLMClient`/`llm.py` layer, not ad-hoc HTTP clients; (f) cognitive artifacts (identity.md, scratchpad, task reflections, review outputs) must NOT use hardcoded `[:N]` truncation — explicit omission notes required; (g) new `get_tools()` exports follow the ToolEntry pattern in registry.py; (h) provider independence — no change may make a core capability (agent loop, multi-model commit review, scope review, or memory/context flows) silently require a second provider or OpenRouter specifically, and every supported single direct provider (local, OpenAI, Anthropic, MiniMax, DeepSeek, Cloud.ru, GigaChat) must keep its model AND review/scope slots self-fillable (see DEVELOPMENT.md "Provider Independence"); (i) a claimed-complete visible UI change includes vision-inspected evidence from at least one relevant real consumer flow. A screenshot file without inspection is insufficient; states/viewports/additional engines are risk-selected, mobile/WebKit are not universal, and an unavailable optional engine alone is not degradation. | critical |
|
||||
| 3 | secrets_check | Are secrets, API keys, .env files, credentials present in the diff? | critical |
|
||||
| 4 | code_quality | Careful code review: bugs, logic errors, crashes, regressions, race conditions, resource leaks? | critical |
|
||||
| 5 | security_issues | Security vulnerabilities: injection, path traversal, secret leakage, unsafe operations? | critical |
|
||||
|
|
@ -564,7 +564,7 @@ and do not return `PASS` for an item that also has a `FAIL` — the concrete
|
|||
| 2 | permissions_honesty | Do the declared `permissions` match what the scripts actually do? Missing permission declaration for an effect the code performs is a concrete FAIL. Examples: `net` must be declared if any script uses `httpx`/`requests`/`socket`/`urllib`; `fs` must be declared if a script writes outside the skill state dir; `subprocess` must be declared if a skill spawns another process. A transport using the `presence` permission must submit only authenticated provider event facts plus an opaque owner-created binding; supplying prompt text, a profile, tools, roots, destinations, authority hashes, or owner commands as trusted fields is a concrete FAIL. | critical |
|
||||
| 3 | no_repo_mutation | Does any script attempt to write to the self-modifying Ouroboros repo (`~/Ouroboros/repo/`)? Import of `write_file`/`commit_reviewed` against the system repo, `git add`/`git commit`, or any path that starts with `OUROBOROS_REPO_DIR` / `~/Ouroboros/repo` is a concrete FAIL. Skills may only propose patches by returning artifact bundles; commits go through the first-party reviewed path. | critical |
|
||||
| 4 | path_confinement | Do scripts stay inside the skill directory and the dedicated state dir (`~/Ouroboros/data/state/skills/<name>/`)? Absolute paths, `..` traversal, and writes to arbitrary user home subdirs are concrete FAIL. Reading from outside the skill dir is OK for read-only lookups (e.g. system info), write-path confinement is the strict rule. | critical |
|
||||
| 5 | env_allowlist | Is `env_from_settings` a short, justified list of settings keys? Core keys in `FORBIDDEN_SKILL_SETTINGS` (`OPENROUTER_API_KEY`, `OPENAI_API_KEY`, `OPENAI_COMPATIBLE_API_KEY`, `CLOUDRU_FOUNDATION_MODELS_API_KEY`, `GIGACHAT_CREDENTIALS`, `GIGACHAT_PASSWORD`, `ANTHROPIC_API_KEY`, `MINIMAX_API_KEY`, `TELEGRAM_BOT_TOKEN`, `GITHUB_TOKEN`, `OUROBOROS_NETWORK_PASSWORD`) may be declared only when the skill genuinely needs that provider/token for its stated purpose; runtime forwards them only after a fresh executable review and a content-bound desktop-launcher owner grant. v5.2.2 dual-track grants: both `type: script` skills (forwarded by `_scrub_env`) and `type: extension` skills (forwarded by `PluginAPIImpl.get_settings`) are eligible; `type: instruction` skills cannot receive core keys. Mark unjustified core-key requests or non-forbidden secrets unrelated to the purpose as FAIL. An empty list is the default and always fine. | critical |
|
||||
| 5 | env_allowlist | Is `env_from_settings` a short, justified list of settings keys? Core keys in `FORBIDDEN_SKILL_SETTINGS` (`OPENROUTER_API_KEY`, `OPENAI_API_KEY`, `OPENAI_COMPATIBLE_API_KEY`, `CLOUDRU_FOUNDATION_MODELS_API_KEY`, `GIGACHAT_CREDENTIALS`, `GIGACHAT_PASSWORD`, `ANTHROPIC_API_KEY`, `MINIMAX_API_KEY`, `DEEPSEEK_API_KEY`, `TELEGRAM_BOT_TOKEN`, `GITHUB_TOKEN`, `OUROBOROS_NETWORK_PASSWORD`) may be declared only when the skill genuinely needs that provider/token for its stated purpose; runtime forwards them only after a fresh executable review and a content-bound desktop-launcher owner grant. v5.2.2 dual-track grants: both `type: script` skills (forwarded by `_scrub_env`) and `type: extension` skills (forwarded by `PluginAPIImpl.get_settings`) are eligible; `type: instruction` skills cannot receive core keys. Mark unjustified core-key requests or non-forbidden secrets unrelated to the purpose as FAIL. An empty list is the default and always fine. | critical |
|
||||
| 6 | timeout_and_output_discipline | Is `timeout_sec` reasonable for the stated workload (default 60, hard cap 300)? Do scripts print to stdout in chunks that the runtime can cap, rather than streaming unbounded output? Unbounded loops without a `break`/timeout path are a concrete FAIL. | advisory |
|
||||
| 7 | extension_namespace_discipline | `type: extension` only: does the extension register its tool/route/ws-handler/ui-tab under the namespace derived from its `name` (e.g. provider-safe tool/ws names like `ext_<len>_<token>_<surface>`, route `/api/extensions/<name>/…`)? Tool and WS short names must be alphanumeric/underscore and at most 24 characters. Namespace collisions with built-in surfaces are a concrete FAIL. If the extension uses `api.send_ws_message`, are emitted event names short/provider-safe and paired with reviewed host-owned widget `subscription` components rather than arbitrary same-origin JavaScript? If the extension declares streaming UI, is it a reviewed extension route consumed by a host-owned `stream` component? A reviewed `module` widget may also consume the skill's own routes (including streaming responses) and the skill's namespaced WebSocket events through the host-mediated bridge (`OuroborosWidget.fetch` / `OuroborosWidget.onEvent`), which is not arbitrary same-origin JavaScript. If the extension owns background resources (threads, sockets, EventSource clients, subprocesses), does it register cleanup with `api.on_unload(callback)`? If the extension declares a widget render block, is it one of the host-owned schemas (`iframe`, `module`, or declarative v1: forms/actions, markdown/code, JSON/kv/table, tabs/chart, stream/subscription, progress/poll, file/gallery/media, map/calendar/kanban, group/metric/callout), with media sourced from extension routes or safe data URLs and no arbitrary same-origin JavaScript? Nested interactive group/tab children must use stable identity and one host-owned lifecycle, while `subscription.render` stays transitively passive. For non-extension skills, verdict PASS with reason "Not applicable — type != extension." | severity-driven for applicable extensions |
|
||||
| 8 | widget_module_safety | **v5.7.0+. ``kind: "module"`` widgets only.** The host fetches reviewed ``widget.js`` through ``GET /api/extensions/<skill>/module/<entry>``, embeds the source into a sandboxed opaque-origin ``<iframe srcdoc sandbox="allow-scripts allow-pointer-lock allow-downloads" allow="autoplay; fullscreen; clipboard-write">`` with no ``allow-same-origin`` — ``document.cookie``, ``localStorage``, and ``sessionStorage`` throw ``SecurityError`` there by construction and need no source review — and injects a parent-mediated ``fetch`` bridge that rejects paths outside the owning skill route prefix. Reviewers confirm at the source level what the sandbox cannot: (a) no ``fetch``/``XMLHttpRequest`` URL outside ``/api/extensions/<skill>/`` and no bespoke ``postMessage`` protocol to ``window.parent`` beyond the host bridge; (b) the declared launch policy ``render.start`` (SSOT ``ouroboros/extension_ui_validation.py::WIDGET_START_MODES``; see CREATING_SKILLS "Launch policy") fits the widget's weight — ``auto`` only for a cheap instrument, ``manual`` for a program that should not run all the time, ``retain`` only for a program that genuinely must keep running while the owner is elsewhere and stays cheap while hidden; (c) a widget with state worth keeping registers ``window.__ouroWidgetOnDispose(fn)`` (never assigns over it) and saves that state through the skill's own routes, because the frame is disposable. Acceptable interactions: ``fetch('/api/extensions/<skill>/...')`` (through the host bridge), ``window.OuroborosWidget.fetch('/api/extensions/<skill>/...')``, and host-supplied data attributes. Mark non-module widgets and non-extension skills PASS with reason "Not applicable". | severity-driven when kind=module |
|
||||
|
|
|
|||
|
|
@ -486,7 +486,7 @@ Skills UI.
|
|||
## Grants for protected keys and host permissions
|
||||
|
||||
Some settings keys are protected: `OPENROUTER_API_KEY`,
|
||||
`OPENAI_API_KEY`, `OPENAI_COMPATIBLE_API_KEY`, `ANTHROPIC_API_KEY`, `MINIMAX_API_KEY`,
|
||||
`OPENAI_API_KEY`, `OPENAI_COMPATIBLE_API_KEY`, `ANTHROPIC_API_KEY`, `MINIMAX_API_KEY`, `DEEPSEEK_API_KEY`,
|
||||
`CLOUDRU_FOUNDATION_MODELS_API_KEY`, `GIGACHAT_CREDENTIALS`, `GIGACHAT_PASSWORD`, `TELEGRAM_BOT_TOKEN`,
|
||||
`GITHUB_TOKEN`, `OUROBOROS_NETWORK_PASSWORD`. These keys are NEVER
|
||||
forwarded to a skill by default, even when listed in
|
||||
|
|
|
|||
|
|
@ -314,7 +314,17 @@ OpenAI tool conversations stay on Chat Completions — custom-first when
|
|||
non-`none` reasoning is requested, an exact custom rejection may fall back to
|
||||
function with the same effort, and explicit `none` is a task-local last resort
|
||||
— and send `reasoning_effort` and `max_completion_tokens` provider-wide;
|
||||
model-name prefixes are not admission authority.
|
||||
model-name prefixes are not admission authority. DeepSeek is the second
|
||||
effort-carrying route (`reasoning_effort` beside the compatible-lane
|
||||
`max_tokens` carrier): the canonical tiers are projected onto its documented
|
||||
`low`/`high`/`max` enum at the send boundary (`minimal`→`low`,
|
||||
`medium`/`xhigh`→`high`, `ultra`→`max`), `none` becomes
|
||||
`extra_body.thinking.type=disabled`, a forced tool choice (`required`/named) is
|
||||
served with thinking disabled because thinking mode accepts only `auto`/`none`
|
||||
(live-probed 2026-09-03), and every projection that changes the tier is
|
||||
disclosed on usage as `reasoning_effort_clamped`. The carriage is keyed on the
|
||||
provider id, never a model-name prefix or a target capability field, so a
|
||||
hand-built target cannot silently drop it.
|
||||
|
||||
All learned request-shape adaptation goes through the one provider-neutral
|
||||
request-wire driver (`ouroboros/request_wire_contract.py`: exact-route
|
||||
|
|
|
|||
|
|
@ -1031,9 +1031,14 @@ def probe(
|
|||
return ev
|
||||
|
||||
|
||||
# Cache-inclusive prompt totals are measurable; GigaChat's semantics remain unknown.
|
||||
# Cache-inclusive prompt totals are measurable; GigaChat's and MiniMax's
|
||||
# semantics remain unknown. DeepSeek probed 2026-09-01: prompt_tokens =
|
||||
# prompt_cache_hit_tokens + prompt_cache_miss_tokens, i.e. cache-inclusive —
|
||||
# and its automatic cache makes nearly every warm call cache-bearing, so
|
||||
# excluding it would starve the route of density witnesses entirely.
|
||||
_CACHE_INCLUSIVE_PROMPT_TOKEN_PROVIDERS = frozenset({
|
||||
"openrouter", "openai", "openai-compatible", "cloudru", "local", "anthropic",
|
||||
"deepseek",
|
||||
})
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -22,7 +22,7 @@ DEFAULT_COLAB_APP_ROOT = "/content/drive/MyDrive/Ouroboros"
|
|||
DEFAULT_COLAB_REPO_DIR = "/content/ouroboros_repo"
|
||||
DEFAULT_OFFICIAL_REPO_URL = "https://github.com/razzant/ouroboros.git"
|
||||
|
||||
_SECRET_KEYS = ("OPENROUTER_API_KEY", "OPENAI_API_KEY", "ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "CLOUDRU_FOUNDATION_MODELS_API_KEY", "GITHUB_TOKEN", "TELEGRAM_BOT_TOKEN")
|
||||
_SECRET_KEYS = ("OPENROUTER_API_KEY", "OPENAI_API_KEY", "ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "DEEPSEEK_API_KEY", "CLOUDRU_FOUNDATION_MODELS_API_KEY", "GITHUB_TOKEN", "TELEGRAM_BOT_TOKEN")
|
||||
|
||||
|
||||
def _run_colab_git_network(args: list[str], *, cwd: pathlib.Path | None = None) -> str:
|
||||
|
|
@ -76,15 +76,16 @@ def collect_colab_secrets() -> Dict[str, str]:
|
|||
"""Collect runtime secrets without printing their values.
|
||||
|
||||
The Telegram bot token is required (the bridge needs it). Any supported
|
||||
provider key works — OpenRouter, OpenAI, or Anthropic are collected
|
||||
optionally, and OpenRouter is prompted only if none is found, so an
|
||||
OpenAI-only or Anthropic-only Colab user is not forced to enter OpenRouter.
|
||||
provider key works — OpenRouter, OpenAI, Anthropic, MiniMax, DeepSeek and
|
||||
Cloud.ru are collected optionally, and OpenRouter is prompted only if none
|
||||
is found, so a single-direct-provider Colab user is not forced to enter
|
||||
OpenRouter.
|
||||
The GitHub token is optional (it only enables personal self-modification
|
||||
persistence), so a quick prototype never blocks on a GitHub prompt.
|
||||
"""
|
||||
# Providers the one-click Colab launch can auto-route models for via
|
||||
# apply_runtime_provider_defaults (OpenRouter is the default aggregator;
|
||||
# OpenAI/Anthropic/MiniMax/Cloud.ru have direct model defaults). OpenAI-compatible
|
||||
# OpenAI/Anthropic/MiniMax/DeepSeek/Cloud.ru have direct model defaults). OpenAI-compatible
|
||||
# endpoints have no universal model default and need explicit OUROBOROS_MODEL_*
|
||||
# config, so they are an advanced manual path, not part of the quick launch.
|
||||
provider_keys = (
|
||||
|
|
@ -92,6 +93,7 @@ def collect_colab_secrets() -> Dict[str, str]:
|
|||
"OPENAI_API_KEY",
|
||||
"ANTHROPIC_API_KEY",
|
||||
"MINIMAX_API_KEY",
|
||||
"DEEPSEEK_API_KEY",
|
||||
"CLOUDRU_FOUNDATION_MODELS_API_KEY",
|
||||
)
|
||||
out: Dict[str, str] = {}
|
||||
|
|
|
|||
|
|
@ -79,6 +79,7 @@ SETTINGS_DEFAULTS = {**UPDATE_SETTINGS_DEFAULTS,
|
|||
"ANTHROPIC_API_KEY": "",
|
||||
"MINIMAX_API_KEY": "",
|
||||
"MINIMAX_REGION": "",
|
||||
"DEEPSEEK_API_KEY": "",
|
||||
"OUROBOROS_NETWORK_PASSWORD": "",
|
||||
"OUROBOROS_SERVER_HOST": "127.0.0.1",
|
||||
"OUROBOROS_HOST_SERVICE_PORT": 8767,
|
||||
|
|
@ -535,15 +536,14 @@ def _exclusive_direct_remote_provider_env() -> str:
|
|||
bool(str(os.environ.get("GIGACHAT_USER", "") or "").strip())
|
||||
and bool(str(os.environ.get("GIGACHAT_PASSWORD", "") or "").strip())
|
||||
)
|
||||
# OpenRouter / legacy OpenAI base / OpenAI-compatible all route through the
|
||||
# OpenRouter-style stack, so their presence means "not an exclusive direct
|
||||
# provider". Among the registered direct providers, return one only when
|
||||
# exactly one is configured.
|
||||
# OpenRouter / legacy base / compatible route through the OpenRouter-style
|
||||
# stack → never exclusive; among registered direct providers, exactly one.
|
||||
if has_openrouter or has_legacy_base or has_compatible:
|
||||
return ""
|
||||
direct = [name for name, present in (
|
||||
("openai", has_openai), ("anthropic", has_anthropic), ("minimax", has_minimax),
|
||||
("cloudru", has_cloudru), ("gigachat", has_gigachat),
|
||||
("deepseek", bool(str(os.environ.get("DEEPSEEK_API_KEY", "") or "").strip())),
|
||||
) if present]
|
||||
return direct[0] if len(direct) == 1 else ""
|
||||
|
||||
|
|
@ -597,7 +597,7 @@ def resolve_prompt_cache_ttl() -> str:
|
|||
|
||||
def direct_provider_review_models_fallback(provider: str) -> list[str]:
|
||||
"""Return the exact review-models list a direct-provider fallback emits."""
|
||||
if provider not in ("openai", "anthropic", "minimax", "cloudru", "gigachat"):
|
||||
if provider not in ("openai", "anthropic", "minimax", "cloudru", "gigachat", "deepseek"):
|
||||
return []
|
||||
main_model = str(
|
||||
os.environ.get("OUROBOROS_MODEL", SETTINGS_DEFAULTS["OUROBOROS_MODEL"]) or ""
|
||||
|
|
|
|||
|
|
@ -255,8 +255,10 @@ CHAT_ARCHIVE_SCAN_WARN_BYTES = 100_000_000
|
|||
def estimate_message_chars(messages: Any) -> int:
|
||||
"""Message chars with image blocks at the provider-billing proxy.
|
||||
|
||||
The bounded basis shared by the fit estimator, the density witness and
|
||||
the compaction proxy — image base64 never counts as text here.
|
||||
Serves the local-context compaction proxy (`llm.py`); the remote fit
|
||||
estimator and the density witness measure on `context_fit`'s
|
||||
`estimate_context_prompt_tokens` basis instead, which serializes message
|
||||
dicts recursively. Image base64 never counts as text here.
|
||||
"""
|
||||
total = 0
|
||||
for msg in messages:
|
||||
|
|
@ -271,4 +273,10 @@ def estimate_message_chars(messages: Any) -> int:
|
|||
total += len(str(block.get("text", "")))
|
||||
else:
|
||||
total += len(str(content or ""))
|
||||
# Reasoning kept on canonical assistant turns is replayed verbatim on
|
||||
# the reasoning-echo lane (DeepSeek), so it is real wire prompt. A
|
||||
# mixed transcript sent to a non-echo lane still carries the key here
|
||||
# while the wire copy strips it — a conservative over-count, the safe
|
||||
# direction for a compaction trigger.
|
||||
total += len(str(msg.get("reasoning_content") or ""))
|
||||
return total
|
||||
|
|
|
|||
|
|
@ -23,7 +23,7 @@ PLUGIN_API_VERSION = "1.4"
|
|||
FORBIDDEN_SKILL_SETTINGS: frozenset[str] = frozenset({
|
||||
"OPENROUTER_API_KEY", "OPENAI_API_KEY", "OPENAI_COMPATIBLE_API_KEY",
|
||||
"CLOUDRU_FOUNDATION_MODELS_API_KEY", "GIGACHAT_CREDENTIALS", "GIGACHAT_PASSWORD",
|
||||
"ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "GITHUB_TOKEN",
|
||||
"ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "DEEPSEEK_API_KEY", "GITHUB_TOKEN",
|
||||
"OUROBOROS_NETWORK_PASSWORD",
|
||||
})
|
||||
# Backwards-compatible alias for the extension name.
|
||||
|
|
|
|||
|
|
@ -17,6 +17,7 @@ from ouroboros.observability import redact_projection
|
|||
from ouroboros.provider_models import (
|
||||
ALL_PROVIDER_CREDENTIAL_KEYS,
|
||||
ACTIVE_MODEL_SETTING_KEYS,
|
||||
DEEPSEEK_BASE_URL,
|
||||
DIRECT_PROVIDER_DEFAULTS,
|
||||
MINIMAX_REGION_ENDPOINTS,
|
||||
OPENROUTER_DEFAULTS,
|
||||
|
|
@ -254,6 +255,21 @@ def _provider_specs(
|
|||
),
|
||||
))
|
||||
|
||||
deepseek_api_key = str(settings.get("DEEPSEEK_API_KEY", "") or "").strip()
|
||||
if deepseek_api_key:
|
||||
# DeepSeek serves an OpenAI-compatible GET /models on its one official
|
||||
# host, so the catalog is fetched live like the other remote providers.
|
||||
specs.append((
|
||||
"deepseek",
|
||||
lambda client: _fetch_openai_compatible_model_catalog(
|
||||
client,
|
||||
"deepseek",
|
||||
"DeepSeek",
|
||||
deepseek_api_key,
|
||||
DEEPSEEK_BASE_URL,
|
||||
),
|
||||
))
|
||||
|
||||
compatible_api_key = str(settings.get("OPENAI_COMPATIBLE_API_KEY", "") or "").strip()
|
||||
compatible_base_url = str(settings.get("OPENAI_COMPATIBLE_BASE_URL", "") or "").strip()
|
||||
legacy_base_url = str(settings.get("OPENAI_BASE_URL", "") or "").strip()
|
||||
|
|
|
|||
359
ouroboros/llm.py
359
ouroboros/llm.py
|
|
@ -25,7 +25,7 @@ from ouroboros.anthropic_native_custody import (
|
|||
scrub_native_custody,
|
||||
)
|
||||
from ouroboros.openrouter_attribution import OPENROUTER_APP_HEADERS
|
||||
from ouroboros.provider_models import OPENROUTER_DEFAULTS, PROVIDER_PREFIXES, normalize_anthropic_model_id, normalize_model_identity, resolve_minimax_base_url
|
||||
from ouroboros.provider_models import DEEPSEEK_BASE_URL, OPENROUTER_DEFAULTS, PROVIDER_PREFIXES, normalize_anthropic_model_id, normalize_deepseek_reasoning_effort, normalize_model_identity, resolve_minimax_base_url
|
||||
from ouroboros.reasoning_artifacts import sealed_reasoning_pin_fact, transcript_has_sealed_reasoning
|
||||
from ouroboros.request_wire_recovery import (
|
||||
finalize_wire_response,
|
||||
|
|
@ -150,6 +150,8 @@ def _route_normalizes_cache_breakpoints(target: Dict[str, Any]) -> bool:
|
|||
|
||||
# Pin disclosure slot: a ContextVar isolates threads AND concurrent asyncio tasks.
|
||||
_REASONING_PIN_CVAR = contextvars.ContextVar("ouroboros_reasoning_pin_note", default=None)
|
||||
# Effort clamp/projection disclosure slot: same isolation contract as the pin.
|
||||
_EFFORT_CLAMP_CVAR = contextvars.ContextVar("ouroboros_effort_clamp_note", default=None)
|
||||
|
||||
|
||||
def _pop_reasoning_pin_note() -> Optional[Dict[str, Any]]:
|
||||
|
|
@ -492,188 +494,13 @@ def add_usage(total: Dict[str, Any], usage: Dict[str, Any]) -> None:
|
|||
merge_request_wire_usage(total, usage)
|
||||
|
||||
|
||||
def fetch_openrouter_pricing(*, timeout_sec: float = 5.0) -> Dict[str, Tuple[Optional[float], ...]]:
|
||||
"""Fetch OpenRouter pricing as model_id -> per-1M prices.
|
||||
|
||||
Tuples are ``(input, cached_read, cache_write, output)``. Missing cache
|
||||
prices remain ``None`` instead of inheriting a synthetic coefficient.
|
||||
"""
|
||||
import logging
|
||||
from ouroboros.pricing import PricingSchedule
|
||||
log = logging.getLogger("ouroboros.llm")
|
||||
|
||||
try:
|
||||
import requests
|
||||
except ImportError:
|
||||
log.warning("requests not installed, cannot fetch pricing")
|
||||
return {}
|
||||
|
||||
try:
|
||||
url = "https://openrouter.ai/api/v1/models"
|
||||
resp = requests.get(url, timeout=max(0.1, min(5.0, float(timeout_sec))))
|
||||
resp.raise_for_status()
|
||||
|
||||
data = resp.json()
|
||||
models = data.get("data", [])
|
||||
|
||||
pricing_dict = {}
|
||||
for model in models:
|
||||
model_id = str(model.get("id") or "").strip()
|
||||
|
||||
pricing = model.get("pricing", {})
|
||||
if not pricing or pricing.get("prompt") is None or pricing.get("completion") is None:
|
||||
continue
|
||||
|
||||
raw_prompt = float(pricing.get("prompt", 0))
|
||||
raw_completion = float(pricing.get("completion", 0))
|
||||
raw_cached_str = pricing.get("input_cache_read")
|
||||
raw_cached = float(raw_cached_str) if raw_cached_str is not None else None
|
||||
raw_cache_write_str = pricing.get("input_cache_write")
|
||||
raw_cache_write = float(raw_cache_write_str) if raw_cache_write_str is not None else None
|
||||
if raw_prompt < 0 or raw_completion < 0:
|
||||
continue
|
||||
if raw_cached is not None and raw_cached < 0:
|
||||
raw_cached = None
|
||||
if raw_cache_write is not None and raw_cache_write < 0:
|
||||
raw_cache_write = None
|
||||
|
||||
prompt_price = round(raw_prompt * 1_000_000, 4)
|
||||
completion_price = round(raw_completion * 1_000_000, 4)
|
||||
cached_price = round(raw_cached * 1_000_000, 4) if raw_cached is not None else None
|
||||
cache_write_price = (
|
||||
round(raw_cache_write * 1_000_000, 4)
|
||||
if raw_cache_write is not None else None
|
||||
)
|
||||
|
||||
if prompt_price > 1000 or completion_price > 1000:
|
||||
log.warning(f"Skipping {model_id}: prices seem wrong (prompt={prompt_price}, completion={completion_price})")
|
||||
continue
|
||||
|
||||
row = (prompt_price, cached_price, cache_write_price, completion_price)
|
||||
|
||||
tiers = []
|
||||
raw_overrides = pricing.get("overrides") or []
|
||||
if isinstance(raw_overrides, list):
|
||||
for override in raw_overrides:
|
||||
if not isinstance(override, dict):
|
||||
continue
|
||||
try:
|
||||
min_prompt_tokens = int(override.get("min_prompt_tokens") or 0)
|
||||
if min_prompt_tokens <= 0:
|
||||
continue
|
||||
tier_raw_prompt = float(override.get("prompt", raw_prompt))
|
||||
tier_raw_completion = float(override.get("completion", raw_completion))
|
||||
tier_prompt = round(tier_raw_prompt * 1_000_000, 4)
|
||||
tier_completion = round(tier_raw_completion * 1_000_000, 4)
|
||||
override_cached = override.get("input_cache_read")
|
||||
tier_cached = (
|
||||
round(float(override_cached) * 1_000_000, 4)
|
||||
if override_cached is not None else None
|
||||
)
|
||||
override_write = override.get("input_cache_write")
|
||||
if override_write is not None:
|
||||
tier_write = round(float(override_write) * 1_000_000, 4)
|
||||
else:
|
||||
tier_write = None
|
||||
if tier_prompt > 1000 or tier_completion > 1000:
|
||||
continue
|
||||
tier_row = (tier_prompt, tier_cached, tier_write, tier_completion)
|
||||
tiers.append((min_prompt_tokens, tier_row))
|
||||
except (TypeError, ValueError):
|
||||
log.warning("Skipping malformed pricing override for %s", model_id)
|
||||
if tiers:
|
||||
row = PricingSchedule(row, tuple(tiers))
|
||||
pricing_dict[model_id] = row
|
||||
normalized_model_id = normalize_model_identity(model_id)
|
||||
if normalized_model_id != model_id:
|
||||
pricing_dict[normalized_model_id] = row
|
||||
|
||||
log.info(f"Fetched pricing for {len(pricing_dict)} models from OpenRouter")
|
||||
return pricing_dict
|
||||
|
||||
except (requests.RequestException, ValueError, KeyError) as e:
|
||||
log.warning(f"Failed to fetch OpenRouter pricing: {e}")
|
||||
return {}
|
||||
|
||||
|
||||
def fetch_cloudru_pricing(*, timeout_sec: float = 5.0) -> Dict[str, Tuple[Optional[float], ...]]:
|
||||
"""Fetch cloud.ru Foundation Models pricing as ``cloudru/<id>`` -> per-1M USD.
|
||||
|
||||
cloud.ru's ``GET /v1/models`` returns per-model ``metadata`` with token costs
|
||||
(``prompt_tokens_cost``, ``generated_tokens_cost``, ``cache_read_tokens_cost``,
|
||||
``cache_write_tokens_cost``) in RUB per 1M tokens — i.e. the real resale price
|
||||
the owner pays. We convert to USD via ``OUROBOROS_RUB_USD_RATE`` so the catalog
|
||||
is the SSOT for ALL cloud.ru models (no hardcoded per-model table). Models with
|
||||
``is_billable=false`` is an exact free row; missing billability or an absent
|
||||
explicit ``OUROBOROS_RUB_USD_RATE`` stays unknown. Returns {} when the catalog
|
||||
cannot be queried. Tuples are ``(input, cached_read, cache_write, output)``."""
|
||||
import logging
|
||||
log = logging.getLogger("ouroboros.llm")
|
||||
|
||||
api_key = (os.environ.get("CLOUDRU_FOUNDATION_MODELS_API_KEY", "") or "").strip()
|
||||
if not api_key:
|
||||
return {}
|
||||
try:
|
||||
import requests
|
||||
except ImportError:
|
||||
return {}
|
||||
|
||||
base_url = (
|
||||
os.environ.get("CLOUDRU_FOUNDATION_MODELS_BASE_URL", "") or ""
|
||||
).strip() or "https://foundation-models.api.cloud.ru/v1"
|
||||
try:
|
||||
rate = float(os.environ.get("OUROBOROS_RUB_USD_RATE", ""))
|
||||
except (TypeError, ValueError):
|
||||
return {}
|
||||
if rate <= 0:
|
||||
return {}
|
||||
|
||||
try:
|
||||
resp = requests.get(
|
||||
f"{base_url.rstrip('/')}/models",
|
||||
headers={"Authorization": f"Bearer {api_key}"},
|
||||
timeout=max(0.1, min(5.0, float(timeout_sec))),
|
||||
)
|
||||
resp.raise_for_status()
|
||||
models = resp.json().get("data", []) or []
|
||||
|
||||
def _rub_per_1m_to_usd(value: Any) -> Optional[float]:
|
||||
try:
|
||||
num = float(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
if num < 0: # cloud.ru uses -1 for "n/a" (e.g. embedding output)
|
||||
return None
|
||||
return round(num / rate, 6)
|
||||
|
||||
pricing_dict: Dict[str, Tuple[Optional[float], ...]] = {}
|
||||
for model in models:
|
||||
model_id = str(model.get("id") or "").strip()
|
||||
meta = model.get("metadata") if isinstance(model.get("metadata"), dict) else {}
|
||||
if not model_id or not meta or meta.get("is_billable") is None:
|
||||
continue
|
||||
if meta.get("is_billable") is False:
|
||||
pricing_dict[normalize_model_identity(f"cloudru::{model_id}")] = (0.0, 0.0, 0.0, 0.0)
|
||||
continue
|
||||
prompt_price = _rub_per_1m_to_usd(meta.get("prompt_tokens_cost"))
|
||||
output_price = _rub_per_1m_to_usd(meta.get("generated_tokens_cost"))
|
||||
if prompt_price is None or output_price is None:
|
||||
continue
|
||||
cached_price = _rub_per_1m_to_usd(meta.get("cache_read_tokens_cost"))
|
||||
cache_write_price = _rub_per_1m_to_usd(meta.get("cache_write_tokens_cost"))
|
||||
row = (
|
||||
prompt_price,
|
||||
cached_price,
|
||||
cache_write_price,
|
||||
output_price,
|
||||
)
|
||||
pricing_dict[normalize_model_identity(f"cloudru::{model_id}")] = row
|
||||
|
||||
log.info(f"Fetched pricing for {len(pricing_dict)} models from cloud.ru")
|
||||
return pricing_dict
|
||||
except (requests.RequestException, ValueError, KeyError) as e:
|
||||
log.warning(f"Failed to fetch cloud.ru pricing: {e}")
|
||||
return {}
|
||||
# Live pricing-catalog fetchers moved whole to ouroboros/provider_catalogs.py
|
||||
# at the 200,000-byte module ratchet ceiling; re-exported here so existing
|
||||
# importers (ouroboros.pricing, tests) keep the historical names.
|
||||
from ouroboros.provider_catalogs import ( # noqa: F401,E402
|
||||
fetch_cloudru_pricing,
|
||||
fetch_openrouter_pricing,
|
||||
)
|
||||
|
||||
|
||||
class LLMClient:
|
||||
|
|
@ -995,13 +822,11 @@ class LLMClient:
|
|||
|
||||
def _clamp_effort_for_model(self, model_id: str, effort: str) -> str:
|
||||
"""Legacy clamp plus diagnostic disclosure; production dispatch bypasses it."""
|
||||
if not hasattr(self, "_effort_clamp_tls"):
|
||||
self._effort_clamp_tls = threading.local()
|
||||
self._effort_clamp_tls.pending = None
|
||||
_EFFORT_CLAMP_CVAR.set(None)
|
||||
from ouroboros.config import effort_rank
|
||||
applied = self.clamp_effort_for_route(model_id, effort)
|
||||
if applied != effort:
|
||||
self._effort_clamp_tls.pending = {
|
||||
_EFFORT_CLAMP_CVAR.set({
|
||||
"requested": effort,
|
||||
"applied": applied,
|
||||
"reason": (
|
||||
|
|
@ -1010,12 +835,13 @@ class LLMClient:
|
|||
else "learned_ceiling"
|
||||
),
|
||||
"model": str(model_id or ""),
|
||||
}
|
||||
})
|
||||
return applied
|
||||
|
||||
def _pop_thread_disclosure(self, slot: str) -> Optional[Dict[str, Any]]:
|
||||
"""Take and clear the disclosure staged in thread-local ``slot`` for THIS
|
||||
thread's call; these slots stage before or at send (pin note: ContextVar)."""
|
||||
thread's call; these slots stage before or at send (pin and effort
|
||||
notes: ContextVar)."""
|
||||
tls = getattr(self, slot, None)
|
||||
pending = getattr(tls, "pending", None) if tls is not None else None
|
||||
if tls is not None:
|
||||
|
|
@ -1023,8 +849,10 @@ class LLMClient:
|
|||
return pending if isinstance(pending, dict) else None
|
||||
|
||||
def _pop_effort_clamp_disclosure(self) -> Optional[Dict[str, Any]]:
|
||||
"""The pending clamp record for THIS thread's in-flight call, if any."""
|
||||
return self._pop_thread_disclosure("_effort_clamp_tls")
|
||||
"""The pending clamp record for THIS call's context (thread or asyncio task)."""
|
||||
pending = _EFFORT_CLAMP_CVAR.get()
|
||||
_EFFORT_CLAMP_CVAR.set(None)
|
||||
return pending if isinstance(pending, dict) else None
|
||||
|
||||
@classmethod
|
||||
def _record_effort_ceiling(cls, model_id: str, current_effort: str) -> None:
|
||||
|
|
@ -1304,6 +1132,8 @@ class LLMClient:
|
|||
return f"gigachat/{resolved_model}"
|
||||
if provider == "minimax":
|
||||
return f"minimax/{resolved_model}"
|
||||
if provider == "deepseek":
|
||||
return f"deepseek/{resolved_model}"
|
||||
return f"openai-compatible/{resolved_model}"
|
||||
|
||||
def _resolve_remote_target(
|
||||
|
|
@ -1359,6 +1189,26 @@ class LLMClient:
|
|||
"supports_generation_cost": False,
|
||||
}
|
||||
|
||||
if provider == "deepseek":
|
||||
return {
|
||||
"provider": provider,
|
||||
"resolved_model": resolved_model,
|
||||
"usage_model": usage_model,
|
||||
"api_key": configured("DEEPSEEK_API_KEY", ""),
|
||||
# One official endpoint; no owner-configurable base URL
|
||||
# (proxy/mirror setups belong to the openai-compatible slot).
|
||||
"base_url": DEEPSEEK_BASE_URL,
|
||||
"default_headers": {},
|
||||
# v4 thinks by default and carries reasoning_effort; the
|
||||
# canonical scale is projected onto its low/high/max enum in
|
||||
# _build_remote_kwargs. Tool-bearing requests MUST replay every
|
||||
# previous assistant turn's reasoning_content (v4-pro enforces
|
||||
# with a 400; "" is accepted for foreign turns — probed 2026-09-01).
|
||||
"requires_reasoning_echo": True,
|
||||
"supports_openrouter_extensions": False,
|
||||
"supports_generation_cost": False,
|
||||
}
|
||||
|
||||
if provider == "cloudru":
|
||||
return {
|
||||
"provider": provider,
|
||||
|
|
@ -1569,6 +1419,7 @@ class LLMClient:
|
|||
*,
|
||||
allow_message_cache_control: bool,
|
||||
flatten_tool_content_blocks: bool,
|
||||
flatten_non_user_content_blocks: bool = False,
|
||||
allow_cache_ttl: bool = False,
|
||||
) -> List[Dict[str, Any]]:
|
||||
cleaned = scrub_native_custody(messages)
|
||||
|
|
@ -1576,7 +1427,12 @@ class LLMClient:
|
|||
content = msg.get("content")
|
||||
if not isinstance(content, list):
|
||||
continue
|
||||
if msg.get("role") == "tool" and flatten_tool_content_blocks:
|
||||
role = msg.get("role")
|
||||
if (role == "tool" and flatten_tool_content_blocks) or (
|
||||
role != "user" and flatten_non_user_content_blocks
|
||||
):
|
||||
# String-only roles: text blocks fold into one string, so host
|
||||
# metadata and cache markers never reach the wire either.
|
||||
msg["content"] = "".join(
|
||||
block.get("text", "") if isinstance(block, dict) else str(block)
|
||||
for block in content
|
||||
|
|
@ -1610,7 +1466,12 @@ class LLMClient:
|
|||
_REASONING_CONTENT_BLOCK_TYPES = frozenset({"thinking", "reasoning", "redacted_thinking"})
|
||||
|
||||
@classmethod
|
||||
def _strip_openrouter_roundtrip_metadata(cls, messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
||||
def _strip_openrouter_roundtrip_metadata(
|
||||
cls,
|
||||
messages: List[Dict[str, Any]],
|
||||
*,
|
||||
keep_reasoning_content: bool = False,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Strip provider-private reasoning round-trip artifacts that a DIFFERENT
|
||||
upstream family rejects: assistant-level ``reasoning``/``reasoning_details``/
|
||||
``reasoning_content``/``response_id`` keys AND ``thinking``/``reasoning``
|
||||
|
|
@ -1622,14 +1483,19 @@ class LLMClient:
|
|||
OpenRouter/Anthropic ``reasoning``/``reasoning_details`` shapes. Strict
|
||||
OpenAI-compatible servers (vLLM/SGLang) reject an echoed ``reasoning_content``
|
||||
with HTTP 400 ``Extra inputs are not permitted``, so it must be scrubbed on
|
||||
the cloudru / openai-compatible / local lanes too."""
|
||||
the cloudru / openai-compatible / local lanes too. DeepSeek is the third
|
||||
class — a server that REQUIRES its own echo (tool-bearing requests 400
|
||||
without the previous turns' ``reasoning_content``) — so its lane passes
|
||||
``keep_reasoning_content=True`` to retain that one field while every
|
||||
other round-trip artifact is still stripped."""
|
||||
cleaned = scrub_native_custody(messages)
|
||||
for msg in cleaned:
|
||||
if not isinstance(msg, dict) or msg.get("role") != "assistant":
|
||||
continue
|
||||
msg.pop("reasoning", None)
|
||||
msg.pop("reasoning_details", None)
|
||||
msg.pop("reasoning_content", None)
|
||||
if not keep_reasoning_content:
|
||||
msg.pop("reasoning_content", None)
|
||||
msg.pop("response_id", None)
|
||||
content = msg.get("content")
|
||||
if isinstance(content, list):
|
||||
|
|
@ -3523,7 +3389,20 @@ class LLMClient:
|
|||
# normalized vision prefix, and blinded every direct-provider install —
|
||||
# same identity contract as the browser-screenshot call site (E1).
|
||||
from ouroboros.provider_models import supports_vision
|
||||
if not supports_vision(str(target.get("usage_model") or resolved_model)):
|
||||
# Judge vision on EITHER identity: direct lanes strip the
|
||||
# ``provider::`` prefix from ``resolved_model``, so the bare id never
|
||||
# matched the slash-form vision prefixes and provider-namespaced direct
|
||||
# routes (openai::/deepseek::/...) were treated blind regardless of
|
||||
# real capability — ``usage_model`` carries their qualified spelling.
|
||||
# The BARE id stays in the judgment too, because the openai-compatible
|
||||
# lane's qualifier (``openai-compatible/<id>``) can never match while
|
||||
# a vendor-form bare id (``qwen/qwen2.5-vl-…``) legitimately does —
|
||||
# judging only the qualified name would flip that lane blind. On
|
||||
# OpenRouter both spellings are the same string.
|
||||
if not (
|
||||
supports_vision(str(target.get("usage_model") or resolved_model))
|
||||
or supports_vision(resolved_model)
|
||||
):
|
||||
messages = self._replace_image_blocks_with_placeholder(messages)
|
||||
# Official direct OpenAI Chat uses the current completion-token carrier:
|
||||
# provider-wide; model names are not capability authority across routes.
|
||||
|
|
@ -3539,8 +3418,22 @@ class LLMClient:
|
|||
messages,
|
||||
allow_message_cache_control=False,
|
||||
flatten_tool_content_blocks=True,
|
||||
)
|
||||
# DeepSeek accepts content arrays only on user turns.
|
||||
flatten_non_user_content_blocks=provider == "deepseek",
|
||||
),
|
||||
keep_reasoning_content=bool(target.get("requires_reasoning_echo")),
|
||||
)
|
||||
if target.get("requires_reasoning_echo"):
|
||||
# A reasoning-echo route (DeepSeek) REQUIRES every assistant
|
||||
# turn's ``reasoning_content`` on tool-bearing requests (v4-pro
|
||||
# 400s otherwise; probed 2026-09-01). Foreign or non-string
|
||||
# values become the explicit empty string the gate accepts —
|
||||
# the honest value for reasoning that does not exist. Harmless
|
||||
# without tools (the API ignores the field).
|
||||
for _msg in clean_messages:
|
||||
if isinstance(_msg, dict) and _msg.get("role") == "assistant":
|
||||
if not isinstance(_msg.get("reasoning_content"), str):
|
||||
_msg["reasoning_content"] = ""
|
||||
kwargs: Dict[str, Any] = {
|
||||
"model": resolved_model,
|
||||
"messages": clean_messages,
|
||||
|
|
@ -3557,11 +3450,33 @@ class LLMClient:
|
|||
kwargs["prompt_cache_key"] = cache_identity
|
||||
requested_effort = normalize_reasoning_effort(reasoning_effort)
|
||||
if direct_openai:
|
||||
# Direct-OpenAI route honors the configured OUROBOROS_EFFORT_*
|
||||
# lanes instead of silently dropping them (OpenRouter parity).
|
||||
# Exact-route request-wire evidence, not legacy model-global
|
||||
# rows, owns any provider-required adaptation after this build.
|
||||
# Effort-carrying routes honor the OUROBOROS_EFFORT_* lanes
|
||||
# instead of dropping them like generic compatible lanes.
|
||||
# Keyed on the PROVIDER id, not a target capability field, so
|
||||
# a hand-built target (fixtures, probes) cannot silently drop
|
||||
# the carriage; request-wire recovery adapts on a provider 400.
|
||||
kwargs["reasoning_effort"] = requested_effort
|
||||
elif provider == "deepseek":
|
||||
# Same carriage, projected onto DeepSeek's wire dialect
|
||||
# (low/high/max; thinking is switched off by a toggle, not an
|
||||
# effort value). Thinking mode accepts only tool_choice
|
||||
# auto/none (probed 2026-09-03: required and named 400 on both
|
||||
# v4 models), so a forced tool call is served with thinking
|
||||
# disabled. Any tier change is disclosed on usage as
|
||||
# ``reasoning_effort_clamped``.
|
||||
forced_tool = bool(prepared_tools) and tool_choice not in (None, "", "auto", "none")
|
||||
applied = "none" if forced_tool else normalize_deepseek_reasoning_effort(requested_effort)
|
||||
if applied == "none":
|
||||
kwargs.setdefault("extra_body", {})["thinking"] = {"type": "disabled"}
|
||||
else:
|
||||
kwargs["reasoning_effort"] = applied
|
||||
_EFFORT_CLAMP_CVAR.set(None) # never inherit a stale note
|
||||
if applied != requested_effort:
|
||||
_EFFORT_CLAMP_CVAR.set({
|
||||
"requested": requested_effort, "applied": applied,
|
||||
"reason": "provider_forced_tool_choice" if forced_tool else "provider_wire_mapping",
|
||||
"model": resolved_model,
|
||||
})
|
||||
if temperature is not None:
|
||||
kwargs["temperature"] = temperature
|
||||
if response_format:
|
||||
|
|
@ -3578,6 +3493,24 @@ class LLMClient:
|
|||
_eb["cache"] = {"no-cache": True}
|
||||
return kwargs
|
||||
|
||||
if any(isinstance(m, dict) and "reasoning_content" in m for m in messages):
|
||||
# ``reasoning_content`` in canonical history is direct-DeepSeek
|
||||
# custody (the inbound normalizer pops it from every other lane's
|
||||
# responses, OpenRouter included). OR upstreams of other families
|
||||
# reject the echoed field, and leaving it here would also trip the
|
||||
# replay-artifact pin below (allow_fallbacks=False), silently
|
||||
# killing same-model failover for a mixed transcript. Dropping it
|
||||
# from the OR physical copy restores the exact pre-DeepSeek OR
|
||||
# wire; the canonical transcript is untouched. Keyed on key
|
||||
# PRESENCE, not truthiness: an empty-string echo (a legal kept
|
||||
# value) must not ride the OR wire either.
|
||||
messages = [
|
||||
(
|
||||
{k: v for k, v in m.items() if k != "reasoning_content"}
|
||||
if isinstance(m, dict) else m
|
||||
)
|
||||
for m in messages
|
||||
]
|
||||
effort = normalize_reasoning_effort(reasoning_effort)
|
||||
raw_return_reasoning = os.environ.get("OUROBOROS_RETURN_REASONING")
|
||||
return_reasoning = (
|
||||
|
|
@ -3705,6 +3638,7 @@ class LLMClient:
|
|||
usage.pop("response_finish_reason", None)
|
||||
usage.pop("response_provider", None)
|
||||
usage.pop("reasoning_pin", None)
|
||||
usage.pop("reasoning_effort_clamped", None)
|
||||
usage.pop("provider_error", None)
|
||||
# An HTTP-200 that carried a provider body-error (OpenRouter passes
|
||||
# 429/5xx through the body) reaches here only when a same-model reroute
|
||||
|
|
@ -3773,12 +3707,33 @@ class LLMClient:
|
|||
# their OWN echoed ``reasoning_content`` with a 400 ``Extra inputs are not
|
||||
# permitted`` on the very next same-model turn. Drop it here so it never enters
|
||||
# the canonical transcript; the outbound scrubber is the second layer.
|
||||
msg.pop("reasoning_content", None)
|
||||
# DeepSeek is the inverse class: its documented tool contract REQUIRES the
|
||||
# previous turns' ``reasoning_content`` back on every tools-bearing request
|
||||
# (v4-pro enforces with a 400), so that lane KEEPS the field on the canonical
|
||||
# assistant message — the same-family-continuity treatment ``reasoning_details``
|
||||
# already gets. Cross-family sends strip it (sanitize_reasoning_on_model_switch
|
||||
# + the outbound scrubber), and the deepseek outbound build replays it.
|
||||
if str(target.get("provider") or "") != "deepseek":
|
||||
msg.pop("reasoning_content", None)
|
||||
elif not isinstance(msg.get("reasoning_content", ""), str):
|
||||
# The SDK surfaces server extras verbatim (same hazard the
|
||||
# refusal/annotations pops above guard): a null here would live on
|
||||
# the canonical assistant turn forever and the direct lane has no
|
||||
# message-level 400 recovery. Only strings enter the transcript.
|
||||
msg.pop("reasoning_content", None)
|
||||
|
||||
if not usage.get("cached_tokens"):
|
||||
prompt_details = usage.get("prompt_tokens_details") or {}
|
||||
if isinstance(prompt_details, dict) and prompt_details.get("cached_tokens"):
|
||||
usage["cached_tokens"] = int(prompt_details["cached_tokens"])
|
||||
if not usage.get("cached_tokens") and usage.get("prompt_cache_hit_tokens"):
|
||||
# DeepSeek mirrors its automatic-cache split as top-level
|
||||
# prompt_cache_hit/miss_tokens beside the details block; the
|
||||
# details block wins when present, this is the fallback.
|
||||
try:
|
||||
usage["cached_tokens"] = int(usage["prompt_cache_hit_tokens"])
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
# LM Studio MLX exposes prefix-cache hits only in stderr/logs, not
|
||||
# OpenAI-compatible usage; cached_tokens=0 is therefore expected.
|
||||
|
||||
|
|
|
|||
|
|
@ -315,6 +315,7 @@ def probe_provider_readiness(
|
|||
|
||||
if provider in {
|
||||
"openrouter", "openai", "openai-compatible", "minimax", "cloudru",
|
||||
"deepseek",
|
||||
}:
|
||||
remote_client = client._new_remote_client(target)
|
||||
|
||||
|
|
|
|||
|
|
@ -185,7 +185,7 @@ def estimate_cost_optional(model: str, prompt_tokens: int, completion_tokens: in
|
|||
def infer_api_key_type(model: str, provider: Optional[str] = None) -> str:
|
||||
"""Infer which API key is used based on model name."""
|
||||
provider_name = str(provider or "").strip().lower()
|
||||
if provider_name in {"local", "openrouter", "openai", "anthropic", "openai-compatible", "cloudru", "gigachat", "minimax"}:
|
||||
if provider_name in {"local", "openrouter", "openai", "anthropic", "openai-compatible", "cloudru", "gigachat", "minimax", "deepseek"}:
|
||||
return provider_name
|
||||
raw_model = str(model or "").strip()
|
||||
direct_provider = provider_for_model(raw_model)
|
||||
|
|
@ -207,7 +207,7 @@ def infer_api_key_type(model: str, provider: Optional[str] = None) -> str:
|
|||
# slash-form ids stay router-style by design (direct routing uses minimax::,
|
||||
# already resolved by provider_for_model above). Classifying minimax/ as the
|
||||
# direct key would make safety.py demand MINIMAX_API_KEY on OpenRouter installs.
|
||||
if normalized.startswith(("anthropic/", "google/", "openai/", "x-ai/", "qwen/", "minimax/")):
|
||||
if normalized.startswith(("anthropic/", "google/", "openai/", "x-ai/", "qwen/", "minimax/", "deepseek/")):
|
||||
return "openrouter"
|
||||
if "claude" in normalized.lower():
|
||||
return "anthropic"
|
||||
|
|
@ -217,7 +217,10 @@ def infer_api_key_type(model: str, provider: Optional[str] = None) -> str:
|
|||
def infer_provider_from_model(model: str) -> str:
|
||||
"""Derive the billing provider string from a model identifier.
|
||||
|
||||
Rules (same prefix logic as infer_api_key_type, returns canonical provider name):
|
||||
Rules (same prefix logic as infer_api_key_type, returns canonical provider name;
|
||||
the registry drives it, so every direct prefix — anthropic::, openai::,
|
||||
openai-compatible::, cloudru::, gigachat::, minimax::, deepseek:: — maps to
|
||||
its provider):
|
||||
anthropic::* → "anthropic"
|
||||
openai::* → "openai"
|
||||
openai-compatible::* → "openai-compatible"
|
||||
|
|
|
|||
199
ouroboros/provider_catalogs.py
Normal file
199
ouroboros/provider_catalogs.py
Normal file
|
|
@ -0,0 +1,199 @@
|
|||
"""Live provider pricing-catalog fetchers (OpenRouter + cloud.ru).
|
||||
|
||||
Extracted whole from ``llm.py`` at the 200,000-byte module ratchet ceiling;
|
||||
``llm.py`` re-exports both historical names so existing importers
|
||||
(``ouroboros.pricing``, tests) keep one surface. The one deliberate change:
|
||||
the module logger is now ``ouroboros.provider_catalogs`` (was ``ouroboros.llm``).
|
||||
These are the only two routes with a machine-readable tariff catalog
|
||||
(DEVELOPMENT.md "Pricing and admission": no hand-maintained tables — every
|
||||
other provider stays nullable/unknown).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
|
||||
from ouroboros.provider_models import normalize_model_identity
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def fetch_openrouter_pricing(*, timeout_sec: float = 5.0) -> Dict[str, Tuple[Optional[float], ...]]:
|
||||
"""Fetch OpenRouter pricing as model_id -> per-1M prices.
|
||||
|
||||
Tuples are ``(input, cached_read, cache_write, output)``. Missing cache
|
||||
prices remain ``None`` instead of inheriting a synthetic coefficient.
|
||||
"""
|
||||
from ouroboros.pricing import PricingSchedule
|
||||
|
||||
try:
|
||||
import requests
|
||||
except ImportError:
|
||||
log.warning("requests not installed, cannot fetch pricing")
|
||||
return {}
|
||||
|
||||
try:
|
||||
url = "https://openrouter.ai/api/v1/models"
|
||||
resp = requests.get(url, timeout=max(0.1, min(5.0, float(timeout_sec))))
|
||||
resp.raise_for_status()
|
||||
|
||||
data = resp.json()
|
||||
models = data.get("data", [])
|
||||
|
||||
pricing_dict = {}
|
||||
for model in models:
|
||||
model_id = str(model.get("id") or "").strip()
|
||||
|
||||
pricing = model.get("pricing", {})
|
||||
if not pricing or pricing.get("prompt") is None or pricing.get("completion") is None:
|
||||
continue
|
||||
|
||||
raw_prompt = float(pricing.get("prompt", 0))
|
||||
raw_completion = float(pricing.get("completion", 0))
|
||||
raw_cached_str = pricing.get("input_cache_read")
|
||||
raw_cached = float(raw_cached_str) if raw_cached_str is not None else None
|
||||
raw_cache_write_str = pricing.get("input_cache_write")
|
||||
raw_cache_write = float(raw_cache_write_str) if raw_cache_write_str is not None else None
|
||||
if raw_prompt < 0 or raw_completion < 0:
|
||||
continue
|
||||
if raw_cached is not None and raw_cached < 0:
|
||||
raw_cached = None
|
||||
if raw_cache_write is not None and raw_cache_write < 0:
|
||||
raw_cache_write = None
|
||||
|
||||
prompt_price = round(raw_prompt * 1_000_000, 4)
|
||||
completion_price = round(raw_completion * 1_000_000, 4)
|
||||
cached_price = round(raw_cached * 1_000_000, 4) if raw_cached is not None else None
|
||||
cache_write_price = (
|
||||
round(raw_cache_write * 1_000_000, 4)
|
||||
if raw_cache_write is not None else None
|
||||
)
|
||||
|
||||
if prompt_price > 1000 or completion_price > 1000:
|
||||
log.warning(f"Skipping {model_id}: prices seem wrong (prompt={prompt_price}, completion={completion_price})")
|
||||
continue
|
||||
|
||||
row = (prompt_price, cached_price, cache_write_price, completion_price)
|
||||
|
||||
tiers = []
|
||||
raw_overrides = pricing.get("overrides") or []
|
||||
if isinstance(raw_overrides, list):
|
||||
for override in raw_overrides:
|
||||
if not isinstance(override, dict):
|
||||
continue
|
||||
try:
|
||||
min_prompt_tokens = int(override.get("min_prompt_tokens") or 0)
|
||||
if min_prompt_tokens <= 0:
|
||||
continue
|
||||
tier_raw_prompt = float(override.get("prompt", raw_prompt))
|
||||
tier_raw_completion = float(override.get("completion", raw_completion))
|
||||
tier_prompt = round(tier_raw_prompt * 1_000_000, 4)
|
||||
tier_completion = round(tier_raw_completion * 1_000_000, 4)
|
||||
override_cached = override.get("input_cache_read")
|
||||
tier_cached = (
|
||||
round(float(override_cached) * 1_000_000, 4)
|
||||
if override_cached is not None else None
|
||||
)
|
||||
override_write = override.get("input_cache_write")
|
||||
if override_write is not None:
|
||||
tier_write = round(float(override_write) * 1_000_000, 4)
|
||||
else:
|
||||
tier_write = None
|
||||
if tier_prompt > 1000 or tier_completion > 1000:
|
||||
continue
|
||||
tier_row = (tier_prompt, tier_cached, tier_write, tier_completion)
|
||||
tiers.append((min_prompt_tokens, tier_row))
|
||||
except (TypeError, ValueError):
|
||||
log.warning("Skipping malformed pricing override for %s", model_id)
|
||||
if tiers:
|
||||
row = PricingSchedule(row, tuple(tiers))
|
||||
pricing_dict[model_id] = row
|
||||
normalized_model_id = normalize_model_identity(model_id)
|
||||
if normalized_model_id != model_id:
|
||||
pricing_dict[normalized_model_id] = row
|
||||
|
||||
log.info(f"Fetched pricing for {len(pricing_dict)} models from OpenRouter")
|
||||
return pricing_dict
|
||||
|
||||
except (requests.RequestException, ValueError, KeyError) as e:
|
||||
log.warning(f"Failed to fetch OpenRouter pricing: {e}")
|
||||
return {}
|
||||
|
||||
|
||||
def fetch_cloudru_pricing(*, timeout_sec: float = 5.0) -> Dict[str, Tuple[Optional[float], ...]]:
|
||||
"""Fetch cloud.ru Foundation Models pricing as ``cloudru/<id>`` -> per-1M USD.
|
||||
|
||||
cloud.ru's ``GET /v1/models`` returns per-model ``metadata`` with token costs
|
||||
(``prompt_tokens_cost``, ``generated_tokens_cost``, ``cache_read_tokens_cost``,
|
||||
``cache_write_tokens_cost``) in RUB per 1M tokens — i.e. the real resale price
|
||||
the owner pays. We convert to USD via ``OUROBOROS_RUB_USD_RATE`` so the catalog
|
||||
is the SSOT for ALL cloud.ru models (no hardcoded per-model table). Models with
|
||||
``is_billable=false`` is an exact free row; missing billability or an absent
|
||||
explicit ``OUROBOROS_RUB_USD_RATE`` stays unknown. Returns {} when the catalog
|
||||
cannot be queried. Tuples are ``(input, cached_read, cache_write, output)``."""
|
||||
api_key = (os.environ.get("CLOUDRU_FOUNDATION_MODELS_API_KEY", "") or "").strip()
|
||||
if not api_key:
|
||||
return {}
|
||||
try:
|
||||
import requests
|
||||
except ImportError:
|
||||
return {}
|
||||
|
||||
base_url = (
|
||||
os.environ.get("CLOUDRU_FOUNDATION_MODELS_BASE_URL", "") or ""
|
||||
).strip() or "https://foundation-models.api.cloud.ru/v1"
|
||||
try:
|
||||
rate = float(os.environ.get("OUROBOROS_RUB_USD_RATE", ""))
|
||||
except (TypeError, ValueError):
|
||||
return {}
|
||||
if rate <= 0:
|
||||
return {}
|
||||
|
||||
try:
|
||||
resp = requests.get(
|
||||
f"{base_url.rstrip('/')}/models",
|
||||
headers={"Authorization": f"Bearer {api_key}"},
|
||||
timeout=max(0.1, min(5.0, float(timeout_sec))),
|
||||
)
|
||||
resp.raise_for_status()
|
||||
models = resp.json().get("data", []) or []
|
||||
|
||||
def _rub_per_1m_to_usd(value: Any) -> Optional[float]:
|
||||
try:
|
||||
num = float(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
if num < 0: # cloud.ru uses -1 for "n/a" (e.g. embedding output)
|
||||
return None
|
||||
return round(num / rate, 6)
|
||||
|
||||
pricing_dict: Dict[str, Tuple[Optional[float], ...]] = {}
|
||||
for model in models:
|
||||
model_id = str(model.get("id") or "").strip()
|
||||
meta = model.get("metadata") if isinstance(model.get("metadata"), dict) else {}
|
||||
if not model_id or not meta or meta.get("is_billable") is None:
|
||||
continue
|
||||
if meta.get("is_billable") is False:
|
||||
pricing_dict[normalize_model_identity(f"cloudru::{model_id}")] = (0.0, 0.0, 0.0, 0.0)
|
||||
continue
|
||||
prompt_price = _rub_per_1m_to_usd(meta.get("prompt_tokens_cost"))
|
||||
output_price = _rub_per_1m_to_usd(meta.get("generated_tokens_cost"))
|
||||
if prompt_price is None or output_price is None:
|
||||
continue
|
||||
cached_price = _rub_per_1m_to_usd(meta.get("cache_read_tokens_cost"))
|
||||
cache_write_price = _rub_per_1m_to_usd(meta.get("cache_write_tokens_cost"))
|
||||
row = (
|
||||
prompt_price,
|
||||
cached_price,
|
||||
cache_write_price,
|
||||
output_price,
|
||||
)
|
||||
pricing_dict[normalize_model_identity(f"cloudru::{model_id}")] = row
|
||||
|
||||
log.info(f"Fetched pricing for {len(pricing_dict)} models from cloud.ru")
|
||||
return pricing_dict
|
||||
except (requests.RequestException, ValueError, KeyError) as e:
|
||||
log.warning(f"Failed to fetch cloud.ru pricing: {e}")
|
||||
return {}
|
||||
|
|
@ -23,6 +23,34 @@ def resolve_minimax_base_url(region: str = "") -> str:
|
|||
return MINIMAX_REGION_ENDPOINTS.get(selected, MINIMAX_REGION_ENDPOINTS[MINIMAX_DEFAULT_REGION])
|
||||
|
||||
|
||||
# DeepSeek serves one official OpenAI-compatible endpoint (no regions, no
|
||||
# owner-configurable base URL — a proxy/mirror setup belongs to the generic
|
||||
# openai-compatible slot). Kept as a module constant so transport and the
|
||||
# provider test resolve the same host; the reviewer-window/base-url fingerprint
|
||||
# maps deliberately have NO deepseek branch (both default to "" consistently —
|
||||
# the fingerprint is already unique per provider+model).
|
||||
DEEPSEEK_BASE_URL = "https://api.deepseek.com/v1"
|
||||
|
||||
# DeepSeek's Chat Completions ``reasoning_effort`` enum is low/high/max
|
||||
# (medium/xhigh are documented aliases of high) and thinking is switched off by
|
||||
# ``thinking.type=disabled``, not by an effort value. This is the wire dialect
|
||||
# of one provider, projected at the physical-send boundary; the canonical
|
||||
# Ouroboros effort scale stays the SSOT everywhere else. Not a model or pricing
|
||||
# table and never an admission gate.
|
||||
DEEPSEEK_REASONING_EFFORT_ALIASES = {
|
||||
"minimal": "low",
|
||||
"medium": "high",
|
||||
"xhigh": "high",
|
||||
"ultra": "max",
|
||||
}
|
||||
|
||||
|
||||
def normalize_deepseek_reasoning_effort(value: str) -> str:
|
||||
"""Project one canonical effort tier onto DeepSeek's Chat wire enum."""
|
||||
normalized = str(value or "").strip().lower()
|
||||
return DEEPSEEK_REASONING_EFFORT_ALIASES.get(normalized, normalized)
|
||||
|
||||
|
||||
# Direct-provider prefix → canonical provider name. Un-prefixed models route
|
||||
# through OpenRouter. Order matters only for readability; prefixes are disjoint.
|
||||
PROVIDER_PREFIXES: tuple[tuple[str, str], ...] = (
|
||||
|
|
@ -31,6 +59,7 @@ PROVIDER_PREFIXES: tuple[tuple[str, str], ...] = (
|
|||
("minimax::", "minimax"),
|
||||
("cloudru::", "cloudru"),
|
||||
("gigachat::", "gigachat"),
|
||||
("deepseek::", "deepseek"),
|
||||
("openai-compatible::", "openai-compatible"),
|
||||
("openrouter::", "openrouter"),
|
||||
)
|
||||
|
|
@ -41,6 +70,7 @@ PROVIDER_ENV_KEYS: dict[str, str] = {
|
|||
"anthropic": "ANTHROPIC_API_KEY",
|
||||
"minimax": "MINIMAX_API_KEY",
|
||||
"cloudru": "CLOUDRU_FOUNDATION_MODELS_API_KEY",
|
||||
"deepseek": "DEEPSEEK_API_KEY",
|
||||
"openrouter": "OPENROUTER_API_KEY",
|
||||
}
|
||||
|
||||
|
|
@ -66,6 +96,7 @@ PROVIDER_CREDENTIAL_GROUPS: dict[str, tuple[str, ...]] = {
|
|||
"anthropic": ("ANTHROPIC_API_KEY",),
|
||||
"minimax": ("MINIMAX_API_KEY", "MINIMAX_REGION"),
|
||||
"cloudru": ("CLOUDRU_FOUNDATION_MODELS_API_KEY", "CLOUDRU_FOUNDATION_MODELS_BASE_URL"),
|
||||
"deepseek": ("DEEPSEEK_API_KEY",),
|
||||
"gigachat": (
|
||||
"GIGACHAT_CREDENTIALS", "GIGACHAT_PASSWORD", "GIGACHAT_USER",
|
||||
"GIGACHAT_BASE_URL", "GIGACHAT_SCOPE", "GIGACHAT_VERIFY_SSL_CERTS",
|
||||
|
|
@ -171,7 +202,7 @@ def local_only_review_route_env() -> bool:
|
|||
provider_has_credentials(provider)
|
||||
for provider in (
|
||||
"openrouter", "openai", "anthropic", "minimax", "cloudru", "gigachat",
|
||||
"openai-compatible",
|
||||
"deepseek", "openai-compatible",
|
||||
)
|
||||
)
|
||||
|
||||
|
|
@ -375,6 +406,24 @@ MINIMAX_DIRECT_DEFAULTS = {
|
|||
# the Cloud.ru/GigaChat clear-instead-of-fill path; owners can opt in manually.
|
||||
}
|
||||
|
||||
DEEPSEEK_DIRECT_DEFAULTS = {
|
||||
"main": "deepseek::deepseek-v4-pro",
|
||||
"heavy": "",
|
||||
"light": "deepseek::deepseek-v4-flash",
|
||||
"fallback": "deepseek::deepseek-v4-flash",
|
||||
# DeepSeek documents a FIRM 1M context on every v4 model (api-docs
|
||||
# 2026-08: "1M context, 384K max output" with no guaranteed-minimum
|
||||
# caveat), so the slot follows the OpenAI/Anthropic fill pattern rather
|
||||
# than the MiniMax clear-instead-of-fill path (whose guaranteed floor was
|
||||
# 512K). The route's /models endpoint publishes NO window metadata, so the
|
||||
# ≥1M authority for blocking deep/scope review in Max mode still requires
|
||||
# the owner capability acknowledgement — until then the gate fails closed
|
||||
# loudly rather than silently degrading (see ARCHITECTURE §7).
|
||||
"deep_self_review": "deepseek::deepseek-v4-pro",
|
||||
# No vision default: deepseek-v4-flash-vision-exp is experimental; it is
|
||||
# recognized by supports_vision() for explicit owner selection only.
|
||||
}
|
||||
|
||||
ANTHROPIC_DIRECT_DEFAULTS = {
|
||||
"main": "anthropic::claude-opus-5",
|
||||
"heavy": "",
|
||||
|
|
@ -392,6 +441,7 @@ DIRECT_PROVIDER_DEFAULTS = {
|
|||
"cloudru": CLOUDRU_DIRECT_DEFAULTS,
|
||||
"gigachat": GIGACHAT_DIRECT_DEFAULTS,
|
||||
"minimax": MINIMAX_DIRECT_DEFAULTS,
|
||||
"deepseek": DEEPSEEK_DIRECT_DEFAULTS,
|
||||
}
|
||||
|
||||
# Review panels are declared as provider ROLE sequences, then compiled against
|
||||
|
|
@ -406,6 +456,9 @@ DIRECT_PROVIDER_REVIEW_ROLES = {
|
|||
"cloudru": ("main", "main", "main"),
|
||||
"gigachat": ("main", "main", "main"),
|
||||
"minimax": ("main", "light", "light"),
|
||||
# Strongest-main ×3 policy (same as OpenAI/Anthropic): an exclusive
|
||||
# DeepSeek install reviews with three independent thinking v4-pro calls.
|
||||
"deepseek": ("main", "main", "main"),
|
||||
}
|
||||
|
||||
DIRECT_PROVIDER_SCOPE_DEFAULTS = {
|
||||
|
|
@ -450,6 +503,12 @@ def migrate_model_value(provider: str, value: str) -> str:
|
|||
if text.startswith("minimax/"):
|
||||
return f"minimax::{text[len('minimax/'):]}"
|
||||
return text
|
||||
if provider == "deepseek":
|
||||
if text.startswith("deepseek::"):
|
||||
return text
|
||||
if text.startswith("deepseek/"):
|
||||
return f"deepseek::{text[len('deepseek/'):]}"
|
||||
return text
|
||||
return text
|
||||
|
||||
|
||||
|
|
@ -487,6 +546,11 @@ _VISION_MODEL_PREFIXES: tuple[str, ...] = (
|
|||
"qwen/qwen-vl", "qwen/qwen2.5-vl", "qwen/qwen3-vl",
|
||||
"mistralai/pixtral", "meta-llama/llama-4", "meta-llama/llama-3.2-90b-vision",
|
||||
"openai/gpt-5.5",
|
||||
# Narrow on purpose: only the dedicated vision variant. Plain deepseek
|
||||
# chat/v4 ids stay non-vision (pinned by tests), and this slash-form
|
||||
# normalized id also names a real OpenRouter vendor namespace — the
|
||||
# OpenRouter /models overlay may refine exact ids either way.
|
||||
"deepseek/deepseek-v4-flash-vision",
|
||||
)
|
||||
|
||||
# Runtime overlay: model_id → bool, fed from OpenRouter /models
|
||||
|
|
@ -537,6 +601,8 @@ def normalize_model_identity(model: str) -> str:
|
|||
return f"gigachat/{text[len('gigachat::'):]}"
|
||||
if text.startswith("minimax::"):
|
||||
return f"minimax/{text[len('minimax::'):]}"
|
||||
if text.startswith("deepseek::"):
|
||||
return f"deepseek/{text[len('deepseek::'):]}"
|
||||
if text.startswith("anthropic::"):
|
||||
return f"anthropic/{normalize_anthropic_model_id(text[len('anthropic::'):])}"
|
||||
if text.startswith("anthropic/"):
|
||||
|
|
|
|||
|
|
@ -46,6 +46,7 @@ OPTIONAL_REQUEST_FIELDS = OPTIONAL_SAMPLING_FIELDS + (
|
|||
"parallel_tool_calls",
|
||||
)
|
||||
NESTED_REASONING_FIELD = "extra_body.reasoning"
|
||||
NESTED_THINKING_FIELD = "extra_body.thinking"
|
||||
ANTHROPIC_REASONING_FIELDS = ("thinking", "output_config")
|
||||
DROP_FIELDS = frozenset((*OPTIONAL_REQUEST_FIELDS, NESTED_REASONING_FIELD))
|
||||
|
||||
|
|
@ -197,6 +198,8 @@ def reasoning_carrier(payload: Mapping[str, Any]) -> str:
|
|||
extra_body = payload.get("extra_body")
|
||||
if isinstance(extra_body, Mapping) and isinstance(extra_body.get("reasoning"), Mapping):
|
||||
return NESTED_REASONING_FIELD
|
||||
if isinstance(extra_body, Mapping) and isinstance(extra_body.get("thinking"), Mapping):
|
||||
return NESTED_THINKING_FIELD
|
||||
return ""
|
||||
|
||||
|
||||
|
|
@ -213,8 +216,15 @@ def payload_effort(payload: Mapping[str, Any]) -> str:
|
|||
if isinstance(thinking, Mapping) and str(thinking.get("type") or "").lower() == "disabled":
|
||||
return "none"
|
||||
extra_body = payload.get("extra_body")
|
||||
reasoning = extra_body.get("reasoning") if isinstance(extra_body, Mapping) else None
|
||||
return str(reasoning.get("effort") or "").strip().lower() if isinstance(reasoning, Mapping) else ""
|
||||
if not isinstance(extra_body, Mapping):
|
||||
return ""
|
||||
reasoning = extra_body.get("reasoning")
|
||||
if isinstance(reasoning, Mapping):
|
||||
return str(reasoning.get("effort") or "").strip().lower()
|
||||
thinking = extra_body.get("thinking")
|
||||
if isinstance(thinking, Mapping) and str(thinking.get("type") or "").lower() == "disabled":
|
||||
return "none"
|
||||
return ""
|
||||
|
||||
|
||||
def infer_tool_dialect(payload: Mapping[str, Any]) -> str:
|
||||
|
|
|
|||
|
|
@ -348,6 +348,12 @@ class NativeToolRoundReviewExecutor(ReviewSlotExecutor):
|
|||
total_usage[_fact] = usage[_fact]
|
||||
content = str(msg.get("content") or "") if isinstance(msg, dict) else ""
|
||||
transcript_chars += len(content)
|
||||
# The reasoning-echo lane (DeepSeek) keeps ``reasoning_content``
|
||||
# on the canonical assistant message appended below and replays
|
||||
# it verbatim on every later send — uncounted, the fail-closed
|
||||
# bound would drift by the entire thinking tail.
|
||||
if isinstance(msg, dict):
|
||||
transcript_chars += len(str(msg.get("reasoning_content") or ""))
|
||||
# Tool-call objects (names + argument JSON) join `messages`
|
||||
# below and ride every later send — the cumulative half of
|
||||
# the previous under-count.
|
||||
|
|
|
|||
|
|
@ -587,6 +587,7 @@ _REMOTE_PROVIDER_KEYS = (
|
|||
"OPENAI_API_KEY",
|
||||
"ANTHROPIC_API_KEY",
|
||||
"MINIMAX_API_KEY",
|
||||
"DEEPSEEK_API_KEY",
|
||||
"OPENAI_COMPATIBLE_API_KEY",
|
||||
"CLOUDRU_FOUNDATION_MODELS_API_KEY",
|
||||
"GIGACHAT_CREDENTIALS",
|
||||
|
|
@ -607,6 +608,7 @@ _PROVIDER_KEY_ENV = {
|
|||
"openai": "OPENAI_API_KEY",
|
||||
"anthropic": "ANTHROPIC_API_KEY",
|
||||
"minimax": "MINIMAX_API_KEY",
|
||||
"deepseek": "DEEPSEEK_API_KEY",
|
||||
"openai-compatible": "OPENAI_COMPATIBLE_API_KEY",
|
||||
"cloudru": "CLOUDRU_FOUNDATION_MODELS_API_KEY",
|
||||
"gigachat": "GIGACHAT_CREDENTIALS",
|
||||
|
|
|
|||
|
|
@ -27,6 +27,7 @@ MASKED_SECRET_SETTING_KEYS = frozenset(
|
|||
"GIGACHAT_PASSWORD",
|
||||
"ANTHROPIC_API_KEY",
|
||||
"MINIMAX_API_KEY",
|
||||
"DEEPSEEK_API_KEY",
|
||||
"GITHUB_TOKEN",
|
||||
"OUROBOROS_NETWORK_PASSWORD",
|
||||
}
|
||||
|
|
|
|||
|
|
@ -30,7 +30,7 @@ _DIRECT_PROVIDER_AUTO_DEFAULTS = {
|
|||
for provider, defaults in DIRECT_PROVIDER_DEFAULTS.items()
|
||||
}
|
||||
# Legacy values that should be auto-replaced with a provider's direct defaults.
|
||||
# Cloud.ru, GigaChat, and MiniMax intentionally have NO entry: such a provider-only user's
|
||||
# Cloud.ru, GigaChat, MiniMax, and DeepSeek intentionally have NO entry: such a provider-only user's
|
||||
# main/code/light slots match the shipped SETTINGS_DEFAULTS or the shared
|
||||
# _PRIOR_SHIPPED_SLOT_DEFAULTS below (google/gemini era) and migrate via the
|
||||
# `current in {"", default, *legacy}` check, and the review/scope slots are
|
||||
|
|
@ -92,7 +92,7 @@ for _legacy_defaults in _DIRECT_PROVIDER_LEGACY_DEFAULTS.values():
|
|||
for _slot in ("OUROBOROS_MODEL", "OUROBOROS_MODEL_HEAVY", "OUROBOROS_MODEL_LIGHT"):
|
||||
_legacy_defaults[_slot].add(_LEGACY_GEMINI_31_FLASH_LITE)
|
||||
# Outgoing SHIPPED OpenRouter defaults (through v6.104), applied for EVERY
|
||||
# exclusive-direct provider (incl. cloudru/gigachat/minimax, which have no per-provider
|
||||
# exclusive-direct provider (incl. cloudru/gigachat/minimax/deepseek, which have no per-provider
|
||||
# legacy table): before each defaults refresh a stored copy of the shipped default matched the
|
||||
# `current in {"", default}` check because SETTINGS_DEFAULTS still carried it;
|
||||
# after the defaults refresh these stored copies are still "the old DEFAULT, not
|
||||
|
|
@ -269,6 +269,7 @@ def _exclusive_direct_remote_provider(settings: dict) -> str:
|
|||
has_official_openai = bool(_setting_text(settings, "OPENAI_API_KEY"))
|
||||
has_anthropic = bool(_setting_text(settings, "ANTHROPIC_API_KEY"))
|
||||
has_minimax = bool(_setting_text(settings, "MINIMAX_API_KEY"))
|
||||
has_deepseek = bool(_setting_text(settings, "DEEPSEEK_API_KEY"))
|
||||
has_legacy_openai_base = bool(_setting_text(settings, "OPENAI_BASE_URL"))
|
||||
has_compatible = bool(_setting_text(settings, "OPENAI_COMPATIBLE_BASE_URL"))
|
||||
has_cloudru = bool(_setting_text(settings, "CLOUDRU_FOUNDATION_MODELS_API_KEY"))
|
||||
|
|
@ -288,6 +289,7 @@ def _exclusive_direct_remote_provider(settings: dict) -> str:
|
|||
("minimax", has_minimax),
|
||||
("cloudru", has_cloudru),
|
||||
("gigachat", has_gigachat),
|
||||
("deepseek", has_deepseek),
|
||||
) if present
|
||||
]
|
||||
return direct[0] if len(direct) == 1 else ""
|
||||
|
|
@ -392,6 +394,7 @@ def has_remote_provider(settings: dict) -> bool:
|
|||
"OPENAI_API_KEY",
|
||||
"ANTHROPIC_API_KEY",
|
||||
"MINIMAX_API_KEY",
|
||||
"DEEPSEEK_API_KEY",
|
||||
"OPENAI_COMPATIBLE_BASE_URL",
|
||||
"CLOUDRU_FOUNDATION_MODELS_API_KEY",
|
||||
"GIGACHAT_CREDENTIALS",
|
||||
|
|
|
|||
|
|
@ -9,6 +9,7 @@ from ouroboros.config import SETTINGS_DEFAULTS, VALID_RUNTIME_MODES
|
|||
from ouroboros.provider_models import (
|
||||
ANTHROPIC_DIRECT_DEFAULTS,
|
||||
CLOUDRU_DIRECT_DEFAULTS,
|
||||
DEEPSEEK_DIRECT_DEFAULTS,
|
||||
MINIMAX_DIRECT_DEFAULTS,
|
||||
MINIMAX_REGION_ENDPOINTS,
|
||||
OPENAI_DIRECT_DEFAULTS,
|
||||
|
|
@ -40,6 +41,7 @@ _MODEL_DEFAULTS = {
|
|||
"openai": {key: value for key, value in OPENAI_DIRECT_DEFAULTS.items() if key != "heavy"},
|
||||
"cloudru": {key: value for key, value in CLOUDRU_DIRECT_DEFAULTS.items() if key != "heavy"},
|
||||
"minimax": {key: value for key, value in MINIMAX_DIRECT_DEFAULTS.items() if key != "heavy"},
|
||||
"deepseek": {key: value for key, value in DEEPSEEK_DIRECT_DEFAULTS.items() if key != "heavy"},
|
||||
"anthropic": {key: value for key, value in ANTHROPIC_DIRECT_DEFAULTS.items() if key != "heavy"},
|
||||
# No defaults: model names are server-specific; user must fill all slots.
|
||||
"openai-compatible": {"main": "", "light": "", "vision": "", "fallback": ""},
|
||||
|
|
@ -73,6 +75,7 @@ _PROVIDER_FIELDS = _rows(("id", "stateKey", "settingKey", "settingsInputId", "la
|
|||
("cloudru-key", "cloudruKey", "CLOUDRU_FOUNDATION_MODELS_API_KEY", "s-cloudru-key", "Cloud.ru Foundation Models API Key", "Cloud.ru API key", "Optional. If this is the only remote key, the next step prefills direct cloudru::... models.", "password", "more"),
|
||||
("minimax-key", "minimaxKey", "MINIMAX_API_KEY", "s-minimax-key", "MiniMax API Key", "MiniMax API key", "Optional. If this is the only remote key, the next step prefills direct minimax::... models.", "password", "more"),
|
||||
("minimax-region", "minimaxRegion", "MINIMAX_REGION", "s-minimax-region", "MiniMax Region", "global_en or cn_zh", "Choose global_en for the global endpoint or cn_zh for the China endpoint.", "text", "more"),
|
||||
("deepseek-key", "deepseekKey", "DEEPSEEK_API_KEY", "s-deepseek-key", "DeepSeek API Key", "sk-...", "Optional. If this is the only remote key, the next step prefills direct deepseek::... models.", "password", "more"),
|
||||
("anthropic-key", "anthropicKey", "ANTHROPIC_API_KEY", "s-anthropic", "Anthropic API Key", "sk-ant-...", "Optional. Saved for direct anthropic::... models and Claude tooling.", "password", "primary"),
|
||||
("openai-compatible-url", "compatibleBaseUrl", "OPENAI_COMPATIBLE_BASE_URL", "s-compatible-url", "OpenAI-compatible Base URL", "http://localhost:11434/v1", "Base URL for your OpenAI-compatible endpoint (e.g. Ollama, LM Studio, vLLM). Required when using openai-compatible:: models.", "url", "more"),
|
||||
("openai-compatible-key", "compatibleApiKey", "OPENAI_COMPATIBLE_API_KEY", "s-compatible-key", "OpenAI-compatible API Key", "Leave empty for no auth", "API key for the endpoint. Leave empty if your server does not require authentication.", "password", "more"),
|
||||
|
|
@ -86,6 +89,7 @@ _PROFILE_SPECS = {
|
|||
"openai": ("OpenAI", "OpenAI is present, so the next step prefills direct openai:: model values.", "OpenAI-only setup detected. These defaults are explicit and official."),
|
||||
"cloudru": ("Cloud.ru Foundation Models", "Cloud.ru is present, so the next step prefills direct cloudru:: model values.", "Cloud.ru-only setup detected. These defaults use explicit cloudru:: model IDs."),
|
||||
"minimax": ("MiniMax", "MiniMax is present, so the next step prefills direct minimax:: model values.", "MiniMax-only setup detected. These defaults include MiniMax-M3 and MiniMax-M2.7."),
|
||||
"deepseek": ("DeepSeek", "DeepSeek is present, so the next step prefills direct deepseek:: model values.", "DeepSeek-only setup detected. These defaults use deepseek-v4-pro for main work and deepseek-v4-flash for the light lane. Blocking deep/scope review in Max context mode additionally needs the owner 1M-window acknowledgement in Settings."),
|
||||
"anthropic": ("Anthropic", "Anthropic is present, so the next step prefills direct anthropic:: model values.", "Anthropic-only setup detected. These defaults are explicit and official."),
|
||||
"openai-compatible": ("OpenAI-compatible endpoint", "An OpenAI-compatible base URL is configured. Enter the model names your server exposes in the next step.", "OpenAI-compatible endpoint detected. Use openai-compatible::your-model-name for every slot. The model list is whatever your server supports."),
|
||||
"direct-multi": ("Direct multi-provider", "Multiple direct providers are present, so the next step keeps your model values editable without forcing one provider family.", "Multiple direct providers are configured. Start here, then split model slots across them if you want."),
|
||||
|
|
@ -187,7 +191,7 @@ _SUBSCRIPTION_FIELDS = _rows(("id", "payloadKey", "label", "note"), (
|
|||
("skip-subscription-presets", SKIP_SUBSCRIPTION_PRESETS_FIELD, "Finish without agent defaults", "Completes onboarding without moving reviewers and subagents onto the connected subscriptions. Everything stays editable in Settings afterwards."),
|
||||
))
|
||||
|
||||
_MODEL_SUGGESTIONS = list(dict.fromkeys(("google/gemini-3.7-flash", "x-ai/grok-4.6", "openai/gpt-5.6-terra", "openai/gpt-5.6-sol", "openai/gpt-5.6-luna", "openai::gpt-5.6-terra", "openai::gpt-5.6-sol", "openai::gpt-5.6-luna", "anthropic/claude-sonnet-5", "anthropic/claude-opus-5", "anthropic::claude-sonnet-5", "anthropic::claude-opus-5", "anthropic::claude-opus-4-6", "deepseek/deepseek-v4-pro", "openai-compatible::meta-llama/compatible", "cloudru::zai-org/GLM-4.7", "minimax::MiniMax-M3", "minimax::MiniMax-M2.7")))
|
||||
_MODEL_SUGGESTIONS = list(dict.fromkeys(("google/gemini-3.7-flash", "x-ai/grok-4.6", "openai/gpt-5.6-terra", "openai/gpt-5.6-sol", "openai/gpt-5.6-luna", "openai::gpt-5.6-terra", "openai::gpt-5.6-sol", "openai::gpt-5.6-luna", "anthropic/claude-sonnet-5", "anthropic/claude-opus-5", "anthropic::claude-sonnet-5", "anthropic::claude-opus-5", "anthropic::claude-opus-4-6", "deepseek/deepseek-v4-pro", "deepseek::deepseek-v4-pro", "deepseek::deepseek-v4-flash", "openai-compatible::meta-llama/compatible", "cloudru::zai-org/GLM-4.7", "minimax::MiniMax-M3", "minimax::MiniMax-M2.7")))
|
||||
|
||||
|
||||
def _string(value: Any) -> str:
|
||||
|
|
@ -262,6 +266,7 @@ def derive_provider_profile(settings: dict) -> str:
|
|||
("OPENAI_API_KEY", "openai"),
|
||||
("CLOUDRU_FOUNDATION_MODELS_API_KEY", "cloudru"),
|
||||
("MINIMAX_API_KEY", "minimax"),
|
||||
("DEEPSEEK_API_KEY", "deepseek"),
|
||||
("ANTHROPIC_API_KEY", "anthropic"),
|
||||
]
|
||||
configured = [name for key, name in direct if flags[key]]
|
||||
|
|
@ -399,8 +404,8 @@ def wizard_authors_safety_light() -> bool:
|
|||
|
||||
Key absence alone is deliberately NOT the test: an older install re-entering the
|
||||
wizard (or one whose file is unreadable) would then be silently moved off the
|
||||
fail-closed ``full``. This predicate is HOST-AGNOSTIC by design and is called by
|
||||
the desktop launcher ONLY (`launcher.py::_run_first_run_wizard`) — the shared
|
||||
fail-closed ``full``. This predicate is HOST-AGNOSTIC by design and its one
|
||||
caller is the onboarding gateway (`gateway/onboarding.py`) — the shared
|
||||
validator must not author it, because web/Docker onboarding posts the same
|
||||
payload through generic `/api/settings`, which is a non-owner path. The persist
|
||||
seam re-proves freshness under the settings lock, so this is the caller-side
|
||||
|
|
@ -415,7 +420,13 @@ def validate_setup_payload(data: dict, current_settings: dict) -> Tuple[dict, st
|
|||
keys: Dict[str, str] = {}
|
||||
for field in _PROVIDER_FIELDS:
|
||||
setting_key = field["settingKey"]
|
||||
value = _string(data.get(setting_key))
|
||||
raw_value = data.get(setting_key)
|
||||
if raw_value is not None and not isinstance(raw_value, str):
|
||||
# A JSON object/array/number here is a malformed API post, never a
|
||||
# credential: str()-ing it would persist unusable garbage as a
|
||||
# "successful" setup. Honest rejection over silent stringification.
|
||||
return {}, f"{field['label']} must be a text value."
|
||||
value = _string(raw_value)
|
||||
# An untouched credential field posts back the marker it was prefilled
|
||||
# with. It means "keep the stored secret" and NEVER "store this string":
|
||||
# resolving it to the stored value here — before the length check, the
|
||||
|
|
@ -459,7 +470,7 @@ def validate_setup_payload(data: dict, current_settings: dict) -> Tuple[dict, st
|
|||
)
|
||||
has_local = bool(local_source)
|
||||
if not has_remote and not has_local:
|
||||
return {}, "Configure OpenRouter, OpenAI, OpenAI-compatible, Cloud.ru, MiniMax, Anthropic, or a local model before continuing."
|
||||
return {}, "Configure OpenRouter, OpenAI, OpenAI-compatible, Cloud.ru, MiniMax, DeepSeek, Anthropic, or a local model before continuing."
|
||||
minimax_region = keys.get("MINIMAX_REGION", "").lower()
|
||||
if minimax_region and minimax_region not in MINIMAX_REGION_ENDPOINTS:
|
||||
return {}, "MiniMax Region must be global_en or cn_zh."
|
||||
|
|
|
|||
|
|
@ -193,6 +193,7 @@ BAND_PATHS = {
|
|||
"tests/test_onboarding_wizard.py": None,
|
||||
"tests/test_owner_stop_s3.py": "Entered the band from 821 lines: the S3 contract suite now covers retry-root aliasing, graceful-to-immediate hardening, stale-control drain races, hard deadline preservation, descendant settlement failure, and late resweep exactly-once root finalization.",
|
||||
"tests/test_packaged_runtime_and_lifecycle.py": None,
|
||||
"tests/test_provider_contract_ci.py": "Provider canary matrix pins grew with the deepseek_direct row; split when the next provider lands.",
|
||||
"tests/test_provider_failure_reporting.py": "Entered the band from 938 lines: the provider-failure honesty rework added the retry-wall marker suites (entry-clear, all-retryable stamp, empty-response provider-failing gate, no-repay rail) beside the cost-validation suites (bool/NaN/inf/negative/huge-int at both boundaries) - one file per failure-reporting surface.",
|
||||
"tests/test_repo_health_smoke.py": "size-ratchet redesign: merge-aware previous, pairwise base-vs-tip, candidate-mode generator contract tests",
|
||||
"tests/test_review_cycles_dispatch.py": "Task acceptance wallet and paid-stamp dispatch regression matrix was integrated from the current managed target alongside the existing review-cycle tests.",
|
||||
|
|
@ -218,6 +219,7 @@ BAND_PATHS = {
|
|||
"web/modules/review_presentation.js": "Review Checkpoint read-side grouping, lifecycle/verdict separation, and keyed disclosure reconciliation remain one pure adapter below the 1500-line band cap.",
|
||||
"web/modules/reviewer_slots.js": "Owner-approved 5A editor: per-row Direct model / Configured subagent source picker with read-only derived disclosure replaces the legacy Claude-SDK advisory input in the same module that owns reviewer-row editing.",
|
||||
"web/modules/settings.js": None,
|
||||
"web/modules/settings_ui.js": "Provider cards grew with the DeepSeek card; extract the card table when the next provider lands.",
|
||||
"web/modules/widgets.js": None,
|
||||
"web/tests/harness_login_cards.test.js": "Login-card suite grew past 1000 lines with the name-the-account face cases (agy pickup, issue #232); split when the next face lands.",
|
||||
"web/tests/review_presentation.test.js": "Review Checkpoint lifecycle and verdict reconciliation remain covered by one focused presentation suite.",
|
||||
|
|
@ -235,6 +237,6 @@ BYTE_BASELINE_DEBT = {
|
|||
BYTE_DEBT = {
|
||||
"ouroboros/loop.py": 284435,
|
||||
"tests/test_delegated_subagent_transport.py": 320337,
|
||||
"tests/test_devtools_benchmarks.py": 328116,
|
||||
"tests/test_devtools_benchmarks.py": 328100,
|
||||
"web/modules/chat.js": 206949,
|
||||
}
|
||||
|
|
|
|||
|
|
@ -108,7 +108,7 @@ python_functions = ["test_*"]
|
|||
# Default local pytest excludes token-burning integration/browser suites.
|
||||
addopts = "-q --tb=short -m 'not integration and not browser and not ui_browser and not ui_browser_docker and not portable_detail and not skill_smoke and not size_ratchet'"
|
||||
markers = [
|
||||
"integration: requires real provider API keys (OPENROUTER_API_KEY / OPENAI_API_KEY / ANTHROPIC_API_KEY / CLOUDRU_FOUNDATION_MODELS_API_KEY); excluded from default local runs, opt in via `pytest -m integration`",
|
||||
"integration: requires real provider API keys (OPENROUTER_API_KEY / OPENAI_API_KEY / ANTHROPIC_API_KEY / MINIMAX_API_KEY / DEEPSEEK_API_KEY / CLOUDRU_FOUNDATION_MODELS_API_KEY / GIGACHAT_CREDENTIALS); excluded from default local runs, opt in via `pytest -m integration`",
|
||||
"browser: launches real Playwright Chromium; excluded from default local runs, opt in via `pytest -m browser`",
|
||||
"ui_browser: launches the real Ouroboros web UI in host/direct mode; excluded from default local runs",
|
||||
"ui_browser_docker: launches UI smoke checks against Docker runtime; excluded from default local runs",
|
||||
|
|
|
|||
|
|
@ -669,6 +669,10 @@ def _handle_llm_usage(evt: Dict[str, Any], ctx: Any) -> None:
|
|||
# Host-owned sealed-reasoning pin fact (issue #468): why same-model provider
|
||||
# failover was withheld on this call. Bounded {"sealed", "artifact"} dict.
|
||||
reasoning_pin = usage.get("reasoning_pin")
|
||||
# Provider wire projection / clamp of the requested reasoning effort
|
||||
# ({requested, applied, reason, model}); persisted so the owner can audit
|
||||
# what tier the provider actually received.
|
||||
effort_clamped = usage.get("reasoning_effort_clamped")
|
||||
|
||||
usage_event = {
|
||||
"ts": evt.get("ts", utc_now_iso()),
|
||||
|
|
@ -700,6 +704,7 @@ def _handle_llm_usage(evt: Dict[str, Any], ctx: Any) -> None:
|
|||
**({"chat_id": evt["chat_id"]} if evt.get("chat_id") is not None else {}),
|
||||
**({"web_search_sources": web_search_sources} if isinstance(web_search_sources, list) and web_search_sources else {}),
|
||||
**({"reasoning_pin": reasoning_pin} if isinstance(reasoning_pin, dict) and reasoning_pin else {}),
|
||||
**({"reasoning_effort_clamped": effort_clamped} if isinstance(effort_clamped, dict) and effort_clamped else {}),
|
||||
}
|
||||
_address_ctx(ctx, usage_event)
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -12,7 +12,11 @@ from enum import Enum
|
|||
|
||||
import pytest
|
||||
|
||||
from ouroboros.provider_models import OPENAI_DIRECT_DEFAULTS, normalize_model_identity
|
||||
from ouroboros.provider_models import (
|
||||
OPENAI_DIRECT_DEFAULTS,
|
||||
normalize_deepseek_reasoning_effort,
|
||||
normalize_model_identity,
|
||||
)
|
||||
from ouroboros.utils import sanitize_tool_result_for_log
|
||||
|
||||
CANARY_TIMEOUT_SEC = 120.0
|
||||
|
|
@ -71,6 +75,14 @@ _OPTIONAL_DIRECT_CANARIES = (
|
|||
"minimax_direct", "minimax::MiniMax-M3", "minimax",
|
||||
"MINIMAX_API_KEY", False, "none",
|
||||
),
|
||||
ProviderCanary(
|
||||
# "medium" ON PURPOSE (unlike its optional siblings): the deepseek lane
|
||||
# CARRIES reasoning_effort, and the canary pins that carriage. The
|
||||
# continuation is mandatory too: DeepSeek's tool contract is defined by
|
||||
# replaying reasoning_content on the second request.
|
||||
"deepseek_direct", "deepseek::deepseek-v4-flash", "deepseek",
|
||||
"DEEPSEEK_API_KEY", False, "medium", continue_to_final=True,
|
||||
),
|
||||
ProviderCanary(
|
||||
"cloudru_direct", "cloudru::zai-org/GLM-4.7", "cloudru",
|
||||
"CLOUDRU_FOUNDATION_MODELS_API_KEY", False, "none",
|
||||
|
|
@ -452,7 +464,7 @@ def assert_openai_canary_usage(usage, model):
|
|||
return disclosure
|
||||
|
||||
|
||||
def assert_canary_usage(usage, canary: ProviderCanary):
|
||||
def assert_canary_usage(usage, canary: ProviderCanary, *, forced_tool_choice: bool = False):
|
||||
def failure(code):
|
||||
return _canary_failure_payload(canary, None, usage, code)
|
||||
assert isinstance(usage, dict), failure("usage_not_mapping")
|
||||
|
|
@ -461,11 +473,28 @@ def assert_canary_usage(usage, canary: ProviderCanary):
|
|||
assert _safe_nonnegative_int(usage.get("prompt_tokens")) > 0, failure("missing_prompt_tokens")
|
||||
assert _safe_nonnegative_int(usage.get("completion_tokens")) > 0, failure("missing_completion_tokens")
|
||||
if canary.reasoning_effort == "medium":
|
||||
assert usage.get("reasoning_effort_clamped") is None, failure("reasoning_effort_clamped")
|
||||
expected_effort = canary.reasoning_effort
|
||||
if canary.expected_provider == "deepseek":
|
||||
# DeepSeek's wire enum aliases medium to high, and its thinking
|
||||
# mode rejects a forced tool choice, so the named first turn runs
|
||||
# with thinking disabled; both projections must be disclosed and
|
||||
# the physical payload carries the projected tier.
|
||||
if forced_tool_choice:
|
||||
expected_effort, reason = "none", "provider_forced_tool_choice"
|
||||
else:
|
||||
expected_effort = normalize_deepseek_reasoning_effort(expected_effort)
|
||||
reason = "provider_wire_mapping"
|
||||
note = usage.get("reasoning_effort_clamped")
|
||||
assert isinstance(note, dict), failure("reasoning_effort_clamped")
|
||||
assert note.get("requested") == canary.reasoning_effort, failure("reasoning_effort_clamped_requested")
|
||||
assert note.get("applied") == expected_effort, failure("reasoning_effort_clamped_applied")
|
||||
assert note.get("reason") == reason, failure("reasoning_effort_clamped_reason")
|
||||
else:
|
||||
assert usage.get("reasoning_effort_clamped") is None, failure("reasoning_effort_clamped")
|
||||
disclosure = usage.get("request_wire")
|
||||
assert isinstance(disclosure, dict), failure("missing_request_wire_disclosure")
|
||||
assert disclosure.get("requested_effort") == "medium", failure("request_wire_requested_effort")
|
||||
assert disclosure.get("applied_effort") == "medium", failure("request_wire_applied_effort")
|
||||
assert disclosure.get("requested_effort") == expected_effort, failure("request_wire_requested_effort")
|
||||
assert disclosure.get("applied_effort") == expected_effort, failure("request_wire_applied_effort")
|
||||
if canary.expected_provider == "openai":
|
||||
return assert_openai_canary_usage(usage, canary.model)
|
||||
return usage.get("request_wire")
|
||||
|
|
@ -701,7 +730,9 @@ def run_provider_contract_canary(
|
|||
usage=usage,
|
||||
)
|
||||
_record_canary_response_warnings(canary, message, usage)
|
||||
first_disclosure = assert_canary_usage(usage, canary)
|
||||
first_disclosure = assert_canary_usage(
|
||||
usage, canary, forced_tool_choice=canary.named_tool_choice,
|
||||
)
|
||||
if not canary.continue_to_final:
|
||||
return message, usage, None, None
|
||||
|
||||
|
|
|
|||
|
|
@ -48,16 +48,11 @@ PROFILE_TARGETS = {
|
|||
"openai/gpt-5.5",
|
||||
}
|
||||
|
||||
_PROVIDER_ROUTE_ENV_KEYS = (
|
||||
"OPENROUTER_API_KEY",
|
||||
"OPENAI_API_KEY",
|
||||
"ANTHROPIC_API_KEY",
|
||||
"OPENAI_BASE_URL",
|
||||
"OPENAI_COMPATIBLE_BASE_URL",
|
||||
"CLOUDRU_FOUNDATION_MODELS_API_KEY",
|
||||
"GIGACHAT_CREDENTIALS",
|
||||
"GIGACHAT_USER",
|
||||
"GIGACHAT_PASSWORD",
|
||||
# Registry-derived so a newly registered provider can never leak ambient routing.
|
||||
from ouroboros.provider_models import PROVIDER_CREDENTIAL_GROUPS as _CRED_GROUPS
|
||||
|
||||
_PROVIDER_ROUTE_ENV_KEYS = tuple(
|
||||
key for group in _CRED_GROUPS.values() for key in group
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
121
tests/test_benchmark_provider_touchpoints.py
Normal file
121
tests/test_benchmark_provider_touchpoints.py
Normal file
|
|
@ -0,0 +1,121 @@
|
|||
"""Benchmark surfaces must recognize every registered direct provider.
|
||||
|
||||
Wave-2 scope review (DeepSeek sprint) found three benchmark enumerations that
|
||||
silently missed DeepSeek (and, in two of them, MiniMax): the CyberGym
|
||||
settings-authoritative env sweep, ProgramBench's exclusive-direct-provider
|
||||
mirror, and Terminal-Bench's container network preflight. These regressions pin
|
||||
the fixes registry-first, so the NEXT provider cannot repeat the class.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import contextlib
|
||||
import io
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from types import SimpleNamespace
|
||||
|
||||
from ouroboros.provider_models import (
|
||||
DEEPSEEK_BASE_URL,
|
||||
PROVIDER_CREDENTIAL_GROUPS,
|
||||
resolve_minimax_base_url,
|
||||
)
|
||||
|
||||
|
||||
def test_authoritative_env_prefixes_cover_every_registered_credential():
|
||||
"""The settings-authoritative sweep promises to remove "provider/SDK
|
||||
families" — so every credential key the runtime registry knows must match
|
||||
one of its strip prefixes (or be an explicit exact/keep entry). DeepSeek's
|
||||
ambient key survived the sweep because DEEPSEEK_ was missing here."""
|
||||
from devtools.benchmarks.common.server_runner import (
|
||||
_AUTHORITATIVE_ENV_EXACT,
|
||||
_AUTHORITATIVE_ENV_KEEP,
|
||||
_AUTHORITATIVE_ENV_PREFIXES,
|
||||
)
|
||||
|
||||
for provider, group in PROVIDER_CREDENTIAL_GROUPS.items():
|
||||
for key in group:
|
||||
covered = (
|
||||
key in _AUTHORITATIVE_ENV_KEEP
|
||||
or key in _AUTHORITATIVE_ENV_EXACT
|
||||
or key.startswith(_AUTHORITATIVE_ENV_PREFIXES)
|
||||
)
|
||||
assert covered, f"{provider}: ambient {key} would survive the authoritative sweep"
|
||||
|
||||
|
||||
def test_programbench_mirror_recognizes_deepseek_and_minimax(monkeypatch):
|
||||
from devtools.benchmarks.programbench.run_programbench_e2e import _active_direct_provider
|
||||
|
||||
for group in PROVIDER_CREDENTIAL_GROUPS.values():
|
||||
for key in group:
|
||||
monkeypatch.delenv(key, raising=False)
|
||||
|
||||
assert _active_direct_provider({"DEEPSEEK_API_KEY": "sk-x"}) == "deepseek"
|
||||
assert _active_direct_provider({"MINIMAX_API_KEY": "mm-x"}) == "minimax"
|
||||
# Two configured direct providers are ambiguous — the OpenRouter-style route.
|
||||
assert _active_direct_provider(
|
||||
{"DEEPSEEK_API_KEY": "sk-x", "OPENAI_API_KEY": "sk-o"}
|
||||
) == ""
|
||||
|
||||
|
||||
class _PreflightEnv:
|
||||
"""Replays the container-side probe script in-process (same harness the
|
||||
existing openai preflight test uses)."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.command = ""
|
||||
|
||||
async def exec(self, *, command, timeout_sec=None, env=None, cwd=None):
|
||||
self.command = command
|
||||
script = command.split("python3 - <<'PY'\n", 1)[1].rsplit("\nPY", 1)[0]
|
||||
stdout = io.StringIO()
|
||||
code = 0
|
||||
try:
|
||||
with contextlib.redirect_stdout(stdout):
|
||||
exec(script, {})
|
||||
except SystemExit as exc:
|
||||
code = int(exc.code or 0)
|
||||
return SimpleNamespace(return_code=code, stdout=stdout.getvalue(), stderr="")
|
||||
|
||||
|
||||
def _run_preflight(tmp_path, monkeypatch, env):
|
||||
import devtools.benchmarks.terminal_bench.harbor_installed_agent as tb_agent
|
||||
|
||||
def fake_urlopen(req, timeout=0):
|
||||
raise urllib.error.HTTPError(req.full_url, 401, "Unauthorized", hdrs=None, fp=None)
|
||||
|
||||
monkeypatch.setattr(urllib.request, "urlopen", fake_urlopen)
|
||||
harness = _PreflightEnv()
|
||||
agent = tb_agent.OuroborosTerminalBenchAgent(logs_dir=tmp_path)
|
||||
asyncio.run(agent._network_preflight(harness, env))
|
||||
report = (tmp_path / "network-preflight.txt").read_text(encoding="utf-8")
|
||||
return harness.command, report
|
||||
|
||||
|
||||
def test_terminal_bench_preflight_probes_deepseek(tmp_path, monkeypatch):
|
||||
command, report = _run_preflight(tmp_path, monkeypatch, {"DEEPSEEK_API_KEY": "sk-x"})
|
||||
assert DEEPSEEK_BASE_URL.rstrip("/") + "/models" in command
|
||||
assert "deepseek_preflight_status 401" in report
|
||||
|
||||
|
||||
def test_terminal_bench_preflight_probes_minimax_by_region(tmp_path, monkeypatch):
|
||||
command, _report = _run_preflight(
|
||||
tmp_path, monkeypatch, {"MINIMAX_API_KEY": "mm-x", "MINIMAX_REGION": "cn_zh"}
|
||||
)
|
||||
assert resolve_minimax_base_url("cn_zh").rstrip("/") + "/models" in command
|
||||
|
||||
command, report = _run_preflight(tmp_path, monkeypatch, {"MINIMAX_API_KEY": "mm-x"})
|
||||
assert resolve_minimax_base_url("").rstrip("/") + "/models" in command
|
||||
assert "minimax_preflight_status 401" in report
|
||||
|
||||
|
||||
def test_terminal_bench_container_env_forwards_minimax_region(tmp_path, monkeypatch):
|
||||
"""The container resolves the MiniMax endpoint from MINIMAX_REGION; without
|
||||
the forward, a cn_zh owner silently probes and routes the global host."""
|
||||
import devtools.benchmarks.terminal_bench.harbor_installed_agent as tb_agent
|
||||
|
||||
agent = tb_agent.OuroborosTerminalBenchAgent(logs_dir=tmp_path)
|
||||
monkeypatch.setattr(agent, "_host_settings", lambda: {})
|
||||
monkeypatch.setattr(agent, "_container_secret_injection_allowed", lambda settings: False)
|
||||
monkeypatch.setenv("MINIMAX_REGION", "cn_zh")
|
||||
env = agent._container_env()
|
||||
assert env.get("MINIMAX_REGION") == "cn_zh"
|
||||
625
tests/test_deepseek_provider.py
Normal file
625
tests/test_deepseek_provider.py
Normal file
|
|
@ -0,0 +1,625 @@
|
|||
"""Direct DeepSeek provider regressions (v4 family, OpenAI-compatible endpoint).
|
||||
|
||||
Covers the single-provider independence contract (DEVELOPMENT.md "Provider
|
||||
Independence") and the two DeepSeek-specific wire classes established by live
|
||||
probes (2026-09-01):
|
||||
|
||||
- the effort-carrying route: DeepSeek's Chat API takes ``low``/``high``/``max``
|
||||
(``medium``/``xhigh`` are aliases of ``high``) and switches thinking off via
|
||||
``thinking.type=disabled``, so the lane projects the canonical scale onto that
|
||||
dialect instead of dropping effort like other generic compatible lanes;
|
||||
- the reasoning-echo REQUIREMENT: tool-bearing requests must pass every
|
||||
assistant turn's ``reasoning_content`` back (v4-pro enforces with a 400;
|
||||
an explicit empty string satisfies the gate for turns produced elsewhere).
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
from ouroboros import provider_models
|
||||
from ouroboros.llm import LLMClient
|
||||
from ouroboros.provider_models import (
|
||||
DEEPSEEK_BASE_URL,
|
||||
DEEPSEEK_DIRECT_DEFAULTS,
|
||||
DIRECT_PROVIDER_DEFAULTS,
|
||||
DIRECT_PROVIDER_REVIEW_ROLES,
|
||||
DIRECT_PROVIDER_SCOPE_DEFAULTS,
|
||||
migrate_model_value,
|
||||
normalize_model_identity,
|
||||
provider_for_model,
|
||||
provider_has_credentials,
|
||||
supports_vision,
|
||||
)
|
||||
from ouroboros.request_wire_contract import payload_effort, reasoning_carrier
|
||||
from ouroboros.request_wire_recovery import (
|
||||
prepare_wire_payload_for_send,
|
||||
request_wire_call_scope,
|
||||
)
|
||||
|
||||
|
||||
def _clear_provider_env(monkeypatch):
|
||||
for key in (
|
||||
"OPENROUTER_API_KEY", "OPENAI_API_KEY", "OPENAI_BASE_URL",
|
||||
"OPENAI_COMPATIBLE_API_KEY", "OPENAI_COMPATIBLE_BASE_URL",
|
||||
"ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "DEEPSEEK_API_KEY",
|
||||
"CLOUDRU_FOUNDATION_MODELS_API_KEY", "GIGACHAT_CREDENTIALS",
|
||||
"GIGACHAT_USER", "GIGACHAT_PASSWORD", "USE_LOCAL_MAIN",
|
||||
):
|
||||
monkeypatch.delenv(key, raising=False)
|
||||
|
||||
|
||||
class TestRegistry:
|
||||
def test_prefix_routes_direct(self):
|
||||
assert provider_for_model("deepseek::deepseek-v4-pro") == "deepseek"
|
||||
|
||||
def test_slash_form_stays_openrouter(self):
|
||||
# deepseek/ is a REAL OpenRouter vendor namespace (the CI canary uses
|
||||
# it); only the :: prefix selects the direct route.
|
||||
assert provider_for_model("deepseek/deepseek-v4-pro") == "openrouter"
|
||||
from ouroboros.pricing import infer_api_key_type
|
||||
assert infer_api_key_type("deepseek/deepseek-v4-pro") == "openrouter"
|
||||
assert infer_api_key_type("deepseek::deepseek-v4-pro") == "deepseek"
|
||||
|
||||
def test_credentials_mapping(self, monkeypatch):
|
||||
_clear_provider_env(monkeypatch)
|
||||
assert provider_has_credentials("deepseek") is False
|
||||
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-x")
|
||||
assert provider_has_credentials("deepseek") is True
|
||||
assert provider_models.model_has_credentials("deepseek::deepseek-v4-flash") is True
|
||||
|
||||
def test_direct_defaults_registered(self):
|
||||
assert DIRECT_PROVIDER_DEFAULTS["deepseek"] is DEEPSEEK_DIRECT_DEFAULTS
|
||||
assert DEEPSEEK_DIRECT_DEFAULTS["main"] == "deepseek::deepseek-v4-pro"
|
||||
assert DEEPSEEK_DIRECT_DEFAULTS["light"] == "deepseek::deepseek-v4-flash"
|
||||
assert DEEPSEEK_DIRECT_DEFAULTS["deep_self_review"] == "deepseek::deepseek-v4-pro"
|
||||
assert DIRECT_PROVIDER_REVIEW_ROLES["deepseek"] == ("main", "main", "main")
|
||||
assert DIRECT_PROVIDER_SCOPE_DEFAULTS["deepseek"] == "deepseek::deepseek-v4-pro"
|
||||
|
||||
def test_migrate_and_normalize_round_trip(self):
|
||||
assert migrate_model_value("deepseek", "deepseek/deepseek-v4-pro") == "deepseek::deepseek-v4-pro"
|
||||
assert migrate_model_value("deepseek", "deepseek::deepseek-v4-pro") == "deepseek::deepseek-v4-pro"
|
||||
assert normalize_model_identity("deepseek::deepseek-v4-flash") == "deepseek/deepseek-v4-flash"
|
||||
|
||||
def test_vision_narrow_prefix(self):
|
||||
assert supports_vision("deepseek::deepseek-v4-flash-vision-exp") is True
|
||||
assert supports_vision("deepseek::deepseek-v4-flash") is False
|
||||
assert supports_vision("deepseek/deepseek-chat") is False
|
||||
|
||||
|
||||
class TestSingleProviderIndependence:
|
||||
def test_exclusive_direct_env_detection(self, monkeypatch):
|
||||
_clear_provider_env(monkeypatch)
|
||||
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-x")
|
||||
from ouroboros.config import _exclusive_direct_remote_provider_env
|
||||
assert _exclusive_direct_remote_provider_env() == "deepseek"
|
||||
|
||||
def test_review_and_scope_fallback_compile(self, monkeypatch):
|
||||
_clear_provider_env(monkeypatch)
|
||||
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-x")
|
||||
monkeypatch.setenv("OUROBOROS_MODEL", "deepseek::deepseek-v4-pro")
|
||||
monkeypatch.setenv("OUROBOROS_MODEL_LIGHT", "deepseek::deepseek-v4-flash")
|
||||
from ouroboros.config import get_review_models, get_scope_review_models
|
||||
assert get_review_models() == ["deepseek::deepseek-v4-pro"] * 3
|
||||
assert get_scope_review_models() == ["deepseek::deepseek-v4-pro"]
|
||||
|
||||
def test_startup_gate_accepts_deepseek_only(self):
|
||||
from ouroboros.server_runtime import (
|
||||
_exclusive_direct_remote_provider,
|
||||
has_remote_provider,
|
||||
has_startup_ready_provider,
|
||||
)
|
||||
settings = {"DEEPSEEK_API_KEY": "sk-x"}
|
||||
assert has_remote_provider(settings) is True
|
||||
assert has_startup_ready_provider(settings) is True
|
||||
assert _exclusive_direct_remote_provider(settings) == "deepseek"
|
||||
|
||||
def test_local_only_review_route_sees_deepseek(self, monkeypatch):
|
||||
_clear_provider_env(monkeypatch)
|
||||
monkeypatch.setenv("USE_LOCAL_MAIN", "1")
|
||||
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-x")
|
||||
# A live remote DeepSeek credential means review slots must NOT be
|
||||
# forced onto the local route.
|
||||
assert provider_models.local_only_review_route_env() is False
|
||||
|
||||
def test_secret_surfaces_cover_deepseek(self):
|
||||
from ouroboros.contracts.plugin_api import FORBIDDEN_SKILL_SETTINGS
|
||||
from ouroboros.secret_masking import MASKED_SECRET_SETTING_KEYS
|
||||
assert "DEEPSEEK_API_KEY" in FORBIDDEN_SKILL_SETTINGS
|
||||
assert "DEEPSEEK_API_KEY" in MASKED_SECRET_SETTING_KEYS
|
||||
from ouroboros.config import SETTINGS_DEFAULTS
|
||||
assert SETTINGS_DEFAULTS["DEEPSEEK_API_KEY"] == ""
|
||||
|
||||
|
||||
class TestWireProjection:
|
||||
def _target(self, monkeypatch, model="deepseek::deepseek-v4-flash"):
|
||||
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-x")
|
||||
return LLMClient()._resolve_remote_target(model)
|
||||
|
||||
def test_resolve_remote_target_shape(self, monkeypatch):
|
||||
target = self._target(monkeypatch)
|
||||
assert target["provider"] == "deepseek"
|
||||
assert target["base_url"] == DEEPSEEK_BASE_URL
|
||||
assert target["api_key"] == "sk-x"
|
||||
assert target["usage_model"] == "deepseek/deepseek-v4-flash"
|
||||
assert target["requires_reasoning_echo"] is True
|
||||
assert target["supports_openrouter_extensions"] is False
|
||||
|
||||
def test_effort_carried_with_max_tokens_carrier(self, monkeypatch):
|
||||
client = LLMClient()
|
||||
target = self._target(monkeypatch)
|
||||
kwargs = client._build_remote_kwargs(
|
||||
target, [{"role": "user", "content": "hi"}], "high", 256, "auto", None, None,
|
||||
)
|
||||
assert kwargs["reasoning_effort"] == "high"
|
||||
assert kwargs["max_tokens"] == 256
|
||||
assert "max_completion_tokens" not in kwargs
|
||||
assert "extra_body" not in kwargs
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("requested", "wire"),
|
||||
[
|
||||
("none", None),
|
||||
("minimal", "low"),
|
||||
("low", "low"),
|
||||
("medium", "high"),
|
||||
("high", "high"),
|
||||
("xhigh", "high"),
|
||||
("max", "max"),
|
||||
("ultra", "max"),
|
||||
],
|
||||
)
|
||||
def test_effort_projection_matches_deepseek_chat_contract(
|
||||
self, monkeypatch, requested, wire,
|
||||
):
|
||||
# Official contract (api-docs.deepseek.com, Thinking Mode): the enum is
|
||||
# low/high/max, medium/xhigh alias high, and ``none`` is the separate
|
||||
# ``thinking.type=disabled`` toggle, never a reasoning_effort value.
|
||||
client = LLMClient()
|
||||
target = self._target(monkeypatch)
|
||||
kwargs = client._build_remote_kwargs(
|
||||
target, [{"role": "user", "content": "hi"}], requested, 256, "auto", None, None,
|
||||
)
|
||||
assert kwargs.get("reasoning_effort") == wire
|
||||
if wire is None:
|
||||
assert kwargs["extra_body"] == {"thinking": {"type": "disabled"}}
|
||||
else:
|
||||
assert "extra_body" not in kwargs
|
||||
note = client._pop_effort_clamp_disclosure()
|
||||
if wire is not None and wire != requested:
|
||||
assert note == {
|
||||
"requested": requested,
|
||||
"applied": wire,
|
||||
"reason": "provider_wire_mapping",
|
||||
"model": "deepseek-v4-flash",
|
||||
}
|
||||
else:
|
||||
assert note is None
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"choice",
|
||||
["required", {"type": "function", "function": {"name": "get_date"}}],
|
||||
)
|
||||
def test_forced_tool_choice_is_served_without_thinking(self, monkeypatch, choice):
|
||||
# Live-probed 2026-09-03 on v4-flash and v4-pro: thinking mode 400s on
|
||||
# tool_choice required/named ("Thinking mode does not support this
|
||||
# tool_choice"), while auto/none work; every form works with thinking
|
||||
# disabled. The caller's structural demand (a tool call WILL come
|
||||
# back) wins over reasoning on that one call, and usage says so.
|
||||
client = LLMClient()
|
||||
target = self._target(monkeypatch)
|
||||
tool = {"type": "function", "function": {
|
||||
"name": "get_date", "parameters": {"type": "object", "properties": {}}}}
|
||||
kwargs = client._build_remote_kwargs(
|
||||
target, [{"role": "user", "content": "hi"}], "high", 256, choice, None, [tool],
|
||||
)
|
||||
assert "reasoning_effort" not in kwargs
|
||||
assert kwargs["extra_body"] == {"thinking": {"type": "disabled"}}
|
||||
assert kwargs["tool_choice"] == choice
|
||||
assert client._pop_effort_clamp_disclosure() == {
|
||||
"requested": "high", "applied": "none",
|
||||
"reason": "provider_forced_tool_choice", "model": "deepseek-v4-flash",
|
||||
}
|
||||
# auto keeps thinking and the effort carriage.
|
||||
kwargs = client._build_remote_kwargs(
|
||||
target, [{"role": "user", "content": "hi"}], "high", 256, "auto", None, [tool],
|
||||
)
|
||||
assert kwargs["reasoning_effort"] == "high"
|
||||
assert "extra_body" not in kwargs
|
||||
|
||||
def test_projection_note_never_inherits_a_stale_entry(self, monkeypatch):
|
||||
# A note staged by an earlier call that never reached usage
|
||||
# normalization must not surface on the next unprojected call.
|
||||
client = LLMClient()
|
||||
target = self._target(monkeypatch)
|
||||
client._build_remote_kwargs(
|
||||
target, [{"role": "user", "content": "hi"}], "medium", 256, "auto", None, None,
|
||||
)
|
||||
client._build_remote_kwargs(
|
||||
target, [{"role": "user", "content": "hi"}], "high", 256, "auto", None, None,
|
||||
)
|
||||
assert client._pop_effort_clamp_disclosure() is None
|
||||
|
||||
def test_projection_note_is_isolated_per_asyncio_task(self, monkeypatch):
|
||||
# chat_async builds the payload before its first await and reads the
|
||||
# note after; two tasks on one loop must each see their own note.
|
||||
import asyncio
|
||||
|
||||
client = LLMClient()
|
||||
target = self._target(monkeypatch)
|
||||
seen = {}
|
||||
|
||||
async def call(name, effort):
|
||||
client._build_remote_kwargs(
|
||||
target, [{"role": "user", "content": "hi"}], effort, 256, "auto", None, None,
|
||||
)
|
||||
await asyncio.sleep(0)
|
||||
seen[name] = client._pop_effort_clamp_disclosure()
|
||||
|
||||
async def main():
|
||||
await asyncio.gather(call("projected", "medium"), call("native", "high"))
|
||||
|
||||
asyncio.run(main())
|
||||
assert seen["native"] is None
|
||||
assert seen["projected"]["requested"] == "medium"
|
||||
assert seen["projected"]["applied"] == "high"
|
||||
|
||||
def test_provider_usage_cannot_spoof_the_effort_note(self, monkeypatch):
|
||||
client = LLMClient()
|
||||
target = self._target(monkeypatch)
|
||||
_msg, usage = client._normalize_remote_response(
|
||||
{"id": "gen-1", "choices": [{"message": {"content": "ok"}, "finish_reason": "stop"}],
|
||||
"usage": {"prompt_tokens": 1, "completion_tokens": 1,
|
||||
"reasoning_effort_clamped": {"requested": "x", "applied": "y",
|
||||
"reason": "forged", "model": "forged"}}},
|
||||
target, skip_cost_fetch=True,
|
||||
)
|
||||
assert "reasoning_effort_clamped" not in usage
|
||||
|
||||
def test_request_wire_reads_disabled_thinking_as_none(self, monkeypatch):
|
||||
client = LLMClient()
|
||||
target = self._target(monkeypatch)
|
||||
kwargs = client._build_remote_kwargs(
|
||||
target, [{"role": "user", "content": "hi"}], "none", 256, "auto", None, None,
|
||||
)
|
||||
with request_wire_call_scope():
|
||||
physical = prepare_wire_payload_for_send(
|
||||
target, kwargs, api_surface="chat.completions",
|
||||
)
|
||||
assert "reasoning_effort" not in physical
|
||||
assert payload_effort(physical) == "none"
|
||||
assert reasoning_carrier(physical) == "extra_body.thinking"
|
||||
|
||||
def test_string_only_roles_flattened_in_send_copy_only(self, monkeypatch):
|
||||
# DeepSeek accepts content arrays only on user turns; the canonical
|
||||
# history (including the context-compaction assistant capsule) keeps
|
||||
# its block form.
|
||||
client = LLMClient()
|
||||
target = self._target(monkeypatch)
|
||||
messages = [
|
||||
{"role": "system", "content": [
|
||||
{"type": "text", "text": "policy", "cache_control": {"type": "ephemeral"}},
|
||||
{"type": "text", "text": " rules"},
|
||||
]},
|
||||
{"role": "user", "content": [
|
||||
{"type": "text", "text": "look"},
|
||||
{"type": "image_url", "image_url": {"url": "data:image/png;base64,AAAA"}},
|
||||
]},
|
||||
{"role": "assistant", "content": [
|
||||
{"type": "text", "text": "summary", "_context_capsule": {"id": "c1"}},
|
||||
]},
|
||||
{"role": "tool", "tool_call_id": "c1", "content": [
|
||||
{"type": "text", "text": "result"},
|
||||
]},
|
||||
]
|
||||
kwargs = client._build_remote_kwargs(
|
||||
target, messages, "high", 256, "auto", None, None,
|
||||
)
|
||||
sent = kwargs["messages"]
|
||||
assert sent[0]["content"] == "policy rules"
|
||||
assert isinstance(sent[1]["content"], list)
|
||||
assert sent[2]["content"] == "summary"
|
||||
assert sent[3]["content"] == "result"
|
||||
assert "_context_capsule" in messages[2]["content"][0]
|
||||
assert "cache_control" in messages[0]["content"][0]
|
||||
|
||||
def test_openai_effort_carriage_survives_minimal_stub_targets(self, monkeypatch):
|
||||
# The carriage is keyed on the provider id, NOT a target capability
|
||||
# field: the wire ladder's eligibility reads provider + payload effort,
|
||||
# and a hand-built target (fixtures, probes) must not silently drop the
|
||||
# effort (regression: test_issue229_synthesis stubs a minimal target).
|
||||
monkeypatch.setenv("OPENAI_API_KEY", "sk-openai")
|
||||
client = LLMClient()
|
||||
minimal_target = {
|
||||
"provider": "openai", "resolved_model": "gpt-5.6-terra",
|
||||
"usage_model": "openai/gpt-5.6-terra", "api_key": "k",
|
||||
"base_url": "https://api.openai.com/v1", "default_headers": {},
|
||||
"supports_openrouter_extensions": False,
|
||||
"supports_generation_cost": False,
|
||||
}
|
||||
kwargs = client._build_remote_kwargs(
|
||||
minimal_target, [{"role": "user", "content": "hi"}], "high", 128, "auto", None, None,
|
||||
)
|
||||
assert kwargs["reasoning_effort"] == "high"
|
||||
|
||||
def test_reasoning_content_replayed_and_gap_filled(self, monkeypatch):
|
||||
client = LLMClient()
|
||||
target = self._target(monkeypatch)
|
||||
messages = [
|
||||
{"role": "user", "content": "weather?"},
|
||||
{ # DeepSeek's own turn: reasoning must ride back verbatim.
|
||||
"role": "assistant", "content": "",
|
||||
"reasoning_content": "call the tool",
|
||||
"tool_calls": [{"id": "c1", "type": "function",
|
||||
"function": {"name": "t", "arguments": "{}"}}],
|
||||
},
|
||||
{"role": "tool", "tool_call_id": "c1", "content": "18C"},
|
||||
{ # Foreign-model turn (no reasoning): the strict v4-pro gate
|
||||
# still demands the field — an explicit "" satisfies it.
|
||||
"role": "assistant", "content": "done",
|
||||
},
|
||||
]
|
||||
kwargs = client._build_remote_kwargs(
|
||||
target, messages, "high", 256, "auto", None,
|
||||
[{"type": "function", "function": {"name": "t", "parameters": {"type": "object", "properties": {}}}}],
|
||||
)
|
||||
sent = [m for m in kwargs["messages"] if m.get("role") == "assistant"]
|
||||
assert sent[0]["reasoning_content"] == "call the tool"
|
||||
assert sent[1]["reasoning_content"] == ""
|
||||
|
||||
def test_other_compatible_lanes_still_strip(self, monkeypatch):
|
||||
monkeypatch.setenv("CLOUDRU_FOUNDATION_MODELS_API_KEY", "k")
|
||||
client = LLMClient()
|
||||
target = client._resolve_remote_target("cloudru::zai-org/GLM-4.7")
|
||||
messages = [
|
||||
{"role": "user", "content": "hi"},
|
||||
{"role": "assistant", "content": "x", "reasoning_content": "glm echo"},
|
||||
]
|
||||
kwargs = client._build_remote_kwargs(
|
||||
target, messages, "high", 256, "auto", None, None,
|
||||
)
|
||||
sent = [m for m in kwargs["messages"] if m.get("role") == "assistant"]
|
||||
assert "reasoning_content" not in sent[0]
|
||||
assert "reasoning_effort" not in kwargs # non-carrying lane unchanged
|
||||
|
||||
def test_normalize_keeps_deepseek_reasoning_in_transcript(self):
|
||||
client = LLMClient(api_key="x")
|
||||
resp = {
|
||||
"id": "1",
|
||||
"choices": [{"message": {
|
||||
"role": "assistant", "content": "4",
|
||||
"reasoning_content": "2+2 -> 4",
|
||||
}}],
|
||||
"usage": {"prompt_tokens": 5, "completion_tokens": 2},
|
||||
}
|
||||
target = {
|
||||
"provider": "deepseek",
|
||||
"usage_model": "deepseek/deepseek-v4-flash",
|
||||
"supports_openrouter_extensions": False,
|
||||
"supports_generation_cost": False,
|
||||
}
|
||||
msg, _usage = client._normalize_remote_response(resp, target, skip_cost_fetch=True)
|
||||
assert msg["reasoning_content"] == "2+2 -> 4"
|
||||
|
||||
def test_normalize_cached_tokens_top_level_fallback(self):
|
||||
client = LLMClient(api_key="x")
|
||||
resp = {
|
||||
"id": "1",
|
||||
"choices": [{"message": {"role": "assistant", "content": "ok"}}],
|
||||
# DeepSeek-shaped usage: top-level hit/miss beside an EMPTY
|
||||
# details block — the fallback must still account the cache hit.
|
||||
"usage": {
|
||||
"prompt_tokens": 446, "completion_tokens": 19,
|
||||
"prompt_tokens_details": {},
|
||||
"prompt_cache_hit_tokens": 384, "prompt_cache_miss_tokens": 62,
|
||||
},
|
||||
}
|
||||
target = {
|
||||
"provider": "deepseek",
|
||||
"usage_model": "deepseek/deepseek-v4-flash",
|
||||
"supports_openrouter_extensions": False,
|
||||
"supports_generation_cost": False,
|
||||
}
|
||||
_msg, usage = client._normalize_remote_response(resp, target, skip_cost_fetch=True)
|
||||
assert usage["cached_tokens"] == 384
|
||||
|
||||
def test_cross_family_switch_strips_deepseek_reasoning(self):
|
||||
messages = [
|
||||
{"role": "assistant", "content": "x", "reasoning_content": "ds"},
|
||||
]
|
||||
out = LLMClient.sanitize_reasoning_on_model_switch(
|
||||
messages, "deepseek::deepseek-v4-pro", "google/gemini-3.7-flash",
|
||||
)
|
||||
assert "reasoning_content" not in out[0]
|
||||
same = LLMClient.sanitize_reasoning_on_model_switch(
|
||||
messages, "deepseek::deepseek-v4-pro", "deepseek::deepseek-v4-flash",
|
||||
)
|
||||
assert same[0].get("reasoning_content") == "ds"
|
||||
|
||||
def test_vision_images_survive_for_vision_variant_only(self, monkeypatch):
|
||||
client = LLMClient()
|
||||
image_msg = [{"role": "user", "content": [
|
||||
{"type": "text", "text": "what is this"},
|
||||
{"type": "image_url", "image_url": {"url": "data:image/png;base64,AAAA"}},
|
||||
]}]
|
||||
vision_target = self._target(monkeypatch, "deepseek::deepseek-v4-flash-vision-exp")
|
||||
kwargs = client._build_remote_kwargs(
|
||||
vision_target, image_msg, "high", 128, "auto", None, None,
|
||||
)
|
||||
blocks = kwargs["messages"][0]["content"]
|
||||
assert any(isinstance(b, dict) and b.get("type") == "image_url" for b in blocks)
|
||||
|
||||
blind_target = self._target(monkeypatch, "deepseek::deepseek-v4-flash")
|
||||
kwargs = client._build_remote_kwargs(
|
||||
blind_target, image_msg, "high", 128, "auto", None, None,
|
||||
)
|
||||
blocks = kwargs["messages"][0]["content"]
|
||||
assert not any(isinstance(b, dict) and b.get("type") == "image_url" for b in blocks)
|
||||
|
||||
def test_openrouter_lane_neither_leaks_nor_pins_on_deepseek_residue(self, monkeypatch):
|
||||
# A mixed transcript (direct-DeepSeek turns replayed on an OpenRouter
|
||||
# model without passing a switch seam) must NOT forward the
|
||||
# deepseek-private reasoning_content to OpenRouter, and must NOT trip
|
||||
# the replay-artifact allow_fallbacks=False pin off that residue.
|
||||
monkeypatch.setenv("OPENROUTER_API_KEY", "sk-or-x")
|
||||
client = LLMClient()
|
||||
target = client._resolve_remote_target("google/gemini-3.7-flash")
|
||||
messages = [
|
||||
{"role": "user", "content": "hi"},
|
||||
{"role": "assistant", "content": "x", "reasoning_content": "ds thoughts"},
|
||||
{"role": "user", "content": "again"},
|
||||
]
|
||||
kwargs = client._build_remote_kwargs(
|
||||
target, messages, "high", 128, "auto", None, None,
|
||||
)
|
||||
sent = [m for m in kwargs["messages"] if m.get("role") == "assistant"]
|
||||
assert all("reasoning_content" not in m for m in sent)
|
||||
provider_body = (kwargs.get("extra_body") or {}).get("provider") or {}
|
||||
assert provider_body.get("allow_fallbacks") is not False
|
||||
# The canonical transcript is untouched.
|
||||
assert messages[1]["reasoning_content"] == "ds thoughts"
|
||||
|
||||
def test_openrouter_lane_strips_falsy_reasoning_residue_too(self, monkeypatch):
|
||||
# The scrub is keyed on key PRESENCE: an empty-string echo (a legal
|
||||
# kept value on the deepseek lane) and a legacy null must both stay
|
||||
# off the OpenRouter wire, not just truthy thoughts.
|
||||
monkeypatch.setenv("OPENROUTER_API_KEY", "sk-or-x")
|
||||
client = LLMClient()
|
||||
target = client._resolve_remote_target("google/gemini-3.7-flash")
|
||||
messages = [
|
||||
{"role": "user", "content": "hi"},
|
||||
{"role": "assistant", "content": "x", "reasoning_content": ""},
|
||||
{"role": "assistant", "content": "y", "reasoning_content": None},
|
||||
]
|
||||
kwargs = client._build_remote_kwargs(
|
||||
target, messages, "high", 128, "auto", None, None,
|
||||
)
|
||||
sent = [m for m in kwargs["messages"] if m.get("role") == "assistant"]
|
||||
assert all("reasoning_content" not in m for m in sent)
|
||||
|
||||
def test_openai_compatible_vendor_form_vision_stays_sighted(self, monkeypatch):
|
||||
# The qualified identity (openai-compatible/<id>) can never match the
|
||||
# vision prefixes; the BARE vendor-form id legitimately does. The
|
||||
# either-identity judgment keeps that lane sighted.
|
||||
monkeypatch.setenv("OPENAI_COMPATIBLE_API_KEY", "k")
|
||||
monkeypatch.setenv("OPENAI_COMPATIBLE_BASE_URL", "http://localhost:1234/v1")
|
||||
client = LLMClient()
|
||||
target = client._resolve_remote_target("openai-compatible::qwen/qwen2.5-vl-7b")
|
||||
image_msg = [{"role": "user", "content": [
|
||||
{"type": "image_url", "image_url": {"url": "data:image/png;base64,AAAA"}},
|
||||
]}]
|
||||
kwargs = client._build_remote_kwargs(
|
||||
target, image_msg, "high", 128, "auto", None, None,
|
||||
)
|
||||
blocks = kwargs["messages"][0]["content"]
|
||||
assert any(isinstance(b, dict) and b.get("type") == "image_url" for b in blocks)
|
||||
|
||||
def test_direct_openai_vision_judged_on_qualified_identity(self, monkeypatch):
|
||||
# Regression for the latent direct-lane class: the bare resolved id
|
||||
# never matched the slash-form vision prefixes, so every direct route
|
||||
# was captioned/placeholder'd regardless of real capability.
|
||||
monkeypatch.setenv("OPENAI_API_KEY", "sk-openai")
|
||||
client = LLMClient()
|
||||
target = client._resolve_remote_target("openai::gpt-5.5")
|
||||
image_msg = [{"role": "user", "content": [
|
||||
{"type": "image_url", "image_url": {"url": "data:image/png;base64,AAAA"}},
|
||||
]}]
|
||||
kwargs = client._build_remote_kwargs(
|
||||
target, image_msg, "high", 128, "auto", None, None,
|
||||
)
|
||||
blocks = kwargs["messages"][0]["content"]
|
||||
assert any(isinstance(b, dict) and b.get("type") == "image_url" for b in blocks)
|
||||
|
||||
|
||||
class TestReviewWaveHardening:
|
||||
"""Phase A review-wave fixes: witness eligibility, null reasoning, fit basis."""
|
||||
|
||||
def test_deepseek_usage_is_density_witness_eligible(self, tmp_path):
|
||||
# DeepSeek's automatic cache makes nearly every warm call cache-bearing;
|
||||
# excluding it from the cache-inclusive set starved the route of
|
||||
# witnesses after the first cold call (probed: prompt_tokens = hit + miss).
|
||||
from ouroboros import usage_accounting as ua
|
||||
from ouroboros.capability_evidence import _DENSITY_MEMO, get_token_density
|
||||
from ouroboros.provider_models import normalize_model_identity
|
||||
|
||||
_DENSITY_MEMO.clear()
|
||||
try:
|
||||
root = tmp_path / "deepseek"
|
||||
ua._observe_token_density(
|
||||
ua.AttemptRequest(
|
||||
model="deepseek/deepseek-v4-pro",
|
||||
provider="deepseek",
|
||||
prompt_tokens_estimate=1_000_000,
|
||||
drive_root=root,
|
||||
),
|
||||
{"prompt_tokens": 1_500_000, "cached_tokens": 900_000},
|
||||
)
|
||||
measured = get_token_density(root, normalize_model_identity("deepseek/deepseek-v4-pro"))
|
||||
assert abs(measured - 1.5) < 1e-6
|
||||
finally:
|
||||
# The memo is process-global and keyed WITHOUT drive_root: leaving
|
||||
# this tmp_path measurement behind would poison a co-located test
|
||||
# reading density for the same model id.
|
||||
_DENSITY_MEMO.clear()
|
||||
|
||||
def test_normalize_drops_non_string_reasoning_content(self):
|
||||
# A server-emitted null would live on the canonical assistant turn and
|
||||
# replay as JSON null against a gate probed only for strings, with no
|
||||
# message-level 400 recovery on the direct lane.
|
||||
client = LLMClient(api_key="x")
|
||||
resp = {
|
||||
"id": "1",
|
||||
"choices": [{"message": {
|
||||
"role": "assistant", "content": "4",
|
||||
"reasoning_content": None,
|
||||
}}],
|
||||
"usage": {"prompt_tokens": 5, "completion_tokens": 2},
|
||||
}
|
||||
target = {
|
||||
"provider": "deepseek",
|
||||
"usage_model": "deepseek/deepseek-v4-flash",
|
||||
"supports_openrouter_extensions": False,
|
||||
"supports_generation_cost": False,
|
||||
}
|
||||
msg, _usage = client._normalize_remote_response(resp, target, skip_cost_fetch=True)
|
||||
assert "reasoning_content" not in msg
|
||||
|
||||
def test_echo_fill_coerces_non_string_reasoning(self, monkeypatch):
|
||||
# An imported/legacy transcript can already carry a null; setdefault
|
||||
# cannot repair an existing key, so the fill must coerce by type.
|
||||
monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-x")
|
||||
client = LLMClient()
|
||||
target = client._resolve_remote_target("deepseek::deepseek-v4-pro")
|
||||
messages = [
|
||||
{"role": "user", "content": "hi"},
|
||||
{"role": "assistant", "content": "x", "reasoning_content": None},
|
||||
]
|
||||
kwargs = client._build_remote_kwargs(
|
||||
target, messages, "high", 256, "auto", None,
|
||||
[{"type": "function", "function": {"name": "t", "parameters": {"type": "object", "properties": {}}}}],
|
||||
)
|
||||
sent = [m for m in kwargs["messages"] if m.get("role") == "assistant"]
|
||||
assert sent[0]["reasoning_content"] == ""
|
||||
|
||||
def test_estimate_message_chars_counts_replayed_reasoning(self):
|
||||
# Replayed reasoning is real wire prompt on the echo lane: the planning
|
||||
# basis must see it or fit drift grows with transcript length.
|
||||
from ouroboros.context_budget import estimate_message_chars
|
||||
|
||||
plain = [{"role": "assistant", "content": "ab"}]
|
||||
with_reasoning = [{"role": "assistant", "content": "ab", "reasoning_content": "cdef"}]
|
||||
assert estimate_message_chars(with_reasoning) == estimate_message_chars(plain) + 4
|
||||
|
||||
def test_remote_fit_estimator_counts_replayed_reasoning(self):
|
||||
# The PRODUCTION remote-fit path: estimate_context_prompt_tokens
|
||||
# serializes message dicts recursively, so the deepseek lane's kept
|
||||
# reasoning_content must grow the estimate — this is the guarantee the
|
||||
# fit machinery actually runs on for remote sends.
|
||||
from ouroboros.context_fit import estimate_context_prompt_tokens
|
||||
|
||||
plain = [{"role": "assistant", "content": "ab"}]
|
||||
with_reasoning = [
|
||||
{"role": "assistant", "content": "ab", "reasoning_content": "r" * 4000},
|
||||
]
|
||||
assert (
|
||||
estimate_context_prompt_tokens(with_reasoning, provider="deepseek")
|
||||
> estimate_context_prompt_tokens(plain, provider="deepseek") + 500
|
||||
)
|
||||
|
|
@ -1599,16 +1599,11 @@ def test_programbench_submit_and_wait_stale_checkpoint_falls_back_to_fresh_submi
|
|||
assert json.loads(checkpoint.read_text(encoding="utf-8"))["task_id"] == "task-new"
|
||||
|
||||
|
||||
_PROVIDER_ROUTE_ENV_KEYS = (
|
||||
"OPENROUTER_API_KEY",
|
||||
"OPENAI_API_KEY",
|
||||
"ANTHROPIC_API_KEY",
|
||||
"OPENAI_BASE_URL",
|
||||
"OPENAI_COMPATIBLE_BASE_URL",
|
||||
"CLOUDRU_FOUNDATION_MODELS_API_KEY",
|
||||
"GIGACHAT_CREDENTIALS",
|
||||
"GIGACHAT_USER",
|
||||
"GIGACHAT_PASSWORD",
|
||||
# Registry-derived so a newly registered provider can never leak ambient routing.
|
||||
from ouroboros.provider_models import PROVIDER_CREDENTIAL_GROUPS as _CRED_GROUPS
|
||||
|
||||
_PROVIDER_ROUTE_ENV_KEYS = tuple(
|
||||
key for group in _CRED_GROUPS.values() for key in group
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -25,15 +25,15 @@ def test_architecture_mentions_shared_log_grouping_and_direct_provider_review_fa
|
|||
assert "log_events.js" in arch
|
||||
assert "live task card" in arch
|
||||
assert "grouped task cards" in arch
|
||||
# Direct-provider fallback covers official OpenAI, Anthropic, MiniMax, Cloud.ru,
|
||||
# and GigaChat, while still excluding OpenRouter/OpenAI-compatible/mixed-provider configs.
|
||||
# Direct-provider fallback covers official OpenAI, Anthropic, MiniMax, DeepSeek,
|
||||
# Cloud.ru, and GigaChat, while still excluding OpenRouter/OpenAI-compatible/mixed-provider configs.
|
||||
# Keep the generalized name ("Direct-provider review fallback") and a
|
||||
# reference to the legacy "OpenAI-only review fallback" phrase for
|
||||
# discoverability, and pin the honest scope language so the doc cannot
|
||||
# silently re-expand to claim symmetric coverage it does not have yet.
|
||||
assert "Direct-provider review fallback" in arch
|
||||
assert "OpenAI-only review fallback" in arch # legacy name still referenced for discoverability
|
||||
assert "official OpenAI, Anthropic, MiniMax, Cloud.ru, and GigaChat" in arch
|
||||
assert "official OpenAI, Anthropic, MiniMax, DeepSeek, Cloud.ru, and GigaChat" in arch
|
||||
assert "_exclusive_direct_remote_provider_env" in arch
|
||||
# v4.34.0: direct-provider fallback now documents the
|
||||
# `main_model.startswith(provider_prefix)` guard in get_review_models —
|
||||
|
|
|
|||
|
|
@ -460,12 +460,18 @@ def test_usage_accounting_abort_discards_clamp_note(
|
|||
client._create_chat_completion_with_retries(lambda **kw: {"ok": True}, {"model": "m"}, target)
|
||||
assert client._pop_effort_clamp_disclosure() is None
|
||||
|
||||
assert client._clamp_effort_for_model("vendor/uae-test", "none") == "low"
|
||||
with _pytest.raises(UsageAccountingError):
|
||||
asyncio.run(client._create_chat_completion_with_retries_async(
|
||||
lambda **kw: {"ok": True}, {"model": "m"}, target,
|
||||
))
|
||||
assert client._pop_effort_clamp_disclosure() is None
|
||||
# The note lives in a ContextVar (isolated per thread AND per asyncio
|
||||
# task, like the reasoning pin), so stage it where chat_async does: inside
|
||||
# the coroutine that runs the driver.
|
||||
async def _stage_then_abort():
|
||||
assert client._clamp_effort_for_model("vendor/uae-test", "none") == "low"
|
||||
with _pytest.raises(UsageAccountingError):
|
||||
await client._create_chat_completion_with_retries_async(
|
||||
lambda **kw: {"ok": True}, {"model": "m"}, target,
|
||||
)
|
||||
assert client._pop_effort_clamp_disclosure() is None
|
||||
|
||||
asyncio.run(_stage_then_abort())
|
||||
|
||||
|
||||
def test_uae_on_floored_resend_discards_clamp_note(
|
||||
|
|
|
|||
|
|
@ -58,6 +58,27 @@ def test_llm_usage_writes_cached_tokens_and_cache_write_tokens(tmp_path):
|
|||
assert ctx.last_usage["prompt_cache_ttl"] == "default"
|
||||
|
||||
|
||||
def test_llm_usage_persists_reasoning_effort_projection(tmp_path):
|
||||
from supervisor import events as ev_module
|
||||
|
||||
(tmp_path / "logs").mkdir()
|
||||
|
||||
class FakeCtx:
|
||||
DRIVE_ROOT = tmp_path
|
||||
|
||||
def update_budget_from_usage(self, usage):
|
||||
self.last_usage = usage
|
||||
|
||||
note = {"requested": "medium", "applied": "high",
|
||||
"reason": "provider_wire_mapping", "model": "deepseek-v4-flash"}
|
||||
ev_module._handle_llm_usage(
|
||||
{"type": "llm_usage", "task_id": "t", "usage": {"prompt_tokens": 3, "reasoning_effort_clamped": note}},
|
||||
FakeCtx(),
|
||||
)
|
||||
written = json.loads((tmp_path / "logs" / "events.jsonl").read_text(encoding="utf-8"))
|
||||
assert written["reasoning_effort_clamped"] == note
|
||||
|
||||
|
||||
def test_llm_usage_preserves_unknown_cost_as_null(tmp_path):
|
||||
from supervisor import events as ev_module
|
||||
|
||||
|
|
|
|||
|
|
@ -27,6 +27,7 @@ def test_model_catalog_tags_provider_values(monkeypatch):
|
|||
"CLOUDRU_FOUNDATION_MODELS_API_KEY": "cloudru-key",
|
||||
"GIGACHAT_CREDENTIALS": "giga-creds",
|
||||
"MINIMAX_API_KEY": "minimax-key",
|
||||
"DEEPSEEK_API_KEY": "deepseek-key",
|
||||
})
|
||||
|
||||
async def fake_openrouter(_client, _api_key):
|
||||
|
|
@ -46,6 +47,7 @@ def test_model_catalog_tags_provider_values(monkeypatch):
|
|||
"openai-compatible": "compatible-pro",
|
||||
"cloudru": "cloudru-pro",
|
||||
"minimax": "MiniMax-M3",
|
||||
"deepseek": "deepseek-v4-pro",
|
||||
}[provider_id]
|
||||
return [model_catalog_api._build_model_catalog_entry(provider_id, provider_label, model_id, model_id)]
|
||||
|
||||
|
|
@ -68,9 +70,10 @@ def test_model_catalog_tags_provider_values(monkeypatch):
|
|||
assert "openai-compatible::compatible-pro" in values
|
||||
assert "cloudru::cloudru-pro" in values
|
||||
assert "gigachat::giga-pro" in values
|
||||
# MiniMax rides the shared OpenAI-compatible live fetcher (GET /v1/models on
|
||||
# the region host), so the fake returns exactly one model for it.
|
||||
# MiniMax and DeepSeek ride the shared OpenAI-compatible live fetcher
|
||||
# (GET /v1/models on their fixed hosts), so the fake returns one model each.
|
||||
assert "minimax::MiniMax-M3" in values
|
||||
assert "deepseek::deepseek-v4-pro" in values
|
||||
assert payload["errors"] == []
|
||||
|
||||
openrouter_item = items_by_value["anthropic/claude-sonnet-4-6"]
|
||||
|
|
|
|||
|
|
@ -175,6 +175,36 @@ def test_transcript_counter_includes_system_schemas_and_args(subject_repo, monke
|
|||
assert not llm.calls, "the send bound must refuse before paying for a send"
|
||||
|
||||
|
||||
def test_transcript_counter_includes_replayed_reasoning(subject_repo, monkeypatch):
|
||||
"""The reasoning-echo lane (DeepSeek) keeps ``reasoning_content`` on the
|
||||
canonical assistant message the loop appends, and replays it verbatim on
|
||||
every later send. The SEND bound must count it like content — previously
|
||||
the whole dict joined ``messages`` while the counter saw only content and
|
||||
tool-call JSON, so a large thinking tail drifted past the promised bound
|
||||
unmeasured.
|
||||
"""
|
||||
import ouroboros.review_native_episode as native_episode
|
||||
|
||||
llm = _ScriptedLLM([
|
||||
{
|
||||
"tool_calls": [_tool_call("read_file", {"path": "greeting.txt"})],
|
||||
"reasoning_content": "r" * 1_000_000,
|
||||
},
|
||||
# No second entry ON PURPOSE: the counter must refuse before send 2.
|
||||
])
|
||||
executor = NativeToolRoundReviewExecutor(_assignment(subject_repo, llm), llm=llm)
|
||||
# Generously admits the first send (prompt + ~9K system/schema cost) and
|
||||
# anything the round adds EXCEPT the megachar reasoning tail.
|
||||
cap = len(executor.episode_prompt) + 200_000
|
||||
monkeypatch.setattr(
|
||||
native_episode, "review_native_max_transcript_chars", lambda: cap
|
||||
)
|
||||
with pytest.raises(ReviewRouteUnavailable) as exc:
|
||||
executor.execute()
|
||||
assert exc.value.code == "native_transcript_cap_exceeded"
|
||||
assert len(llm.calls) == 1, "the bound must refuse before paying for send 2"
|
||||
|
||||
|
||||
def test_uninspectable_tool_is_refused_in_episode(subject_repo):
|
||||
llm = _ScriptedLLM([
|
||||
{"tool_calls": [_tool_call("write_file", {"path": "greeting.txt", "content": "hacked"})]},
|
||||
|
|
|
|||
|
|
@ -41,7 +41,7 @@ def test_prepare_onboarding_settings_requires_runnable_config():
|
|||
prepared, error = prepare_onboarding_settings(_base_payload(), {})
|
||||
|
||||
assert prepared == {}
|
||||
assert "Configure OpenRouter, OpenAI, OpenAI-compatible, Cloud.ru, MiniMax, Anthropic, or a local model" in error
|
||||
assert "Configure OpenRouter, OpenAI, OpenAI-compatible, Cloud.ru, MiniMax, DeepSeek, Anthropic, or a local model" in error
|
||||
|
||||
|
||||
def test_prepare_onboarding_settings_accepts_openai_only_setup():
|
||||
|
|
@ -140,6 +140,38 @@ def test_prepare_onboarding_settings_accepts_minimax_only_setup():
|
|||
assert prepared["OUROBOROS_MODEL_LIGHT"] == "minimax::MiniMax-M2.7"
|
||||
|
||||
|
||||
def test_prepare_onboarding_settings_rejects_non_string_credentials():
|
||||
# A JSON object/array/number in a credential slot is a malformed API post:
|
||||
# str()-ing it used to persist "{'nested': ...}" as a working key with no
|
||||
# error. The validator now refuses with an honest message instead.
|
||||
for bad in ({"nested": "abcdefghij"}, ["sk-x"], 123, True):
|
||||
payload = _base_payload()
|
||||
payload.update({
|
||||
"DEEPSEEK_API_KEY": bad,
|
||||
"OUROBOROS_MODEL": "deepseek::deepseek-v4-pro",
|
||||
})
|
||||
prepared, error = prepare_onboarding_settings(payload, {})
|
||||
assert prepared == {}
|
||||
assert error == "DeepSeek API Key must be a text value."
|
||||
|
||||
|
||||
def test_prepare_onboarding_settings_accepts_deepseek_only_setup():
|
||||
payload = _base_payload()
|
||||
payload.update({
|
||||
"DEEPSEEK_API_KEY": "sk-deepseek-key-1234567890",
|
||||
"OUROBOROS_MODEL": "deepseek::deepseek-v4-pro",
|
||||
"OUROBOROS_MODEL_LIGHT": "deepseek::deepseek-v4-flash",
|
||||
"OUROBOROS_MODEL_FALLBACKS": "deepseek::deepseek-v4-flash",
|
||||
})
|
||||
|
||||
prepared, error = prepare_onboarding_settings(payload, {})
|
||||
|
||||
assert error is None
|
||||
assert prepared["DEEPSEEK_API_KEY"] == "sk-deepseek-key-1234567890"
|
||||
assert prepared["OUROBOROS_MODEL"] == "deepseek::deepseek-v4-pro"
|
||||
assert prepared["OUROBOROS_MODEL_LIGHT"] == "deepseek::deepseek-v4-flash"
|
||||
|
||||
|
||||
def test_prepare_onboarding_settings_rejects_unknown_minimax_region():
|
||||
payload = _base_payload()
|
||||
payload["MINIMAX_API_KEY"] = "minimax-key-1234567890"
|
||||
|
|
@ -300,7 +332,7 @@ def test_prepare_onboarding_settings_rejects_openai_compatible_key_without_base_
|
|||
prepared, error = prepare_onboarding_settings(payload, {})
|
||||
|
||||
assert prepared == {}
|
||||
assert "Configure OpenRouter, OpenAI, OpenAI-compatible, Cloud.ru, MiniMax, Anthropic, or a local model" in error
|
||||
assert "Configure OpenRouter, OpenAI, OpenAI-compatible, Cloud.ru, MiniMax, DeepSeek, Anthropic, or a local model" in error
|
||||
|
||||
|
||||
def test_onboarding_frontend_uses_base_url_first_compatible_validation():
|
||||
|
|
@ -580,6 +612,7 @@ def test_setup_contract_groups_rarely_used_providers():
|
|||
"CLOUDRU_FOUNDATION_MODELS_API_KEY": "more",
|
||||
"MINIMAX_API_KEY": "more",
|
||||
"MINIMAX_REGION": "more",
|
||||
"DEEPSEEK_API_KEY": "more",
|
||||
"ANTHROPIC_API_KEY": "primary",
|
||||
"OPENAI_COMPATIBLE_BASE_URL": "more",
|
||||
"OPENAI_COMPATIBLE_API_KEY": "more",
|
||||
|
|
@ -621,6 +654,8 @@ def test_setup_contract_has_no_secret_values():
|
|||
assert "anthropic::claude-sonnet-5" in suggestions
|
||||
assert "minimax::MiniMax-M3" in suggestions
|
||||
assert "minimax::MiniMax-M2.7" in suggestions
|
||||
assert "deepseek::deepseek-v4-pro" in suggestions
|
||||
assert "deepseek::deepseek-v4-flash" in suggestions
|
||||
assert empty_bootstrap["initialState"]["totalBudget"] == 200.0
|
||||
assert empty_bootstrap["initialState"]["perTaskCostUsd"] == 50.0
|
||||
assert budget_fields["TOTAL_BUDGET"]["default"] == 200.0
|
||||
|
|
@ -633,6 +668,12 @@ def test_setup_contract_has_no_secret_values():
|
|||
assert initial["mainModel"] == "minimax::MiniMax-M3"
|
||||
assert initial["lightModel"] == "minimax::MiniMax-M2.7"
|
||||
|
||||
deepseek_bootstrap = build_setup_bootstrap({"DEEPSEEK_API_KEY": "ds-hidden-value"}, "web")
|
||||
deepseek_initial = deepseek_bootstrap["initialState"]
|
||||
assert deepseek_initial["providerProfile"] == "deepseek"
|
||||
assert deepseek_initial["mainModel"] == "deepseek::deepseek-v4-pro"
|
||||
assert deepseek_initial["lightModel"] == "deepseek::deepseek-v4-flash"
|
||||
|
||||
|
||||
# --- The served page must not hand back a stored credential -----------------
|
||||
# The onboarding page is an unauthenticated GET on every host, and a non-loopback
|
||||
|
|
@ -645,6 +686,7 @@ _SECRET_CANARIES = {
|
|||
"OPENAI_COMPATIBLE_API_KEY": "compat-SECRETCANARY125",
|
||||
"CLOUDRU_FOUNDATION_MODELS_API_KEY": "cloudru-SECRETCANARY126",
|
||||
"MINIMAX_API_KEY": "minimax-SECRETCANARY127",
|
||||
"DEEPSEEK_API_KEY": "sk-ds-SECRETCANARY133",
|
||||
"ANTHROPIC_API_KEY": "sk-ant-SECRETCANARY128",
|
||||
"GIGACHAT_CREDENTIALS": "giga-SECRETCANARY129",
|
||||
"GIGACHAT_PASSWORD": "gigapw-SECRETCANARY130",
|
||||
|
|
|
|||
|
|
@ -132,7 +132,7 @@ def test_onboarding_compact_access_step_keeps_default_width_two_column():
|
|||
|
||||
|
||||
def test_settings_more_providers_collapse_keeps_inputs_mounted():
|
||||
"""Rarely used provider cards (Cloud.ru, MiniMax, GigaChat) collapse under a
|
||||
"""Rarely used provider cards (Cloud.ru, MiniMax, DeepSeek, GigaChat) collapse under a
|
||||
"More providers" details wrapper, but their inputs must stay mounted:
|
||||
settings.js applyInputValue has no null guard, so a missing input id
|
||||
breaks settings load. The wrapper auto-opens when configured."""
|
||||
|
|
@ -141,7 +141,7 @@ def test_settings_more_providers_collapse_keeps_inputs_mounted():
|
|||
css = _read("web/settings.css")
|
||||
|
||||
assert 'id="settings-more-providers"' in ui
|
||||
assert ui.count("advanced: true") == 3
|
||||
assert ui.count("advanced: true") == 4
|
||||
assert "PROVIDER_CARDS.filter((card) => !card.advanced)" in ui
|
||||
assert "PROVIDER_CARDS.filter((card) => card.advanced)" in ui
|
||||
assert "syncMoreProvidersDisclosure" in settings
|
||||
|
|
|
|||
|
|
@ -9,7 +9,11 @@ import json
|
|||
|
||||
import pytest
|
||||
|
||||
from ouroboros.provider_models import OPENAI_DIRECT_DEFAULTS, normalize_model_identity
|
||||
from ouroboros.provider_models import (
|
||||
OPENAI_DIRECT_DEFAULTS,
|
||||
normalize_deepseek_reasoning_effort,
|
||||
normalize_model_identity,
|
||||
)
|
||||
from ouroboros.request_wire_contract import canonical_sha256
|
||||
from ouroboros.request_wire_receipts import (
|
||||
WireCandidateSpec,
|
||||
|
|
@ -177,6 +181,7 @@ def test_exact_provider_canary_matrix_logical_turns_and_attempt_bound():
|
|||
("openai_direct_fallback", "openai::gpt-5.6-sol"),
|
||||
("anthropic_direct", "anthropic::claude-sonnet-5"),
|
||||
("minimax_direct", "minimax::MiniMax-M3"),
|
||||
("deepseek_direct", "deepseek::deepseek-v4-flash"),
|
||||
("cloudru_direct", "cloudru::zai-org/GLM-4.7"),
|
||||
("gigachat_direct", "gigachat::GigaChat-2-Max"),
|
||||
]
|
||||
|
|
@ -191,16 +196,17 @@ def test_exact_provider_canary_matrix_logical_turns_and_attempt_bound():
|
|||
"openai_direct_light",
|
||||
"openai_direct_fallback",
|
||||
"anthropic_direct",
|
||||
"deepseek_direct",
|
||||
}
|
||||
assert [row.canary_id for row in matrix if row.continue_to_final] == [
|
||||
"openai_direct_main"
|
||||
"openai_direct_main", "deepseek_direct"
|
||||
]
|
||||
assert [row.canary_id for row in matrix if not row.named_tool_choice] == [
|
||||
"gigachat_direct"
|
||||
]
|
||||
logical_turns = sum(1 + int(row.continue_to_final) for row in matrix)
|
||||
assert logical_turns == 13
|
||||
assert logical_turns * CANARY_EMPTY_RESPONSE_MAX_ATTEMPTS == 26
|
||||
assert logical_turns == 15
|
||||
assert logical_turns * CANARY_EMPTY_RESPONSE_MAX_ATTEMPTS == 30
|
||||
assert sum(
|
||||
1 + int(row.continue_to_final)
|
||||
for row in matrix
|
||||
|
|
@ -585,9 +591,23 @@ def _fake_usage(canary: ProviderCanary, ordinal: int):
|
|||
}
|
||||
if canary.expected_provider != "openai":
|
||||
if canary.reasoning_effort == "medium":
|
||||
applied_effort = canary.reasoning_effort
|
||||
if canary.expected_provider == "deepseek":
|
||||
forced = ordinal == 1 and canary.named_tool_choice
|
||||
applied_effort = (
|
||||
"none" if forced else normalize_deepseek_reasoning_effort(applied_effort)
|
||||
)
|
||||
usage["reasoning_effort_clamped"] = {
|
||||
"requested": canary.reasoning_effort,
|
||||
"applied": applied_effort,
|
||||
"reason": "provider_forced_tool_choice" if forced else "provider_wire_mapping",
|
||||
"model": canary.model.split("::", 1)[-1],
|
||||
}
|
||||
usage["request_wire"] = {
|
||||
"requested_effort": "medium",
|
||||
"applied_effort": "medium",
|
||||
"requested_effort": applied_effort,
|
||||
"applied_effort": applied_effort,
|
||||
# A continuation must bind a fresh physical candidate.
|
||||
"candidate_sha256": ("c" if ordinal == 1 else "d") * 64,
|
||||
}
|
||||
return usage
|
||||
usage["request_wire"] = {
|
||||
|
|
|
|||
|
|
@ -695,7 +695,7 @@ def test_response_log_and_accounting_expose_only_controlled_error(
|
|||
def test_provider_test_registry_is_derived_from_provider_defaults():
|
||||
assert provider_api._PROVIDER_TEST_KNOWN_IDS == {
|
||||
"openrouter", "openai", "anthropic", "cloudru", "gigachat", "minimax",
|
||||
"openai-compatible",
|
||||
"deepseek", "openai-compatible",
|
||||
}
|
||||
assert provider_api._PROVIDER_TEST_OVERRIDE_KEYS == provider_api.ALL_PROVIDER_CREDENTIAL_KEYS
|
||||
|
||||
|
|
|
|||
|
|
@ -64,7 +64,8 @@ import { installAltMenuSuppression, installDesktopShellLinkInterceptor } from '.
|
|||
modelsDirty: false,
|
||||
localSourceOpen: Boolean(INITIAL_STATE.localSource),
|
||||
moreProvidersOpen: Boolean(
|
||||
INITIAL_STATE.cloudruKey || INITIAL_STATE.minimaxKey || INITIAL_STATE.compatibleBaseUrl || INITIAL_STATE.compatibleApiKey,
|
||||
INITIAL_STATE.cloudruKey || INITIAL_STATE.minimaxKey || INITIAL_STATE.deepseekKey
|
||||
|| INITIAL_STATE.compatibleBaseUrl || INITIAL_STATE.compatibleApiKey,
|
||||
),
|
||||
localStatusText: 'Status: Offline',
|
||||
localStatusTone: 'muted',
|
||||
|
|
@ -144,6 +145,7 @@ import { installAltMenuSuppression, installDesktopShellLinkInterceptor } from '.
|
|||
['OPENAI_API_KEY', 'openai'],
|
||||
['CLOUDRU_FOUNDATION_MODELS_API_KEY', 'cloudru'],
|
||||
['MINIMAX_API_KEY', 'minimax'],
|
||||
['DEEPSEEK_API_KEY', 'deepseek'],
|
||||
['ANTHROPIC_API_KEY', 'anthropic'],
|
||||
].filter(([settingKey]) => configured[settingKey]);
|
||||
if (hasOpenrouter) return 'openrouter';
|
||||
|
|
@ -443,6 +445,7 @@ import { installAltMenuSuppression, installDesktopShellLinkInterceptor } from '.
|
|||
if (trim(state.openaiKey)) rows.splice(1, 0, ['OpenAI', 'configured']);
|
||||
if (trim(state.cloudruKey)) rows.splice(1, 0, ['Cloud.ru', 'configured']);
|
||||
if (trim(state.minimaxKey)) rows.splice(1, 0, ['MiniMax', 'configured']);
|
||||
if (trim(state.deepseekKey)) rows.splice(1, 0, ['DeepSeek', 'configured']);
|
||||
if (trim(state.anthropicKey)) rows.splice(1, 0, ['Anthropic', 'configured']);
|
||||
if (hasLocalModel()) {
|
||||
rows.splice(
|
||||
|
|
@ -656,7 +659,7 @@ import { installAltMenuSuppression, installDesktopShellLinkInterceptor } from '.
|
|||
note: slot.note,
|
||||
})).join('')}
|
||||
</div>
|
||||
<div class="wizard-inline-note">Direct providers use explicit <code>provider::model</code> values, including <code>minimax::MiniMax-M3</code> and <code>minimax::MiniMax-M2.7</code>. OpenAI-compatible endpoints use <code>openai-compatible::your-model-name</code>. Plain slash-form model IDs stay router-style by design.</div>
|
||||
<div class="wizard-inline-note">Direct providers use explicit <code>provider::model</code> values, including <code>minimax::MiniMax-M3</code>, <code>minimax::MiniMax-M2.7</code>, <code>deepseek::deepseek-v4-pro</code> and <code>deepseek::deepseek-v4-flash</code>. OpenAI-compatible endpoints use <code>openai-compatible::your-model-name</code>. Plain slash-form model IDs stay router-style by design.</div>
|
||||
`;
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -329,6 +329,8 @@ const SETTINGS_FALLBACK_MODELS = [
|
|||
'anthropic::claude-opus-5',
|
||||
'anthropic::claude-opus-4-6',
|
||||
'deepseek/deepseek-v4-pro',
|
||||
'deepseek::deepseek-v4-pro',
|
||||
'deepseek::deepseek-v4-flash',
|
||||
'minimax::MiniMax-M3',
|
||||
'minimax::MiniMax-M2.7',
|
||||
];
|
||||
|
|
@ -343,11 +345,12 @@ let settingsModelCatalogItems = SETTINGS_FALLBACK_MODELS.map((value) => ({ value
|
|||
* Exported for dependency-free node tests.
|
||||
*/
|
||||
export function moreProvidersCredentialConfigured({
|
||||
cloudruKey = '', minimaxKey = '', gigachatCredentials = '', gigachatUser = '', gigachatPassword = '',
|
||||
cloudruKey = '', minimaxKey = '', deepseekKey = '', gigachatCredentials = '', gigachatUser = '', gigachatPassword = '',
|
||||
} = {}) {
|
||||
const has = (v) => Boolean(String(v ?? '').trim());
|
||||
return has(cloudruKey)
|
||||
|| has(minimaxKey)
|
||||
|| has(deepseekKey)
|
||||
|| has(gigachatCredentials)
|
||||
|| (has(gigachatUser) && has(gigachatPassword));
|
||||
}
|
||||
|
|
@ -639,6 +642,7 @@ export function initSettings({ state, setBeforePageLeave, ws } = {}) {
|
|||
if (moreProvidersCredentialConfigured({
|
||||
cloudruKey: value('s-cloudru-key'),
|
||||
minimaxKey: value('s-minimax-key'),
|
||||
deepseekKey: value('s-deepseek-key'),
|
||||
gigachatCredentials: value('s-gigachat-credentials'),
|
||||
gigachatUser: value('s-gigachat-user'),
|
||||
gigachatPassword: value('s-gigachat-password'),
|
||||
|
|
|
|||
|
|
@ -122,6 +122,15 @@ const PROVIDER_CARDS = [
|
|||
testInputs: { 's-minimax-key': 'MINIMAX_API_KEY', 's-minimax-region': 'MINIMAX_REGION' },
|
||||
note: 'Use <code>minimax::MiniMax-M3</code> or <code>minimax::MiniMax-M2.7</code> in the Models tab. Leave Region empty for <code>global_en</code>; use <code>cn_zh</code> for the China endpoint.',
|
||||
},
|
||||
{
|
||||
id: 'deepseek', title: 'DeepSeek', icon: '', hint: 'Direct OpenAI-compatible runtime (v4 family)', advanced: true,
|
||||
fields: [
|
||||
{ id: 's-deepseek-key', settingKey: 'DEEPSEEK_API_KEY', label: 'API Key', placeholder: 'sk-...' },
|
||||
],
|
||||
testProvider: 'deepseek',
|
||||
testInputs: { 's-deepseek-key': 'DEEPSEEK_API_KEY' },
|
||||
note: 'Use <code>deepseek::deepseek-v4-pro</code> or <code>deepseek::deepseek-v4-flash</code> in the Models tab. Blocking deep/scope review in Max context mode additionally needs the owner 1M-window acknowledgement.',
|
||||
},
|
||||
{
|
||||
id: 'gigachat', title: 'GigaChat', icon: '/static/providers/gigachat.svg', hint: 'Sber GigaChat via the gigachat library', advanced: true,
|
||||
fields: [
|
||||
|
|
@ -238,6 +247,7 @@ export const SECRET_KEYS = [
|
|||
['GIGACHAT_PASSWORD', 'GigaChat Password (basic auth)', 'password'],
|
||||
['ANTHROPIC_API_KEY', 'Anthropic API Key', 'sk-ant-...'],
|
||||
['MINIMAX_API_KEY', 'MiniMax API Key', 'MiniMax key'],
|
||||
['DEEPSEEK_API_KEY', 'DeepSeek API Key', 'sk-...'],
|
||||
['GITHUB_TOKEN', 'GitHub Token', 'ghp_...'],
|
||||
['OUROBOROS_NETWORK_PASSWORD', 'Network Password', 'Required for LAN/Docker binds'],
|
||||
];
|
||||
|
|
@ -314,7 +324,7 @@ export function renderSettingsPage() {
|
|||
<details class="settings-more-providers" id="settings-more-providers">
|
||||
<summary>
|
||||
<span class="settings-provider-title"><span>More providers</span></span>
|
||||
<span class="settings-provider-hint">Cloud.ru Foundation Models, MiniMax, and GigaChat</span>
|
||||
<span class="settings-provider-hint">Cloud.ru Foundation Models, MiniMax, DeepSeek, and GigaChat</span>
|
||||
</summary>
|
||||
<div class="settings-more-providers-body">
|
||||
${PROVIDER_CARDS.filter((card) => card.advanced).map(providerSettingsCard).join('')}
|
||||
|
|
|
|||
|
|
@ -8,7 +8,7 @@ test('defaults-only values keep the More providers section closed', () => {
|
|||
// Base URLs / scope / TLS carry shipped defaults and are deliberately not
|
||||
// part of the predicate's inputs — an unconfigured install stays closed.
|
||||
assert.equal(moreProvidersCredentialConfigured({
|
||||
cloudruKey: '', minimaxKey: '', gigachatCredentials: '', gigachatUser: '', gigachatPassword: '',
|
||||
cloudruKey: '', minimaxKey: '', deepseekKey: '', gigachatCredentials: '', gigachatUser: '', gigachatPassword: '',
|
||||
}), false);
|
||||
assert.equal(moreProvidersCredentialConfigured({ cloudruKey: ' ' }), false);
|
||||
});
|
||||
|
|
@ -16,6 +16,7 @@ test('defaults-only values keep the More providers section closed', () => {
|
|||
test('each usable credential path opens the section', () => {
|
||||
assert.equal(moreProvidersCredentialConfigured({ cloudruKey: 'ck-123' }), true);
|
||||
assert.equal(moreProvidersCredentialConfigured({ minimaxKey: 'mm-123' }), true);
|
||||
assert.equal(moreProvidersCredentialConfigured({ deepseekKey: 'sk-ds' }), true);
|
||||
assert.equal(moreProvidersCredentialConfigured({ gigachatCredentials: 'base64pair' }), true);
|
||||
// Masked secrets from the server ("***set***" / prefixed) are non-empty
|
||||
// strings and count as configured.
|
||||
|
|
|
|||
|
|
@ -45,7 +45,7 @@ test('provider test responses apply only to the exact draft generation', () => {
|
|||
test('every provider test button warns that one charged request is sent', () => {
|
||||
const html = renderSettingsPage();
|
||||
const buttons = [...html.matchAll(/data-provider-test="[^"]+"[^>]*>/g)];
|
||||
assert.equal(buttons.length, 7);
|
||||
assert.equal(buttons.length, 8);
|
||||
for (const [button] of buttons) {
|
||||
assert.match(
|
||||
button,
|
||||
|
|
@ -58,10 +58,10 @@ test('provider actions use the shared status-first action row contract', () => {
|
|||
const html = renderSettingsPage();
|
||||
const rows = [...html.matchAll(/<div class="settings-action-row(?:"|\s)[\s\S]*?<\/div>/g)]
|
||||
.map(([row]) => row);
|
||||
// Seven provider probes plus the catalog action. The Claude-runtime
|
||||
// Eight provider probes plus the catalog action. The Claude-runtime
|
||||
// status/Repair panel is retired with its product surface (the advisory
|
||||
// pre-review runs on a configured routed model or agent session now).
|
||||
assert.equal(rows.length, 8, 'seven provider probes plus the catalog action');
|
||||
assert.equal(rows.length, 9, 'eight provider probes plus the catalog action');
|
||||
assert.doesNotMatch(html, /settings-claude-code/);
|
||||
assert.doesNotMatch(html, /settings-ghost-btn/);
|
||||
for (const row of rows) {
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue