diff --git a/devtools/benchmarks/common/secrets.py b/devtools/benchmarks/common/secrets.py index d36e2c310..8bb4b3374 100644 --- a/devtools/benchmarks/common/secrets.py +++ b/devtools/benchmarks/common/secrets.py @@ -15,6 +15,7 @@ SECRET_KEYS = ( "ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "DEEPSEEK_API_KEY", + "ZAI_API_KEY", "GITHUB_TOKEN", ) diff --git a/devtools/benchmarks/common/server_runner.py b/devtools/benchmarks/common/server_runner.py index 2a9ae7a5b..0716b5acf 100644 --- a/devtools/benchmarks/common/server_runner.py +++ b/devtools/benchmarks/common/server_runner.py @@ -112,6 +112,7 @@ _PROVIDER_ENV_KEYS = frozenset({ "OPENROUTER_API_KEY", "OPENAI_API_KEY", "OPENAI_COMPATIBLE_API_KEY", "CLOUDRU_FOUNDATION_MODELS_API_KEY", "ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "DEEPSEEK_API_KEY", + "ZAI_API_KEY", "GIGACHAT_CREDENTIALS", "GIGACHAT_PASSWORD", }) @@ -145,6 +146,7 @@ _AUTHORITATIVE_ENV_PREFIXES = ( "ANTHROPIC_", "MINIMAX_", "DEEPSEEK_", + "ZAI_", "CLOUDRU_", "GIGACHAT_", "CLAUDE_", diff --git a/devtools/benchmarks/programbench/run_programbench_e2e.py b/devtools/benchmarks/programbench/run_programbench_e2e.py index 921e5d2ae..13d661d66 100644 --- a/devtools/benchmarks/programbench/run_programbench_e2e.py +++ b/devtools/benchmarks/programbench/run_programbench_e2e.py @@ -115,6 +115,7 @@ def _active_direct_provider(settings: dict[str, Any]) -> str: ("minimax", "MINIMAX_API_KEY"), ("cloudru", "CLOUDRU_FOUNDATION_MODELS_API_KEY"), ("deepseek", "DEEPSEEK_API_KEY"), + ("zai", "ZAI_API_KEY"), ) if _setting_or_env(settings, key) ] diff --git a/devtools/benchmarks/terminal_bench/scrub_submission_secrets.py b/devtools/benchmarks/terminal_bench/scrub_submission_secrets.py index d06d98058..3f185f7a2 100644 --- a/devtools/benchmarks/terminal_bench/scrub_submission_secrets.py +++ b/devtools/benchmarks/terminal_bench/scrub_submission_secrets.py @@ -62,6 +62,7 @@ EXTRA_SECRET_FIELDS = ( "CLOUDRU_FOUNDATION_MODELS_API_KEY", "MINIMAX_API_KEY", "DEEPSEEK_API_KEY", + "ZAI_API_KEY", "GIGACHAT_PASSWORD", "OUROBOROS_NETWORK_PASSWORD", "TELEGRAM_BOT_TOKEN", diff --git a/docs/CREATING_SKILLS.md b/docs/CREATING_SKILLS.md index 11ea4b5f5..3abc644d4 100644 --- a/docs/CREATING_SKILLS.md +++ b/docs/CREATING_SKILLS.md @@ -603,7 +603,7 @@ Skills UI. ## Grants for protected keys and host permissions Some settings keys are protected: `OPENROUTER_API_KEY`, -`OPENAI_API_KEY`, `OPENAI_COMPATIBLE_API_KEY`, `ANTHROPIC_API_KEY`, `MINIMAX_API_KEY`, `DEEPSEEK_API_KEY`, +`OPENAI_API_KEY`, `OPENAI_COMPATIBLE_API_KEY`, `ANTHROPIC_API_KEY`, `MINIMAX_API_KEY`, `DEEPSEEK_API_KEY`, `ZAI_API_KEY`, `CLOUDRU_FOUNDATION_MODELS_API_KEY`, `GIGACHAT_CREDENTIALS`, `GIGACHAT_PASSWORD`, `TELEGRAM_BOT_TOKEN`, `GITHUB_TOKEN`, `OUROBOROS_NETWORK_PASSWORD`. These keys are NEVER forwarded to a skill by default, even when listed in diff --git a/docs/DESIGN.md b/docs/DESIGN.md index 1e27baf3f..4c98b0430 100644 --- a/docs/DESIGN.md +++ b/docs/DESIGN.md @@ -645,7 +645,11 @@ answer keep both forms readable. Anatomy, top to bottom: contain paragraphs, lists, checklists, tables and code; those blocks keep the shared rich-content gutter, rhythm and bounded code scrolling. The card does not infer a title from the first line or rewrite authored Markdown to - make it fit. + make it fit, and the question has no quiz-specific length cap. Directly under + it, a muted plain-text host line (`.chat-quiz-host-facts`, `--type-meta`, + `--text-meta`) states what only the host knows: the asking task, how its run + started and when the owner last wrote in this chat, with unknown facts said + as unknown; the line is absent when the card carries no `host_facts`. 3. **Stake** — optional one-liner (`At stake: …`), `--type-meta`, `--text-meta`. 4. **Options** — real owner actions: buttons with `--text-primary` labels, legible at rest; an optional per-option detail steps down to meta ink. diff --git a/docs/DOMAIN_MAP.md b/docs/DOMAIN_MAP.md index f5e9d930b..f804aec59 100644 --- a/docs/DOMAIN_MAP.md +++ b/docs/DOMAIN_MAP.md @@ -22,13 +22,13 @@ The manifest is the SSOT of the module→domain assignment (1:1, complete over t | D12 | Settings & configuration | 15 | 0 | | D13 | Safety, guards & runtime mode | 9 | 0 | | D14 | Skills & extensions | 56 | 0 | -| D15 | Memory, knowledge, consciousness & self-evolution | 22 | 0 | +| D15 | Memory, knowledge, consciousness & self-evolution | 23 | 0 | | D16 | Observability, usage accounting & cost | 11 | 0 | | D17 | Projects, workspaces & task results | 23 | 0 | | D18 | Launcher, packaging, platform & shared substrate | 15 | 0 | | D19 | Frozen contracts (ABI) | 10 | 0 | | D20 | Presence | 10 | 0 | -| **total** | | **574** | **0** | +| **total** | | **575** | **0** | ## Dependency direction matrix (strict, pinned) @@ -720,6 +720,7 @@ No function body (≥ 10 normalized lines) is shared verbatim across domains. Ne - `ouroboros/knowledge.py` - `ouroboros/memory.py` - `ouroboros/memory_journal_compaction.py` +- `ouroboros/memory_nomination_receipts.py` - `ouroboros/post_task_evolution.py` - `ouroboros/project_facts.py` - `ouroboros/reflection.py` diff --git a/docs/PERSISTENCE.md b/docs/PERSISTENCE.md index d0468b59f..5df607d40 100644 --- a/docs/PERSISTENCE.md +++ b/docs/PERSISTENCE.md @@ -22,7 +22,7 @@ scanned data-relative path to be covered by a row here (count-anchored both ways keys migrate). Governs subagent worktrees, headless/task drives, task trees, service logs, consumed schedule receipts, confirmed capability probes, delegate recovery/supervision sweeps, code_intel - and reconcile-failed prunes, memory-journal digesting and agent media. + reconcile-failed prunes, and agent media. Memory journals retain full new rows independently of this knob. - **Rotation** — `supervisor/state.py::rotate_jsonl_log_if_needed`: >800 KB → atomic rename to `archive/_.jsonl` under the append lock. Applied on the supervisor tick to `chat.jsonl`, `progress.jsonl`, @@ -143,10 +143,10 @@ scanned data-relative path to be covered by a row here (count-anchored both ways | `memory/scratchpad.md` + `scratchpad_blocks.json` | `ouroboros/memory.py` (derived, regenerated from blocks under lock) | none | bounded: 10 blocks, eviction journaled first (fail-closed) | regenerated; evicted history in journal | | `memory/WORLD.md` | `ouroboros/world_profiler.py` (write-once) | none | fixed | regenerates on restart — deletion IS the refresh mechanism | | `memory/registry.md`, `memory/deep_review.md` | `ouroboros/tools/memory_tools.py` (section RMW), `ouroboros/agent.py` (overwrite) | none | unbounded / last-wins — accepted | recreated lazily | -| `memory/dialogue_blocks.json` + `dialogue_meta.json` | `ouroboros/consolidator.py` (locked atomic) | none | bounded by era compression (10 blocks, oldest 4 compressed) | blocks: compressed biography irreproducible; meta: full re-consolidation (cost, not loss) | +| `memory/dialogue_blocks.json` + `dialogue_meta.json` | `ouroboros/consolidator.py`, `memory_nomination_receipts.py` (locked atomic) | `pending_knowledge_nominations` source-entry IDs; legacy `last_unpublished_nominations` preserved | blocks bounded by era compression (10 blocks, oldest 4); unresolved nomination index unbounded; no tool-level resolver yet, later success never retires old debt | blocks: compressed biography irreproducible; meta: cursor and unpublished-obligation evidence lost | | `memory/dialogue_summary.md` | none — legacy read-only (reader in context.py) | none | frozen | legacy artifact; nothing writes it | | `memory/knowledge/**` (topic .md + `index-full.md` + `patterns.md`) | `ouroboros/tools/knowledge.py`, `consolidator.py` (index rebuild), `reflection.py` (patterns CAS rewrite) | none | topic files unbounded — accepted (curated by consolidation); backlog topic merge-only fail-closed | recreated lazily; knowledge lost | -| `memory/*_journal.jsonl`, `memory/knowledge_history.jsonl`, `memory/knowledge/patterns_history.jsonl` | `ouroboros/memory.py`, `tools/control_runtime.py`, `tools/knowledge.py`, `reflection.py` — every append through the `append_jsonl` sidecar-lock seam | scratchpad journal: `type` rows; others unversioned full-text snapshots; digested rows carry `content_digested: true` | full old+new text only inside GC retention: older identity/knowledge/patterns rows go digest-only (sha256+len) at startup (`memory_journal_compaction.py`, under the append lock, unreadable lines byte-preserved); scratchpad journal keeps its own eviction contract | undo/provenance record lost (live .md survives); eviction/rewrite paths fail closed when journal append fails; digested history is irreversible by design | +| `memory/*_journal.jsonl`, `memory/knowledge_history.jsonl`, `memory/knowledge/patterns_history.jsonl` | `ouroboros/memory.py`, `tools/control_runtime.py`, `tools/knowledge.py`, `reflection.py` — every append through the `append_jsonl` sidecar-lock seam | scratchpad journal: `type` rows; others unversioned full-text snapshots; historical digested rows retain `content_digested: true` | complete new old+new snapshots are retained indefinitely; `memory_journal_compaction.py` is a read-only compatibility entry point, not a source rewriter; existing digest-only rows cannot be restored; the `memory_journal_observation` startup event gives byte sizes (or missing/unreadable) for the three named journals; scratchpad keeps its eviction journal | deleting the journals loses undo/provenance; eviction/rewrite paths fail closed when journal append fails; historically digested content remains irrecoverable | | `memory/owner_mailbox/.jsonl` + `.acks.jsonl` | `ouroboros/owner_mailbox.py` (append-only; revocation appends, reader resolves) | `kind` discriminator | lifecycle-bounded: unlinked at task terminal; a startup sweep unlinks mailboxes whose task has a SETTLED durable result (no result / non-terminal keeps the mailbox fail-closed) | undelivered owner directives + restart-surviving hurry latch lost; acks lost ⇒ re-delivery | ## 7. Skills payloads, tasks, uploads, projects, services @@ -180,13 +180,15 @@ scanned data-relative path to be covered by a row here (count-anchored both ways Always safe (pure caches, recreated): `state/pycache`, `state/code_intel`, `state/evolution_metrics_cache.json`, `playwright-browsers/`, `state/cx`, `state/betterleaks`, lock files, `state/server_port`. -Safe with bounded cost: `WORLD.md` (regenerates), `dialogue_meta.json` -(re-consolidation), `state/usage_import_watermark.json` (safe re-import), +Safe with bounded cost: `WORLD.md` (regenerates), +`state/usage_import_watermark.json` (safe re-import), `ui_preferences.json`, `auth_secret.key` (one re-login). Fail-closed losses (system stays correct, work/authority is forgone): skill state dirs, `advisory_review.json`, `capability_evidence.json`, `pending_restart_verify.json`. Dangerous (authority/history destruction): `settings.json`, `state/usage_attempts.jsonl`, `task_results/**`, `logs/events.jsonl`, -`memory/**`, `archive/**`, `observability/**`, `state/subagent_worktrees.json` +`memory/**` (including `dialogue_meta.json`: deletion erases cursor and pending +nomination obligations; re-consolidation cannot reconstruct the old IDs), +`archive/**`, `observability/**`, `state/subagent_worktrees.json` (leak), `claudexor/**`, `state/python-userbase` (real deps). diff --git a/docs/architecture/01-high-level-architecture.md b/docs/architecture/01-high-level-architecture.md index 3da7dcd89..6b4a830f0 100644 --- a/docs/architecture/01-high-level-architecture.md +++ b/docs/architecture/01-high-level-architecture.md @@ -161,7 +161,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de ├── delegate_source_coverage.py ← Oversized-work-order source custody: interval union/completeness, durable receipt bounds, replay-safe start binding — incomplete source cannot authorize a terminal PASS/apply; no alternate store ├── delegate_evidence.py ← Read-side execution evidence over custody rows (`task_execution_evidence`; `delegate_start_attempted` counts blocked and uncustodied attempts), the stamp writers `record_nanny_nudge_stamp`/`record_start_blocked`, `applied_access_profiles`, `acceptance_patch_dispositions` (cap 20, `unreviewed_delegated_apply`); an unreadable log is `evidence_read_failed`, never clean (§6 Delegated subagents; §11.1) ├── synthesis_cost_text.py ← Synthesis-prompt renderers for the pre-synthesis cost/outcome snapshot over the SSOT `cost_display` - ├── llm.py ← Multi-provider LLM routing (OpenRouter/OpenAI/compatible/Cloud.ru/MiniMax/DeepSeek/GigaChat/Anthropic); canonical conversations stay function-shaped while the physical-send seam delegates exact-route request adaptation to the request-wire leaves below + ├── llm.py ← Multi-provider LLM routing (OpenRouter/OpenAI/compatible/Cloud.ru/MiniMax/DeepSeek/Z.ai/GigaChat/Anthropic); canonical conversations stay function-shaped while the physical-send seam delegates exact-route request adaptation to the request-wire leaves below ├── llm_routing.py, llm_attempt.py, llm_messages.py, llm_capability_policy.py, llm_fallback.py, llm_pricing.py, llm_openai_compatible.py, llm_anthropic.py, llm_gigachat.py, llm_local.py, llm_claudexor.py, llm_substitution.py ← The client's leaves behind that facade: target resolution, client construction and route affinity; physical-attempt candidates and send-time prompt-cache policy; wire transcript shaping and the reasoning-artifact contract; capability metadata and the learned parameter/effort policy; the recovery ladder; live price catalogs (OpenRouter, Cloud.ru); the wire lanes — OpenAI-compatible, native Anthropic, GigaChat, local llama.cpp, caller-owned Claudexor model operations; which account a route must not prefer next and the refusal of a round ANOTHER model answered (§6 Context fitting, retry, and compaction; Caller-owned subscription model calls; §7 Direct-provider routes) ├── llm_stream.py ← Complete Chat Completions and native Messages SSE assembly inside one physical attempt; private wire/partial evidence and terminal framing (§6 Streams and transport waits) ├── net_transport.py ← Shared httpx transport construction for remote LLM clients; TCP-keepalive socket options (§6 Context fitting, retry, and compaction) @@ -186,11 +186,12 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de ├── consciousness_wake.py ← The wake-up MESSAGE (`prompts/CONSCIOUSNESS.md` rendered as the turn's USER message, cuts disclosed as `(+N more)`) and the origin/authority envelope `wake_task_metadata` ├── consciousness_authority.py ← The three autonomy levels of a wake (observe/act/full) and their consequences — `disabled_tools`, bound at dispatch only so the prompt prefix matches an owner turn's, `runtime_mode_cap=light` below Full, and Observe's argument-level narrowing of the mutating names it keeps (§6 Background consciousness and Evolution) ├── consciousness_allowance.py ← Rolling-24h consciousness spend read off the usage ledger; typed `allowance_unknown` on a read failure; read by the alarm and the single admission door in `supervisor/queue.py` - ├── room_consolidation.py ← Per-room memory: one Light draft + one source-grounded correction per room, deterministic assembly of typed room sections into one block/era; no cross-room LLM recombine, legacy blocks keep unknown provenance (§6) - ├── consolidator.py ← Dialogue consolidation with a generation-aware cursor; an unfindable generation appends a loud `[MEMORY GAP]` block, never a silent offset reset; `last_consolidation_error` / `last_unpublished_nominations` in `dialogue_meta.json` (§6 Durable memory and project focus) + ├── room_consolidation.py ← Per-room Light draft/correction and deterministic assembly; no cross-room LLM recombine (§6) + ├── consolidator.py ← Generation cursor, explicit `[MEMORY GAP]`, and knowledge nomination outcomes in `dialogue_meta.json` (§6) + ├── memory_nomination_receipts.py ← Source-addressed pending nominations; no cross-batch retirement (§6) ├── memory.py ← Scratchpad, identity, chat history ├── knowledge.py ← `ouroboros/knowledge.py`: linked-Markdown note addressing, exact source reads, generated shelf indexes for global and project knowledge, and revision-checked writes, so concurrent cognition cannot silently overwrite a newer note (§6 Durable memory and project focus) - ├── memory_journal_compaction.py ← Digest-only compaction of old memory-journal snapshots: the digest replaces the snapshots it summarizes, never a silent drop + ├── memory_journal_compaction.py ← Startup read-only size facts (`memory_journal_observation`); new history stays complete, old digests unrecoverable ├── project_facts.py ← project_id resolution (explicit `--project-id` or workspace-path hash); per-project knowledge dir `projects//knowledge` isolated from `memory/knowledge`; journal/workpad helpers ├── task_tree_ledger.py ← Append-only `data/task_trees//blackboard.jsonl`: EPHEMERAL typed swarm coordination (`tree_note`/`tree_read`), mirrored into the durable project journal at root completion; pruned on root terminal ├── projects_registry.py ← Durable `data/state/projects.json`: 80-char names, `active|deleting|tombstoned`; deletion preserves bindings/history/folder/memory; a tombstone blocks resurrection; reconcile NEVER prunes (§6 Project registry and lease) diff --git a/docs/architecture/02-startup-onboarding-flow.md b/docs/architecture/02-startup-onboarding-flow.md index 4a0e32560..8d5b80cc9 100644 --- a/docs/architecture/02-startup-onboarding-flow.md +++ b/docs/architecture/02-startup-onboarding-flow.md @@ -18,7 +18,7 @@ Completion is one HTTP conversation on every host — `POST /api/onboarding/comp Validation (`settings_setup_contract.validate_setup_payload`, shared by the desktop and web wizard through `onboarding_wizard.py`) is structural: at least one exposed remote configuration, selected managed model source or local model source; local-only setup routes at least one active lane locally; Main is required, while Light, Vision, Consciousness and Fallback keep inheritance/empty semantics and Heavy is readable only for bounded migration into an explicit API actor; enforcement and runtime mode are closed enums, budgets finite and positive, the MiniMax region closed, a Hugging Face local source needs a filename. Credential length is checked only on fields changed in the payload: rejecting an unchanged short legacy value would discard the whole form, including its own repair. -Provider readiness and provider defaulting are separate. With no OpenRouter, legacy OpenAI base or OpenAI-compatible endpoint, exactly one registered direct provider (OpenAI, Anthropic, Cloud.ru, GigaChat, MiniMax, DeepSeek) receives its provider-prefixed defaults and migration of untouched shipped/legacy slot values. Multiple direct providers stay owner-editable, and OpenRouter keeps router-style routing. An arbitrary OpenAI-compatible endpoint gets no guessed model ids — compatible servers have no universal safe name; the owner selects explicit `openai-compatible::...` routes. A local-source install with no remote provider clears only untouched shipped remote Light/Fallback values that would be unreachable. This is migration of defaults, not a model allowlist, and never proof the local server is running. +Provider readiness and provider defaulting are separate. With no OpenRouter, legacy OpenAI base or OpenAI-compatible endpoint, exactly one registered direct provider (OpenAI, Anthropic, Cloud.ru, GigaChat, MiniMax, DeepSeek, Z.ai) receives its provider-prefixed defaults and migration of untouched shipped/legacy slot values. Multiple direct providers stay owner-editable, and OpenRouter keeps router-style routing. An arbitrary OpenAI-compatible endpoint gets no guessed model ids — compatible servers have no universal safe name; the owner selects explicit `openai-compatible::...` routes. A local-source install with no remote provider clears only untouched shipped remote Light/Fallback values that would be unreachable. This is migration of defaults, not a model allowlist, and never proof the local server is running. `scripts/build_repo_bundle.py` creates the packaged seed only from a clean named checkout, writes a git bundle of that commit/tags, and records schema, version, source SHA, release tag, bundle hash and managed branch/remote metadata (the release-tag check itself: §8). The launcher validates the manifest fields and the bundle SHA-256 but no per-file member set: clone-time Git verification proves the manifest source object exists and checked-out HEAD equals it. diff --git a/docs/architecture/06-agent-core.md b/docs/architecture/06-agent-core.md index f666f7028..9a73ba0e4 100644 --- a/docs/architecture/06-agent-core.md +++ b/docs/architecture/06-agent-core.md @@ -557,9 +557,9 @@ Every IMPLICIT claim — the UI conversion, that admission, the reaper's retry a `context.py` assembles static governance, semi-stable memory, and dynamic task evidence without treating truncation as forgetting; the recent-activity sections are each task's OWN newest rows (progress 50 rendered; tools 20 selected, 10 rendered and 20 scanned for review markers; events 200 counted by type) through the bounded reader `jsonl_tail.py` (`Memory.read_task_recent`: a doubling live tail plus at most three newest archives), never a global tail filtered afterwards (issue #131), and their header's coverage line names the rows, the window and any unopened archives while `read_file` pages the rest; a subagent child gets the same three windows beside its `## Working sources` block, its tools and events read from its own execution drive (its worker rows; host-side rows such as waits stay in the canonical log, as the header says) and progress from the canonical log; the Development context matrix and `context_layout.py` own which reference form is resident. When the rendered scratchpad exceeds `SCRATCHPAD_SECTION_BUDGET_CHARS`, `context.py` keeps the newest whole blocks that fit and drops the oldest behind an in-band gap marker naming `memory/scratchpad.md` as the live source; no block is retired by a context build, and scratchpad replacement keeps its explicit summary and source-journal provenance. -`consolidator.py` publishes a dialogue block only after every part succeeds, retaining raw generations and their cursor; an unfindable generation appends `[MEMORY GAP]`, never silently resets the offset. `context_fit` measures full Light requests against fresh route/account capacity and calibrated density (`llm_local.local_context_limits` owns local output reservation; missing/stale evidence remains unknown). `room_consolidation.py` drafts and corrects each room separately, then deterministically assembles the sections. Episodic text is grounded in that room's source; cumulative knowledge replacements require the complete current note plus the episode in BOTH stages. Corrected entries bind to the corrector's complete delivered read, never the draft's revision credit. A narrow or older episode cannot negate prior facts or later receipts; supported corrections and removals remain model judgment. Range reads, authored views, CAS and old/new history remain the publication path; no new stage or store. Failed, empty or truncated correction withholds the chunk, cursor and nominations. +`consolidator.py` publishes a block and advances its generation-aware cursor only after complete room draft and correction; a missing generation appends `[MEMORY GAP]` instead of resetting. `context_fit` measures Light against fresh route/account capacity and calibrated density; `llm_local` owns the local output reserve, and absent evidence stays unknown. `room_consolidation.py` processes each room separately and assembles sections deterministically. Both knowledge stages receive the entire current note and source episode; the corrector's complete read, not the draft's, binds revised entries. Older episodes cannot negate newer facts; model judgment governs supported corrections. Source range reads and CAS preserve old/new history. Startup compaction is a no-op; earlier digests cannot be reversed. A failed correction withholds the chunk and cursor. -Oversized source splits without clipping, including inside an entry; continuation context stays outside source bytes. A real refusal records its source hash and strictly smaller same-route byte bound in `dialogue_meta.json` (`consolidation_retry`); changed source, route, capacity or output reserve invalidates it. Era compression regroups each recorded room across blocks and reassembles deterministically; legacy untyped blocks remain explicitly unknown provenance. Failed/overflowed eras preserve old blocks. Failures retain `last_consolidation_error`, cleared by an advance without a new failure; incomplete knowledge publication retains `last_unpublished_nominations`. Unknown spend remains nullable and control/resource/unknown model errors retain `propagate_model_error` semantics. +Oversized sources split without clipping, including within an entry. `consolidation_retry` records source hash and a smaller same-route bound, invalidated by source/route/capacity/reserve changes. Era compression regroups rooms deterministically; failed eras retain blocks and legacy provenance stays unknown. `last_consolidation_error` clears after a failure-free advance. `pending_knowledge_nominations` records each source entry BEFORE note publication; an unrelated successful batch cannot remove one. Legacy `last_unpublished_nominations` persists. Health shows three distinct abbreviated source+position IDs and omitted count; full proposals live in `knowledge_history.jsonl`. Unreadable meta preserves debts and warns in Health; Nano pressure records no-progress instead of aborting Main. Invalid legacy receipts warn separately. Old digests and debts need explicit resolution. Spend remains nullable; model-control errors follow `propagate_model_error`. Consolidation labels every chronological source message through `dialogue_provenance.RoomLabelResolver`, using the actual `chat_id`, never lineage `project_id`; one read-only registry snapshot supplies the window. Main is named only for the actual Main id, a resolved project uses its current registry name and stable chat id, and missing, unknown or ambiguous rooms stay explicit. Ephemeral formatter offsets carry the original room/author/direction/transport header into split continuations without parsing message bodies or duplicating their bytes. Room draft and correction prompts require meaningful decisions, approvals, outcomes and unresolved commitments of that room, retaining source distinctions (who decided, what was authorized, what stays owed) and one first-person Ouroboros voice. Length adapts to content within the existing output-token ceiling; no per-room word quota, semantic gate or absent room is imposed. Labels establish provenance, not summary success. The mixed Main recent view opts into the same labels; focused Project rendering, membership and explicit `chat_history` retain their existing behavior and bytes. diff --git a/docs/architecture/07-configuration.md b/docs/architecture/07-configuration.md index 7c8845311..e223c337c 100644 --- a/docs/architecture/07-configuration.md +++ b/docs/architecture/07-configuration.md @@ -67,6 +67,8 @@ A registry of `config.SETTINGS_DEFAULTS` (exact defaults canonical in `settings_ | MINIMAX_API_KEY | "" | MiniMax credential | | MINIMAX_REGION | "" | MiniMax region (empty resolves `global_en`) | | DEEPSEEK_API_KEY | "" | Optional DeepSeek direct key (`deepseek::...` values; route below) | +| ZAI_API_KEY | "" | Optional Z.ai (GLM) direct key (`zai::...` values; route below) | +| ZAI_PLAN | "" | Z.ai endpoint plan: empty/`payg` = pay-as-you-go, `coding` = Coding Plan | | OUROBOROS_NETWORK_PASSWORD | "" | Non-localhost HTTP gate password (`server_auth.py`; unset only warns — §8) | | OUROBOROS_SERVER_HOST | 127.0.0.1 | HTTP bind host (`0.0.0.0` for Docker/non-loopback) | | OUROBOROS_UPDATE_CHANNEL | `stable` | Update channel: stable/qa/development (§8) | @@ -219,10 +221,12 @@ A registry of `config.SETTINGS_DEFAULTS` (exact defaults canonical in `settings_ #### Direct-provider routes -Direct-provider review fallback (legacy name: OpenAI-only review fallback): with exactly one official direct provider configured, `config.get_review_models()` compiles that provider's declarative reviewer-role sequence from provider-prefixed model IDs. Scope covers official OpenAI, Anthropic, MiniMax, DeepSeek, Cloud.ru, and GigaChat; OpenRouter, legacy-base, OpenAI-compatible and mixed configurations stay outside. Per-provider role coverage differs — three independent Main slots down to one role model for every slot (`provider_models.compute_direct_review_models_fallback`). `_exclusive_direct_remote_provider_env` returns empty when OpenRouter, legacy `OPENAI_BASE_URL`, OpenAI-compatible keys or several direct providers are present, and the fallback requires `provider_models.migrate_model_value` to make the main model already start with the exclusive provider prefix, so free text cannot silently enter a single-provider route (DEVELOPMENT "Provider Independence"). +Direct-provider review fallback (legacy name: OpenAI-only review fallback): with exactly one official direct provider configured, `config.get_review_models()` compiles that provider's declarative reviewer-role sequence from provider-prefixed model IDs. Scope covers official OpenAI, Anthropic, MiniMax, DeepSeek, Z.ai, Cloud.ru, and GigaChat; OpenRouter, legacy-base, OpenAI-compatible and mixed configurations stay outside. Per-provider role coverage differs — three independent Main slots down to one role model for every slot (`provider_models.compute_direct_review_models_fallback`). `_exclusive_direct_remote_provider_env` returns empty when OpenRouter, legacy `OPENAI_BASE_URL`, OpenAI-compatible keys or several direct providers are present, and the fallback requires `provider_models.migrate_model_value` to make the main model already start with the exclusive provider prefix, so free text cannot silently enter a single-provider route (DEVELOPMENT "Provider Independence"). DeepSeek (`deepseek::`): the OpenAI-compatible endpoint is the fixed constant `provider_models.DEEPSEEK_BASE_URL`; a proxy or mirror belongs to `openai-compatible::`, and slash-form `deepseek/...` stays OpenRouter. The canonical effort scale is projected onto the provider's wire dialect at the send boundary, a forced tool choice is served with thinking disabled (thinking accepts only `auto`/`none`), and every tier change is disclosed as `reasoning_effort_clamped` (projection table: `provider_models.DEEPSEEK_REASONING_EFFORT_ALIASES`). `reasoning_content` stays on CANONICAL assistant turns for strict v4 replay (an explicit empty string marks a turn produced without provider reasoning); other lanes strip the field from their physical send copy and cross-family switches scrub it. On the send copy only, system/assistant/tool content arrays are flattened to strings — the API accepts arrays on user turns alone (`llm_openai_compatible.py`). Caching is automatic, cost stays nullable without an exact catalog, and the 1M context claim needs route-fingerprinted evidence or owner acknowledgement. +Z.ai (`zai::`): GLM through Z.ai's OpenAI-compatible API. `ZAI_PLAN` selects the endpoint (`provider_models.resolve_zai_base_url`: empty/`payg` = `api.z.ai/api/paas/v4`, `coding` = the Coding Plan endpoint); a proxy or the China host belongs to `openai-compatible::`, and slash-form `zai/...` stays OpenRouter. An absent `reasoning_effort` is served at the provider's MAX, so the canonical scale is always projected onto Z.ai's own `low/high/max` enum at the send boundary (`provider_models.ZAI_REASONING_EFFORT_ALIASES`: none/minimal→low, medium→high, xhigh/ultra→max — GLM-5.3 rejects every other value and cannot disable thinking, HTTP 400 code 1210) and every tier change is disclosed as `reasoning_effort_clamped`; a forced tool choice keeps its tier (no DeepSeek-style suppression). HTTP 429 code 1113 "Insufficient balance" is billing (also a Coding Plan key on the pay-as-you-go endpoint), so the provider Test reports it as `No credits`, not "Rate limited". + GigaChat (`gigachat::`): the native `gigachat` library, not OpenAI-compatible — OpenAI `tools` map to GigaChat `functions`, one `function_call` per turn (parallel `tool_calls` collapse to the first), `tool` results become role `function` and must be valid JSON (plain text is wrapped as `{"result": ...}`), and `system` must come first, so later system-reminders demote to `user` (`llm.py::_chat_gigachat`). `reasoning_effort` is deliberately omitted: hidden reasoning can consume the whole output budget and return empty content. No live cost source exists, so cost stays nullable/unknown, never a hand-maintained tariff. A GigaChat scope row runs native retrieval on its own window, with at most one function call per turn. Reading gaps are diagnostic and never remove its response from quorum; the agent decides whether more reading is needed. Missing inspection tools still produce `native_inspection_unavailable`, not a completed review, and the blocking triad continues to review the full staged diff (§6 Review stack). --- diff --git a/docs/architecture/11-frozen-contracts-v1.md b/docs/architecture/11-frozen-contracts-v1.md index 51e65a950..f1b4fd640 100644 --- a/docs/architecture/11-frozen-contracts-v1.md +++ b/docs/architecture/11-frozen-contracts-v1.md @@ -28,7 +28,7 @@ This chapter owns the ABI promise: which typed shapes and their parsing, normali | `ChatOutbound.initiator` — additive origin label of a self-initiated turn (`"consciousness"` on every frame and chat/progress/summary row of a consciousness wake-up and of the roots it starts; absent on an owner's turn); stamped by the turn's own event queue and the agent's frame meta, persisted by `log_chat`/the authored summary row/the task result, replayed by history on each row | `ouroboros/gateway/contracts.py`, `supervisor/log_addressing.py`, `ouroboros/subagent_messages.py`, `supervisor/message_bus.py`, `ouroboros/gateway/history.py`, `web/modules/api_types.js` | `tests/test_consciousness_initiator_label.py`, `tests/test_consciousness_wake_lane.py`, `web/tests/consciousness_label.test.js` | | `project_thread` stamp on all seven outbound frame types, stamped at the message-bus broadcast choke; a stamped frame is never adopted by Main (`chat_activity.mainThreadAccepts`) | `supervisor/message_bus.py`, `ouroboros/projects_registry.py` | `tests/test_message_bus.py`, `web/tests/chat_thread_routing.test.js` | | Media/link envelopes — media `task_id`/`size_bytes`/`download_url`; `LinkAction {label,url}` with at most twelve absolute HTTP(S) actions; `links` in `WS_MESSAGE_TYPES`; `chat.links` host topic | `ouroboros/gateway/contracts.py`, `ouroboros/tools/core.py`, `ouroboros/event_bus.py` | `tests/test_contracts.py` | -| Owner quiz ABI — `QuizOption {label, detail?, recommended?}` (the asker marks its recommendation on that option; the web card badges it, Telegram stars its button, the durable block keeps `recommended_index`), `QuizOutbound` (quiz_id, question, options, stake, `assumption` (required for optional clarification), additive `wait_for_answer` for a live pooled or ordinary-conversation root that must wait, lifecycle state open/answered/expired_terminal/superseded), separate `QuizStateOutbound` discriminator, `chat.quiz` host topic; the producer is the one escalation verb `escalate(question, options, stake, assumption, wait_for_answer=False, max_wait_minutes=None)` (the bound applies to a required wait only, never past the task's own deadline: the wait resumes with a system notice and the card stays open; named on an optional question it takes the omitted path and the asker's receipt says so) — a ROOT asks the owner, a SUBAGENT delivers a typed frame to its nearest LIVE ancestor, which answers via `forward_to_worker` or escalates verbatim, so the owner sees only what no ancestor answered; answers arrive through the ONE ingress `POST /api/decisions` (family ids `quiz:{task_id}:{quiz_id}`, `routing:{client_message_id}:{routing_token}`; `interaction:` reserved), request-id idempotent, first answer wins, validated against the STORED options; `option_index` is optional for the quiz family alone — a comment-only answer writes NO `answered_index`, because a stored 0 would replay as "chose the first option"; injected as the typed `KIND_QUIZ_ANSWER` mailbox control and broadcast as `quiz_state` (carrying the recorded `comment` when the owner answered in their own words, so the live card shows `Owner's answer:` exactly as replay does); expiry is structural only (the task-done seam flips open quizzes to `expired_terminal`, and the SAME reconcile closes the paired `owner_wait` so a terminal task never projects `quiz=expired_terminal` beside `owner_wait=waiting`; `owner_wait.set_owner_wait`'s refusal to continue waiting on a terminal result is preserved, not caught); a LATE answer to such an expired card is nevertheless ACCEPTED at the same ingress (the projection records `answered_after_terminal`) and, because no mailbox will ever be drained, is delivered as the owner's OWN message into the card's chat through the named ingress `supervisor.message_bus.accept_local_message`, idempotent on `client_message_id = quiz_late_answer::`, its provenance in the message's own `late_answer` metadata rather than a substituted `client_surface`; the 2xx says `forwarded` so no surface claims a delivery that did not happen, and 409 is left for what is genuinely settled (an already answered card, a non-root addressee). History replay merges the projection state | `ouroboros/gateway/contracts.py`, `ouroboros/gateway/task_decision.py`, `ouroboros/owner_quiz.py`, `ouroboros/tools/core.py` | `tests/test_gateway_parity.py`, `tests/test_quiz_display.py`, `tests/test_quiz_answer.py`, `web/tests/chat_decision.test.js` | +| Owner quiz ABI — `QuizOption {label, detail?, recommended?}` (the asker marks its recommendation on that option; the web card badges it, Telegram stars its button, the durable block keeps `recommended_index`), `QuizOutbound` (quiz_id, question, options, stake, `assumption` (required for optional clarification), additive `wait_for_answer` for a live pooled or ordinary-conversation root that must wait, lifecycle state open/answered/expired_terminal/superseded, optional host-written `host_facts` sentence, also on history rows and Main pointers), separate `QuizStateOutbound` discriminator, `chat.quiz` host topic (+ optional event-only `project_name`); the producer is the one escalation verb `escalate(question, options, stake, assumption, wait_for_answer=False, max_wait_minutes=None)` (the bound applies to a required wait only, never past the task's own deadline: the wait resumes with a system notice and the card stays open; named on an optional question it takes the omitted path and the asker's receipt says so) — a ROOT asks the owner, a SUBAGENT delivers a typed frame to its nearest LIVE ancestor, which answers via `forward_to_worker` or escalates verbatim, so the owner sees only what no ancestor answered; answers arrive through the ONE ingress `POST /api/decisions` (family ids `quiz:{task_id}:{quiz_id}`, `routing:{client_message_id}:{routing_token}`; `interaction:` reserved), request-id idempotent, first answer wins, validated against the STORED options; `option_index` is optional for the quiz family alone — a comment-only answer writes NO `answered_index`, because a stored 0 would replay as "chose the first option"; injected as the typed `KIND_QUIZ_ANSWER` mailbox control and broadcast as `quiz_state` (carrying the recorded `comment` when the owner answered in their own words, so the live card shows `Owner's answer:` exactly as replay does); expiry is structural only (the task-done seam flips open quizzes to `expired_terminal`, and the SAME reconcile closes the paired `owner_wait` so a terminal task never projects `quiz=expired_terminal` beside `owner_wait=waiting`; `owner_wait.set_owner_wait`'s refusal to continue waiting on a terminal result is preserved, not caught); a LATE answer to such an expired card is nevertheless ACCEPTED at the same ingress (the projection records `answered_after_terminal`) and, because no mailbox will ever be drained, is delivered as the owner's OWN message into the card's chat through the named ingress `supervisor.message_bus.accept_local_message`, idempotent on `client_message_id = quiz_late_answer::`, its provenance in the message's own `late_answer` metadata rather than a substituted `client_surface`; the 2xx says `forwarded` so no surface claims a delivery that did not happen, and 409 is left for what is genuinely settled (an already answered card, a non-root addressee). History replay merges the projection state | `ouroboros/gateway/contracts.py`, `ouroboros/gateway/task_decision.py`, `ouroboros/owner_quiz.py`, `ouroboros/tools/core.py` | `tests/test_gateway_parity.py`, `tests/test_quiz_display.py`, `tests/test_quiz_answer.py`, `web/tests/chat_decision.test.js` | | Managed update ABI — preflight, `UpdateMergePlan`, pinned apply, process-local `update_progress`, `update_progress_changed` invalidation and boot-only `update_status_ready` | `ouroboros/gateway/contracts.py` | `tests/test_update_apply_routing.py` | | `ChatOutbound.review_projection` — bounded actor findings via `utils.truncate_review_artifact`, at most `MAX_PROJECTED_ACTOR_FINDINGS` rows (`review_execution_projection.py`) | `ouroboros/gateway/contracts.py` | `tests/test_review_substrate_v2.py`, `web/tests/review_truth.test.js` | | Skill preflight statuses — `preflight_failed` is fresh-only; a stale failure surfaces as `preflight_failed_stale`; absence means the caller could not know | `ouroboros/skill_review_status.py` | `tests/test_skill_preflight_repair.py`, `web/tests/skill_preflight_repair.test.js` | diff --git a/docs/development/02-naming-and-boundaries.md b/docs/development/02-naming-and-boundaries.md index 54c93cb43..3f46b8685 100644 --- a/docs/development/02-naming-and-boundaries.md +++ b/docs/development/02-naming-and-boundaries.md @@ -385,6 +385,7 @@ rows — review-only maintenance. | `ouroboros/reviewer_slot_config.py::_ACCEPTANCE_API_PANEL_MEASURED` | Historical API-panel comparison: approximately 12 s / $0.07 per model row per task (median of the 2026-09-01 OSWorld traces); 75 s / $0.82 for a three-row panel on ProgramBench | Workload and route dependent | The named measurement constant used by the one-time delivery disclosure | Repeat the same workload with recorded model, route and usage | An old comparison can be mistaken for a current tariff or a subscription-cost estimate | Keep the date and workload visible; current usage owns money, and session delivery spends subscription time | | `ouroboros/llm_claudexor.py::cache_key_for_model` | The 2026-09-17 measurement found Codex prefix reuse across conversations requires one `prompt_cache_key` + `session_id`, while per-conversation turn states remain valid under that shared session | Provider dependent | Dated measurement beside the key derivation | Re-measure cache reads and turn state across two conversations | A stale positive pays cold prefixes or breaks turn state | Re-measure before changing the key scope | | `ouroboros/llm_openai_compatible.py` DeepSeek send projection | The 2026-09-03 probe found thinking accepts only `auto`/`none` tool choice; required/named calls returned 400 on both probed v4 models | Provider dependent | Dated probe recorded beside the send projection and its transport tests | Re-probe the exact endpoint/model when that dialect changes | Removing the projection too early breaks forced calls; keeping it after a provider change may suppress supported thinking | Revalidate the wire contract before changing the projection; keep its effect disclosed | +| `ouroboros/provider_models.py::ZAI_REASONING_EFFORT_ALIASES` (Z.ai send projection) | The 2026-09-21 contributor probe (PR #1207, Coding Plan key, glm-5.3): only `low`/`high`/`max` are accepted, an absent tier is served at max, thinking cannot be disabled (400 code 1210), and forced tool_choice works with thinking on; GLM-5.2 accepts the wider scale | Provider dependent | Dated probe recorded beside the projection and its tests | Re-probe the exact endpoint/model when Z.ai changes the enum or a GLM release changes semantics | Dropping the projection bills every call at max; a stale one rejects tiers the provider would accept | Revalidate the wire contract before changing the projection; keep its effect disclosed | ### Provider Independence @@ -435,9 +436,9 @@ slug, not an official OpenAI model id, so a direct OpenAI Chat slot uses the pla Sol id (the slug in Chat Completions is a guaranteed 404) — a compatibility constraint, not a mutable capability table; direct OpenAI tool conversations stay on Chat Completions and a model-name prefix is never admission authority; -DeepSeek is the second effort-carrying route, its `reasoning_effort` keyed on the -provider id rather than a name prefix or capability field, so a hand-built target -cannot silently drop it; direct Anthropic is the deliberate exception to a purely +DeepSeek and Z.ai carry `reasoning_effort` through provider-specific projections +keyed on the provider id rather than a name prefix or capability field, so a +hand-built target cannot silently drop it; direct Anthropic is the deliberate exception to a purely reconstructed provider transcript, and no effort-to-`budget_tokens` policy is synthesized (ARCHITECTURE §6 "Context fitting, retry, and compaction", ARCHITECTURE §7 "LLM output token budgets"). A provider-specific optional feature may be unavailable elsewhere, but diff --git a/docs/inventories/DATA_LAYOUT_INVENTORY.md b/docs/inventories/DATA_LAYOUT_INVENTORY.md index 3d2743af0..5d89f3a6f 100644 --- a/docs/inventories/DATA_LAYOUT_INVENTORY.md +++ b/docs/inventories/DATA_LAYOUT_INVENTORY.md @@ -2,7 +2,7 @@ Machine extraction of the `docs/ARCHITECTURE.md` "Data layout (`~/Ouroboros/`)" tree — the durable-file orientation carrier (this tree's counterpart of the reference PERSISTENCE_OWNERS derivation checklist) — regenerated by `python scripts/regenerate_inventories.py`. Do not edit. Every entry is probed against reality: repo entries must exist as tracked paths; data-plane entries must appear as a literal in the runtime sources that construct them. A durable file renamed or removed in code while its tree row survives = red (`tests/test_generated_inventories.py`). -Source: `docs/architecture/01-high-level-architecture.md`, physical LF lines 602-691; UTF-8 SHA-256 `1839b923a939496f18c4d428807d8c876ca14422dd09c33001bf39a2b4a0e0ba`. +Source: `docs/architecture/01-high-level-architecture.md`, physical LF lines 603-692; UTF-8 SHA-256 `7954873ca14387e7d615b148fc6a97444eb6ff922837f3a33a3bedc034362b3f`. - entries: **79** (code-ref: 72, repo-dir: 6, repo-path: 1) diff --git a/docs/inventories/FACADE_INVENTORY.md b/docs/inventories/FACADE_INVENTORY.md index a5a32adf2..0fd0ae01e 100644 --- a/docs/inventories/FACADE_INVENTORY.md +++ b/docs/inventories/FACADE_INVENTORY.md @@ -2,7 +2,7 @@ AST-derived inventory of compatibility facades, regenerated by `python scripts/regenerate_inventories.py`. Do not edit. A facade row is any runtime module whose top-level `from import ...` statements carry the `noqa: F401` re-export marker — the codebase's declared "this binding exists for its binding, not for this module's own use" convention (reference FACADE_CONSUMERS method). Leaf domains come from `ouroboros/domains.toml`; a leaf outside the facade's domain is marked ✗ (that edge also appears in the manifest's pinned direction matrix). `tests/test_generated_inventories.py` pins byte-identity, so any re-export surface change must regenerate this file. -- facade modules: **63**; marked re-export bindings: **2349**; cross-domain facade→leaf pairs: **133** +- facade modules: **63**; marked re-export bindings: **2350**; cross-domain facade→leaf pairs: **133** | facade | domain | bindings | leaves | |---|---|---:|---| @@ -43,7 +43,7 @@ AST-derived inventory of compatibility facades, regenerated by `python scripts/r | `ouroboros/tool_access.py` | D04 | 43 | `ouroboros/contracts/task_constraint.py` (2 ✗D19)
`ouroboros/tool_access_paths.py` (10)
`ouroboros/tool_access_roots.py` (9)
`ouroboros/tool_access_types.py` (15)
`ouroboros/tool_access_user_files.py` (5)
`ouroboros/tool_capabilities.py` (2) | | `ouroboros/tools/claude_advisory_review.py` | D06 | 50 | `ouroboros/commit_admission.py` (3)
`ouroboros/config.py` (1 ✗D12)
`ouroboros/deadline_utils.py` (1 ✗D01)
`ouroboros/skill_review_status.py` (1 ✗D14)
`ouroboros/tools/preflight_review_prompt.py` (7)
`ouroboros/tools/preflight_review_run.py` (19)
`ouroboros/tools/review_helpers.py` (16)
`ouroboros/triad_review.py` (2) | | `ouroboros/tools/control.py` | D08 | 113 | `ouroboros/config.py` (4 ✗D12)
`ouroboros/contracts/task_contract.py` (3 ✗D19)
`ouroboros/depth_evidence.py` (1 ✗D07)
`ouroboros/headless.py` (2 ✗D17)
`ouroboros/outcomes.py` (1 ✗D01)
`ouroboros/subagent_runtime.py` (3 ✗D07)
`ouroboros/subagents.py` (2 ✗D07)
`ouroboros/task_results.py` (5 ✗D17)
`ouroboros/task_status.py` (2 ✗D17)
`ouroboros/tool_capabilities.py` (2 ✗D04)
`ouroboros/tools/control_delegation.py` (8 ✗D07)
`ouroboros/tools/control_events.py` (9)
`ouroboros/tools/control_routing.py` (10)
`ouroboros/tools/control_runtime.py` (13)
`ouroboros/tools/control_scheduling.py` (18 ✗D07)
`ouroboros/tools/control_subagent_spec.py` (6 ✗D07)
`ouroboros/tools/control_task_results.py` (15 ✗D07)
`ouroboros/tools/registry.py` (4 ✗D04)
`ouroboros/utils.py` (5 ✗D18) | -| `ouroboros/tools/core.py` | D05 | 85 | `ouroboros/code_search_rg.py` (5)
`ouroboros/contracts/skill_payload_policy.py` (13 ✗D19)
`ouroboros/project_facts.py` (1 ✗D15)
`ouroboros/tool_access.py` (9 ✗D04)
`ouroboros/tools/core_artifacts.py` (17)
`ouroboros/tools/core_file_tools.py` (32)
`ouroboros/tools/registry.py` (3 ✗D04)
`ouroboros/utils.py` (5 ✗D18) | +| `ouroboros/tools/core.py` | D05 | 86 | `ouroboros/code_search_rg.py` (5)
`ouroboros/contracts/skill_payload_policy.py` (13 ✗D19)
`ouroboros/project_facts.py` (1 ✗D15)
`ouroboros/tool_access.py` (9 ✗D04)
`ouroboros/tools/core_artifacts.py` (18)
`ouroboros/tools/core_file_tools.py` (32)
`ouroboros/tools/registry.py` (3 ✗D04)
`ouroboros/utils.py` (5 ✗D18) | | `ouroboros/tools/core_file_tools.py` | D05 | 8 | `ouroboros/credential_shapes.py` (1 ✗D13)
`ouroboros/tools/core_secret_paths.py` (7) | | `ouroboros/tools/delegate.py` | D07 | 62 | `ouroboros/delegate_containment.py` (5)
`ouroboros/delegate_interactions.py` (8)
`ouroboros/delegate_output.py` (14)
`ouroboros/delegate_shared.py` (6)
`ouroboros/delegate_source_coverage.py` (3)
`ouroboros/subagent_runtime.py` (2)
`ouroboros/subagent_work_order.py` (1)
`ouroboros/tools/delegate_integration.py` (14)
`ouroboros/tools/delegate_terminal_evidence.py` (9) | | `ouroboros/tools/delegate_integration.py` | D07 | 10 | `ouroboros/delegate_target_drift.py` (3)
`ouroboros/tools/delegate_payload_patch.py` (7) | diff --git a/docs/inventories/FROZEN_CONTRACTS_INVENTORY.md b/docs/inventories/FROZEN_CONTRACTS_INVENTORY.md index c39b7c927..effbc5075 100644 --- a/docs/inventories/FROZEN_CONTRACTS_INVENTORY.md +++ b/docs/inventories/FROZEN_CONTRACTS_INVENTORY.md @@ -2,7 +2,7 @@ Machine extraction of `docs/ARCHITECTURE.md` §11.1 (the frozen-ABI SSOT), regenerated by `python scripts/regenerate_inventories.py`. Do not edit — edit the owning chapter named in the Source line and regenerate; `tests/test_generated_inventories.py` pins byte-identity and the resolution invariants (a §11.1 row whose owner or anchor file disappeared from the tree = red). -Source: `docs/architecture/11-frozen-contracts-v1.md`, physical LF lines 7-40; UTF-8 SHA-256 `902c0a2935eb46a9cea18f4d0405f27e5facf7e400c7c8efcd914dafca74ffd4`. +Source: `docs/architecture/11-frozen-contracts-v1.md`, physical LF lines 7-40; UTF-8 SHA-256 `f6567210a91c233940f20d8118e4fac610e62d653581d969046b11e09d82d526`. - table rows: **29** - browser-envelope prose owners: diff --git a/ouroboros/capability_evidence.py b/ouroboros/capability_evidence.py index 23270ef6b..457d2e9f3 100644 --- a/ouroboros/capability_evidence.py +++ b/ouroboros/capability_evidence.py @@ -1340,7 +1340,7 @@ def probe( # excluding it would starve the route of density witnesses entirely. _CACHE_INCLUSIVE_PROMPT_TOKEN_PROVIDERS = frozenset({ "openrouter", "openai", "openai-compatible", "cloudru", "local", "anthropic", - "deepseek", + "deepseek", "zai", }) diff --git a/ouroboros/colab_bootstrap.py b/ouroboros/colab_bootstrap.py index c3057c12c..762ac6c5b 100644 --- a/ouroboros/colab_bootstrap.py +++ b/ouroboros/colab_bootstrap.py @@ -24,7 +24,7 @@ DEFAULT_COLAB_APP_ROOT = "/content/drive/MyDrive/Ouroboros" DEFAULT_COLAB_REPO_DIR = "/content/ouroboros_repo" DEFAULT_OFFICIAL_REPO_URL = "https://github.com/razzant/ouroboros.git" -_SECRET_KEYS = ("OPENROUTER_API_KEY", "OPENAI_API_KEY", "ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "DEEPSEEK_API_KEY", "CLOUDRU_FOUNDATION_MODELS_API_KEY", "GITHUB_TOKEN", "TELEGRAM_BOT_TOKEN") +_SECRET_KEYS = ("OPENROUTER_API_KEY", "OPENAI_API_KEY", "ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "DEEPSEEK_API_KEY", "ZAI_API_KEY", "CLOUDRU_FOUNDATION_MODELS_API_KEY", "GITHUB_TOKEN", "TELEGRAM_BOT_TOKEN") def _run_colab_git_network(args: list[str], *, cwd: pathlib.Path | None = None) -> str: @@ -87,7 +87,7 @@ def collect_colab_secrets() -> Dict[str, str]: """ # Providers the one-click Colab launch can auto-route models for via # apply_runtime_provider_defaults (OpenRouter is the default aggregator; - # OpenAI/Anthropic/MiniMax/DeepSeek/Cloud.ru have direct model defaults). OpenAI-compatible + # OpenAI/Anthropic/MiniMax/DeepSeek/Z.ai/Cloud.ru have direct model defaults). OpenAI-compatible # endpoints have no universal model default and need explicit OUROBOROS_MODEL_* # config, so they are an advanced manual path, not part of the quick launch. provider_keys = ( @@ -96,6 +96,7 @@ def collect_colab_secrets() -> Dict[str, str]: "ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "DEEPSEEK_API_KEY", + "ZAI_API_KEY", "CLOUDRU_FOUNDATION_MODELS_API_KEY", ) out: Dict[str, str] = {} diff --git a/ouroboros/consciousness_wake.py b/ouroboros/consciousness_wake.py index 2f565db7c..05f126932 100644 --- a/ouroboros/consciousness_wake.py +++ b/ouroboros/consciousness_wake.py @@ -2,10 +2,10 @@ A wake-up is an ordinary Main turn nobody typed: ``prompts/CONSCIOUSNESS.md`` is its USER message (system prompt, memory and tools are Main's own, owner decision В15). -``render_wake_message`` fills its placeholders from existing readers (tasks settled since the last -wake, open owner quiz cards, the count of owner messages) and truncates the event list with an -explicit source pointer, never silently (BIBLE P1). ``wake_task_metadata`` is -the wake's origin/authority envelope for ``handle_wake_direct`` (``consciousness_authority``). +``render_wake_message`` fills its placeholders from existing readers (tasks settled and owner +cards answered since the last wake, open owner quiz cards, the count of owner messages) and +truncates the event list with an explicit source pointer, never silently (BIBLE P1). +``wake_task_metadata`` is the wake's origin/authority envelope for ``handle_wake_direct`` (``consciousness_authority``). """ from __future__ import annotations @@ -165,16 +165,43 @@ def _card_line(task_id: str, quiz_id: str, block: Dict[str, Any], *, now: float, return f"- owner card {quiz_id} on task {task_id}: {label}{age}; {preview}" +def _answered_card_line(task_id: str, quiz_id: str, block: Dict[str, Any], *, now: float) -> str: + """One answered card of the window: stamps, the recorded answer, a question preview. + + The owner's own words are the answer itself, so the comment is rendered whole; + only the question is a (named) preview. No verdict on what the answer meant. + """ + parts = [] + for key, word in (("answered_at", "answered"), ("asked_at", "asked")): + stamp = _parse_iso(block.get(key)) + parts.append(f"{word} {_ago(now - stamp)}" if stamp is not None else f"{word} at an unknown time") + options = block.get("options") if isinstance(block.get("options"), list) else [] + index = block.get("answered_index") + comment = str(block.get("comment") or "") + if isinstance(index, int) and not isinstance(index, bool): + label = str(options[index]) if 0 <= index < len(options) else "label unavailable" + answer = f"chose option {index + 1}: {label}" + if comment.strip(): + answer += f"; with the words: {comment}" + elif comment.strip(): + answer = f"answered in own words: {comment}" + else: + answer = "answer text unavailable" + preview = _clip_preview(block.get("question") or "question text unavailable") + return f"- owner card {quiz_id} on task {task_id}: {'; '.join(parts)}; {answer}; question: {preview}" + + def wake_events( drive_root: Any, *, since: float, now: float, reason: str = "", exclude_task_id: str = "", ) -> List[str]: """Render a trigger-first, bounded view of fresh facts and outstanding cards. - Settled task rows are filtered by ``since``. Answerable cards intentionally span the - full store because ``expired_terminal`` still accepts a late owner answer (В17a), - but they are rendered after the fresh trigger/facts and carry their semantic state. + Settled task rows and answered cards are filtered by ``since`` (an answer is an + event of the window, sorted with the settled facts by ``answered_at``). Answerable + cards intentionally span the full store because ``expired_terminal`` still accepts + a late owner answer (В17a), but they are rendered after the fresh trigger/facts and carry their semantic state. """ - from ouroboros.owner_quiz import STATE_EXPIRED_TERMINAL, STATE_OPEN + from ouroboros.owner_quiz import STATE_ANSWERED, STATE_EXPIRED_TERMINAL, STATE_OPEN from ouroboros.task_results import list_task_results from ouroboros.task_status import SETTLED_STATUSES from ouroboros.task_finalization import HOST_AUTHORED_TERMINAL_ORIGINS @@ -191,7 +218,16 @@ def wake_events( continue quizzes = row.get("owner_quiz") if isinstance(row.get("owner_quiz"), dict) else {} for quiz_id, block in quizzes.items(): - if not isinstance(block, dict) or block.get("answered_at"): + if not isinstance(block, dict): + continue + if block.get("answered_at"): + # An answer given inside the window is an event of the window: it sorts + # with the settled facts by its own stamp (no task id, so the trigger's + # de-duplication never hides it). Older answers are not news. + answered_at = str(block.get("answered_at") or "") + answered_ts = _parse_iso(answered_at) # an unreadable stamp cannot be placed in the window + if block.get("state") == STATE_ANSWERED and answered_ts is not None and answered_ts >= since: + settled.append((_iso(answered_ts), "", _answered_card_line(task_id, str(quiz_id), block, now=now))) continue if block.get("state") in (STATE_OPEN, STATE_EXPIRED_TERMINAL): cards.append(( diff --git a/ouroboros/consolidator.py b/ouroboros/consolidator.py index 8c7d12785..33a2cccb1 100644 --- a/ouroboros/consolidator.py +++ b/ouroboros/consolidator.py @@ -10,7 +10,6 @@ from ouroboros import room_consolidation from ouroboros.utils import ( append_jsonl, atomic_write_json, - read_json_dict, replace_atomic, utc_now_iso, read_text, @@ -388,6 +387,10 @@ def _run_block_consolidation( return total_usage for nominated_block, _entries in pending_knowledge: nominated_block["knowledge_source_ref"] = ref + if knowledge_context is not None: + from ouroboros.memory_nomination_receipts import prepare + pending_ids = prepare(meta, source_id, pending_knowledge) + atomic_write_json(meta_path, meta) # Debt precedes block and note publication. existing_blocks = _load_blocks(blocks_path) all_blocks = existing_blocks + new_blocks @@ -435,21 +438,11 @@ def _run_block_consolidation( "source_ref": block["knowledge_source_ref"], "outcomes": block["knowledge_writes"], }) if pending_knowledge: - # Nominations were durable before mutation. Outcome facts belong to - # the same blocks, so a failed write is available to later learning. _write_locked_json(blocks_path, all_blocks) - # Era compression later replaces these blocks with one object carrying no - # knowledge_writes, so this batch receipt lives in meta, not in a scan of - # dialogue_blocks.json. A fully published batch clears it; a run with no - # nominations at all leaves the older receipt standing. - # Count what was NOMINATED, not only what produced an outcome: an entry - # the writer skipped as malformed was not published either. - nominated = sum(len(entries) for _block, entries in pending_knowledge) - failed = nominated - sum(1 for outcome in published if outcome["ok"]) - meta.pop("last_unpublished_nominations", None) - if failed > 0: - meta["last_unpublished_nominations"] = {"entry_id": ref["entry_id"], - "failed": failed, "total": nominated} + from ouroboros.memory_nomination_receipts import settle + settle(meta, pending_ids, published) + # Legacy batch-only receipts remain open: no positional evidence can + # prove which old entry a later successful nomination resolved. _advance_cursor(meta, segments, segment_sigs, segment_entries, last_offset + processed) if not run_failed: # An advance by a run that recorded no failure retires a stale error. @@ -1019,8 +1012,17 @@ def maintain_memory_pressure(memory: Any, llm_client: Any, context: Any, *, identity_ref = retain_memory_source(context, "maintenance_identity", memory.identity_path().read_bytes()) identity += "\nExact identity source, available through read_file; no identity rewrite is authorized here:\n" + json.dumps(identity_ref) if chat.exists() or blocks.exists(): - usage = consolidate(chat, blocks, meta, llm_client, identity, knowledge_context=context, - force_tail=True, compact_chronicle=True, pressure_fits=fits) + from ouroboros.memory_nomination_receipts import DialogueMetaUnreadable + + try: + usage = consolidate(chat, blocks, meta, llm_client, identity, knowledge_context=context, + force_tail=True, compact_chronicle=True, pressure_fits=fits) + except DialogueMetaUnreadable as exc: + # A damaged existing cursor is neither empty nor permission to rewrite + # memory. Keep the original context available to Main, with a typed + # maintenance gap instead of aborting its first round. + usage = {"_consolidation_errors": [{"kind": "dialogue_meta_unreadable", + "message": str(exc)}]} if usage is not None: usages.append(usage) actions.append({"owner": "dialogue_consolidation", "usage": usage}) @@ -1237,7 +1239,9 @@ def _advance_cursor( def _load_meta(path: pathlib.Path) -> Dict[str, Any]: - return read_json_dict(path) or {} + from ouroboros.memory_nomination_receipts import load_meta + + return load_meta(path) from ouroboros.utils import jsonl_generation_signature as _chat_log_signature @@ -1452,9 +1456,11 @@ def _write_knowledge_entries( outcomes = [] for entry in entries: if not isinstance(entry, dict): + outcomes.append({"topic": "", "ok": False, "reason": "malformed_nomination"}) continue topic, content = entry.get("topic"), entry.get("content") if not isinstance(content, str) or not content.strip(): + outcomes.append({"topic": topic, "ok": False, "reason": "empty_nomination"}) continue try: topic = sanitize_topic(topic) diff --git a/ouroboros/context_health.py b/ouroboros/context_health.py index 9ef599aa7..b8ec6a73e 100644 --- a/ouroboros/context_health.py +++ b/ouroboros/context_health.py @@ -232,26 +232,47 @@ def _memory_health_lines(env: Any) -> List[str]: except Exception: pass + from ouroboros.memory_nomination_receipts import DialogueMetaUnreadable, load_meta + try: - meta = read_json_dict(env.drive_path("memory/dialogue_meta.json")) or {} - receipt = meta.get("last_unpublished_nominations") - if isinstance(receipt, dict) and int(receipt.get("failed") or 0) > 0: - # The recovery route is named because the reader may hold no read_file: - # an external-channel turn has the cognitive memory tools and nothing else. + meta = load_meta(env.drive_path("memory/dialogue_meta.json")) + except DialogueMetaUnreadable: + # A broken existing meta file is not an empty nomination/cursor state. + lines.append("WARNING: DIALOGUE META UNREADABLE — memory/dialogue_meta.json; " + "consolidation withheld to preserve existing bytes") + else: + pending = meta.get("pending_knowledge_nominations") + if pending: + # load_meta already validates the whole list; malformed state raises. + sample = ", ".join(row["id"].split(":")[0][:12] + ":" + + ":".join(row["id"].split(":")[-2:]) + for row in pending[:3]) lines.append( - f"WARNING: LAST DIALOGUE KNOWLEDGE PUBLICATION INCOMPLETE — {receipt.get('failed')} of " - f"{receipt.get('total')} nominations from the latest consolidation batch were not published " - f"(entry_id {receipt.get('entry_id')}); from the main chat, read_file(root='runtime_data', " - "path='memory/knowledge_history.jsonl') and publish what still holds" + f"WARNING: DIALOGUE KNOWLEDGE PUBLICATION OPEN — {len(pending)} source-addressed " + f"nominations (first {min(3, len(pending))}: {sample}; omitted {max(0, len(pending)-3)}). " + "Read memory/dialogue_meta.json and memory/knowledge_history.jsonl for full source. " + "No automatic or tool-level discharge exists yet; later successes cannot retire older entries." ) + receipt = meta.get("last_unpublished_nominations") + if isinstance(receipt, dict): + failed = receipt.get("failed") + if type(failed) is int and failed > 0: + # This old batch receipt cannot be retired by a later new-source success. + lines.append( + f"WARNING: LAST DIALOGUE KNOWLEDGE PUBLICATION INCOMPLETE — {failed} of " + f"{receipt.get('total')} nominations in a legacy consolidation batch remain unresolved " + f"(entry_id {receipt.get('entry_id')}); from the main chat, read_file(root='runtime_data', " + "path='memory/knowledge_history.jsonl') and publish what still holds" + ) + elif failed is not None and failed != 0: + lines.append("WARNING: DIALOGUE LEGACY NOMINATION RECEIPT INVALID — " + "memory/dialogue_meta.json; inspect the original receipt") error = meta.get("last_consolidation_error") if isinstance(error, dict): lines.append( f"WARNING: LAST DIALOGUE CONSOLIDATION FAILED — kind={error.get('kind') or 'unknown'} " f"at cursor {error.get('cursor_offset')}" ) - except Exception: - pass return lines diff --git a/ouroboros/contracts/plugin_api.py b/ouroboros/contracts/plugin_api.py index 22a608c0c..438f77aa6 100644 --- a/ouroboros/contracts/plugin_api.py +++ b/ouroboros/contracts/plugin_api.py @@ -37,7 +37,7 @@ LEGACY_PLUGIN_API_GENERATION = "1.3" FORBIDDEN_SKILL_SETTINGS: frozenset[str] = frozenset({ "OPENROUTER_API_KEY", "OPENAI_API_KEY", "OPENAI_COMPATIBLE_API_KEY", "CLOUDRU_FOUNDATION_MODELS_API_KEY", "GIGACHAT_CREDENTIALS", "GIGACHAT_PASSWORD", - "ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "DEEPSEEK_API_KEY", "GITHUB_TOKEN", + "ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "DEEPSEEK_API_KEY", "ZAI_API_KEY", "GITHUB_TOKEN", "OUROBOROS_NETWORK_PASSWORD", }) diff --git a/ouroboros/domains.toml b/ouroboros/domains.toml index c15f438ce..876ba6283 100644 --- a/ouroboros/domains.toml +++ b/ouroboros/domains.toml @@ -273,6 +273,7 @@ D20 = "Presence" "ouroboros/mcp_client.py" = "D05" "ouroboros/memory.py" = "D15" "ouroboros/memory_journal_compaction.py" = "D15" +"ouroboros/memory_nomination_receipts.py" = "D15" "ouroboros/model_concurrency.py" = "D02" "ouroboros/model_send_seal.py" = "D16" "ouroboros/mutation_attribution.py" = "D01" diff --git a/ouroboros/gateway/contracts.py b/ouroboros/gateway/contracts.py index cf83d24d1..d49585aa5 100644 --- a/ouroboros/gateway/contracts.py +++ b/ouroboros/gateway/contracts.py @@ -148,6 +148,7 @@ class ChatOutbound(TypedDict): recommended_index: NotRequired[int] answered_index: NotRequired[int] comment: NotRequired[str] + host_facts: NotRequired[str] # the mirrored card's host sentence (QuizOutbound.host_facts) task_incident: NotRequired[str] # A cancellation fault names the PHYSICAL task it could not settle when it differs from the logical task id. cancel_physical_task_id: NotRequired[str] @@ -387,8 +388,7 @@ class QuizOutbound(TypedDict): Optional questions continue under ``assumption``; ``wait_for_answer`` marks required waiting without an implied answer. ``state`` carries lifecycle. Replay merges the stored ``answered_index`` and verbatim ``comment``; a - comment without an index is the owner's whole answer, not an option choice. - """ + comment without an index is the owner's whole answer, not an option choice.""" type: Literal["quiz"] role: Literal["assistant"] @@ -402,6 +402,7 @@ class QuizOutbound(TypedDict): ts: str answered_index: NotRequired[int] comment: NotRequired[str] + host_facts: NotRequired[str] # host-written, never the model's: asking task, run start, owner's last message chat_id: NotRequired[int] task_id: NotRequired[str] project_thread: NotRequired[bool] diff --git a/ouroboros/gateway/history.py b/ouroboros/gateway/history.py index e212d91d2..78617c395 100644 --- a/ouroboros/gateway/history.py +++ b/ouroboros/gateway/history.py @@ -881,6 +881,8 @@ def _collect_chat_rows( for key in ("answered_index", "comment", "wait_ended_at"): # the answer, the closed bound if key in _live: quiz[key] = _live[key] + if _live.get("host_facts") and not quiz.get("host_facts"): + quiz["host_facts"] = str(_live["host_facts"]) # the ask-time sentence the block holds if "wait_for_answer" not in _live: quiz.pop("wait_for_answer", None) # the bound closed: the card no longer waits if quiz.get("wait_for_answer") or quiz.get("wait_ended_at"): diff --git a/ouroboros/gateway/models.py b/ouroboros/gateway/models.py index ee0572e3c..d1466b640 100644 --- a/ouroboros/gateway/models.py +++ b/ouroboros/gateway/models.py @@ -18,8 +18,10 @@ from ouroboros.provider_models import ( ALL_PROVIDER_CREDENTIAL_KEYS, ACTIVE_MODEL_SETTING_KEYS, DEEPSEEK_BASE_URL, + resolve_zai_base_url, DIRECT_PROVIDER_DEFAULTS, MINIMAX_REGION_ENDPOINTS, + ZAI_PLAN_ENDPOINTS, OPENROUTER_DEFAULTS, provider_for_model, resolve_minimax_base_url, @@ -41,6 +43,7 @@ def _provider_label_from_model_id(model_id: str) -> str: "qwen": "Qwen", "mistralai": "Mistral", "deepseek": "DeepSeek", + "z-ai": "Z.ai (GLM)", "perplexity": "Perplexity", }.get(prefix, prefix.title() if prefix else "Other") @@ -269,6 +272,20 @@ def _provider_specs( DEEPSEEK_BASE_URL, ), )) + zai_api_key = str(settings.get("ZAI_API_KEY", "") or "").strip() + if zai_api_key: + # Z.ai serves an OpenAI-compatible GET /models on its plan-selected + # official host, so the catalog is fetched live like the other providers. + specs.append(( + "zai", + lambda client: _fetch_openai_compatible_model_catalog( + client, + "zai", + "Z.ai (GLM)", + zai_api_key, + resolve_zai_base_url(str(settings.get("ZAI_PLAN", "") or "")), + ), + )) compatible_api_key = str(settings.get("OPENAI_COMPATIBLE_API_KEY", "") or "").strip() compatible_base_url = str(settings.get("OPENAI_COMPATIBLE_BASE_URL", "") or "").strip() @@ -740,6 +757,9 @@ def _run_provider_test(provider_id: str, overrides: dict[str, str]) -> dict: minimax_region = str(settings.get("MINIMAX_REGION", "") or "").strip().lower() if provider_id == "minimax" and minimax_region and minimax_region not in MINIMAX_REGION_ENDPOINTS: return {"error": "unknown MiniMax region", "_http_status": 400} + zai_plan = str(settings.get("ZAI_PLAN", "") or "").strip().lower() + if provider_id == "zai" and zai_plan and zai_plan not in ZAI_PLAN_ENDPOINTS: + return {"error": "unknown Z.ai plan", "_http_status": 400} return _run_provider_test_with_settings(provider_id, settings) diff --git a/ouroboros/gateway/settings.py b/ouroboros/gateway/settings.py index 6bbd27e48..ab354e2dd 100644 --- a/ouroboros/gateway/settings.py +++ b/ouroboros/gateway/settings.py @@ -38,7 +38,9 @@ from ouroboros.gateway.owner_settings import ( ) from ouroboros.onboarding_wizard import build_onboarding_html from ouroboros.platform_layer import is_container_env -from ouroboros.provider_models import MINIMAX_REGION_ENDPOINTS, resolve_minimax_base_url +from ouroboros.provider_models import ( + MINIMAX_REGION_ENDPOINTS, ZAI_PLAN_ENDPOINTS, resolve_minimax_base_url, resolve_zai_base_url, +) from ouroboros.secret_masking import ( MCP_RESPONSE_ONLY_FIELDS, is_custom_secret_setting_key, @@ -558,6 +560,8 @@ def _active_main_route( "cloudru": "CLOUDRU_FOUNDATION_MODELS_BASE_URL", "gigachat": "GIGACHAT_BASE_URL"}.get(provider) if provider == "minimax": base_url = resolve_minimax_base_url(settings.get("MINIMAX_REGION") or "") + elif provider == "zai": + base_url = resolve_zai_base_url(settings.get("ZAI_PLAN") or "") else: base_url = str(settings.get(base_url_key) or "") if base_url_key else "" # CW7 (v6.34.0): honour the USE_LOCAL_MAIN routing setting — a local-routed main @@ -1266,6 +1270,10 @@ def _api_settings_post_locked(request: Request, body: Any) -> JSONResponse: if minimax_region and minimax_region not in MINIMAX_REGION_ENDPOINTS: return unsaved_error("MINIMAX_REGION must be global_en or cn_zh.", 400) current["MINIMAX_REGION"] = minimax_region + zai_plan = str(current.get("ZAI_PLAN") or "").strip().lower() + if zai_plan and zai_plan not in ZAI_PLAN_ENDPOINTS: + return unsaved_error("ZAI_PLAN must be payg or coding.", 400) + current["ZAI_PLAN"] = zai_plan # Generic settings saves operate on the current boot baseline. A pending # next-boot mode written by /api/owner/runtime-mode is preserved on disk # below, but never hot-applied to this process/env. diff --git a/ouroboros/gateway/state.py b/ouroboros/gateway/state.py index d3d5d1468..cbf1b0bb9 100644 --- a/ouroboros/gateway/state.py +++ b/ouroboros/gateway/state.py @@ -322,7 +322,8 @@ def _task_activity_facts(drive_root: Any, task_id: str) -> dict: # (project_dialogue.project_question_pointer): display fields ride along. "quiz": {key: quiz[key] for key in ("quiz_id", "state", "asked_at", "wait_for_answer", "question", "options", "option_details", "stake", "assumption", - "recommended_index", "answered_index", "comment", "wait_ended_at") + "recommended_index", "answered_index", "comment", "wait_ended_at", + "host_facts") if isinstance(quiz, dict) and key in quiz}} if len(_FINALIZING_MEMO) >= _FINALIZING_MEMO_MAX: _FINALIZING_MEMO.clear() diff --git a/ouroboros/gateway/task_decision.py b/ouroboros/gateway/task_decision.py index 69fa96333..b5a41d83a 100644 --- a/ouroboros/gateway/task_decision.py +++ b/ouroboros/gateway/task_decision.py @@ -107,33 +107,12 @@ def _quiz_answer_frame( ) -> str: """Host-authored structural frame around the owner's VERBATIM choice. - The asked/answered timestamps ride inside so the MODEL judges freshness - itself (owner decision 30=A — no host staleness verdict). With no - ``option_index`` the owner took none of the offered options and wrote - their own answer — say exactly that, so the model never reads the free - answer as a gloss on a chosen option.""" - options = block.get("options") if isinstance(block.get("options"), list) else [] - lines = [ - f"[Owner quiz answer] quiz {block.get('quiz_id')} — asked {block.get('asked_at')}, " - f"answered {block.get('answered_at')}.", - f"Question was: {block.get('question')}", - ] - if option_index is None: - lines.append( - "The owner rejected all offered options and answered verbatim: " - f"{comment}" - ) - else: - label = str(options[option_index]) if 0 <= option_index < len(options) else "" - lines.append(f"The owner chose option {option_index + 1}: {label}") - if comment: - lines.append(f"Owner comment (verbatim): {comment}") - if str(block.get("assumption") or "") and not block.get("wait_for_answer"): - lines.append( - f"You continued under the assumption: {block.get('assumption')} — " - "judge yourself whether work has moved past the answered fork." - ) - return "\n".join(lines) + Thin alias of the ONE shared builder ``owner_quiz.quiz_answer_frame``, so + the live mailbox control here and the late-answer delivery in the drain + render the same words.""" + from ouroboros.owner_quiz import quiz_answer_frame + + return quiz_answer_frame(block, option_index, comment) def _refused(message: str, status: int, **extra: Any) -> Tuple[int, Dict[str, Any]]: @@ -224,8 +203,14 @@ def _forward_late_quiz_answer( is the acceptance receipt and whose ``client_message_id`` is the idempotency key (a crash after acceptance never authorizes another enqueue), so a retry of this request re-enters the same delivery instead of - duplicating it. The text is the same full ``_quiz_answer_frame`` the live - mailbox control carries — nothing is trimmed. + duplicating it. + + The canonical chat row and the owner's bubble carry the HUMAN's words only + (``owner_quiz.late_answer_owner_text``: the verbatim comment, else the + pressed option as ``{n}. {label}``), never the host frame. The typed + ``late_answer`` provenance rides the message's metadata; the receiving turn + rebuilds the FULL frame from the stored block for the model + (``owner_quiz.late_answer_model_text``), so the model still reads the card. Disclosed property: in a PROJECT room that currently has exactly one live steerable root task, the ordinary routing delivers this message into THAT @@ -237,21 +222,32 @@ def _forward_late_quiz_answer( chat_id, reason = _late_answer_destination(drive_root, task_id, block) if chat_id is None: return False, reason - index = block.get("answered_index") - text = _quiz_answer_frame( - block, index if isinstance(index, int) else None, - str(block.get("comment") or ""), - ) + from ouroboros.owner_quiz import late_answer_owner_text + + text = late_answer_owner_text(block) + if not text: + # An answered block always has a comment or a valid index; a block + # that has neither cannot be spoken as the owner's words. + return False, "answer_unreadable" client_message_id = f"quiz_late_answer:{task_id}:{quiz_id}" bridge = message_bus.get_bridge() - row, rejoined = message_bus.accept_local_message( - bridge, drive_root, text, - chat_id=chat_id, user_id=1, source=str(source or "web"), - client_message_id=client_message_id, - # Provenance rides its OWN field; the real transport (the web card, or - # the skill that relayed the owner's tap) is the message's source. - task_metadata={"late_answer": {"task_id": task_id, "quiz_id": quiz_id}}, - ) + try: + row, rejoined = message_bus.accept_local_message( + bridge, drive_root, text, + chat_id=chat_id, user_id=1, source=str(source or "web"), + client_message_id=client_message_id, + # Provenance rides its OWN field; the real transport (the web card, or + # the skill that relayed the owner's tap) is the message's source. + task_metadata={"late_answer": {"task_id": task_id, "quiz_id": quiz_id}}, + ) + except ValueError: + # This id was already accepted, but under other bytes: a delivery + # accepted before the row carried only the owner's words (it carried + # the host frame then). It WAS delivered once; a retry rejoins it + # instead of failing forever or enqueueing a second owner turn. + if message_bus.accepted_chat_message(drive_root, chat_id, client_message_id) is None: + raise + return True, "" if not rejoined: # The named ingress accepts and enqueues but does not echo; give the # owner the SAME user bubble their own typing produces (a rejoin diff --git a/ouroboros/llm_openai_compatible.py b/ouroboros/llm_openai_compatible.py index 5a8ee463a..770e9ab5d 100644 --- a/ouroboros/llm_openai_compatible.py +++ b/ouroboros/llm_openai_compatible.py @@ -27,7 +27,7 @@ from ouroboros.llm_capability_policy import ( ) from ouroboros.reasoning_artifacts import transcript_has_sealed_reasoning from ouroboros.llm_routing import _resolve_or_provider -from ouroboros.provider_models import normalize_deepseek_reasoning_effort +from ouroboros.provider_models import normalize_deepseek_reasoning_effort, normalize_zai_reasoning_effort from ouroboros.request_wire_recovery import ( finalize_wire_response, note_provider_metadata_drop_fields, @@ -206,6 +206,26 @@ class _OpenAICompatibleLaneMixin: "reason": "provider_forced_tool_choice" if forced_tool else "provider_wire_mapping", "model": resolved_model, }) + elif provider == "zai": + # Same carriage family, Z.ai's OWN projection table (NOT + # DeepSeek's: medium does not exist at Z.ai and xhigh maps to + # max, not high). GLM reasoning cannot be disabled — the + # DeepSeek ``thinking={"type":"disabled"}`` arm answers + # 400 code 1210 ("please use low, high or max") on PAYG — and + # forced tool_choice WORKS with thinking enabled (measured + # 2026-09-21), so there is no forced-tool exception either. + # An absent parameter is served at MAX: dropping the tier + # silently billed every call at max. Any tier change is + # disclosed on usage as ``reasoning_effort_clamped``. + applied = normalize_zai_reasoning_effort(requested_effort) + kwargs["reasoning_effort"] = applied + _EFFORT_CLAMP_CVAR.set(None) # never inherit a stale note + if applied != requested_effort: + _EFFORT_CLAMP_CVAR.set({ + "requested": requested_effort, "applied": applied, + "reason": "provider_wire_mapping", + "model": resolved_model, + }) if temperature is not None: kwargs["temperature"] = temperature if response_format: diff --git a/ouroboros/llm_probe.py b/ouroboros/llm_probe.py index bfbd2baad..68e22cd62 100644 --- a/ouroboros/llm_probe.py +++ b/ouroboros/llm_probe.py @@ -208,6 +208,10 @@ def controlled_probe_error(exc: BaseException) -> dict[str, Any]: """Map typed transport facts to one bounded, provider-neutral reason.""" status, code, error_type = _error_facts(exc) credit_codes = { + # Z.ai answers plan exhaustion as HTTP 429 code 1113 "Insufficient + # balance" (billing, not rate limiting; a Coding Plan key on the + # pay-as-you-go endpoint lands here too). + "1113", "billing_hard_limit_reached", "credit_balance_too_low", "credits_exhausted", @@ -325,7 +329,7 @@ def probe_provider_readiness( if provider in { "openrouter", "openai", "openai-compatible", "minimax", "cloudru", - "deepseek", + "deepseek", "zai", }: remote_client = client._new_remote_client(target) diff --git a/ouroboros/llm_routing.py b/ouroboros/llm_routing.py index f2979923a..fd625ba30 100644 --- a/ouroboros/llm_routing.py +++ b/ouroboros/llm_routing.py @@ -22,6 +22,7 @@ from ouroboros.model_wait import dispatch_deadline_remaining_sec from ouroboros.openrouter_attribution import OPENROUTER_APP_HEADERS from ouroboros.provider_models import ( DEEPSEEK_BASE_URL, + resolve_zai_base_url, PROVIDER_PREFIXES, normalize_anthropic_model_id, normalize_model_identity, @@ -304,6 +305,8 @@ class _ProviderRoutingMixin: return f"minimax/{resolved_model}" if provider == "deepseek": return f"deepseek/{resolved_model}" + if provider == "zai": + return f"zai/{resolved_model}" if provider == "claudexor": return f"claudexor::{resolved_model}" return f"openai-compatible/{resolved_model}" @@ -391,6 +394,20 @@ class _ProviderRoutingMixin: "supports_generation_cost": False, } + if provider == "zai": + return { + "provider": provider, + "resolved_model": resolved_model, + "usage_model": usage_model, + "api_key": configured("ZAI_API_KEY", ""), + # Plan-selected official endpoint (PAYG default; the Coding + # Plan endpoint is intended for supported tools only). + "base_url": resolve_zai_base_url(configured("ZAI_PLAN", "")), + "default_headers": {}, + "supports_openrouter_extensions": False, + "supports_generation_cost": False, + } + if provider == "cloudru": return { "provider": provider, diff --git a/ouroboros/loop_llm_call.py b/ouroboros/loop_llm_call.py index 18b4476e0..2a5d90efe 100644 --- a/ouroboros/loop_llm_call.py +++ b/ouroboros/loop_llm_call.py @@ -493,6 +493,7 @@ _NON_RETRYABLE_PROVIDER_MARKERS = { "insufficient credits", "insufficient_credit", "insufficient_quota", + "1113", # Z.ai "Insufficient balance" (HTTP 429): billing, not a rate limit "quota exceeded", "billing", "payment required", diff --git a/ouroboros/loop_round_limits.py b/ouroboros/loop_round_limits.py index cccd84469..f5d999815 100644 --- a/ouroboros/loop_round_limits.py +++ b/ouroboros/loop_round_limits.py @@ -197,10 +197,21 @@ def _drain_incoming_messages( ) acknowledge_transcript_entry(drive_root, task_id, entry) continue + # A LATE quiz answer: the owner's row and this entry's text are the + # owner's own words; the model reads (and the owner corpus keeps) the + # FULL card frame rebuilt from the stored block on the canonical root. + model_msg = dmsg + if entry.get("late_answer") is not None: + from ouroboros.owner_quiz import late_answer_model_text + + model_msg = late_answer_model_text( + str(getattr(owner_ctx, "budget_drive_root", "") or "") or drive_root, + entry.get("late_answer"), dmsg, + ) _loop()._record_owner_directive( owner_ctx, source="owner_mailbox", - content=dmsg, + content=model_msg, msg_id=str(entry.get("msg_id") or ""), ) _stamp_owner_delivery( @@ -213,7 +224,7 @@ def _drain_incoming_messages( from ouroboros.client_surface import noted_owner_text _loop()._append_or_merge_user_message( - messages, _loop()._owner_marked_content(noted_owner_text(owner_ctx, entry, dmsg)), + messages, _loop()._owner_marked_content(noted_owner_text(owner_ctx, entry, model_msg)), slot=owner_ctx, ) acknowledge_transcript_entry(drive_root, task_id, entry) diff --git a/ouroboros/memory.py b/ouroboros/memory.py index 7b7deaace..56ef15b44 100644 --- a/ouroboros/memory.py +++ b/ouroboros/memory.py @@ -439,9 +439,26 @@ class Memory: log.warning("Corrupt blocks file %s", path) return [] + @staticmethod + def era_host_note(block: Dict[str, Any]) -> str: + """Host-authored framing for an era block: what it is, and its known coverage.""" + return ( + "Host note: compression of older dialogue blocks; an interpretation, not a grant " + f"or a standing rule. Range: {block.get('range') or 'unknown'}; " + f"source messages: {block.get('message_count') or 'unknown'}." + ) + @staticmethod def format_blocks_as_markdown(blocks: List[Dict[str, Any]]) -> str: - return "\n\n".join(b.get("content", "") for b in blocks) + """Render dialogue blocks for the task context (``context.py`` is the only caller). + + An era block gets the host note on its own line first; other blocks are unchanged. + """ + return "\n\n".join( + Memory.era_host_note(b) + "\n" + str(b.get("content", "")) + if b.get("type") == "era" else b.get("content", "") + for b in blocks + ) def load_identity(self) -> str: path = self.identity_path() diff --git a/ouroboros/memory_journal_compaction.py b/ouroboros/memory_journal_compaction.py index 097f0dd0f..57335f2a9 100644 --- a/ouroboros/memory_journal_compaction.py +++ b/ouroboros/memory_journal_compaction.py @@ -1,182 +1,20 @@ -"""Digest-only compaction of old memory-journal snapshots (CPL4-C16, owner 4A). +"""Compatibility entry point for the retired destructive memory-journal sweep. -``memory/identity_journal.jsonl``, ``memory/knowledge_history.jsonl`` and -``memory/knowledge/patterns_history.jsonl`` record the FULL old+new document -text on every write — O(doc×edits) growth, the worst byte offenders in the -memory plane. Owner decision 4A: entries younger than the unified GC -retention keep their full text; older entries become digest-only — the -content keys are replaced by their sha256 + length (existing hashes are -never overwritten) and the row is marked ``content_digested``. - -Strictly fail-closed per line: an unparseable line, a row without a -readable ``ts``, a row with nothing to digest, or a row whose STORED digest -disagrees with the text it claims to describe is carried through -BYTE-IDENTICAL. The scratchpad journal (typed rows, its own eviction -contract) is deliberately NOT in scope. - -This is the only sweep that DESTROYS content rather than whole dead files, -so its three guards are load-bearing (audit #15-11): - -* **Digest truth before deletion.** The digest becomes the only surviving - record of the text, so a stored ``*_sha256``/``*_len`` that does not match - the text is never published over it: the row keeps its full content and the - mismatch is reported as a typed fact (``digest_mismatch``, surfaced on the - ``memory_journal_compaction`` event). -* **A lock nobody can steal.** The append lock is taken ``owner_aware_stale`` - so elapsed time alone can never hand a second writer the same journal. -* **Publish only an unchanged source.** ``append_jsonl`` falls back to an - UNLOCKED append after its own lock timeout, so a concurrent row can still - land while this rewrite streams. The file is re-identified (size + inode) - against the bytes actually consumed immediately before ``os.replace``; any - delta aborts the publish and the journal stays as the appender left it. - -The rewrite streams line by line into the temp sibling — the journals are the -worst byte offenders in the memory plane and must never be loaded whole. +The startup/maintenance caller still invokes this function. Retaining that call +keeps old integrations working while new knowledge, identity and Pattern Register +history stays complete. A previously digested row cannot be reconstructed; +no new row loses its old/new text merely because of its age. """ from __future__ import annotations -import hashlib -import json -import logging -import os -import pathlib -from typing import Any, Dict, Optional, Tuple - -from ouroboros.deadline_utils import parse_deadline_ts -from ouroboros.platform_layer import acquire_exclusive_file_lock, release_exclusive_file_lock -from ouroboros.utils import jsonl_append_lock_path - -log = logging.getLogger(__name__) - -_JOURNAL_RELS = ( - pathlib.Path("memory") / "identity_journal.jsonl", - pathlib.Path("memory") / "knowledge_history.jsonl", - pathlib.Path("memory") / "knowledge" / "patterns_history.jsonl", -) -_CONTENT_KEYS = ("old_content", "new_content") +from pathlib import Path +from typing import Any, Dict, Optional +import stat -def _digest_row(row: Dict[str, Any]) -> str: - """Replace full-text keys with sha256+len. - - Returns ``"digested"`` when text was dropped, ``"mismatch"`` when a STORED - digest or length contradicts the text it describes, ``""`` when there was - nothing to digest. On a mismatch the row is left EXACTLY as found: the - digest is about to become the only surviving record of that content, and - publishing a digest already known to be false while deleting the last - correct copy is unrecoverable. - """ - dropped = [] - for key in _CONTENT_KEYS: - value = row.get(key) - if not isinstance(value, str): - continue - prefix = key[: -len("_content")] - digest = hashlib.sha256(value.encode("utf-8")).hexdigest() if value else "" - stored_digest = row.get(f"{prefix}_sha256") - stored_len = row.get(f"{prefix}_len") - if isinstance(stored_digest, str) and stored_digest != digest: - return "mismatch" - if isinstance(stored_len, int) and not isinstance(stored_len, bool) and stored_len != len(value): - return "mismatch" - dropped.append((key, prefix, digest, len(value))) - if not dropped: - return "" - for key, prefix, digest, length in dropped: - row[f"{prefix}_sha256"] = digest - row[f"{prefix}_len"] = length - del row[key] - row["content_digested"] = True - return "digested" - - -def _digest_line(raw: bytes, cutoff: float) -> Tuple[bytes, str]: - """One journal line, transformed or carried through byte-identical.""" - stripped = raw.strip() - if not stripped: - return raw, "" - try: - row = json.loads(stripped.decode("utf-8")) - except (UnicodeDecodeError, ValueError): - return raw, "" # fail-closed: never rewrite what cannot be read - if not isinstance(row, dict): - return raw, "" - parsed_ts = parse_deadline_ts(str(row.get("ts") or "")) - if parsed_ts is None or parsed_ts.timestamp() >= cutoff: - return raw, "" # fresh, or age unknowable: keep full text - outcome = _digest_row(row) - if outcome != "digested": - return raw, outcome - return json.dumps(row, ensure_ascii=False).encode("utf-8") + b"\n", "digested" - - -def _publish_if_unchanged( - path: pathlib.Path, tmp: pathlib.Path, expected: Tuple[int, int, int], -) -> bool: - """Swap the rewritten journal in ONLY if the source is still what we read. - - ``append_jsonl`` appends WITHOUT the sidecar lock once its own acquisition - times out, so a concurrent row can land while this rewrite streams. The - identity is (bytes consumed, device, inode): a grown file means an append - we did not carry over, a different inode means the journal was replaced - outright. Either way the rewrite is dropped and the appender's file stands. - """ - try: - stat = path.stat() - except OSError: - return False - if (int(stat.st_size), int(stat.st_dev), int(stat.st_ino)) != expected: - return False - os.replace(tmp, path) - return True - - -def _compact_one(path: pathlib.Path, cutoff: float) -> Tuple[int, int, str]: - """Digest one journal in place. - - Returns ``(digested, mismatched, error)``; a nonempty ``error`` - (``lock_unavailable`` / ``source_changed``) means nothing was published and - the journal is byte-identical to what the appenders left. - """ - lock_path = jsonl_append_lock_path(path) - lock_fd = acquire_exclusive_file_lock( - lock_path, timeout_sec=2.0, stale_sec=10.0, owner_aware_stale=True, - ) - if lock_fd is None: - return 0, 0, "lock_unavailable" - tmp = path.with_name(path.name + ".compact.tmp") - published = False - try: - digested = 0 - mismatched = 0 - consumed = 0 - with path.open("rb") as source: - start = os.fstat(source.fileno()) - with tmp.open("wb") as sink: - for raw in source: # streaming: one line in flight, never the file - consumed += len(raw) - out, outcome = _digest_line(raw, cutoff) - sink.write(out) - if outcome == "digested": - digested += 1 - elif outcome == "mismatch": - mismatched += 1 - if not digested: - return 0, mismatched, "" - published = _publish_if_unchanged( - path, tmp, (consumed, int(start.st_dev), int(start.st_ino)), - ) - if not published: - return 0, 0, "source_changed" - return digested, mismatched, "" - finally: - if not published: - try: - tmp.unlink() - except OSError: - log.debug("Failed to drop the journal compaction temp file", exc_info=True) - release_exclusive_file_lock(lock_path, lock_fd) +_JOURNALS = ("memory/identity_journal.jsonl", "memory/knowledge_history.jsonl", + "memory/knowledge/patterns_history.jsonl") def compact_memory_journal_snapshots( @@ -185,33 +23,26 @@ def compact_memory_journal_snapshots( *, now: Optional[float] = None, ) -> Dict[str, Any]: - """Digest old full-text snapshots in the three memory journals.""" - from ouroboros.retention import age_cutoff, get_gc_retention_days + """Preserve every journal byte, including malformed and historical rows. - if retention_days is None: - retention_days = get_gc_retention_days() - cutoff = age_cutoff(retention_days, now) - report: Dict[str, Any] = {"digested": {}, "digest_mismatch": {}, "errors": []} - root = pathlib.Path(drive_root) - for rel in _JOURNAL_RELS: - path = root / rel - if not path.exists(): - continue + The arguments keep the previous call contract. Size facts in the existing + startup report measure growth; a missing journal is not a measured zero. + """ + sizes: Dict[str, Optional[int]] = {} + errors: list[str] = [] + for relative in _JOURNALS: try: - digested, mismatched, error = _compact_one(path, cutoff) - except OSError: - report["errors"].append({"journal": rel.as_posix(), "error": "io_error"}) - continue - if error: - report["errors"].append({"journal": rel.as_posix(), "error": error}) - if digested: - report["digested"][rel.as_posix()] = digested - if mismatched: - # Typed fact, not a silent skip: a stored digest that contradicts - # its own text means one of the two is already corrupt, and the - # content stays in full until a human looks. - report["digest_mismatch"][rel.as_posix()] = mismatched - return report + info = (Path(drive_root) / relative).lstat() + sizes[relative] = info.st_size if stat.S_ISREG(info.st_mode) else None + if sizes[relative] is None: + errors.append(f"{relative}: not_regular") + except FileNotFoundError: + sizes[relative] = None + except OSError as exc: + sizes[relative] = None + errors.append(f"{relative}: {type(exc).__name__}") + return {"digested": {}, "digest_mismatch": {}, "errors": errors, + "journal_bytes": sizes} __all__ = ["compact_memory_journal_snapshots"] diff --git a/ouroboros/memory_nomination_receipts.py b/ouroboros/memory_nomination_receipts.py new file mode 100644 index 000000000..4bcc98f0e --- /dev/null +++ b/ouroboros/memory_nomination_receipts.py @@ -0,0 +1,104 @@ +"""Source-addressed pending knowledge nominations in dialogue_meta.json. + +The history log carries full nomination bytes. This compact index makes an +unpublished nomination visible even when its summary block becomes an era. +A later unrelated success cannot discharge an older source identity. +""" +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any + + +KEY = "pending_knowledge_nominations" + + +class DialogueMetaUnreadable(ValueError): + """Existing cursor or nomination obligations cannot be safely interpreted.""" + + +def load_meta(path: Path) -> dict[str, Any]: + """An absent cursor is new; an unreadable existing cursor is not empty. + + This meta file now owns durable pending obligations. A permissive JSON read + would erase them on the next consolidation. Reject duplicate keys as well: + the second copy of an obligation field cannot silently replace the first. + """ + def unique_pairs(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + result: dict[str, Any] = {} + for key, value in pairs: + if key in result: + raise ValueError(f"Duplicate dialogue meta key: {key}") + result[key] = value + return result + + try: + with path.open("r", encoding="utf-8") as source: + value = json.load(source, object_pairs_hook=unique_pairs) + except FileNotFoundError: + # A dangling link is an existing, unreadable source, not a new cursor. + try: + path.lstat() + except FileNotFoundError: + return {} + raise DialogueMetaUnreadable("Dialogue meta exists but cannot be read") from None + except (OSError, UnicodeError, ValueError) as exc: + raise DialogueMetaUnreadable(f"Dialogue meta unreadable: {type(exc).__name__}") from exc + if not isinstance(value, dict): + raise DialogueMetaUnreadable("Dialogue meta must be a JSON object") + _pending(value) # Refuse corrupt obligations before the first paid correction call. + return value + + +def _pending(meta: dict[str, Any]) -> dict[str, dict[str, Any]]: + rows = meta.get(KEY, []) + if not isinstance(rows, list) or any(not isinstance(row, dict) or not isinstance(row.get("id"), str) + for row in rows): + raise DialogueMetaUnreadable("Unreadable nomination obligations; refusing to replace their bytes") + if len({row["id"] for row in rows}) != len(rows): + raise DialogueMetaUnreadable("Duplicate nomination obligation IDs") + return {row["id"]: row for row in rows} + + +def prepare(meta: dict[str, Any], source_id: str, batches: list[tuple[Any, list[Any]]]) -> list[str]: + """Record every proposed entry before publication; return positional IDs.""" + pending = _pending(meta) + ids: list[str] = [] + for block_index, (_block, entries) in enumerate(batches): + for entry_index, entry in enumerate(entries): + identifier = f"{source_id}:{block_index}:{entry_index}" + ids.append(identifier) + if identifier not in pending: + pending[identifier] = { + "id": identifier, + "scope": str(entry.get("scope") or "default") if isinstance(entry, dict) else "invalid", + "topic": str(entry.get("topic") or "") if isinstance(entry, dict) else "", + "reason": "publication_pending", + } + meta[KEY] = list(pending.values()) + return ids + + +def settle(meta: dict[str, Any], ids: list[str], outcomes: list[dict[str, Any]]) -> None: + """Only this exact source's successful entries retire; missing outcomes stay owed. + + A failed or unobserved entry never expires. A future source-grounded + resolution must address its ID explicitly; same-topic later writes cannot. + """ + pending = _pending(meta) + for index, identifier in enumerate(ids): + outcome = outcomes[index] if index < len(outcomes) else {} + if outcome.get("ok") is True: + pending.pop(identifier, None) + elif identifier in pending: + row = pending[identifier] + row["reason"] = str(outcome.get("reason") or "outcome_missing") + if isinstance(outcome.get("scope"), str): + row["scope"] = outcome["scope"] + if isinstance(outcome.get("topic"), str): + row["topic"] = outcome["topic"] + if pending: + meta[KEY] = list(pending.values()) + else: + meta.pop(KEY, None) diff --git a/ouroboros/observability.py b/ouroboros/observability.py index d93e63ae5..8f6e8b695 100644 --- a/ouroboros/observability.py +++ b/ouroboros/observability.py @@ -189,7 +189,7 @@ _GENERIC_KV_SECRET_KEY_HINTS = ( "key", "token", "secret", "auth", "bearer", "cred", "password", "passwd", "passphrase", "apikey", "access_token", "openrouter", "openai", "anthropic", "cloudru", "cloud_ru", "gigachat", "groq", "deepseek", "together", "fireworks", - "mistral", "cohere", "perplexity", "replicate", "huggingface", "azure", "xai", + "mistral", "cohere", "perplexity", "replicate", "huggingface", "azure", "xai", "zai", ) diff --git a/ouroboros/owner_mailbox.py b/ouroboros/owner_mailbox.py index 1ba08493c..fb1f2cb95 100644 --- a/ouroboros/owner_mailbox.py +++ b/ouroboros/owner_mailbox.py @@ -206,6 +206,7 @@ def write_owner_message( client_surface: Optional[Dict[str, Any]] = None, attachment_manifest: Optional[List[Dict[str, Any]]] = None, client_message_id: str = "", + late_answer: Optional[Dict[str, Any]] = None, ) -> bool: """Write an owner message or typed control entry to a task's mailbox. @@ -213,6 +214,11 @@ def write_owner_message( (additively, like ``client_surface``) only when the writer knows it STRUCTURALLY — never parsed back out of ``msg_id``, whose shape is a transport key each producer composes for its own dedupe. + + ``late_answer`` is the typed ``{task_id, quiz_id}`` provenance of an owner + message that answers a finished task's quiz card (stored additively when + valid): the drain rebuilds the card's frame for the model from it, while + ``text`` stays the owner's own words. """ path = _mailbox_path(drive_root, task_id) path.parent.mkdir(parents=True, exist_ok=True) @@ -224,6 +230,11 @@ def write_owner_message( } if str(client_message_id or ""): entry["client_message_id"] = str(client_message_id) + from ouroboros.owner_quiz import late_answer_ref + + late_ref = late_answer_ref(late_answer) + if late_ref is not None: + entry["late_answer"] = late_ref if isinstance(client_surface, dict) and client_surface: # Owner Surface Fact (additive, like ``ts``): which client surface sent # this follow-up, so the loop can note a mid-task device change. @@ -680,6 +691,10 @@ def drain_owner_entries( # out here is a written fact nobody can read. if str(entry.get("client_message_id") or ""): drained["client_message_id"] = str(entry["client_message_id"]) + # Same for a late quiz answer's typed provenance: the drain + # rebuilds the card frame for the model from it. + if isinstance(entry.get("late_answer"), dict): + drained["late_answer"] = dict(entry["late_answer"]) if isinstance(entry.get("attachment_manifest"), list): drained["attachment_manifest"] = [ dict(item) for item in entry["attachment_manifest"] diff --git a/ouroboros/owner_quiz.py b/ouroboros/owner_quiz.py index 8ecb5dfaf..a49212651 100644 --- a/ouroboros/owner_quiz.py +++ b/ouroboros/owner_quiz.py @@ -17,7 +17,7 @@ like the hurry projection): "assumption", "state": open|answered|expired_terminal, "asked_at", "answered_at"?, "answered_index"?, "request_id"?, "comment"?, "reconciled_at"?, "chat_id"?, "max_wait_minutes"?, - "answered_after_terminal"?, + "answered_after_terminal"?, "host_facts"?, }, ... } @@ -121,6 +121,7 @@ def record_asked( recommended_index: Optional[int] = None, chat_id: Optional[int] = None, max_wait_minutes: Optional[int] = None, + host_facts: str = "", ) -> Dict[str, Any]: """Worker-side projection write at ask time. @@ -132,7 +133,10 @@ def record_asked( real hidden partition, never "no chat"): a late answer arriving after the task is gone is delivered there as an ordinary owner message instead of into a mailbox nobody drains. ``max_wait_minutes`` records the bound a - waiting asker chose, so replay can say what the task waited for.""" + waiting asker chose, so replay can say what the task waited for. + ``host_facts`` is the host-written sentence the card shows under the + question (asking task, how its run started, the owner's last message in + the chat); stored only when non-empty.""" if option_details is not None and ( not isinstance(option_details, list) or len(option_details) != len(options) or not all(isinstance(value, str) for value in option_details) @@ -150,6 +154,7 @@ def record_asked( **({"chat_id": int(chat_id)} if isinstance(chat_id, int) and not isinstance(chat_id, bool) else {}), **({"max_wait_minutes": int(max_wait_minutes)} if isinstance(max_wait_minutes, int) and not isinstance(max_wait_minutes, bool) else {}), + **({"host_facts": str(host_facts)} if str(host_facts or "") else {}), } refused: Dict[str, str] = {} @@ -336,3 +341,106 @@ def quiz_states(drive_root: Any, task_id: str) -> Dict[str, Dict[str, Any]]: if not isinstance(quizzes, dict): return {} return {str(k): dict(v) for k, v in quizzes.items() if isinstance(v, dict)} + + +def quiz_answer_frame( + block: Dict[str, Any], option_index: Optional[int], comment: str, +) -> str: + """Host-authored structural frame around the owner's VERBATIM choice. + + The ONE frame builder shared by the live mailbox control (the decision + ingress) and the late-answer delivery (the drain / direct turn), so the + model reads the same words whichever way the answer arrives. + + The asked/answered timestamps ride inside so the MODEL judges freshness + itself (owner decision 30=A — no host staleness verdict). With no + ``option_index`` the owner took none of the offered options and wrote + their own answer — say exactly that, without calling it a rejection the + owner never stated, so the model never reads the free answer as a gloss on + a chosen option.""" + options = block.get("options") if isinstance(block.get("options"), list) else [] + lines = [ + f"[Owner quiz answer] quiz {block.get('quiz_id')} — asked {block.get('asked_at')}, " + f"answered {block.get('answered_at')}.", + f"Question was: {block.get('question')}", + ] + if option_index is None: + lines.append( + "The owner answered in their own words without choosing an offered " + f"option. Verbatim: {comment}" + ) + else: + label = str(options[option_index]) if 0 <= option_index < len(options) else "" + lines.append(f"The owner chose option {option_index + 1}: {label}") + if comment: + lines.append(f"Owner comment (verbatim): {comment}") + if str(block.get("assumption") or "") and not block.get("wait_for_answer"): + lines.append( + f"You continued under the assumption: {block.get('assumption')} — " + "judge yourself whether work has moved past the answered fork." + ) + return "\n".join(lines) + + +def recorded_answer_frame(block: Dict[str, Any]) -> str: + """The frame of an ANSWERED block, read from what the projection recorded.""" + index = block.get("answered_index") + return quiz_answer_frame( + block, index if isinstance(index, int) and not isinstance(index, bool) else None, + str(block.get("comment") or ""), + ) + + +def late_answer_owner_text(block: Dict[str, Any]) -> str: + """What the HUMAN said on the card, for the owner's own chat row and bubble. + + The verbatim comment when the owner wrote one; otherwise the pressed option + exactly as the button showed it, ``{n}. {label}``. Never the host frame: + the bubble is the owner's words, the frame is for the model.""" + comment = str(block.get("comment") or "") + if comment.strip(): + return comment + options = block.get("options") if isinstance(block.get("options"), list) else [] + index = block.get("answered_index") + if isinstance(index, int) and not isinstance(index, bool) and 0 <= index < len(options): + return f"{index + 1}. {options[index]}" + return "" + + +def late_answer_ref(value: Any) -> Optional[Dict[str, str]]: + """The typed ``late_answer`` provenance ``{task_id, quiz_id}``, or None.""" + if not isinstance(value, dict): + return None + task_id = str(value.get("task_id") or "").strip() + quiz_id = str(value.get("quiz_id") or "").strip() + if not task_id or not quiz_id: + return None + return {"task_id": task_id, "quiz_id": quiz_id} + + +def late_answer_model_text(drive_root: Any, late_answer: Any, owner_text: str) -> str: + """The model-facing delivery of a LATE quiz answer (owner message ``owner_text``). + + The owner's chat row carries only their own words; the receiving model + needs the card they answered. The frame is rebuilt from the stored block on + the canonical ``drive_root`` (the same projection the ingress recorded). + When that block cannot be read (evicted, unanswered, or unreadable), the + owner's words are delivered with one host line saying so — disclosed, never + a silent loss of the card's context. A message with no valid ``late_answer`` + provenance is returned unchanged.""" + ref = late_answer_ref(late_answer) + if ref is None: + return owner_text + block: Dict[str, Any] = {} + try: + block = quiz_states(drive_root, ref["task_id"]).get(ref["quiz_id"]) or {} + except Exception: + block = {} + if isinstance(block, dict) and str(block.get("state") or "") == STATE_ANSWERED: + return recorded_answer_frame(block) + return ( + f"{owner_text}\n" + f"[Host note] This owner message answers quiz {ref['quiz_id']} of task " + f"{ref['task_id']}; that card could not be read, so only the owner's own " + "words are shown." + ) diff --git a/ouroboros/pricing.py b/ouroboros/pricing.py index de4f85c53..283e80a1b 100644 --- a/ouroboros/pricing.py +++ b/ouroboros/pricing.py @@ -223,7 +223,7 @@ def _cost_from_pricing(pricing: tuple, prompt_tokens: int, completion_tokens: in def infer_api_key_type(model: str, provider: Optional[str] = None) -> str: """Infer which API key is used based on model name.""" provider_name = str(provider or "").strip().lower() - if provider_name in {"local", "openrouter", "openai", "anthropic", "openai-compatible", "cloudru", "gigachat", "minimax", "deepseek"}: + if provider_name in {"local", "openrouter", "openai", "anthropic", "openai-compatible", "cloudru", "gigachat", "minimax", "deepseek", "zai"}: return provider_name raw_model = str(model or "").strip() direct_provider = provider_for_model(raw_model) diff --git a/ouroboros/project_dialogue.py b/ouroboros/project_dialogue.py index f703d57cb..3e3218953 100644 --- a/ouroboros/project_dialogue.py +++ b/ouroboros/project_dialogue.py @@ -134,6 +134,7 @@ def project_question_pointer(row: Dict[str, Any], block: Any, project: Any, question = str(quiz.get("question") or row.get("text") or block.get("question") or "") assumption = str(quiz.get("assumption") or block.get("assumption") or "") stake = str(quiz.get("stake") or block.get("stake") or "") + host_facts = str(quiz.get("host_facts") or block.get("host_facts") or "") recommended = block.get("recommended_index") if not isinstance(recommended, int) or isinstance(recommended, bool): recommended = next((i for i, option in enumerate(options) @@ -151,6 +152,7 @@ def project_question_pointer(row: Dict[str, Any], block: Any, project: Any, **({"option_details": details} if details else {}), **({"stake": stake} if stake else {}), **({"assumption": assumption} if assumption else {}), + **({"host_facts": host_facts} if host_facts else {}), **({"recommended_index": recommended} if recommended is not None else {}), **facts, **({"source_status": "unavailable"} if not known else {}), diff --git a/ouroboros/provider_models.py b/ouroboros/provider_models.py index 9540fc185..5f0d9ee23 100644 --- a/ouroboros/provider_models.py +++ b/ouroboros/provider_models.py @@ -34,6 +34,24 @@ def resolve_minimax_base_url(region: str = "") -> str: # the fingerprint is already unique per provider+model). DEEPSEEK_BASE_URL = "https://api.deepseek.com/v1" +# Z.ai (Zhipu / GLM) serves one OpenAI-compatible API surface on two plans that +# share the same key: pay-as-you-go and the subscription Coding Plan. The plan +# selects the endpoint (analogous to MiniMax regions); the PAYG endpoint is the +# default because the Coding Plan endpoint is officially intended for supported +# coding tools only. +ZAI_PLAN_ENDPOINTS: dict[str, str] = { + "payg": "https://api.z.ai/api/paas/v4", + "coding": "https://api.z.ai/api/coding/paas/v4", +} +ZAI_DEFAULT_PLAN = "payg" + + +def resolve_zai_base_url(plan: str = "") -> str: + """Return the configured Z.ai OpenAI-compatible endpoint for the plan.""" + selected = str(plan or "").strip().lower() or ZAI_DEFAULT_PLAN + return ZAI_PLAN_ENDPOINTS.get(selected, ZAI_PLAN_ENDPOINTS[ZAI_DEFAULT_PLAN]) + + # DeepSeek's Chat Completions ``reasoning_effort`` enum is low/high/max # (medium/xhigh are documented aliases of high) and thinking is switched off by # ``thinking.type=disabled``, not by an effort value. This is the wire dialect @@ -54,6 +72,39 @@ def normalize_deepseek_reasoning_effort(value: str) -> str: return DEEPSEEK_REASONING_EFFORT_ALIASES.get(normalized, normalized) +# Z.ai (GLM) serves the same Chat Completions ``reasoning_effort`` shape but a +# DIFFERENT enum mapping than DeepSeek — do not reuse the DeepSeek table. The +# provider has exactly three tiers (low | high | max); ``medium`` does not +# exist, thinking cannot be disabled (``thinking={"type":"disabled"}`` answers +# 400 code 1210 "please use low, high or max" on PAYG), and an ABSENT +# parameter is served at MAX — so a silently dropped tier means every call +# runs (and bills) at max. Projection of the canonical scale, measured live +# 2026-09-21 on the Coding Plan endpoint: none/minimal/low -> low, +# medium/high -> high, xhigh/ultra -> max. +ZAI_REASONING_EFFORT_ALIASES = { + "none": "low", + "minimal": "low", + "low": "low", + "medium": "high", + "high": "high", + "xhigh": "max", + "max": "max", + "ultra": "max", +} + + +def normalize_zai_reasoning_effort(value: str) -> str: + """Project one canonical effort tier onto Z.ai's Chat wire enum. + + Every canonical tier maps to a concrete provider tier; unlike DeepSeek + there is no "off" arm — GLM reasoning cannot be disabled, so an unmapped + value still resolves to a tier rather than being dropped (a dropped tier + is served at max). + """ + normalized = str(value or "").strip().lower() + return ZAI_REASONING_EFFORT_ALIASES.get(normalized, "low") + + # Direct-provider prefix → canonical provider name. Un-prefixed models route # through OpenRouter. Order matters only for readability; prefixes are disjoint. PROVIDER_PREFIXES: tuple[tuple[str, str], ...] = ( @@ -64,6 +115,7 @@ PROVIDER_PREFIXES: tuple[tuple[str, str], ...] = ( ("cloudru::", "cloudru"), ("gigachat::", "gigachat"), ("deepseek::", "deepseek"), + ("zai::", "zai"), ("openai-compatible::", "openai-compatible"), ("openrouter::", "openrouter"), ) @@ -75,6 +127,7 @@ PROVIDER_ENV_KEYS: dict[str, str] = { "minimax": "MINIMAX_API_KEY", "cloudru": "CLOUDRU_FOUNDATION_MODELS_API_KEY", "deepseek": "DEEPSEEK_API_KEY", + "zai": "ZAI_API_KEY", "openrouter": "OPENROUTER_API_KEY", } @@ -101,6 +154,7 @@ PROVIDER_CREDENTIAL_GROUPS: dict[str, tuple[str, ...]] = { "minimax": ("MINIMAX_API_KEY", "MINIMAX_REGION"), "cloudru": ("CLOUDRU_FOUNDATION_MODELS_API_KEY", "CLOUDRU_FOUNDATION_MODELS_BASE_URL"), "deepseek": ("DEEPSEEK_API_KEY",), + "zai": ("ZAI_API_KEY", "ZAI_PLAN"), "gigachat": ( "GIGACHAT_CREDENTIALS", "GIGACHAT_PASSWORD", "GIGACHAT_USER", "GIGACHAT_BASE_URL", "GIGACHAT_SCOPE", "GIGACHAT_VERIFY_SSL_CERTS", @@ -268,7 +322,7 @@ def local_only_review_route_env() -> bool: provider_has_credentials(provider) for provider in ( "openrouter", "openai", "anthropic", "minimax", "cloudru", "gigachat", - "deepseek", "openai-compatible", + "deepseek", "zai", "openai-compatible", ) ) @@ -440,6 +494,16 @@ MINIMAX_DIRECT_DEFAULTS = { # the Cloud.ru/GigaChat clear-instead-of-fill path; owners can opt in manually. } +ZAI_DIRECT_DEFAULTS = { + "main": "zai::glm-5.3", + "heavy": "", + "light": "zai::glm-5.3-flash", + "vision": "", + "fallback": "zai::glm-5.3-flash", + # No deep_review default: the route publishes no window metadata and no live + # measurement exists, so the slot follows the MiniMax clear-instead-of-fill path. +} + DEEPSEEK_DIRECT_DEFAULTS = { "main": "deepseek::deepseek-v4-pro", "heavy": "", @@ -475,6 +539,7 @@ DIRECT_PROVIDER_DEFAULTS = { "gigachat": GIGACHAT_DIRECT_DEFAULTS, "minimax": MINIMAX_DIRECT_DEFAULTS, "deepseek": DEEPSEEK_DIRECT_DEFAULTS, + "zai": ZAI_DIRECT_DEFAULTS, } # Review panels are declared as provider ROLE sequences, then compiled against @@ -492,6 +557,7 @@ DIRECT_PROVIDER_REVIEW_ROLES = { # Strongest-main ×3 policy (same as OpenAI/Anthropic): an exclusive # DeepSeek install reviews with three independent thinking v4-pro calls. "deepseek": ("main", "main", "main"), + "zai": ("main", "main", "main"), } DIRECT_PROVIDER_SCOPE_DEFAULTS = { @@ -542,6 +608,10 @@ def migrate_model_value(provider: str, value: str) -> str: if text.startswith("deepseek/"): return f"deepseek::{text[len('deepseek/'):]}" return text + if provider == "zai": + if text.startswith("zai/"): + return f"zai::{text[len('zai/'):]}" + return text return text @@ -665,6 +735,8 @@ def normalize_model_identity(model: str) -> str: return f"minimax/{text[len('minimax::'):]}" if text.startswith("deepseek::"): return f"deepseek/{text[len('deepseek::'):]}" + if text.startswith("zai::"): + return f"zai/{text[len('zai::'):]}" if text.startswith("anthropic::"): return f"anthropic/{normalize_anthropic_model_id(text[len('anthropic::'):])}" if text.startswith("anthropic/"): diff --git a/ouroboros/review_model_routes.py b/ouroboros/review_model_routes.py index 51d3a54f8..1a5ea610d 100644 --- a/ouroboros/review_model_routes.py +++ b/ouroboros/review_model_routes.py @@ -51,13 +51,14 @@ def _exclusive_direct_remote_provider_env() -> str: ("openai", has_openai), ("anthropic", has_anthropic), ("minimax", has_minimax), ("cloudru", has_cloudru), ("gigachat", has_gigachat), ("deepseek", bool(str(runtime_setting("DEEPSEEK_API_KEY", "") or "").strip())), + ("zai", bool(str(runtime_setting("ZAI_API_KEY", "") or "").strip())), ) if present] return direct[0] if len(direct) == 1 else "" def direct_provider_review_models_fallback(provider: str) -> list[str]: """Return the exact review-models list a direct-provider fallback emits.""" - if provider not in ("openai", "anthropic", "minimax", "cloudru", "gigachat", "deepseek"): + if provider not in ("openai", "anthropic", "minimax", "cloudru", "gigachat", "deepseek", "zai"): return [] main_model = str( runtime_setting("OUROBOROS_MODEL", SETTINGS_DEFAULTS["OUROBOROS_MODEL"]) or "" diff --git a/ouroboros/reviewer_window.py b/ouroboros/reviewer_window.py index 118fc75dd..8cf2f5b79 100644 --- a/ouroboros/reviewer_window.py +++ b/ouroboros/reviewer_window.py @@ -151,6 +151,11 @@ def reviewer_route(model_id: str, *, session: bool = False) -> tuple: return provider, str( resolve_minimax_base_url(runtime_settings().get("MINIMAX_REGION") or "") or "") + if provider == "zai": + # Z.ai's base url is selected by the plan, the same way MiniMax's is by region. + from ouroboros.provider_models import resolve_zai_base_url + + return provider, resolve_zai_base_url(runtime_settings().get("ZAI_PLAN") or "") settings_key = { "openai": "OPENAI_BASE_URL", "openai-compatible": "OPENAI_COMPATIBLE_BASE_URL", diff --git a/ouroboros/safety.py b/ouroboros/safety.py index 77f79e438..cc52cba4d 100644 --- a/ouroboros/safety.py +++ b/ouroboros/safety.py @@ -625,6 +625,7 @@ _REMOTE_PROVIDER_KEYS = ( "ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "DEEPSEEK_API_KEY", + "ZAI_API_KEY", "OPENAI_COMPATIBLE_API_KEY", "CLOUDRU_FOUNDATION_MODELS_API_KEY", "GIGACHAT_CREDENTIALS", @@ -646,6 +647,7 @@ _PROVIDER_KEY_ENV = { "anthropic": "ANTHROPIC_API_KEY", "minimax": "MINIMAX_API_KEY", "deepseek": "DEEPSEEK_API_KEY", + "zai": "ZAI_API_KEY", "openai-compatible": "OPENAI_COMPATIBLE_API_KEY", "cloudru": "CLOUDRU_FOUNDATION_MODELS_API_KEY", "gigachat": "GIGACHAT_CREDENTIALS", diff --git a/ouroboros/secret_masking.py b/ouroboros/secret_masking.py index 7682a28ca..8dc392457 100644 --- a/ouroboros/secret_masking.py +++ b/ouroboros/secret_masking.py @@ -29,6 +29,7 @@ MASKED_SECRET_SETTING_KEYS = frozenset( "ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "DEEPSEEK_API_KEY", + "ZAI_API_KEY", "GITHUB_TOKEN", "OUROBOROS_NETWORK_PASSWORD", } diff --git a/ouroboros/server_maintenance.py b/ouroboros/server_maintenance.py index b1e9366c9..d4b7fc678 100644 --- a/ouroboros/server_maintenance.py +++ b/ouroboros/server_maintenance.py @@ -452,9 +452,8 @@ def _startup_retired_settings_notice(settings: dict) -> None: def _prune_event(event_type: str, keys: tuple, **reports: dict) -> None: - """One ``events.jsonl`` row for a GC/sweep step that did or failed something: - ``keys`` are its own evidence of material work, read across every report it - hands in, so a healthy no-op pass stays silent instead of rowing every boot.""" + """Emit when a report has evidence under ``keys``. GC no-ops stay silent; + an observation report with measured journal sizes intentionally rows at boot.""" from supervisor.state import append_jsonl if any(report.get(key) for report in reports.values() for key in keys): @@ -565,11 +564,11 @@ def _startup_prune_sweeps(*, preserve_task_sources: bool = False) -> None: except Exception: log.debug("Stale cache prune failed", exc_info=True) try: - # CPL4-C16 (owner 4A): memory-journal snapshots older than GC retention - # become digest-only (sha256 + length); fresh entries keep full text. + # TZ-3 11A/B5 supersedes old age-digestion: measure journal growth + # without touching historical or new full-text snapshots. from ouroboros.memory_journal_compaction import compact_memory_journal_snapshots - _prune_event("memory_journal_compaction", ("digested", "digest_mismatch", "errors"), + _prune_event("memory_journal_observation", ("journal_bytes", "errors"), report=compact_memory_journal_snapshots(DATA_DIR)) except Exception: log.debug("Memory journal compaction failed", exc_info=True) diff --git a/ouroboros/server_owner_routing.py b/ouroboros/server_owner_routing.py index 9d7e6a879..3c9790023 100644 --- a/ouroboros/server_owner_routing.py +++ b/ouroboros/server_owner_routing.py @@ -248,6 +248,9 @@ def _route_project_chat_to_running_task( else None ), attachment_manifest=staged_manifest if staged_manifest else None, + late_answer=( + task_metadata.get("late_answer") if isinstance(task_metadata, dict) else None + ), ): return "" message_written = True diff --git a/ouroboros/server_runtime.py b/ouroboros/server_runtime.py index dee64e6c1..3700e938f 100644 --- a/ouroboros/server_runtime.py +++ b/ouroboros/server_runtime.py @@ -281,6 +281,7 @@ def _exclusive_direct_remote_provider(settings: dict) -> str: has_anthropic = bool(_setting_text(settings, "ANTHROPIC_API_KEY")) has_minimax = bool(_setting_text(settings, "MINIMAX_API_KEY")) has_deepseek = bool(_setting_text(settings, "DEEPSEEK_API_KEY")) + has_zai = bool(_setting_text(settings, "ZAI_API_KEY")) has_legacy_openai_base = bool(_setting_text(settings, "OPENAI_BASE_URL")) has_compatible = bool(_setting_text(settings, "OPENAI_COMPATIBLE_BASE_URL")) has_cloudru = bool(_setting_text(settings, "CLOUDRU_FOUNDATION_MODELS_API_KEY")) @@ -301,6 +302,7 @@ def _exclusive_direct_remote_provider(settings: dict) -> str: ("cloudru", has_cloudru), ("gigachat", has_gigachat), ("deepseek", has_deepseek), + ("zai", has_zai), ) if present ] return direct[0] if len(direct) == 1 else "" @@ -415,6 +417,7 @@ def has_remote_provider(settings: dict) -> bool: "ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "DEEPSEEK_API_KEY", + "ZAI_API_KEY", "OPENAI_COMPATIBLE_BASE_URL", "CLOUDRU_FOUNDATION_MODELS_API_KEY", "GIGACHAT_CREDENTIALS", diff --git a/ouroboros/settings_defaults.py b/ouroboros/settings_defaults.py index 613a82e81..39207ddbe 100644 --- a/ouroboros/settings_defaults.py +++ b/ouroboros/settings_defaults.py @@ -75,6 +75,8 @@ SETTINGS_DEFAULTS = {**UPDATE_SETTINGS_DEFAULTS, "MINIMAX_API_KEY": "", "MINIMAX_REGION": "", "DEEPSEEK_API_KEY": "", + "ZAI_API_KEY": "", + "ZAI_PLAN": "", "OUROBOROS_NETWORK_PASSWORD": "", "OUROBOROS_SERVER_HOST": "127.0.0.1", "OUROBOROS_HOST_SERVICE_PORT": 8767, diff --git a/ouroboros/settings_setup_contract.py b/ouroboros/settings_setup_contract.py index 1b31b3dd8..2578cb41d 100644 --- a/ouroboros/settings_setup_contract.py +++ b/ouroboros/settings_setup_contract.py @@ -11,6 +11,7 @@ from ouroboros.provider_models import ( ANTHROPIC_DIRECT_DEFAULTS, CLOUDRU_DIRECT_DEFAULTS, DEEPSEEK_DIRECT_DEFAULTS, + ZAI_DIRECT_DEFAULTS, MINIMAX_DIRECT_DEFAULTS, MINIMAX_REGION_ENDPOINTS, OPENAI_DIRECT_DEFAULTS, @@ -122,6 +123,7 @@ _MODEL_DEFAULTS = { "cloudru": {key: value for key, value in CLOUDRU_DIRECT_DEFAULTS.items() if key != "heavy"}, "minimax": {key: value for key, value in MINIMAX_DIRECT_DEFAULTS.items() if key != "heavy"}, "deepseek": {key: value for key, value in DEEPSEEK_DIRECT_DEFAULTS.items() if key != "heavy"}, + "zai": {key: value for key, value in ZAI_DIRECT_DEFAULTS.items() if key != "heavy"}, "anthropic": {key: value for key, value in ANTHROPIC_DIRECT_DEFAULTS.items() if key != "heavy"}, # No defaults: model names are server-specific; user must fill all slots. "openai-compatible": {"main": "", "light": "", "vision": "", "fallback": ""}, @@ -150,6 +152,8 @@ _PROVIDER_FIELDS = _rows(("id", "stateKey", "settingKey", "settingsInputId", "la ("minimax-key", "minimaxKey", "MINIMAX_API_KEY", "s-minimax-key", "MiniMax API Key", "MiniMax API key", "Optional. If this is the only remote key, the next step prefills MiniMax's own model ids.", "password", "more"), ("minimax-region", "minimaxRegion", "MINIMAX_REGION", "s-minimax-region", "MiniMax Region", "global_en or cn_zh", "Choose global_en for the global endpoint or cn_zh for the China endpoint.", "text", "more"), ("deepseek-key", "deepseekKey", "DEEPSEEK_API_KEY", "s-deepseek-key", "DeepSeek API Key", "sk-...", "Optional. If this is the only remote key, the next step prefills DeepSeek's own model ids.", "password", "more"), + ("zai-key", "zaiKey", "ZAI_API_KEY", "s-zai-key", "Z.ai API Key (GLM)", "...", "Optional. If this is the only remote key, the next step prefills Z.ai's own GLM model ids.", "password", "more"), + ("zai-plan", "zaiPlan", "ZAI_PLAN", "s-zai-plan", "Z.ai Plan", "payg or coding", "Choose payg (pay-as-you-go, default) or coding (Coding Plan endpoint; officially intended for supported coding tools only).", "text", "more"), ("anthropic-key", "anthropicKey", "ANTHROPIC_API_KEY", "s-anthropic", "Anthropic API Key", "sk-ant-...", "Optional. Saved for models routed straight to Anthropic, and for Claude tooling.", "password", "primary"), ("openai-compatible-url", "compatibleBaseUrl", "OPENAI_COMPATIBLE_BASE_URL", "s-compatible-url", "OpenAI-compatible Base URL", "http://localhost:11434/v1", "Base URL for your OpenAI-compatible endpoint (e.g. Ollama, LM Studio, vLLM). Required whenever a slot uses the OpenAI-compatible endpoint as its source.", "url", "more"), ("openai-compatible-key", "compatibleApiKey", "OPENAI_COMPATIBLE_API_KEY", "s-compatible-key", "OpenAI-compatible API Key", "Leave empty for no auth", "API key for the endpoint. Leave empty if your server does not require authentication.", "password", "more"), @@ -164,6 +168,7 @@ _PROFILE_SPECS = { "cloudru": ("Cloud.ru Foundation Models", "Cloud.ru is present, so the next step prefills Cloud.ru's own model ids.", "Cloud.ru-only setup detected. These defaults use Cloud.ru's own model ids."), "minimax": ("MiniMax", "MiniMax is present, so the next step prefills MiniMax's own model ids.", "MiniMax-only setup detected. These defaults include MiniMax-M3 and MiniMax-M2.7."), "deepseek": ("DeepSeek", "DeepSeek is present, so the next step prefills DeepSeek's own model ids.", "DeepSeek-only setup detected. These defaults use deepseek-v4-pro for main work and deepseek-v4-flash for the light lane."), + "zai": ("Z.ai (GLM)", "Z.ai is present, so the next step prefills Z.ai's own GLM model ids.", "Z.ai-only setup detected. These defaults use glm-5.3 for main work and glm-5.3-flash for the light lane. The plan setting selects the pay-as-you-go or Coding Plan endpoint."), "anthropic": ("Anthropic", "Anthropic is present, so the next step prefills Anthropic's own model ids.", "Anthropic-only setup detected. These defaults are explicit and official."), "openai-compatible": ("OpenAI-compatible endpoint", "An OpenAI-compatible base URL is configured. Enter the model names your server exposes in the next step.", "OpenAI-compatible endpoint detected. Choose the OpenAI-compatible endpoint as the source and enter the model names your server exposes. The model list is whatever your server supports."), "direct-multi": ("Direct multi-provider", "Multiple direct providers are present, so the next step keeps your model values editable without forcing one provider family.", "Multiple direct providers are configured. Start here, then split model slots across them if you want."), @@ -266,7 +271,7 @@ _SUBSCRIPTION_FIELDS = _rows(("id", "payloadKey", "label", "note"), ( ("skip-subscription-presets", SKIP_SUBSCRIPTION_PRESETS_FIELD, "Finish without agent defaults", "Completes onboarding without moving reviewers and subagents onto the connected subscriptions. Everything stays editable in Settings afterwards."), )) -_MODEL_SUGGESTIONS = list(dict.fromkeys(("google/gemini-3.8-flash", "x-ai/grok-4.6", "openai/gpt-5.6-terra", "openai/gpt-5.6-sol", "openai/gpt-5.6-luna", "openai::gpt-5.6-terra", "openai::gpt-5.6-sol", "openai::gpt-5.6-luna", "anthropic/claude-sonnet-5", "anthropic/claude-opus-5", "anthropic::claude-sonnet-5", "anthropic::claude-opus-5", "anthropic::claude-opus-4-6", "deepseek/deepseek-v4-pro", "deepseek::deepseek-v4-pro", "deepseek::deepseek-v4-flash", "openai-compatible::meta-llama/compatible", "cloudru::zai-org/GLM-4.7", "minimax::MiniMax-M3", "minimax::MiniMax-M2.7"))) +_MODEL_SUGGESTIONS = list(dict.fromkeys(("google/gemini-3.8-flash", "x-ai/grok-4.6", "openai/gpt-5.6-terra", "openai/gpt-5.6-sol", "openai/gpt-5.6-luna", "openai::gpt-5.6-terra", "openai::gpt-5.6-sol", "openai::gpt-5.6-luna", "anthropic/claude-sonnet-5", "anthropic/claude-opus-5", "anthropic::claude-sonnet-5", "anthropic::claude-opus-5", "anthropic::claude-opus-4-6", "deepseek/deepseek-v4-pro", "deepseek::deepseek-v4-pro", "deepseek::deepseek-v4-flash", "zai::glm-5.3", "zai::glm-5.3-flash", "openai-compatible::meta-llama/compatible", "cloudru::zai-org/GLM-4.7", "minimax::MiniMax-M3", "minimax::MiniMax-M2.7"))) def _string(value: Any) -> str: @@ -342,6 +347,7 @@ def derive_provider_profile(settings: dict) -> str: ("CLOUDRU_FOUNDATION_MODELS_API_KEY", "cloudru"), ("MINIMAX_API_KEY", "minimax"), ("DEEPSEEK_API_KEY", "deepseek"), + ("ZAI_API_KEY", "zai"), ("ANTHROPIC_API_KEY", "anthropic"), ] configured = [name for key, name in direct if flags[key]] @@ -556,7 +562,7 @@ def validate_setup_payload(data: dict, current_settings: dict) -> Tuple[dict, st has_remote = any( value for setting_key, value in keys.items() - if setting_key not in {"OPENAI_COMPATIBLE_API_KEY", "MINIMAX_REGION"} + if setting_key not in {"OPENAI_COMPATIBLE_API_KEY", "MINIMAX_REGION", "ZAI_PLAN"} ) has_local = bool(local_source) if not has_remote and not has_local and not (pending_subscription or selected_subscription): diff --git a/ouroboros/tools/core.py b/ouroboros/tools/core.py index fd506d3ed..1126e10f5 100644 --- a/ouroboros/tools/core.py +++ b/ouroboros/tools/core.py @@ -4,6 +4,7 @@ from __future__ import annotations from ouroboros.tools.tool_result import ToolResult, _publish_tool_result +import copy import logging import os import pathlib @@ -94,6 +95,7 @@ from ouroboros.tools.core_artifacts import ( # noqa: F401 QuizValidationError, _MAX_LINK_ACTIONS, _MAX_QUIZ_OPTIONS, + ESCALATE_TOOL_SCHEMA, _escalate, _send_links, validate_link_actions, @@ -1445,36 +1447,8 @@ def get_tools() -> List[ToolEntry]: "include": {"type": "string", "default": "", "description": "Basename glob, including brace alternatives (e.g. '*.py', '*.{js,css}'); applies on every backend"}, }, "required": ["query"]}, }, _code_search), - ToolEntry("escalate", { - "name": "escalate", - "description": ( - "Escalate a decision up the responsibility chain instead of guessing. " - "List 2-6 real options, mark your recommendation with recommended=true on that option, " - "and let each option's detail name what it gains and what it costs. " - "A root task asks the OWNER (a typed quiz card with option buttons); " - "a subagent asks its PARENT task (a typed mailbox frame the parent " - "answers with forward_to_worker or escalates higher, verbatim). " - "By default name your recommended option as the assumption and keep working: " - "the card stays answerable, and a late answer still arrives. " - "Set wait_for_answer=true (a live root, ordinary conversation included) when the next step " - "is irreversible or costly to redo, or the choice is the owner's to make " - "(spending, publishing, deleting); your judgment decides. The task then waits after the " - "current tool batch without model calls; waiting questions in one batch share one wait, " - "which ends on the first incoming message." - ), - "parameters": {"type": "object", "properties": { - "question": {"type": "string", "description": "The decision being escalated (markdown renders in chat)"}, - "options": {"type": "array", "items": {"type": "object", "properties": { - "label": {"type": "string", "description": "Short option label (button text, max 120)"}, - "detail": {"type": "string", "description": "Optional one-line consequence of this option (max 500)"}, - "recommended": {"type": "boolean", "description": "True on the ONE option you recommend"}, - }, "required": ["label"]}, "description": "2-6 mutually exclusive options"}, - "stake": {"type": "string", "description": "What depends on this decision (optional, max 500)"}, - "assumption": {"type": "string", "description": "For optional clarification, the assumption you continue under (max 500); may be empty for required waiting."}, - "wait_for_answer": {"type": "boolean", "default": False, "description": "Live roots: wait for addressed owner input before another model round."}, - "max_wait_minutes": {"type": "integer", "description": "Optional bound for wait_for_answer: resume with a system notice after N minutes if no answer arrives (the card stays open)."}, - }, "required": ["question", "options"]}, - }, _escalate), + # A fresh copy per catalog, as the inline literals are. + ToolEntry("escalate", copy.deepcopy(ESCALATE_TOOL_SCHEMA), _escalate), ToolEntry("forward_to_worker", { "name": "forward_to_worker", "description": ( diff --git a/ouroboros/tools/core_artifacts.py b/ouroboros/tools/core_artifacts.py index 713aaf61f..5895d13d1 100644 --- a/ouroboros/tools/core_artifacts.py +++ b/ouroboros/tools/core_artifacts.py @@ -294,7 +294,8 @@ def validate_link_actions(actions: Any) -> List[Dict[str, str]]: _MAX_QUIZ_OPTIONS = 6 -_MAX_QUIZ_QUESTION_CHARS = 2000 +# A label is button text; it is refused beyond this bound, never sliced. +_MAX_QUIZ_LABEL_CHARS = 120 class QuizValidationError(ValueError): @@ -318,13 +319,16 @@ def validate_quiz_payload( ``max_wait_minutes`` bounds a required wait only, and never past the task's absolute wall-clock ceiling: beyond it the ceiling would end the task first, so a larger bound would be a promise the runtime cannot keep. + + The authored text (question, option details, stake, assumption) is kept + whole: the card may be the only explanation its reader gets, so there is no + quiz-specific length cap and nothing is silently cut (BIBLE P1). Only a + label, which is button text, has a bound, and an over-long label refuses + the card instead of being sliced. """ q_text = str(question or "").strip() - if not q_text or len(q_text) > _MAX_QUIZ_QUESTION_CHARS: - raise QuizValidationError( - "QUIZ_QUESTION_INVALID", - f"question must be 1..{_MAX_QUIZ_QUESTION_CHARS} characters.", - ) + if not q_text: + raise QuizValidationError("QUIZ_QUESTION_INVALID", "question must be non-empty.") if not isinstance(options, list) or not 2 <= len(options) <= _MAX_QUIZ_OPTIONS: raise QuizValidationError( "QUIZ_OPTIONS_INVALID", @@ -344,9 +348,14 @@ def validate_quiz_payload( raise QuizValidationError( "QUIZ_OPTIONS_INVALID", "each option needs a non-empty label." ) - option: Dict[str, Any] = {"label": label[:120]} + if len(label) > _MAX_QUIZ_LABEL_CHARS: + raise QuizValidationError( + "QUIZ_OPTIONS_INVALID", + f"option labels must be at most {_MAX_QUIZ_LABEL_CHARS} characters.", + ) + option: Dict[str, Any] = {"label": label} if detail: - option["detail"] = detail[:500] + option["detail"] = detail if item.get("recommended") is True: # the asker's recommendation rides with its option option["recommended"] = True cleaned.append(option) @@ -367,12 +376,73 @@ def validate_quiz_payload( return { "question": q_text, "options": cleaned, - "stake": str(stake or "").strip()[:500], - "assumption": assumption_text[:500], + "stake": str(stake or "").strip(), + "assumption": assumption_text, **({"max_wait_minutes": bound} if bound is not None else {}), } +# The escalate catalog entry lives beside its validator and handler: the +# description and field texts define the same wire shape validate_quiz_payload +# enforces (core.get_tools registers it). +ESCALATE_TOOL_SCHEMA: Dict[str, Any] = { + "name": "escalate", + "description": ( + "Escalate one decision up the responsibility chain instead of guessing. " + "A root task asks its human through a quiz card; a subagent asks its parent " + "through a typed mailbox frame, which the parent answers with forward_to_worker " + "or escalates higher verbatim.\n\n" + "The card may be the only thing your human sees. It may be read later, outside " + "this room, by someone who remembers the purpose of the work without remembering " + "its mechanisms or your earlier discussion. Issue, specification and question " + "numbers, function names and reviewer names belong to your working context; they " + "are not shared context by themselves. The card carries the explanation needed to " + "understand this decision and its consequences.\n\n" + "The source of the fork belongs in that explanation, in ordinary prose: the human's " + "words, a document they supplied, your idea, or a reviewer's proposal. Human words " + "the fork rests on are quoted exactly when available; missing wording is disclosed, " + "and your paraphrase is identified as your interpretation. Your own proposal is a " + "distinct option, not an assumed premise of every option.\n\n" + "Offer 2-6 real alternatives for this decision and mark your recommendation with " + "recommended=true. By default, state the assumption you continue under and keep " + "working; the card stays answerable and a late answer still arrives. Set " + "wait_for_answer=true on a live root, including an ordinary conversation, when the " + "next step is irreversible or costly to redo, or the choice belongs to the human; " + "your judgment decides. Waiting begins after the current tool batch, without model " + "calls. Waiting questions in one batch share one wait, ending on the first incoming " + "message. How many decisions to raise and when remains your judgment." + ), + "parameters": {"type": "object", "properties": { + "question": {"type": "string", "description": ( + "The self-contained explanation of one decision, in the reader's language. " + "Markdown renders in chat. No quiz-specific character limit.")}, + "options": {"type": "array", "items": {"type": "object", "properties": { + "label": {"type": "string", "description": ( + "Short name of the choice, understandable to the reader on a button " + "(max 120 characters).")}, + "detail": {"type": "string", "description": ( + "What choosing this option changes for the human: what they gain and what " + "they give up, including any relevant cost, delay or lost capability (optional).")}, + "recommended": {"type": "boolean", "description": ( + "True on the one option you recommend; omitted or false otherwise.")}, + }, "required": ["label"]}, "description": "2-6 mutually exclusive alternatives for this decision."}, + "stake": {"type": "string", "description": ( + "What depends on this decision for the human or the work (optional).")}, + "assumption": {"type": "string", "description": ( + "What you will do while the question remains unanswered. Required when continuing " + "without a wait; may be empty when wait_for_answer=true. It is your assumption, " + "not the human's answer.")}, + "wait_for_answer": {"type": "boolean", "default": False, "description": ( + "Live roots: wait for addressed owner input before another model round. Default " + "false; an unanswered card remains answerable either way.")}, + "max_wait_minutes": {"type": "integer", "description": ( + "Optional positive whole-minute bound for wait_for_answer, within the task's " + "existing lifetime limit. On expiry, resume with a system notice; the card stays " + "answerable. Silence is not an answer.")}, + }, "required": ["question", "options"]}, +} + + def _validate_wait_bound(max_wait_minutes: Any, *, wait_for_answer: bool) -> Optional[int]: """The optional wait bound in whole minutes, capped by the task ceiling.""" if max_wait_minutes is None or not wait_for_answer: @@ -436,6 +506,62 @@ def _send_links( return "OK: link buttons queued for delivery to owner." +def _quiz_host_facts(ctx: ToolContext, canonical_root: pathlib.Path, task_id: str, chat_id: int) -> str: + """The host's one-sentence account under a root's owner card: which task asks, + how its run started, and when the owner last wrote in this chat. Read from + typed records only (the task record's ``run_origin`` provenance and the chat + log tail), never from the question text; an unrecorded fact says unknown.""" + from ouroboros.consciousness_authority import CONSCIOUSNESS_INITIATOR + from ouroboros.deadline_utils import parse_deadline_ts as moment, utc_now + from ouroboros.dialogue_provenance import run_origin + from ouroboros.task_results import load_task_result + from ouroboros.tools.followup import FOLLOWUP_SOURCE + from ouroboros.utils import iter_jsonl_objects + + def shown(when: Any) -> str: + return when.strftime("%Y-%m-%d %H:%M UTC") + + meta = getattr(ctx, "task_metadata", {}) if isinstance(getattr(ctx, "task_metadata", {}), dict) else {} + try: + record = load_task_result(canonical_root, task_id) or {} + except Exception: + record = {} + # The live metadata (which carries the owner door's stamp) laid over the + # persisted record's own, as the post-task synthesis reads the same origin. + metadata = {**(record.get("metadata") if isinstance(record.get("metadata"), dict) else {}), **meta} + origin = run_origin({**record, "metadata": metadata}) + ref = metadata.get("origin_message_ref") or record.get("origin_message_ref") + origin_task = str(origin.get("origin_task_id") or "") + if origin.get("owner_ingress"): + sent = moment(ref.get("ts")) if isinstance(ref, dict) else None + started = "started by your message" + (f" of {shown(sent)}" if sent else "") + elif origin.get("source") == FOLLOWUP_SOURCE and origin_task: + started = f"started as a scheduled follow-up of task {origin_task}" + elif origin.get("initiator") == CONSCIOUSNESS_INITIATOR: + started = "started by background consciousness" + elif origin.get("source") == "promote_chat_to_task": + started = "started by promotion" + (f" from task {origin_task}" if origin_task else "") + elif origin.get("schedule_id"): + started = f"started by schedule {origin['schedule_id']}" + elif origin_task: + started = f"started from task {origin_task}" + else: + started = "origin unknown" + (f" (recorded source: {origin['source']})" if origin.get("source") else "") + last = None + try: + for entry in iter_jsonl_objects(canonical_root / "logs" / "chat.jsonl", tail_bytes=512_000): + if entry.get("direction") == "in" and str(entry.get("chat_id")) == str(chat_id): + last = moment(entry.get("ts")) or last + except Exception: + last = None + if last is None: + seen = "your last message in this chat: unknown" + else: + minutes = max(0, int((utc_now() - last).total_seconds() // 60)) + seen = f"your last message in this chat: {shown(last)} ({minutes} minutes before this question)" + return f"Asked by task {task_id}, {started}; {seen}." + + def _escalate( ctx: ToolContext, question: str, @@ -581,6 +707,7 @@ def _escalate( card_chat_id = int(getattr(ctx, "current_chat_id", None) or 0) except (TypeError, ValueError): card_chat_id = 0 + host_facts = _quiz_host_facts(ctx, canonical_root, task_id, card_chat_id) asked = record_asked( canonical_root, task_id, quiz_id=quiz_id, question=payload["question"], @@ -590,6 +717,7 @@ def _escalate( stake=payload["stake"], assumption=payload["assumption"], wait_for_answer=wait_for_answer, chat_id=card_chat_id, max_wait_minutes=payload.get("max_wait_minutes"), + host_facts=host_facts, ) if asked.get("refused"): if wait_for_answer: @@ -610,6 +738,7 @@ def _escalate( "assumption": payload["assumption"], "state": "open", "task_id": task_id, + "host_facts": host_facts, **({"wait_for_answer": True} if wait_for_answer else {}), }) delivered = "delivered to the owner" if mode == "live" else "queued for the owner" diff --git a/skills/telegram/SKILL.md b/skills/telegram/SKILL.md index f6f6b399f..7b89acda4 100644 --- a/skills/telegram/SKILL.md +++ b/skills/telegram/SKILL.md @@ -1,7 +1,7 @@ --- name: telegram description: Owner-only Telegram text bridge and Mini App gateway for the existing Ouroboros interface. -version: 1.2.4 +version: 1.2.5 type: extension entry: plugin.py plugin_api: "2.0" @@ -52,6 +52,10 @@ form shows saved values before Save, and shortens the four long option labels. Version 1.2.4 always sends one short line when a task does not finish cleanly, with the same status word and reason sentence the task card shows; the task-completion toggle now only adds the clean finishes. +Version 1.2.5 shows the whole quiz card: its project, the host's facts about the +asking task, and every option's detail, with localized field names; a card too +long for one Telegram message arrives as ordered parts followed by the keyboard +message, and nothing authored is cut. The Mini App exposes the unchanged Ouroboros SPA through the established owner-authenticated sidecar and a pinned Cloudflare Quick Tunnel. It is enabled diff --git a/skills/telegram/lib/telegram_quiz.py b/skills/telegram/lib/telegram_quiz.py index 0cc1602b5..a0ea27189 100644 --- a/skills/telegram/lib/telegram_quiz.py +++ b/skills/telegram/lib/telegram_quiz.py @@ -11,6 +11,16 @@ of those happened. The only state kept here maps a short callback token and the sent message to that identity: Telegram caps ``callback_data`` at 64 bytes, too short for the ids themselves. Nothing here parses the owner's words; a reply is delivered verbatim as their own answer. + +The card carries the whole authored text: the project it belongs to, the host's +facts about the asking task, the question, the stake, and every option with its +detail. A card longer than one Telegram message is never cut: the explanation +goes out first as ordered plain messages through the client's existing chunker, +and a compact message with the project line, the numbered option labels, the +hint and the keyboard follows last. Only that last message is remembered, so a +tap, a reply to it and the answered-edit keep working exactly as for a short +card; a reply to one of the earlier explanation parts is an ordinary owner +message, deliberately without per-part bookkeeping. """ from __future__ import annotations @@ -19,6 +29,7 @@ import hashlib import json from typing import Any, Awaitable, Callable, Dict, List, Optional, Tuple +from .telegram_api import _TELEGRAM_TEXT_LIMIT, _u16len from .telegram_state import _read_json_file, _state_file _QUIZ_STATE_FILE = "quiz_state.json" @@ -26,6 +37,9 @@ _MAX_REMEMBERED = 50 _CALLBACK_PREFIX = "qz:" _BUTTON_LABEL_MAX = 40 _ANSWER_ECHO_MAX = 200 +# The answered edit appends "Answered: " to the remembered single message; a card is sent +# as ONE message only when that edit still fits Telegram's limit (echo + label + prefix + newline). +_ANSWERED_EDIT_RESERVE = _ANSWER_ECHO_MAX + 32 HostPost = Callable[[Any, str, Dict[str, Any]], Awaitable[Tuple[int, Dict[str, Any]]]] @@ -40,6 +54,11 @@ _TEXTS = { "gone": "This question is no longer known to Ouroboros.", "failed": "Could not deliver the answer (HTTP {status}). Try again.", "answered_line": "Answered: {answer}", + "project": "Project", + "question": "Question", + "stake": "At stake", + "meanwhile": "Continuing meanwhile", + "waiting": "Waiting for your answer; Stop and the task deadline still apply.", }, "ru": { "hint": "Нажмите вариант или ответьте на это сообщение своим текстом.", @@ -51,6 +70,11 @@ _TEXTS = { "gone": "Этот вопрос Ouroboros больше не знает.", "failed": "Не удалось передать ответ (HTTP {status}). Попробуйте ещё раз.", "answered_line": "Ответ: {answer}", + "project": "Проект", + "question": "Вопрос", + "stake": "Что на кону", + "meanwhile": "Пока продолжаю так", + "waiting": "Жду вашего ответа; Stop и срок задачи по-прежнему действуют.", }, } @@ -68,6 +92,28 @@ def mint_token(task_id: str, quiz_id: str) -> str: return hashlib.sha256(f"{task_id}:{quiz_id}".encode("utf-8")).hexdigest()[:12] +def card_options(raw_options: Any, *, limit: int) -> Tuple[List[str], List[str], Optional[int]]: + """Labels, details and the recommended index of the event's options.""" + labels: List[str] = [] + details: List[str] = [] + recommended: Optional[int] = None + for option in raw_options if isinstance(raw_options, list) else []: + label = str(option.get("label") or "").strip() if isinstance(option, dict) else "" + if not label or len(labels) >= limit: + continue + if option.get("recommended") is True and recommended is None: + recommended = len(labels) + labels.append(label) + details.append(str(option.get("detail") or "").strip()) + return labels, details, recommended + + +def button_labels(labels: List[str], recommended_index: Optional[int]) -> List[str]: + """Button captions: the recommended option carries a leading star.""" + return [f"★ {label}" if index == recommended_index else label + for index, label in enumerate(labels)] + + def quiz_keyboard(token: str, labels: List[str]) -> List[List[dict]]: """One button row per option; ``callback_data`` = ``qz::``.""" return [ @@ -77,19 +123,72 @@ def quiz_keyboard(token: str, labels: List[str]) -> List[List[dict]]: ] +def _project_line(project_name: str, lang: str) -> List[str]: + name = str(project_name or "").strip() + return [f"{_texts(lang)['project']}: {name}"] if name else [] + + +def _option_line(index: int, label: str, *, detail: str = "", recommended: bool = False) -> str: + line = f"{index}. {'★ ' if recommended else ''}{label}" + return f"{line} — {detail}" if detail else line + + def render_quiz_text(question: str, labels: List[str], stake: str, assumption: str, - *, wait_for_answer: bool = False) -> str: - lines = [f"Question: {question}"] + *, wait_for_answer: bool = False, project_name: str = "", + option_details: Optional[List[str]] = None, + recommended_index: Optional[int] = None, + host_facts: str = "", lang: str = "en") -> str: + """The whole card body, never shortened: every authored field is kept.""" + texts = _texts(lang) + details = list(option_details or []) + lines = _project_line(project_name, lang) + if host_facts: + lines.append(host_facts) + lines.append(f"{texts['question']}: {question}") if stake: - lines.append(f"At stake: {stake}") - lines.extend(f"{index}. {label}" for index, label in enumerate(labels, 1)) + lines.append(f"{texts['stake']}: {stake}") + lines.extend( + _option_line(index, label, + detail=str(details[index - 1] or "") if index - 1 < len(details) else "", + recommended=recommended_index == index - 1) + for index, label in enumerate(labels, 1) + ) if wait_for_answer: - lines.append("Waiting for your answer; Stop and the task deadline still apply.") + lines.append(texts["waiting"]) elif assumption: - lines.append(f"Continuing meanwhile: {assumption}") + lines.append(f"{texts['meanwhile']}: {assumption}") return "\n".join(lines) +def render_compact_text(labels: List[str], *, project_name: str = "", + recommended_index: Optional[int] = None, lang: str = "en") -> str: + """The keyboard message of an overflowing card: project and numbered labels.""" + lines = _project_line(project_name, lang) + lines.extend(_option_line(index, label, recommended=recommended_index == index - 1) + for index, label in enumerate(labels, 1)) + return "\n".join(lines) + + +async def send_quiz_card(client, chat_id: int, *, body: str, compact: str, hint_text: str, + keyboard: List[List[dict]]) -> Tuple[int, bool]: + """Send the card; return the keyboard message id and whether it overflowed. + + A card that fits Telegram's per-message limit (UTF-16 units) WITH room for + its later answered edit is one message with the keyboard. A longer one is + sent as ordered plain parts through the client's chunker, then the compact + keyboard message; nothing authored is truncated. + """ + full = f"{body}\n{hint_text}" + if _u16len(full) + _ANSWERED_EDIT_RESERVE <= _TELEGRAM_TEXT_LIMIT: + message_id = await client.send_message_with_inline_keyboard( + chat_id, full, keyboard, parse_mode="") + return int(message_id or 0), False + await client.send_message(chat_id, body, parse_mode="") + message_id = await client.send_message_with_inline_keyboard( + chat_id, f"{compact}\n{hint_text}", keyboard, parse_mode="") + return int(message_id or 0), True + + def _load(api) -> Dict[str, Any]: data = _read_json_file(_state_file(api, _QUIZ_STATE_FILE)) quizzes = data.get("quizzes") if isinstance(data, dict) else None diff --git a/skills/telegram/plugin.py b/skills/telegram/plugin.py index 53d62d888..88de29aa1 100644 --- a/skills/telegram/plugin.py +++ b/skills/telegram/plugin.py @@ -1307,14 +1307,10 @@ def _make_quiz(api): return question = str(event.get("question") or "").strip() raw_options = event.get("options") if isinstance(event.get("options"), list) else [] - labels = [] - for option in raw_options: - if isinstance(option, dict): - label = str(option.get("label") or "").strip() - if label: - labels.append(f"★ {label}" if option.get("recommended") is True else label) # Shared quiz contract cap: ouroboros.tools.core._MAX_QUIZ_OPTIONS. - labels = labels[:6] + plain_labels, details, recommended_index = telegram_quiz.card_options(raw_options, limit=6) + # Buttons and the remembered record keep the starred caption. + labels = telegram_quiz.button_labels(plain_labels, recommended_index) if not question or len(labels) < 2: return task_id = str(event.get("task_id") or "").strip() @@ -1326,21 +1322,36 @@ def _make_quiz(api): assumption = str(event.get("assumption") or "").strip() lang = _poller_preferences(api)[4] wait_for_answer = event.get("wait_for_answer") is True + # Both optional: events from an older host carry neither. + project_name = str(event.get("project_name") or "").strip() + host_facts = str(event.get("host_facts") or "").strip() + card = { + "project_name": project_name, "option_details": details, + "recommended_index": recommended_index, "host_facts": host_facts, "lang": lang, + } body = telegram_quiz.render_quiz_text( - question, labels, stake, assumption, wait_for_answer=wait_for_answer) + question, plain_labels, stake, assumption, wait_for_answer=wait_for_answer, **card) + compact = telegram_quiz.render_compact_text( + plain_labels, project_name=project_name, recommended_index=recommended_index, lang=lang) token = telegram_quiz.mint_token(task_id, quiz_id) # One button per option; a reply to the card is a free-form answer. - # Both reach the host's decision ingress (#472). - message_id = await client.send_message_with_inline_keyboard( - chat_id, f"{body}\n{telegram_quiz.hint(lang)}", - telegram_quiz.quiz_keyboard(token, labels), parse_mode="", + # Both reach the host's decision ingress (#472). An overflowing card + # arrives as plain parts followed by the compact keyboard message. + message_id, overflowed = await telegram_quiz.send_quiz_card( + client, chat_id, body=body, compact=compact, hint_text=telegram_quiz.hint(lang), + keyboard=telegram_quiz.quiz_keyboard(token, labels), ) + if overflowed: + settled_text = compact + elif wait_for_answer: + # The answer edit keeps optional history, not a live waiting claim. + settled_text = telegram_quiz.render_quiz_text(question, plain_labels, stake, "", **card) + else: + settled_text = body telegram_quiz.remember_quiz(api, token, { "task_id": task_id, "quiz_id": quiz_id, "chat_id": chat_id, "message_id": int(message_id or 0), "options": labels, - # The answer edit keeps optional history, not a live waiting claim. - "text": (telegram_quiz.render_quiz_text(question, labels, stake, "") - if wait_for_answer else body), + "text": settled_text, }) except Exception as exc: api.log("error", f"Telegram quiz error: {exc}") diff --git a/supervisor/events_chat_delivery.py b/supervisor/events_chat_delivery.py index dd5ef71c6..a5d967e18 100644 --- a/supervisor/events_chat_delivery.py +++ b/supervisor/events_chat_delivery.py @@ -457,6 +457,7 @@ def _handle_send_quiz(evt: Dict[str, Any], ctx: Any) -> None: assumption=str(evt.get("assumption") or ""), state=str(evt.get("state") or "open"), task_id=str(evt.get("task_id") or ""), + host_facts=str(evt.get("host_facts") or ""), **({"wait_for_answer": True} if evt.get("wait_for_answer") is True else {}), ) if not ok: diff --git a/supervisor/message_bus.py b/supervisor/message_bus.py index e61784c43..391f680c7 100644 --- a/supervisor/message_bus.py +++ b/supervisor/message_bus.py @@ -957,8 +957,12 @@ class LocalChatBridge: state: str = "open", task_id: str = "", wait_for_answer: bool = False, + host_facts: str = "", ) -> Tuple[bool, str]: - """Send an owner quiz card to the UI and host event subscribers.""" + """Send an owner quiz card to the UI and host event subscribers. + + ``host_facts`` (the host's sentence under the question) rides the frame, the + event and the chat row when non-empty; ``project_name`` rides the EVENT only.""" if is_a2a_chat_id(chat_id): return True, "ok" qid = str(quiz_id or "").strip() @@ -988,7 +992,18 @@ class LocalChatBridge: "chat_id": int(chat_id or 0), "task_id": str(task_id or ""), } + if host_facts: + msg["host_facts"] = str(host_facts) # the envelope literal keeps constant keys (contract scan) stamp_project_thread(DATA_DIR, msg) + project = None + if msg.get("project_thread"): + try: + from ouroboros.projects_registry import list_reserved_projects + + project = next((row for row in list_reserved_projects(DATA_DIR) + if row.get("chat_id") == int(chat_id)), None) + except Exception: + log.debug("Quiz project lookup failed", exc_info=True) if self._broadcast_fn: self._broadcast_fn(msg) quiz_transport = dict(self._chat_transports.get(int(chat_id or 0), {}) or {}) @@ -1004,6 +1019,8 @@ class LocalChatBridge: "assumption": payload["assumption"], "state": str(state or "open"), "ts": ts, + **({"host_facts": str(host_facts)} if host_facts else {}), + **({"project_name": str(project["name"])} if project and project.get("name") else {}), }) try: owner_id = int(load_state().get("owner_id") or 0) @@ -1019,6 +1036,7 @@ class LocalChatBridge: "stake": payload["stake"], "assumption": payload["assumption"], "state": str(state or "open"), + **({"host_facts": str(host_facts)} if host_facts else {}), }, ) _advance_project_visible_revision(chat_id) @@ -1026,10 +1044,7 @@ class LocalChatBridge: try: from ouroboros.owner_quiz import quiz_states from ouroboros.project_dialogue import project_question_pointer - from ouroboros.projects_registry import list_reserved_projects - project = next((row for row in list_reserved_projects(DATA_DIR) - if row.get("chat_id") == int(chat_id)), None) pointer = project_question_pointer(msg, quiz_states(DATA_DIR, task_id).get(qid), project) if pointer: frame = { @@ -1045,7 +1060,7 @@ class LocalChatBridge: # The complete pointer row (ChatOutbound mirrors): present only when known. for key in ("question", "options", "option_details", "stake", "assumption", "recommended_index", "answered_index", "comment", "wait_for_answer", "wait_ended_at", - "owner_wait_resume_reason"): + "owner_wait_resume_reason", "host_facts"): if key in pointer: frame[key] = pointer[key] self._broadcast_fn(frame) diff --git a/supervisor/worker_chat_lane.py b/supervisor/worker_chat_lane.py index ac82043bd..6714affb2 100644 --- a/supervisor/worker_chat_lane.py +++ b/supervisor/worker_chat_lane.py @@ -298,6 +298,15 @@ def _admit_chat_task( task["task_constraint"] = dict(task_constraint) if task_metadata: task["metadata"] = dict(task_metadata) + if task_metadata.get("late_answer") is not None: + # A LATE quiz answer: the owner's row holds only their words; the + # model reads the card they answered, rebuilt from the stored + # block (disclosed when unreadable) -- the one shared builder. + from ouroboros.owner_quiz import late_answer_model_text + + task["text"] = late_answer_model_text( + _pool().DRIVE_ROOT, task_metadata.get("late_answer"), str(text or ""), + ) # The ingress-captured origin identity rides on the TASK RECORD so a # later post-hoc "Turn into project" reads it from the persisted # result instead of re-deriving identity from content. diff --git a/tests/test_config_extraction.py b/tests/test_config_extraction.py index 1396530c2..a80814e9f 100644 --- a/tests/test_config_extraction.py +++ b/tests/test_config_extraction.py @@ -318,5 +318,6 @@ def test_settings_extraction_size_bounds_have_meaningful_headroom(): } assert counts["ouroboros.config"] <= 1000 assert all(count <= 1000 for count in counts.values()) - assert counts["ouroboros.settings_defaults"] <= 500 + # 500 -> 520: the Z.ai direct provider adds its key and plan rows to the leaf (PR #1207). + assert counts["ouroboros.settings_defaults"] <= 520 assert (PACKAGE / "config.py").is_file() diff --git a/tests/test_consciousness_wake.py b/tests/test_consciousness_wake.py index 9bf883092..3520457a1 100644 --- a/tests/test_consciousness_wake.py +++ b/tests/test_consciousness_wake.py @@ -90,6 +90,59 @@ def test_events_list_settled_tasks_open_cards_and_owner_messages_since_the_last_ assert wake.wake_events(tmp_path / "missing", since=since, now=T0) == [] + +def test_answered_cards_of_the_window_are_events_with_their_facts(tmp_path): + since = T0 - 3600 + long_comment = "Keep the old parser.\n\nReason: " + "the migration cost is too high; " * 120 + "END-OF-COMMENT" + _result(tmp_path, "early", ts=since + 100, cost=0.25, description="earlier settled work") + _result(tmp_path, "late", ts=since + 3000, cost=0.75, description="later settled work") + _result(tmp_path, "asker", ts=since - 500, status="completed", quizzes={ + # Answered by a button inside the window: stamps, the chosen label, a question preview. + "qbtn": {"state": "answered", "asked_at": _iso(since - 7200), "answered_at": _iso(since + 600), + "options": ["Rewrite it", "Keep it"], "answered_index": 1, + "question": "Should the parser be rewritten before the release?"}, + # Answered in the owner's own words inside the window: the words arrive whole. + "qown": {"state": "answered", "asked_at": _iso(since + 60), "answered_at": _iso(since + 2400), + "options": ["A", "B"], "comment": long_comment, "question": "Which parser?"}, + # Answered before the window: not news for this wake. + "qold": {"state": "answered", "asked_at": _iso(since - 9000), "answered_at": _iso(since - 60), + "options": ["A", "B"], "answered_index": 0, "question": "Old question"}, + # Still unanswered after its task finished: listed as before, in the card list. + "qopen": {"state": "expired_terminal", "asked_at": _iso(since - 300), "question": "Still waiting?"}, + }) + lines = wake.wake_events(tmp_path, since=since, now=T0, reason="task_finished:asker:completed") + btn = next(line for line in lines if line.startswith("- owner card qbtn ")) + assert btn == ("- owner card qbtn on task asker: answered 50 min ago; asked 3 h 0 min ago; " + "chose option 2: Keep it; question: Should the parser be rewritten before the release?") + own = next(line for line in lines if line.startswith("- owner card qown ")) + assert own.startswith("- owner card qown on task asker: answered 20 min ago; asked 59 min ago; " + "answered in own words: Keep the old parser.") + assert long_comment in own and "END-OF-COMMENT; question: Which parser?" in own # whole, never clipped + assert not any("qold" in line for line in lines) + assert any(line.startswith("- owner card qopen on task asker: Unanswered · the task finished") for line in lines) + # Answers sort with the settled facts by their own stamp (newest first), after the trigger; + # the trigger's de-duplication of its own task never hides an answer on that task. + assert lines[0].startswith("- wake cause: task asker finished") + events = [line.split()[1:4] for line in lines[1:] if not line.startswith("- owner card qopen ")] + assert events == [["task", "late", "completed,"], ["owner", "card", "qown"], + ["owner", "card", "qbtn"], ["task", "early", "completed,"]] + # No verdict words about what the owner meant. + assert not any(word in btn + own for word in ("understood", "confused", "misunderstood")) + + +def test_answered_card_line_renders_mixed_answers_and_missing_facts_honestly(): + now = T0 + both = wake._answered_card_line("t1", "q1", { + "answered_at": _iso(now - 120), "asked_at": _iso(now - 240), "options": ["Go", "Stop"], + "answered_index": 0, "comment": "but only on weekdays", "question": "Deploy?"}, now=now) + assert both == ("- owner card q1 on task t1: answered 2 min ago; asked 4 min ago; " + "chose option 1: Go; with the words: but only on weekdays; question: Deploy?") + bare = wake._answered_card_line("t1", "q2", {"answered_at": "not a stamp", "answered_index": 5}, now=now) + assert bare == ("- owner card q2 on task t1: answered at an unknown time; asked at an unknown time; " + "chose option 6: label unavailable; question: question text unavailable") + empty = wake._answered_card_line("t1", "q3", {"answered_at": _iso(now - 60)}, now=now) + assert "answer text unavailable" in empty + @pytest.mark.parametrize("status, origin", [("failed", "host_notice"), ("cancelled", ""), ("completed", "host_notice"), ("failed", "model_final")]) def test_failed_inline_presence_is_visible_on_regular_wake_without_reviving_owner_turns(tmp_path, status, origin): diff --git a/tests/test_consolidation_honesty.py b/tests/test_consolidation_honesty.py index 966e331f4..e19ca65fe 100644 --- a/tests/test_consolidation_honesty.py +++ b/tests/test_consolidation_honesty.py @@ -198,18 +198,39 @@ def test_partial_publication_records_the_batch_receipt_in_meta(tmp_path, fit, mo lambda *_a, **_k: [{"topic": "people/alex", "ok": False, "reason": "revision_conflict"}]) ctx = ToolContext(repo_dir=tmp_path, drive_root=tmp_path, task_id="partial") c.consolidate(chat, blocks, meta, _Nominating(), knowledge_context=ctx) - receipt = json.loads(meta.read_text())["last_unpublished_nominations"] - assert receipt["failed"] == 1 and receipt["total"] == 1 and receipt["entry_id"] + pending = json.loads(meta.read_text())["pending_knowledge_nominations"] + assert len(pending) == 1 + assert pending[0]["topic"] == "people/alex" and pending[0]["reason"] == "revision_conflict" + assert pending[0]["id"].endswith(":0:0") -def test_a_fully_published_batch_clears_the_receipt(tmp_path, fit): +def test_new_success_does_not_erase_an_old_failed_entry_or_legacy_receipt(tmp_path, fit, monkeypatch): chat, blocks, meta = _paths(tmp_path) _write_chat(chat, count=100, text_size=0) meta.parent.mkdir(parents=True, exist_ok=True) - c.atomic_write_json(meta, {"last_unpublished_nominations": {"entry_id": "old", "failed": 3, "total": 4}}) + legacy = {"entry_id": "old", "failed": 3, "total": 4} + c.atomic_write_json(meta, {"last_unpublished_nominations": legacy}) ctx = ToolContext(repo_dir=tmp_path, drive_root=tmp_path, task_id="clean") + original = c._write_knowledge_entries + calls = 0 + + def fail_once(*args, **kwargs): + nonlocal calls + calls += 1 + if calls == 1: + return [{"topic": "people/alex", "scope": "global", "ok": False, + "reason": "revision_conflict"}] + return original(*args, **kwargs) + + monkeypatch.setattr(c, "_write_knowledge_entries", fail_once) c.consolidate(chat, blocks, meta, _Nominating(), knowledge_context=ctx) - assert "last_unpublished_nominations" not in json.loads(meta.read_text()) + older = json.loads(meta.read_text())["pending_knowledge_nominations"][0] + _write_chat(chat, count=200, text_size=0) + c.consolidate(chat, blocks, meta, _Nominating(), knowledge_context=ctx) + saved = json.loads(meta.read_text()) + assert saved["last_unpublished_nominations"] == legacy + assert saved["pending_knowledge_nominations"] == [older] + assert calls == 2 def test_a_run_without_nominations_leaves_the_receipt_alone(tmp_path, fit): @@ -235,8 +256,9 @@ def test_the_receipt_survives_era_compression(tmp_path, fit, monkeypatch): saved_blocks = json.loads(blocks.read_text()) assert saved_blocks[0]["type"] == "era" assert "knowledge_writes" not in saved_blocks[0] - receipt = json.loads(meta.read_text())["last_unpublished_nominations"] - assert receipt["failed"] == 11 and receipt["total"] == 11 + pending = json.loads(meta.read_text())["pending_knowledge_nominations"] + assert len(pending) == 11 + assert len({row["id"] for row in pending}) == 11 # --- the Health block is where stale memory becomes visible ----------------------- @@ -297,3 +319,99 @@ def test_unreadable_receipts_do_not_raise_or_shout(tmp_path, payload): env = _health_env(tmp_path) c.atomic_write_json(tmp_path / "memory" / "dialogue_meta.json", payload) assert not any("DIALOGUE" in line for line in context_health._memory_health_lines(env)) + + +def test_invalid_legacy_receipt_does_not_impersonate_unreadable_meta(tmp_path): + env = _health_env(tmp_path) + c.atomic_write_json(tmp_path / "memory" / "dialogue_meta.json", + {"last_unpublished_nominations": {"failed": "many", "total": 2}}) + lines = context_health._memory_health_lines(env) + assert any("LEGACY NOMINATION RECEIPT INVALID" in line for line in lines) + assert not any("DIALOGUE META UNREADABLE" in line for line in lines) + + +def test_pending_receipt_precedes_the_note_writer_and_cannot_be_replaced_by_corrupt_meta(tmp_path, fit, monkeypatch): + chat, blocks, meta = _paths(tmp_path) + _write_chat(chat, count=100, text_size=0) + ctx = ToolContext(repo_dir=tmp_path, drive_root=tmp_path, task_id="interrupted") + + def interrupted(*_args, **_kwargs): + saved = json.loads(meta.read_text()) + assert len(saved["pending_knowledge_nominations"]) == 1 + assert saved["pending_knowledge_nominations"][0]["reason"] == "publication_pending" + raise RuntimeError("simulated stop after pending publication") + + monkeypatch.setattr(c, "_write_knowledge_entries", interrupted) + with pytest.raises(RuntimeError, match="simulated stop"): + c.consolidate(chat, blocks, meta, _Nominating(), knowledge_context=ctx) + saved = json.loads(meta.read_text()) + assert saved["pending_knowledge_nominations"][0]["reason"] == "publication_pending" + assert saved.get("last_consolidated_offset", 0) == 0 + + +def test_health_projects_three_owed_addresses_and_omission_count(tmp_path): + env = _health_env(tmp_path) + rows = [{"id": f"source{i}:0:0", "scope": "global", "topic": f"people/{i}", + "reason": "revision_conflict"} for i in range(5)] + c.atomic_write_json(tmp_path / "memory" / "dialogue_meta.json", + {"pending_knowledge_nominations": rows}) + lines = context_health._memory_health_lines(env) + row = next(line for line in lines if "KNOWLEDGE PUBLICATION OPEN" in line) + assert "5 source-addressed" in row and "first 3" in row and "omitted 2" in row + assert "source0" in row and "source2" in row and "source3" not in row + assert "memory/knowledge_history.jsonl" in row + + +def test_malformed_nomination_keeps_its_position_and_cannot_retire_another_entry(tmp_path): + from ouroboros.memory_nomination_receipts import prepare, settle + + meta = {} + ids = prepare(meta, "source", [({}, [None, {"topic": "people/alex", "content": "Valid"}])]) + outcomes = c._write_knowledge_entries(tmp_path / "memory" / "knowledge", [None, + {"topic": "people/alex", "content": "Valid"}]) + assert len(outcomes) == 2 and outcomes[0]["reason"] == "malformed_nomination" + assert outcomes[1]["ok"] + settle(meta, ids, outcomes) + assert [row["id"] for row in meta["pending_knowledge_nominations"]] == ["source:0:0"] + + +def test_corrupt_obligation_index_refuses_replacement(tmp_path, fit): + chat, blocks, meta = _paths(tmp_path) + _write_chat(chat, count=100, text_size=0) + meta.parent.mkdir(parents=True, exist_ok=True) + c.atomic_write_json(meta, {"pending_knowledge_nominations": {"not": "a list"}}) + ctx = ToolContext(repo_dir=tmp_path, drive_root=tmp_path, task_id="corrupt") + with pytest.raises(ValueError, match="refusing to replace"): + c.consolidate(chat, blocks, meta, _Nominating(), knowledge_context=ctx) + assert not blocks.exists() + assert json.loads(meta.read_text())["pending_knowledge_nominations"] == {"not": "a list"} + + +@pytest.mark.parametrize("bad_bytes", [b'{"pending_knowledge_nominations":[{"id":"old"}]', + b'["wrong top-level type"]', + b'{"pending_knowledge_nominations":[],"pending_knowledge_nominations":[]}']) +def test_unreadable_existing_meta_cannot_erase_obligations(tmp_path, fit, bad_bytes): + chat, blocks, meta = _paths(tmp_path) + _write_chat(chat, count=100, text_size=0) + meta.parent.mkdir(parents=True, exist_ok=True) + meta.write_bytes(bad_bytes) + ctx = ToolContext(repo_dir=tmp_path, drive_root=tmp_path, task_id="corrupt") + with pytest.raises(ValueError): + c.should_consolidate(meta, chat) + with pytest.raises(ValueError): + c.consolidate(chat, blocks, meta, _Nominating(), knowledge_context=ctx) + assert meta.read_bytes() == bad_bytes + assert not blocks.exists() + assert any("DIALOGUE META UNREADABLE" in line for line in + context_health._memory_health_lines(_health_env(tmp_path))) + + +def test_pending_health_disambiguates_two_entries_from_one_source(tmp_path): + env = _health_env(tmp_path) + c.atomic_write_json(tmp_path / "memory" / "dialogue_meta.json", { + "pending_knowledge_nominations": [ + {"id": "a" * 64 + f":{index}:0", "reason": "revision_conflict"} + for index in (0, 1)]}) + row = next(line for line in context_health._memory_health_lines(env) + if "KNOWLEDGE PUBLICATION OPEN" in line) + assert "aaaaaaaaaaaa:0:0" in row and "aaaaaaaaaaaa:1:0" in row diff --git a/tests/test_context_memory.py b/tests/test_context_memory.py index 13d1299f9..0dd033b07 100644 --- a/tests/test_context_memory.py +++ b/tests/test_context_memory.py @@ -388,6 +388,30 @@ def test_retired_dialogue_summary_remains_visible_when_blocks_exist(tmp_path): assert "legacy dialogue" in combined +def test_era_block_carries_a_host_note_and_other_blocks_stay_unchanged(tmp_path): + from ouroboros.context import build_memory_sections + from ouroboros.memory import Memory + + note = ("Host note: compression of older dialogue blocks; an interpretation, not a grant " + "or a standing rule. Range: {range}; source messages: {count}.") + memory_dir = tmp_path / "memory" + memory_dir.mkdir(parents=True, exist_ok=True) + (memory_dir / "dialogue_blocks.json").write_text(json.dumps([ + {"type": "era", "range": "2026-07-01 to 2026-08-31", "message_count": 412, + "content": "### Era: summer\nThe owner preferred X."}, + {"type": "era", "content": "### Era: undated\nOlder stuff."}, + {"type": "summary", "range": "2026-09-01", "message_count": 30, "content": "### Block: 2026-09-01\nRecent."}, + ]), encoding="utf-8") + combined = "\n\n".join(build_memory_sections(Memory(drive_root=tmp_path), partition="volatile")) + + dated = note.format(range="2026-07-01 to 2026-08-31", count=412) + "\n### Era: summer\nThe owner preferred X." + undated = note.format(range="unknown", count="unknown") + "\n### Era: undated\nOlder stuff." + assert "## Dialogue History\n\n" + dated + "\n\n" + undated + "\n\n### Block: 2026-09-01\nRecent." in combined + assert combined.count("Host note:") == 2 # the summary block gets no note + # The renderer itself: a summary-only list is byte-identical to its contents. + assert Memory.format_blocks_as_markdown([{"type": "summary", "content": "a"}, {"content": "b"}]) == "a\n\nb" + + def test_retired_dialogue_summary_fallback_preserves_continuity_without_blocks(tmp_path): from ouroboros.context import build_memory_sections from ouroboros.memory import Memory diff --git a/tests/test_context_memory_preparation.py b/tests/test_context_memory_preparation.py index 3d7e69a15..27519fdf8 100644 --- a/tests/test_context_memory_preparation.py +++ b/tests/test_context_memory_preparation.py @@ -2,6 +2,8 @@ import json +import pytest + from ouroboros import context from ouroboros.context_fit import estimate_context_prompt_tokens from ouroboros.tools.registry import ToolContext @@ -50,6 +52,30 @@ def test_actual_nano_preparation_consolidates_complete_source_before_returning(t assert len(actor.calls) == calls +@pytest.mark.parametrize("broken", [b"{bad", b"[]", b'{"pending_knowledge_nominations":[],"pending_knowledge_nominations":[]}']) +def test_nano_unreadable_dialogue_meta_withholds_maintenance_not_main(tmp_path, fit, monkeypatch, broken): + env, memory, task, ctx, chat_before = _setup(tmp_path) + meta = env.drive_root / "memory/dialogue_meta.json" + meta.write_bytes(broken) + monkeypatch.setattr(context, "get_context_mode", lambda: "nano") + actor = SourceReader(env.drive_root, fit.window) + messages, info = context.build_llm_messages( + env, memory, task, ctx=ctx, llm=actor, tool_schemas=[], + fit_candidate=lambda _messages, _tools: {"accepted": False}, + ) + receipt = info["context_memory_maintenance"] + assert receipt["status"] == "no_progress" + assert receipt["usage"]["_consolidation_errors"][0]["kind"] == "dialogue_meta_unreadable" + assert messages[-1]["content"] == task["text"] + assert meta.read_bytes() == broken + assert (env.drive_root / "logs/chat.jsonl").read_text() == chat_before + assert not actor.calls + events = [json.loads(line) for line in (env.drive_root / "logs/events.jsonl").read_text().splitlines()] + assert any(row.get("type") == "context_memory_maintenance" and + row["usage"]["_consolidation_errors"][0]["kind"] == "dialogue_meta_unreadable" + for row in events) + + def test_max_and_pure_preview_never_start_a_maintenance_model(tmp_path, fit, monkeypatch): env, memory, task, ctx, _raw = _setup(tmp_path) actor = SourceReader(env.drive_root, 50000) diff --git a/tests/test_core_extraction.py b/tests/test_core_extraction.py index 74fc03e64..c2d379c56 100644 --- a/tests/test_core_extraction.py +++ b/tests/test_core_extraction.py @@ -122,9 +122,14 @@ def test_core_catalog_schema_bytes_and_handler_owners_are_stable(): # names its peer addressees (your own parent or a sibling, delivered as a message from a # peer task naming the relation; relay refused there), the 8000-char body bound and the # await_messages companion (395 -> 698 bytes); the `message` parameter description states - # the bound. Diffing the whole catalog base to head shows exactly those edits and nothing else. + # the bound. Rolled again for the truthful owner-question work (PR1, owner 1D/2A): the + # escalate description and its nine field descriptions were replaced (the card is written + # for a reader outside the room, names the source of the fork, and the question has no + # quiz-specific length cap); schema shape, types, defaults and required keys are unchanged, + # and the entry's literal moved beside its validator in core_artifacts (byte-identical + # serialization). Diffing the whole catalog base to head shows exactly those edits and nothing else. assert hashlib.sha256(schema_bytes).hexdigest() == ( - "7195a7459f276f4bdf8758864b079218f518ae446613cc8351967692f31472cd" + "86cf723bf9aa0744b379a0e584034519bc178db1787f82dfd33ed6cddfa6f4d4" ) assert { entry.name: (entry.handler.__module__, entry.handler.__name__) diff --git a/tests/test_docs_sync.py b/tests/test_docs_sync.py index 91a3120a0..c784913ed 100644 --- a/tests/test_docs_sync.py +++ b/tests/test_docs_sync.py @@ -224,7 +224,7 @@ def test_architecture_mentions_shared_log_grouping_and_direct_provider_review_fa # silently re-expand to claim symmetric coverage it does not have yet. assert "Direct-provider review fallback" in arch assert "OpenAI-only review fallback" in arch # legacy name still referenced for discoverability - assert "official OpenAI, Anthropic, MiniMax, DeepSeek, Cloud.ru, and GigaChat" in arch + assert "official OpenAI, Anthropic, MiniMax, DeepSeek, Z.ai, Cloud.ru, and GigaChat" in arch assert "_exclusive_direct_remote_provider_env" in arch # v4.34.0: direct-provider fallback now documents the # `main_model.startswith(provider_prefix)` guard in get_review_models — diff --git a/tests/test_knowledge_consolidation.py b/tests/test_knowledge_consolidation.py index 36de700ea..12165ca14 100644 --- a/tests/test_knowledge_consolidation.py +++ b/tests/test_knowledge_consolidation.py @@ -229,5 +229,7 @@ def test_era_compression_cannot_erase_unpublished_knowledge_proposals(tmp_path, # The era object carries no knowledge_writes, so the batch receipt lives in meta: # without it the incomplete publication would vanish from every resident surface. assert "knowledge_writes" not in saved[0] - receipt = json.loads(meta.read_text())["last_unpublished_nominations"] - assert receipt == {"entry_id": nominations["entry_id"], "failed": 11, "total": 11} + pending = json.loads(meta.read_text())["pending_knowledge_nominations"] + assert len(pending) == 11 + assert all(row["id"].startswith(nominations["entry_id"] + ":") for row in pending) + assert all(row["reason"] == "revision_required" for row in pending) diff --git a/tests/test_memory_journal_compaction.py b/tests/test_memory_journal_compaction.py index c994c3beb..b44738c25 100644 --- a/tests/test_memory_journal_compaction.py +++ b/tests/test_memory_journal_compaction.py @@ -1,223 +1,92 @@ -"""CPL4-C16 pins (owner batch №8, 4A): old journal snapshots go digest-only. +"""Old memory-journal snapshots remain readable after every maintenance pass. -Fresh entries keep their full old/new text; entries older than GC retention -keep only sha256 + length and gain ``content_digested``. Unparseable lines and -rows without a readable ``ts`` survive byte-identical; the consciousness -observation inbox is out of scope. - -Audit #15-11 corrective lane: this compactor is the one sweep that destroys -CONTENT, so it also pins that a stored digest is verified before the text it -describes is deleted, that the rewrite publishes only a source nothing else -touched, and that the journal is never loaded whole. +Previously this startup sweep digested old knowledge, identity and Pattern +Register old/new contents. A digest cannot restore the complete source after +retention; the compatibility entry point is intentionally non-destructive. """ from __future__ import annotations -import hashlib +import inspect import json -import os -import pathlib import pytest -from ouroboros import memory_journal_compaction as mjc from ouroboros.memory_journal_compaction import compact_memory_journal_snapshots -from ouroboros.utils import utc_now_iso -_OLD_TS = "2020-01-01T00:00:00+00:00" +_JOURNALS = ( + "memory/identity_journal.jsonl", + "memory/knowledge_history.jsonl", + "memory/knowledge/patterns_history.jsonl", + "projects/example/knowledge_history.jsonl", + "memory/scratchpad_journal.jsonl", +) -def _journal(tmp_path, rel): - path = tmp_path / rel +@pytest.mark.parametrize("journal", _JOURNALS) +def test_old_journal_bytes_remain_complete_through_repeated_maintenance(tmp_path, journal): + path = tmp_path / journal path.parent.mkdir(parents=True, exist_ok=True) - return path + row = {"ts": "2020-01-01T00:00:00+00:00", "old_content": "old\nwith Unicode Я", + "new_content": "new\nwith Unicode Ё", "old_sha256": "legacy-mismatch"} + original = (json.dumps(row, ensure_ascii=False) + "\n{legacy broken row\n").encode("utf-8") + path.write_bytes(original) + + for _ in range(2): + report = compact_memory_journal_snapshots(tmp_path, retention_days=0, + now=2_000_000_000.0) + assert report["digested"] == report["digest_mismatch"] == {} + assert report["errors"] == [] + assert report["journal_bytes"].get(journal) == (len(original) if journal in + ("memory/identity_journal.jsonl", "memory/knowledge_history.jsonl", + "memory/knowledge/patterns_history.jsonl") else None) + assert path.read_bytes() == original + # Content digested in an older release cannot be restored, but is not deleted either. + assert b"old_content" in path.read_bytes() and b"new_content" in path.read_bytes() -def test_old_rows_digested_fresh_rows_kept_full(tmp_path): - path = _journal(tmp_path, "memory/identity_journal.jsonl") - old_row = { - "ts": _OLD_TS, "old_content": "I was v1", "new_content": "I am v2", - "old_sha256": hashlib.sha256(b"I was v1").hexdigest(), "old_len": 8, - } - fresh_row = {"ts": utc_now_iso(), "old_content": "I am v2", "new_content": "I am v3"} - broken_line = "{not json at all\n" - path.write_text( - json.dumps(old_row) + "\n" + broken_line + json.dumps(fresh_row) + "\n", - encoding="utf-8", - ) - - report = compact_memory_journal_snapshots(tmp_path) - - lines = path.read_text(encoding="utf-8").splitlines() - digested = json.loads(lines[0]) - assert "old_content" not in digested and "new_content" not in digested - assert digested["content_digested"] is True - assert digested["old_sha256"] == hashlib.sha256(b"I was v1").hexdigest() - assert digested["new_sha256"] == hashlib.sha256(b"I am v2").hexdigest() - assert digested["new_len"] == len("I am v2") - assert lines[1] == broken_line.rstrip("\n") # unreadable: byte-identical - kept = json.loads(lines[2]) - assert kept["old_content"] == "I am v2" and kept["new_content"] == "I am v3" - assert report["digested"] == {"memory/identity_journal.jsonl": 1} - assert not report["digest_mismatch"] and not report["errors"] +def test_maintenance_does_not_create_missing_journals_or_directories(tmp_path): + root = tmp_path / "absent" + compact_memory_journal_snapshots(root) + assert not root.exists() -@pytest.mark.parametrize("false_fact", [ - {"old_sha256": "pinned-old-hash"}, - {"old_len": 999}, -]) -def test_a_false_stored_digest_never_costs_the_text(tmp_path, false_fact): - """Audit #15-11: the compactor used ``setdefault``, so a stored digest that - contradicted its own text was KEPT while the only correct copy of the - content was deleted — the lie became the whole record. The pre-fix pin in - this file asserted exactly that behavior (``old_sha256 == "pinned-old-hash"`` - survives the deletion of ``old_content``); it was cementing the defect and - is reshaped above to a truthful stored digest. +def test_startup_prune_still_reaches_compatibility_entry_point(): + import ouroboros.server_maintenance as maintenance - A row whose stored fact does not match its text now keeps its FULL content - and is reported as a typed fact.""" - path = _journal(tmp_path, "memory/identity_journal.jsonl") - row = {"ts": _OLD_TS, "old_content": "I was v1", "new_content": "I am v2", **false_fact} - original = json.dumps(row) + "\n" - path.write_text(original, encoding="utf-8") - - report = compact_memory_journal_snapshots(tmp_path) - - assert path.read_text(encoding="utf-8") == original # byte-identical - assert not report["digested"] - assert report["digest_mismatch"] == {"memory/identity_journal.jsonl": 1} + assert "compact_memory_journal_snapshots" in inspect.getsource(maintenance._startup_prune_sweeps) -def test_a_concurrent_append_is_never_dropped_by_the_rewrite(tmp_path): - """``append_jsonl`` appends WITHOUT the sidecar lock once its own - acquisition times out, so a row can land mid-rewrite. Whether this pass - carries it over or abandons the rewrite, the row must survive.""" - path = _journal(tmp_path, "memory/knowledge_history.jsonl") - old_row = {"ts": _OLD_TS, "old_content": "a", "new_content": "b"} - path.write_text(json.dumps(old_row) + "\n", encoding="utf-8") - racing = {"ts": utc_now_iso(), "old_content": "b", "new_content": "c"} - real_digest_line = mjc._digest_line - fired = {"done": False} - - def racing_digest_line(raw, cutoff): - result = real_digest_line(raw, cutoff) - if not fired["done"]: - fired["done"] = True - with path.open("ab") as unlocked_appender: - unlocked_appender.write(json.dumps(racing).encode("utf-8") + b"\n") - return result - - mjc._digest_line = racing_digest_line +def test_size_observation_does_not_follow_a_journal_symlink(tmp_path): + target = tmp_path / "elsewhere" + target.write_bytes(b"secret data") + link = tmp_path / "memory" / "knowledge_history.jsonl" + link.parent.mkdir() try: - compact_memory_journal_snapshots(tmp_path) - finally: - mjc._digest_line = real_digest_line - - rows = [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines()] - assert len(rows) == 2 - assert rows[1] == racing # the concurrent row is intact whichever branch ran - - -@pytest.mark.skipif(os.name == "nt", reason="the reader holds the journal open; Windows refuses to unlink it (no FILE_SHARE_DELETE)") -def test_a_replaced_source_aborts_the_publish(tmp_path): - """Identity, not just size: if the journal is swapped for a different file - under the rewrite, the finished temp must be dropped, not published over - whatever now lives there.""" - path = _journal(tmp_path, "memory/knowledge/patterns_history.jsonl") - path.write_text(json.dumps({"ts": _OLD_TS, "old_content": "a", "new_content": "b"}) + "\n", - encoding="utf-8") - replacement = json.dumps({"ts": _OLD_TS, "topic": "someone else's file"}) + "\n" - real_digest_line = mjc._digest_line - fired = {"done": False} - - def swapping_digest_line(raw, cutoff): - result = real_digest_line(raw, cutoff) - if not fired["done"]: - fired["done"] = True - path.unlink() - path.write_text(replacement, encoding="utf-8") - return result - - mjc._digest_line = swapping_digest_line - try: - report = compact_memory_journal_snapshots(tmp_path) - finally: - mjc._digest_line = real_digest_line - - assert path.read_text(encoding="utf-8") == replacement - assert not report["digested"] - assert report["errors"] == [ - {"journal": "memory/knowledge/patterns_history.jsonl", "error": "source_changed"}, - ] - assert not list(path.parent.glob("*.compact.tmp")) - - -def test_the_journal_is_never_loaded_whole(tmp_path, monkeypatch): - """Bounded/streaming (audit #15-11c): these journals are the worst byte - offenders in the memory plane; a whole-file read is the thing being fixed. - Poison the whole-file readers and the compaction must still work.""" - path = _journal(tmp_path, "memory/identity_journal.jsonl") - rows = [{"ts": _OLD_TS, "old_content": f"o{i}", "new_content": f"n{i}"} for i in range(50)] - path.write_text("".join(json.dumps(row) + "\n" for row in rows), encoding="utf-8") - - def _boom(self, *args, **kwargs): - raise AssertionError(f"whole-file read of {self}") - - monkeypatch.setattr(pathlib.Path, "read_bytes", _boom) + link.symlink_to(target) + except (OSError, NotImplementedError): + pytest.skip("symlink creation unavailable") report = compact_memory_journal_snapshots(tmp_path) - - assert report["digested"] == {"memory/identity_journal.jsonl": 50} + assert report["journal_bytes"]["memory/knowledge_history.jsonl"] is None + assert "memory/knowledge_history.jsonl: not_regular" in report["errors"] + assert target.read_bytes() == b"secret data" -def test_the_rewrite_takes_an_unstealable_lock(): - """Owner-aware stale: elapsed time alone must never hand a second writer - the journal this destructive rewrite is holding.""" - import inspect - - src = inspect.getsource(mjc._compact_one) - assert "owner_aware_stale=True" in src - - -def test_patterns_history_gains_derived_digests(tmp_path): - path = _journal(tmp_path, "memory/knowledge/patterns_history.jsonl") - path.write_text(json.dumps({ - "ts": _OLD_TS, "task_id": "t", "markers": ["m"], - "old_content": "old body", "new_content": "new body\n", - }) + "\n", encoding="utf-8") - - compact_memory_journal_snapshots(tmp_path) - - row = json.loads(path.read_text(encoding="utf-8")) - assert row["old_sha256"] == hashlib.sha256(b"old body").hexdigest() - assert row["new_len"] == len("new body\n") - assert "old_content" not in row and row["content_digested"] is True - - -def test_row_without_readable_ts_keeps_full_text(tmp_path): - path = _journal(tmp_path, "memory/knowledge_history.jsonl") - original = json.dumps({"topic": "x", "old_content": "a", "new_content": "b"}) + "\n" - path.write_text(original, encoding="utf-8") +def test_startup_event_publishes_normal_journal_sizes(tmp_path, monkeypatch): + import ouroboros.server_maintenance as maintenance + import supervisor.state as state + journal = tmp_path / "memory" / "knowledge_history.jsonl" + journal.parent.mkdir() + journal.write_bytes(b"full historical text\n") + rows = [] + monkeypatch.setattr(maintenance, "DATA_DIR", tmp_path) + monkeypatch.setattr(state, "append_jsonl", lambda _path, row: rows.append(row)) report = compact_memory_journal_snapshots(tmp_path) - - assert path.read_text(encoding="utf-8") == original - assert not report["digested"] and not report["errors"] - - -def test_observation_inbox_is_out_of_scope(tmp_path): - inbox = tmp_path / "state" / "consciousness_observations.jsonl" - inbox.parent.mkdir(parents=True) - original = json.dumps({"ts": _OLD_TS, "op": "enqueue", "payload": "keep me"}) + "\n" - inbox.write_text(original, encoding="utf-8") - - compact_memory_journal_snapshots(tmp_path) - - assert inbox.read_text(encoding="utf-8") == original - - -def test_startup_prune_sweeps_run_the_compaction(): - import inspect - - import ouroboros.server_maintenance as sm - - assert "compact_memory_journal_snapshots" in inspect.getsource(sm._startup_prune_sweeps) + maintenance._prune_event("memory_journal_observation", ("journal_bytes", "errors"), report=report) + assert len(rows) == 1 + assert rows[0]["report"]["journal_bytes"]["memory/knowledge_history.jsonl"] == len(b"full historical text\n") + assert journal.read_bytes() == b"full historical text\n" + # Pin the real startup caller, not only this unit invocation. + source = inspect.getsource(maintenance._startup_prune_sweeps) + assert '_prune_event("memory_journal_observation", ("journal_bytes", "errors")' in source diff --git a/tests/test_multiprovider_conformance.py b/tests/test_multiprovider_conformance.py index b2f06512d..b1c66c25c 100644 --- a/tests/test_multiprovider_conformance.py +++ b/tests/test_multiprovider_conformance.py @@ -176,6 +176,8 @@ PROVIDER_DRIVERS: Dict[str, ProviderDriver] = { "minimax", "minimax::model-x", {"MINIMAX_API_KEY": "minimax-conformance-key"}), "deepseek": _openai_family( "deepseek", "deepseek::model-x", {"DEEPSEEK_API_KEY": "deepseek-conformance-key"}), + "zai": _openai_family( + "zai", "zai::model-x", {"ZAI_API_KEY": "zai-conformance-key"}), "anthropic": ProviderDriver( "anthropic", model="anthropic::claude-x", env={"ANTHROPIC_API_KEY": "anthropic-conformance-key"}, diff --git a/tests/test_onboarding_wizard.py b/tests/test_onboarding_wizard.py index 1fb6ab579..21ef54f1f 100644 --- a/tests/test_onboarding_wizard.py +++ b/tests/test_onboarding_wizard.py @@ -338,7 +338,7 @@ def test_prepare_onboarding_settings_rejects_openai_compatible_key_without_base_ def test_onboarding_frontend_uses_base_url_first_compatible_validation(): source = (REPO / "web/modules/onboarding_wizard.js").read_text(encoding="utf-8") - assert "!['OPENAI_COMPATIBLE_API_KEY', 'MINIMAX_REGION'].includes(field.settingKey)" in source + assert "!['OPENAI_COMPATIBLE_API_KEY', 'MINIMAX_REGION', 'ZAI_PLAN'].includes(field.settingKey)" in source assert "const hasRemote = keyValues.some(([, value]) => value);" not in source @@ -617,6 +617,8 @@ def test_setup_contract_groups_rarely_used_providers(): "MINIMAX_API_KEY": "more", "MINIMAX_REGION": "more", "DEEPSEEK_API_KEY": "more", + "ZAI_API_KEY": "more", + "ZAI_PLAN": "more", "ANTHROPIC_API_KEY": "primary", "OPENAI_COMPATIBLE_BASE_URL": "more", "OPENAI_COMPATIBLE_API_KEY": "more", @@ -691,6 +693,7 @@ _SECRET_CANARIES = { "CLOUDRU_FOUNDATION_MODELS_API_KEY": "cloudru-SECRETCANARY126", "MINIMAX_API_KEY": "minimax-SECRETCANARY127", "DEEPSEEK_API_KEY": "sk-ds-SECRETCANARY133", + "ZAI_API_KEY": "sk-zai-SECRETCANARY134", "ANTHROPIC_API_KEY": "sk-ant-SECRETCANARY128", "GIGACHAT_CREDENTIALS": "giga-SECRETCANARY129", "GIGACHAT_PASSWORD": "gigapw-SECRETCANARY130", diff --git a/tests/test_page_chrome_static.py b/tests/test_page_chrome_static.py index e79d0a5bf..9dd8bb199 100644 --- a/tests/test_page_chrome_static.py +++ b/tests/test_page_chrome_static.py @@ -132,7 +132,7 @@ def test_onboarding_compact_access_step_keeps_default_width_two_column(): def test_settings_more_providers_collapse_keeps_inputs_mounted(): - """Rarely used provider cards (Cloud.ru, MiniMax, DeepSeek, GigaChat) collapse under a + """Rarely used provider cards (Cloud.ru, MiniMax, DeepSeek, Z.ai, GigaChat) collapse under a "More providers" details wrapper, but their inputs must stay mounted: settings.js applyInputValue has no null guard, so a missing input id breaks settings load. The wrapper auto-opens when configured.""" @@ -141,7 +141,7 @@ def test_settings_more_providers_collapse_keeps_inputs_mounted(): css = _read("web/settings.css") assert 'id="settings-more-providers"' in ui - assert ui.count("advanced: true") == 4 + assert ui.count("advanced: true") == 5 assert "PROVIDER_CARDS.filter((card) => !card.advanced)" in ui assert "PROVIDER_CARDS.filter((card) => card.advanced)" in ui assert "syncMoreProvidersDisclosure" in settings diff --git a/tests/test_per_skill_version_resync.py b/tests/test_per_skill_version_resync.py index fe43ab712..904334202 100644 --- a/tests/test_per_skill_version_resync.py +++ b/tests/test_per_skill_version_resync.py @@ -374,14 +374,14 @@ def test_telegram_owner_wait_upgrade_reseeds_current_version(tmp_path, fake_log) ) assert upgraded == 1 - assert "version: 1.2.4" in (installed / "SKILL.md").read_text(encoding="utf-8") + assert "version: 1.2.5" in (installed / "SKILL.md").read_text(encoding="utf-8") for path in ("plugin.py", "lib/telegram_quiz.py"): assert (installed / path).read_bytes() == (seed_dir / "telegram" / path).read_bytes() @pytest.mark.serial @pytest.mark.parametrize("name,source,old_version,new_version", [ - ("telegram", "d5418e05b822feaf6aaa652e8cdc5b53af1232cc", "1.2.1", "1.2.4"), + ("telegram", "d5418e05b822feaf6aaa652e8cdc5b53af1232cc", "1.2.1", "1.2.5"), ("unix_computer_use", "162ad3fe6791fcaf6cf625e6b0c50d3a2a27e7f8", "0.4.1", "0.4.2"), ]) def test_resync_delivers_payload_from_real_previous_seed(tmp_path, fake_log, name, source, old_version, new_version): diff --git a/tests/test_persistence_inventory.py b/tests/test_persistence_inventory.py index 2c65cf0b5..02c6700f0 100644 --- a/tests/test_persistence_inventory.py +++ b/tests/test_persistence_inventory.py @@ -571,7 +571,10 @@ def scan_data_paths(root: pathlib.Path = REPO) -> frozenset[str]: # one rebuildable projection per conversation written by presence_runner at the end of an executed # turn; it has its own row in section 2. # 293 -> 295: the disposable test-environment caches (``cache/pip``, ``cache/uv``; test root only). -EXPECTED_SCAN_PATHS = 295 +# 295 -> 294: TZ-3 removed the destructive memory journal rewrite and its +# ``.compact.tmp`` sibling path; PERSISTENCE.md keeps the journals, now +# read-only observed and never age-digested. +EXPECTED_SCAN_PATHS = 294 # Scanned paths that must always be present — guards the scanner itself # against a silent regression that would shrink coverage while keeping counts diff --git a/tests/test_project_question_pointer.py b/tests/test_project_question_pointer.py index edb3933e4..4903c5da7 100644 --- a/tests/test_project_question_pointer.py +++ b/tests/test_project_question_pointer.py @@ -328,11 +328,16 @@ def test_project_question_pointer_display_fields_share_one_contract(): # History: the producer itself emits every field the browser merges. block = {"quiz_id": "q", "state": "open", "question": "Which?", "options": ["a", "b"], "option_details": ["A detail", ""], "stake": "What rides on it", "assumption": "a", - "recommended_index": 0} + "recommended_index": 0, "host_facts": "Asked by task t, origin unknown."} pointer = project_question_pointer({"task_id": "t", "quiz_id": "q"}, block, {"id": "p", "chat_id": 12, "name": "Project"}, None) assert browser_fields <= set(pointer), sorted(browser_fields - set(pointer)) assert pointer["option_details"] == ["A detail", ""] and pointer["stake"] == "What rides on it" + assert pointer["host_facts"] == "Asked by task t, origin unknown." + # A block that recorded no host sentence yields a pointer without the key, never an empty one. + bare = project_question_pointer({"task_id": "t", "quiz_id": "q"}, {**block, "host_facts": ""}, + {"id": "p", "chat_id": 12, "name": "Project"}, None) + assert "host_facts" not in bare # Live delivery: the frame literal plus its copied key loop, all declared in ChatOutbound. send_quiz = _function_node(REPO_ROOT / "supervisor" / "message_bus.py", "send_quiz") diff --git a/tests/test_provider_key_test.py b/tests/test_provider_key_test.py index f193bd295..4dbd0a65c 100644 --- a/tests/test_provider_key_test.py +++ b/tests/test_provider_key_test.py @@ -699,7 +699,7 @@ def test_response_log_and_accounting_expose_only_controlled_error( def test_provider_test_registry_is_derived_from_provider_defaults(): assert provider_api._PROVIDER_TEST_KNOWN_IDS == { "openrouter", "openai", "anthropic", "cloudru", "gigachat", "minimax", - "deepseek", "openai-compatible", + "deepseek", "zai", "openai-compatible", } assert provider_api._PROVIDER_TEST_OVERRIDE_KEYS == provider_api.ALL_PROVIDER_CREDENTIAL_KEYS diff --git a/tests/test_quiz_answer.py b/tests/test_quiz_answer.py index f17555058..aa8137023 100644 --- a/tests/test_quiz_answer.py +++ b/tests/test_quiz_answer.py @@ -693,8 +693,10 @@ def test_own_answer_needs_no_option_index(tmp_path, monkeypatch): entries = drain_owner_entries(tmp_path, "task-1", set()) frame_text = [e for e in entries if e.get("kind") == KIND_QUIZ_ANSWER][0]["text"] - assert ("The owner rejected all offered options and answered verbatim: " - "neither — use duckdb") in frame_text + assert ("The owner answered in their own words without choosing an offered " + "option. Verbatim: neither — use duckdb") in frame_text + # The host never words the free answer as a rejection the owner did not state. + assert "rejected" not in frame_text assert "chose option" not in frame_text diff --git a/tests/test_quiz_display.py b/tests/test_quiz_display.py index 5d1691344..bf5fa9812 100644 --- a/tests/test_quiz_display.py +++ b/tests/test_quiz_display.py @@ -72,11 +72,66 @@ class TestValidateQuizPayload: validate_quiz_payload("q", ["a", "b"], "", "", wait_for_answer=True, max_wait_minutes=bad) assert "omit it for an unbounded wait" in str(err.value) - def test_empty_or_oversized_question_refused(self): - with pytest.raises(QuizValidationError): - validate_quiz_payload("", ["a", "b"], "", "assume") - with pytest.raises(QuizValidationError): - validate_quiz_payload("q" * 2001, ["a", "b"], "", "assume") + def test_question_has_no_length_cap_but_must_be_non_empty(self): + """The card may be the only explanation its reader gets: a long + question is accepted whole; only an empty one is refused.""" + long_question = "q" * 2001 + assert validate_quiz_payload(long_question, ["a", "b"], "", "assume")["question"] == long_question + for empty in ("", " \n\t "): + with pytest.raises(QuizValidationError) as err: + validate_quiz_payload(empty, ["a", "b"], "", "assume") + assert err.value.code == "QUIZ_QUESTION_INVALID" + assert str(err.value) == "question must be non-empty." + + def test_over_long_label_is_refused_not_sliced(self): + at_bound = "L" * 120 + payload = validate_quiz_payload("q", [at_bound, "b"], "", "assume") + assert payload["options"][0]["label"] == at_bound + with pytest.raises(QuizValidationError) as err: + validate_quiz_payload("q", ["L" * 121, "b"], "", "assume") + assert err.value.code == "QUIZ_OPTIONS_INVALID" + assert str(err.value) == "option labels must be at most 120 characters." + # A dict option takes the same bound. + with pytest.raises(QuizValidationError) as err: + validate_quiz_payload("q", [{"label": "L" * 121, "detail": "d"}, "b"], "", "assume") + assert err.value.code == "QUIZ_OPTIONS_INVALID" + + def test_detail_stake_and_assumption_survive_byte_exact(self): + detail = "детали🙂 " * 75 + "end" # well past the former 500-char slice + stake = "stake-" * 100 + assumption = "assumption-" * 60 + assert min(len(detail), len(stake), len(assumption)) > 500 + payload = validate_quiz_payload( + "q", [{"label": "a", "detail": detail}, "b"], stake, assumption) + assert payload["options"][0]["detail"] == detail.strip() + assert payload["stake"] == stake + assert payload["assumption"] == assumption + + def test_multi_paragraph_unicode_question_survives_into_the_stored_block(self, tmp_path): + """Validator and the root ask path together: nothing between the tool + call and the durable owner_quiz block cuts the authored explanation.""" + from ouroboros.owner_quiz import quiz_states + from tests.test_quiz_answer import _escalate, _tool_ctx + + question = ( + "## Что решаем\n\n" + + "Абзац с объяснением — «кавычки», emoji 🚀, 𐍈. " * 60 + + "\n\nSecond paragraph in English, with `code` and a list:\n- one\n- two" + ) + detail = "Что меняется: " + "x" * 700 + assert len(question) > 2000 + assert validate_quiz_payload(question, ["a", "b"], "", "assume")["question"] == question + ctx = _tool_ctx(tmp_path, role="root") + out = _escalate(ctx, question=question, + options=[{"label": "A", "detail": detail}, {"label": "B"}], + stake="s" * 600, assumption="a" * 600) + assert out.startswith("OK: quiz ") + event = next(e for e in ctx.pending_events if e.get("type") == "send_quiz") + assert event["question"] == question + block = quiz_states(tmp_path, "root-1")[event["quiz_id"]] + assert block["question"] == question + assert block["option_details"] == [detail, ""] + assert block["stake"] == "s" * 600 and block["assumption"] == "a" * 600 @pytest.mark.parametrize("wait_for_answer", [False, True]) @@ -130,6 +185,37 @@ def test_send_quiz_broadcasts_publishes_and_persists_row(monkeypatch, tmp_path, assert live["wait_for_answer"] is payload["wait_for_answer"] is row["quiz"]["wait_for_answer"] is wait_for_answer +def test_send_quiz_accepts_a_long_question_through_the_shared_validator(monkeypatch, tmp_path): + bridge = _make_bridge(monkeypatch) + frames, events = [], [] + bridge._broadcast_fn = frames.append + monkeypatch.setattr(message_bus, "DATA_DIR", tmp_path) + monkeypatch.setattr(message_bus, "load_state", lambda: {"session_id": "s", "owner_id": 7}) + monkeypatch.setattr(message_bus, "_advance_project_visible_revision", lambda _chat_id: None) + monkeypatch.setattr(message_bus, "publish_event", lambda topic, data: events.append((topic, data))) + question = "Длинное объяснение решения. " * 200 + detail = "consequence " * 60 + ok, error = bridge.send_quiz( + 123, quiz_id="qz-long", question=question, + options=[{"label": "Yes", "detail": detail}, {"label": "No"}], + assumption="continuing", task_id="task-quiz", + ) + assert (ok, error) == (True, "ok") + live = next(frame for frame in frames if frame.get("type") == "quiz") + assert live["question"] == question.strip() + assert live["options"][0]["detail"] == detail.strip() + assert events[-1][1]["question"] == question.strip() + row = json.loads((tmp_path / "logs" / "chat.jsonl").read_text(encoding="utf-8").splitlines()[-1]) + assert row["text"] == question.strip() + # The same validator still refuses an over-long label atomically. + ok, error = bridge.send_quiz( + 123, quiz_id="qz-label", question="q", + options=[{"label": "L" * 121}, {"label": "No"}], + assumption="continuing", task_id="task-quiz", + ) + assert not ok and "at most 120 characters" in error + + def test_send_quiz_refuses_invalid_payload_and_missing_ids(monkeypatch, tmp_path): bridge = _make_bridge(monkeypatch) monkeypatch.setattr(message_bus, "DATA_DIR", tmp_path) diff --git a/tests/test_quiz_history_evidence.py b/tests/test_quiz_history_evidence.py index 26db5e0f4..2f53414b1 100644 --- a/tests/test_quiz_history_evidence.py +++ b/tests/test_quiz_history_evidence.py @@ -100,7 +100,9 @@ def test_answers_survive_eighteen_quizzes_mailbox_gc_and_rotation(runtime): assert quiz["comment"] == f" Verbatim choice {index}\nsecond line " assert quiz["request_id"] == f"answer-{index}" if index % 2: - assert "answered_index" not in quiz and "rejected all offered options" in row["text"] + assert "answered_index" not in quiz and ( + "answered in their own words without choosing an offered option" in row["text"]) + assert "rejected" not in row["text"] else: assert quiz["answered_index"] == 0 and "chose option 1: First" in row["text"] assert len([frame for frame in runtime.frames if frame.get("type") == "quiz"]) == 18 @@ -134,8 +136,27 @@ def test_a_late_answer_keeps_its_evidence_row_and_also_enters_dialogue(runtime, if row.get("client_message_id") == "quiz_late_answer:task-quiz:late"] assert len(inbound) == 1 and inbound[0]["direction"] == "in" assert inbound[0]["chat_id"] == 1 and inbound[0]["source"] == "web" - assert "[Owner quiz answer]" in inbound[0]["text"] and "Second" in inbound[0]["text"] - assert [frame["role"] for frame in runtime.frames if frame.get("type") == "chat"] == ["user"] + # The owner's row is the owner's words (the ingress strips edge whitespace + # of every owner row), never the host frame; the frame is for the model. + assert inbound[0]["text"] == "After the fact" + assert "[Owner quiz answer]" not in inbound[0]["text"] + chats = [frame for frame in runtime.frames if frame.get("type") == "chat"] + assert [frame["role"] for frame in chats] == ["user"] + assert chats[0]["content"] == "After the fact" + + # Reload: history replays the late answer as an ordinary user row carrying + # the owner's words; no path re-injects the frame into the bubble. + from ouroboros.gateway.history import make_chat_history_endpoint + + response = asyncio.run(make_chat_history_endpoint(runtime.root)( + SimpleNamespace(query_params={"n_human": "100", "thread": "1"}))) + messages = json.loads(response.body)["messages"] + replayed = [row for row in messages + if row.get("client_message_id") == "quiz_late_answer:task-quiz:late"] + assert len(replayed) == 1 and replayed[0]["role"] == "user" + assert replayed[0]["text"] == "After the fact" + assert not [row for row in messages if row.get("role") == "user" + and "[Owner quiz answer]" in str(row.get("text") or "")] @pytest.mark.parametrize("initial_index", [0, None]) diff --git a/tests/test_quiz_host_facts.py b/tests/test_quiz_host_facts.py new file mode 100644 index 000000000..d144612dc --- /dev/null +++ b/tests/test_quiz_host_facts.py @@ -0,0 +1,199 @@ +"""Host facts on the owner quiz card (4A) and the Project name on the ``chat.quiz`` event. + +The host writes one sentence under the question from facts only it knows — the asking +task, how that task's run started (``dialogue_provenance.run_origin``) and when the +owner last wrote in the card's chat — never from the question text; an unrecorded fact +says unknown. The sentence rides the durable block, the live frame, the host event, the +chat row, history replay and the Main pointer; the Project name rides the event only. +""" + +from __future__ import annotations + +import datetime +import json + +from ouroboros.owner_quiz import quiz_states, record_asked +from ouroboros.task_results import STATUS_RUNNING, write_task_result +from tests.test_quiz_answer import _escalate, _tool_ctx + +def _stamp(minutes_ago: float) -> str: + # Built at CALL time, never at import: the helper measures against the real clock when the + # card is asked, and collection under xdist can run tens of seconds before this test body. + return (datetime.datetime.now(datetime.timezone.utc) - datetime.timedelta(minutes=minutes_ago)).isoformat() + + +def _shown(iso: str) -> str: + return datetime.datetime.fromisoformat(iso).astimezone(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M UTC") + + +def _chat_rows(tmp_path, *rows): + path = tmp_path / "logs" / "chat.jsonl" + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("a", encoding="utf-8") as handle: + for row in rows: + handle.write(json.dumps(row) + "\n") + + +def _ask(ctx): + out = _escalate(ctx, question="Started by your message? Ship it?", options=["Ship", "Hold"], + assumption="Hold meanwhile") + assert out.startswith("OK: quiz "), out + event = next(evt for evt in ctx.pending_events if evt.get("type") == "send_quiz") + return event, quiz_states(ctx.drive_root, ctx.task_id)[event["quiz_id"]] + + +def test_record_asked_stores_host_facts_only_when_present(tmp_path): + with_facts = record_asked(tmp_path, "t-1", quiz_id="q-1", question="Q?", options=["a", "b"], + assumption="a", host_facts="Asked by task t-1, origin unknown.") + assert with_facts["host_facts"] == "Asked by task t-1, origin unknown." + without = record_asked(tmp_path, "t-1", quiz_id="q-2", question="Q?", options=["a", "b"], assumption="a") + assert "host_facts" not in without + stored = quiz_states(tmp_path, "t-1") + assert stored["q-1"]["host_facts"] == "Asked by task t-1, origin unknown." + assert "host_facts" not in stored["q-2"] + + +def test_owner_started_root_names_its_message_and_the_last_owner_message_in_this_chat(tmp_path): + started, last = _stamp(90.3), _stamp(47.3) + ctx = _tool_ctx(tmp_path, chat_id=1) + ctx.task_metadata["origin_message_ref"] = { + "chat_id": 1, "client_message_id": "m-1", "ts": started, "text_sha256": "0" * 64} + _chat_rows( + tmp_path, + {"ts": started, "direction": "in", "chat_id": 1, "text": "please ship"}, + {"ts": last, "direction": "in", "chat_id": 1, "text": "and quickly"}, + # Neither the assistant's own row nor another chat's owner message is "your last message here". + {"ts": _stamp(10), "direction": "out", "chat_id": 1, "text": "working"}, + {"ts": _stamp(5), "direction": "in", "chat_id": 7, "text": "other room"}, + ) + event, block = _ask(ctx) + prefix = (f"Asked by task root-1, started by your message of {_shown(started)}; " + f"your last message in this chat: {_shown(last)} (") + assert block["host_facts"].startswith(prefix) + # 47 minutes and 18 seconds before the ask; a slow machine may cross into the 48th. + assert block["host_facts"][len(prefix):] in ("47 minutes before this question).", "48 minutes before this question).") + assert event["host_facts"] == block["host_facts"] + + +def test_scheduled_follow_up_names_the_task_it_follows(tmp_path): + write_task_result(tmp_path, "root-1", STATUS_RUNNING, + metadata={"source": "task_followup", "origin_task_id": "prev-7", "schedule_id": "followup-prev-7"}) + event, block = _ask(_tool_ctx(tmp_path, chat_id=1)) + assert block["host_facts"] == ("Asked by task root-1, started as a scheduled follow-up of task prev-7; " + "your last message in this chat: unknown.") + assert event["host_facts"] == block["host_facts"] + + +def test_consciousness_origin_is_named(tmp_path): + ctx = _tool_ctx(tmp_path, chat_id=1) + ctx.task_metadata["initiator"] = "consciousness" + _chat_rows(tmp_path, {"ts": _stamp(3.2), "direction": "in", "chat_id": 1, "text": "hi"}) + _event, block = _ask(ctx) + assert block["host_facts"].startswith("Asked by task root-1, started by background consciousness; ") + assert block["host_facts"].endswith(("(3 minutes before this question).", "(4 minutes before this question).")) + + +def test_unknown_origin_and_no_owner_message_are_said_as_unknown(tmp_path): + # An unrelated source marker is shown as recorded, never mapped to a guess. + _event, block = _ask(_tool_ctx(tmp_path, chat_id=1)) + assert block["host_facts"] == "Asked by task root-1, origin unknown; your last message in this chat: unknown." + write_task_result(tmp_path, "root-2", STATUS_RUNNING, metadata={"source": "mystery_lane"}) + _event, block = _ask(_tool_ctx(tmp_path, task_id="root-2", chat_id=1)) + assert block["host_facts"] == ("Asked by task root-2, origin unknown (recorded source: mystery_lane); " + "your last message in this chat: unknown.") + + +def _bridge(tmp_path, monkeypatch): + from supervisor import message_bus, state + + monkeypatch.setattr(message_bus, "DATA_DIR", tmp_path) + state.init(tmp_path, 100) + frames, events = [], [] + bridge = message_bus.LocalChatBridge({}) + bridge._broadcast_fn = frames.append + monkeypatch.setattr(message_bus, "publish_event", lambda topic, data: events.append((topic, data))) + monkeypatch.setattr(message_bus, "get_bridge", lambda: bridge) + return bridge, frames, events + + +_OPTIONS = [{"label": "Local"}, {"label": "Shared"}] + + +def test_send_quiz_carries_host_facts_and_names_the_project_on_the_event_only(tmp_path, monkeypatch): + from ouroboros.event_bus import CHAT_QUIZ + from ouroboros.gateway.history import _assemble_history_response + from ouroboros.projects_registry import create_project + + project = create_project(tmp_path, "facts-project", name="Facts Project") + record_asked(tmp_path, "task-1", quiz_id="q-1", question="Which storage?", options=["Local", "Shared"], + assumption="Local", chat_id=project["chat_id"], host_facts="Asked by task task-1, origin unknown.") + bridge, frames, events = _bridge(tmp_path, monkeypatch) + ok, _ = bridge.send_quiz(project["chat_id"], quiz_id="q-1", question="Which storage?", options=_OPTIONS, + assumption="Local", task_id="task-1", host_facts="Asked by task task-1, origin unknown.") + assert ok + quiz_frame, pointer = frames + assert quiz_frame["type"] == "quiz" and quiz_frame["host_facts"] == "Asked by task task-1, origin unknown." + assert "project_name" not in quiz_frame # the browser wire names the Project through its own pointer + assert pointer["system_type"] == "project_question_pointer" + assert pointer["host_facts"] == "Asked by task task-1, origin unknown." + [(topic, event)] = events + assert topic == CHAT_QUIZ + assert event["host_facts"] == "Asked by task task-1, origin unknown." + assert event["project_name"] == "Facts Project" + stored = [json.loads(line) for line in (tmp_path / "logs/chat.jsonl").read_text(encoding="utf-8").splitlines()] + assert stored[-1]["quiz"]["host_facts"] == "Asked by task task-1, origin unknown." + room = json.loads(_assemble_history_response(tmp_path, project["chat_id"], 10, 0))["messages"] + assert room[0]["msg_type"] == "quiz" and room[0]["quiz"]["host_facts"] == "Asked by task task-1, origin unknown." + main = json.loads(_assemble_history_response(tmp_path, 1, 10, 0))["messages"] + assert main[0]["system_type"] == "project_question_pointer" + assert main[0]["host_facts"] == "Asked by task task-1, origin unknown." + + +def test_send_quiz_without_host_facts_or_project_adds_neither(tmp_path, monkeypatch): + bridge, frames, events = _bridge(tmp_path, monkeypatch) + assert bridge.send_quiz(1, quiz_id="q-2", question="Which storage?", options=_OPTIONS, + assumption="Local", task_id="task-2")[0] + [quiz_frame] = frames + [(_topic, event)] = events + assert "host_facts" not in quiz_frame and "host_facts" not in event and "project_name" not in event + stored = json.loads((tmp_path / "logs/chat.jsonl").read_text(encoding="utf-8").splitlines()[-1]) + assert "host_facts" not in stored["quiz"] + + +def test_history_replays_the_block_sentence_for_a_row_logged_without_it(tmp_path, monkeypatch): + from ouroboros.gateway.history import _assemble_history_response + + record_asked(tmp_path, "task-3", quiz_id="q-3", question="Which storage?", options=["Local", "Shared"], + assumption="Local", chat_id=1, host_facts="Asked by task task-3, origin unknown.") + record_asked(tmp_path, "task-4", quiz_id="q-4", question="Which storage?", options=["Local", "Shared"], + assumption="Local", chat_id=1) + bridge, _frames, _events = _bridge(tmp_path, monkeypatch) + for task_id, quiz_id in (("task-3", "q-3"), ("task-4", "q-4")): + assert bridge.send_quiz(1, quiz_id=quiz_id, question="Which storage?", options=_OPTIONS, + assumption="Local", task_id=task_id)[0] + rows = {row["quiz"]["quiz_id"]: row["quiz"] + for row in json.loads(_assemble_history_response(tmp_path, 1, 10, 0))["messages"] + if row.get("msg_type") == "quiz"} + assert rows["q-3"]["host_facts"] == "Asked by task task-3, origin unknown." + assert "host_facts" not in rows["q-4"] + + +def test_the_supervisor_event_handler_forwards_host_facts_to_send_quiz(): + from types import SimpleNamespace + + from supervisor.events_chat_delivery import _handle_send_quiz + + sent = [] + + class Bridge: + def send_quiz(self, chat_id, **kwargs): + sent.append((chat_id, kwargs)) + return True, "ok" + + ctx = SimpleNamespace(bridge=Bridge()) + evt = {"type": "send_quiz", "chat_id": 1, "quiz_id": "q-5", "question": "Q?", "options": _OPTIONS, + "assumption": "a", "task_id": "t-5", "host_facts": "Asked by task t-5, origin unknown."} + _handle_send_quiz(evt, ctx) + _handle_send_quiz({**evt, "host_facts": ""}, ctx) + assert sent[0][1]["host_facts"] == "Asked by task t-5, origin unknown." + assert sent[1][1]["host_facts"] == "" diff --git a/tests/test_quiz_late_answer.py b/tests/test_quiz_late_answer.py index d1854c9a8..91af90603 100644 --- a/tests/test_quiz_late_answer.py +++ b/tests/test_quiz_late_answer.py @@ -78,17 +78,19 @@ def test_ingress_late_answer_is_accepted_and_delivered_as_an_owner_message(tmp_p [queued] = _inbox(bridge) assert queued["chat_id"] == 1 and queued["user_id"] == 1 and queued["source"] == "web" assert queued["client_message_id"] == source_id - # The FULL existing answer frame, nothing trimmed; provenance is its own - # field and the real transport's client_surface is never substituted. - assert "[Owner quiz answer]" in queued["text"] and "Which db?" in queued["text"] - assert "prod parity" in queued["text"] and "postgres" in queued["text"] + # The owner's row and bubble are the HUMAN's words (the verbatim comment), + # never the host frame; provenance is its own field and the real + # transport's client_surface is never substituted. + assert queued["text"] == "prod parity" + assert "[Owner quiz answer]" not in queued["text"] assert queued["task_metadata"]["late_answer"] == {"task_id": "task-1", "quiz_id": "q1"} assert "client_surface" not in queued["task_metadata"] # The same user bubble the owner's own typing produces. echo = [f for f in frames if f.get("type") == "chat" and f.get("role") == "user"] assert len(echo) == 1 and echo[0]["client_message_id"] == source_id assert echo[0]["chat_id"] == 1 and echo[0]["content"] == queued["text"] - assert len(_accepted_rows(tmp_path, source_id)) == 1 + [row] = _accepted_rows(tmp_path, source_id) + assert row["text"] == echo[0]["content"] == "prod parity" # A retry of the SAME request re-enters delivery; the named ingress rejoins. again = _post(app, {"request_id": "r1", "decision_id": "quiz:task-1:q1", @@ -219,3 +221,222 @@ def test_a_second_answer_to_a_settled_card_is_still_a_first_wins_409(tmp_path, m assert loser.json()["state"] == STATE_ANSWERED assert loser.json()["answered_index"] == 1 assert _inbox(bridge) == [] + + +def test_a_button_only_late_answer_speaks_the_pressed_option_as_the_owner(tmp_path, monkeypatch): + """No comment: the owner's row, queued text and bubble are the pressed + option exactly as the button showed it (the ingress refuses empty text), + and still never the English host frame.""" + record_asked(tmp_path, "task-1", quiz_id="q1", question="Which db?", + options=["sqlite", "postgres"], assumption="sqlite meanwhile", chat_id=1) + reconcile_terminal(tmp_path, "task-1") + bridge, frames = _late_bridge(tmp_path, monkeypatch) + app = _decision_app(tmp_path, monkeypatch, live_task=None) + resp = _post(app, {"request_id": "r1", "decision_id": "quiz:task-1:q1", "option_index": 1}) + assert resp.status_code == 200 and resp.json()["forwarded"] is True + [queued] = _inbox(bridge) + assert queued["text"] == "2. postgres" + [echo] = [f for f in frames if f.get("type") == "chat" and f.get("role") == "user"] + [row] = _accepted_rows(tmp_path, "quiz_late_answer:task-1:q1") + assert echo["content"] == row["text"] == "2. postgres" + assert "[Owner quiz answer]" not in row["text"] + + +def test_a_retry_of_a_late_answer_accepted_with_the_old_frame_text_rejoins(tmp_path, monkeypatch): + """A late answer accepted before the row carried only the owner's words has + the host frame under the same id; a retry now rejoins that delivery instead + of failing forever on the text mismatch or enqueueing a second owner turn.""" + import supervisor.message_bus as mb + + record_asked(tmp_path, "task-1", quiz_id="q1", question="Which db?", + options=["sqlite", "postgres"], assumption="sqlite meanwhile", chat_id=1) + reconcile_terminal(tmp_path, "task-1") + bridge, frames = _late_bridge(tmp_path, monkeypatch) + mb.log_chat("in", 1, 1, "[Owner quiz answer] quiz q1 -- old frame", source="web", + client_message_id="quiz_late_answer:task-1:q1", drive_root=tmp_path, + require_write=True) + app = _decision_app(tmp_path, monkeypatch, live_task=None) + resp = _post(app, {"request_id": "r1", "decision_id": "quiz:task-1:q1", "option_index": 1}) + assert resp.status_code == 200, resp.text + assert resp.json()["forwarded"] is True + assert _inbox(bridge) == [] and not [f for f in frames if f.get("type") == "chat"] + assert len(_accepted_rows(tmp_path, "quiz_late_answer:task-1:q1")) == 1 + + +def _answered_late(tmp_path, *, option_index, comment=""): + from ouroboros.owner_quiz import record_answered + + record_asked(tmp_path, "task-1", quiz_id="q1", question="Which db for the pilot?", + options=["sqlite", "postgres"], assumption="sqlite meanwhile", chat_id=1) + reconcile_terminal(tmp_path, "task-1") + outcome = record_answered(tmp_path, "task-1", quiz_id="q1", option_index=option_index, + request_id="r1", comment=comment, allow_expired=True) + assert outcome["ok"] is True + return outcome["block"] + + +def test_the_drained_late_answer_gives_the_model_the_rebuilt_frame_while_the_row_stays_human(tmp_path): + """Project-room mailbox delivery: the entry (and the owner's row) carry only + the owner's words; the drain rebuilds the FULL card frame from the stored + block for the model and records the owner directive with that frame.""" + import queue as queue_mod + from types import SimpleNamespace + + from ouroboros.loop_round_limits import _drain_incoming_messages + from ouroboros.owner_mailbox import drain_owner_entries, write_owner_message + + block = _answered_late(tmp_path, option_index=None, comment="neither -- use duckdb") + assert write_owner_message( + tmp_path, "neither -- use duckdb", "live-root", msg_id="quiz_late_answer:task-1:q1:live-root", + client_message_id="quiz_late_answer:task-1:q1", + late_answer={"task_id": "task-1", "quiz_id": "q1"}, + ) + [entry] = drain_owner_entries(tmp_path, "live-root", set(), include_acknowledged=True) + assert entry["text"] == "neither -- use duckdb" + assert entry["late_answer"] == {"task_id": "task-1", "quiz_id": "q1"} + + ctx = SimpleNamespace() + messages = [{"role": "user", "content": "Initial requirement"}] + _drain_incoming_messages(messages, queue_mod.Queue(), tmp_path, "live-root", None, set(), + owner_ctx=ctx) + delivered = str(messages[-1]["content"]) + assert "[Owner quiz answer] quiz q1" in delivered + assert f"asked {block['asked_at']}" in delivered and f"answered {block['answered_at']}" in delivered + assert "Question was: Which db for the pilot?" in delivered + assert ("The owner answered in their own words without choosing an offered option. " + "Verbatim: neither -- use duckdb") in delivered + [directive] = [row for row in ctx._owner_directives if row["source"] == "owner_mailbox"] + assert "[Owner quiz answer] quiz q1" in directive["content"] + assert "Verbatim: neither -- use duckdb" in directive["content"] + # The steer relay's delivery fact stays the owner's own message. + assert ctx.last_owner_delivery["text"] == "neither -- use duckdb" + + +def test_an_ordinary_mailbox_message_is_not_reframed(tmp_path): + """The still-working case: a mailbox message without late_answer provenance + reaches the model as the owner's words, unchanged.""" + import queue as queue_mod + from types import SimpleNamespace + + from ouroboros.loop_round_limits import _drain_incoming_messages + from ouroboros.owner_mailbox import write_owner_message + + _answered_late(tmp_path, option_index=1) + assert write_owner_message(tmp_path, "please also fix the test", "live-root", msg_id="m1") + ctx = SimpleNamespace() + messages = [{"role": "user", "content": "Initial requirement"}] + _drain_incoming_messages(messages, queue_mod.Queue(), tmp_path, "live-root", None, set(), + owner_ctx=ctx) + delivered = str(messages[-1]["content"]) + assert "please also fix the test" in delivered and "[Owner quiz answer]" not in delivered + + +def test_an_unreadable_card_is_disclosed_with_the_owners_words(tmp_path): + """The block is gone (evicted, or the task result unreadable): the model + gets the owner's words plus one host line naming the quiz, never silence.""" + import queue as queue_mod + from types import SimpleNamespace + + from ouroboros.loop_round_limits import _drain_incoming_messages + from ouroboros.owner_mailbox import write_owner_message + + assert write_owner_message( + tmp_path, "2. postgres", "live-root", msg_id="late-1", + late_answer={"task_id": "task-gone", "quiz_id": "q9"}, + ) + ctx = SimpleNamespace() + messages = [{"role": "user", "content": "Initial requirement"}] + _drain_incoming_messages(messages, queue_mod.Queue(), tmp_path, "live-root", None, set(), + owner_ctx=ctx) + delivered = str(messages[-1]["content"]) + assert "2. postgres" in delivered + assert "answers quiz q9 of task task-gone; that card could not be read" in delivered + assert "[Owner quiz answer]" not in delivered + + +def test_a_late_answer_direct_turn_starts_with_the_rebuilt_frame(tmp_path, monkeypatch): + """Main / no live root: the late answer starts an ordinary owner turn whose + task carries the late_answer provenance; the model's first user content (and + therefore the run's initial owner directive) is the rebuilt frame, while the + owner's row stays their words. A turn without that provenance is unchanged.""" + import queue as queue_mod + + import supervisor.workers as workers + from ouroboros.context import build_user_content + + block = _answered_late(tmp_path, option_index=1, comment="prod parity") + monkeypatch.setattr(workers, "DRIVE_ROOT", tmp_path) + monkeypatch.setattr(workers, "get_event_q", lambda: queue_mod.Queue()) + + class _Agent: + task = None + + def handle_task(self, task): + self.task = task + return [] + + agent = _Agent() + workers._run_chat_task(agent, 1, "prod parity", None, task_metadata={ + "client_message_id": "quiz_late_answer:task-1:q1", + "late_answer": {"task_id": "task-1", "quiz_id": "q1"}, + }) + content = str(build_user_content(agent.task)) + assert "[Owner quiz answer] quiz q1" in content + assert f"answered {block['answered_at']}" in content + assert "The owner chose option 2: postgres" in content + assert "Owner comment (verbatim): prod parity" in content + + plain = _Agent() + workers._run_chat_task(plain, 1, "prod parity", None, task_metadata={"client_message_id": "m-2"}) + assert plain.task["text"] == "prod parity" + + +def test_a_late_answer_routed_into_a_project_rooms_live_root_carries_its_provenance(tmp_path, monkeypatch): + """End to end through the real bridge intake: a late answer in a Project room + with exactly one live root lands in THAT root's mailbox as the owner's words + with its typed late_answer provenance, so the drain can rebuild the frame.""" + import queue as queue_mod + from types import SimpleNamespace + + import server + import supervisor.message_bus as mb + from ouroboros.loop_round_limits import _drain_incoming_messages + from ouroboros.owner_mailbox import drain_owner_entries + from ouroboros.projects_registry import create_project + + project = create_project(tmp_path, "racer") + chat_id = int(project["chat_id"]) + record_asked(tmp_path, "task-1", quiz_id="q1", question="Which db?", + options=["sqlite", "postgres"], assumption="sqlite meanwhile", chat_id=chat_id) + reconcile_terminal(tmp_path, "task-1") + bridge, _frames = _late_bridge(tmp_path, monkeypatch) + monkeypatch.setattr(mb, "load_state", lambda: {"session_id": "s-1"}) + app = _decision_app(tmp_path, monkeypatch, live_task=None) + resp = _post(app, {"request_id": "r1", "decision_id": "quiz:task-1:q1", + "option_index": 1, "comment": "prod parity"}) + assert resp.status_code == 200 and resp.json()["forwarded"] is True + + pending = [{"id": "pending-root", "chat_id": chat_id, "root_task_id": "pending-root", + "delegation_role": "root", "drive_root": str(tmp_path)}] + ctx = SimpleNamespace( + DRIVE_ROOT=tmp_path, PENDING=pending, RUNNING={}, + load_state=lambda: {"owner_id": 1, "owner_chat_id": 1, "session_id": "s-1"}, + update_state=lambda fn: fn({"owner_id": 1, "owner_chat_id": 1}), + consciousness=SimpleNamespace(inject_observation=lambda _text: None), + get_chat_agent=lambda: SimpleNamespace(_busy=False), + handle_chat_direct=lambda *a, **k: pytest.fail("mailbox delivery must not run a turn"), + send_with_budget=lambda *a, **k: None, + ) + monkeypatch.setattr(bridge, "send_routing_ack", lambda *a, **k: None, raising=False) + server._process_bridge_updates(bridge, 0, ctx) + + [entry] = drain_owner_entries(tmp_path, "pending-root", set(), include_acknowledged=True) + assert entry["text"] == "prod parity" + assert entry["late_answer"] == {"task_id": "task-1", "quiz_id": "q1"} + owner_ctx = SimpleNamespace() + messages = [{"role": "user", "content": "Initial requirement"}] + _drain_incoming_messages(messages, queue_mod.Queue(), tmp_path, "pending-root", None, set(), + owner_ctx=owner_ctx) + delivered = str(messages[-1]["content"]) + assert "The owner chose option 2: postgres" in delivered + assert "Owner comment (verbatim): prod parity" in delivered diff --git a/tests/test_reference_book_budgets.py b/tests/test_reference_book_budgets.py index 130932eab..a8b4cb3dd 100644 --- a/tests/test_reference_book_budgets.py +++ b/tests/test_reference_book_budgets.py @@ -162,7 +162,10 @@ CHAPTER_BYTE_BUDGETS: dict[str, int] = { "docs/architecture/06-agent-core.md": 310100, # 36991 -> 37300: the facade paragraph names the three loop constants runtime_limits.py # gained (events batch bound, budget-projection retry interval); no older text to displace. - "docs/architecture/07-configuration.md": 37300, + # 37300 -> 38400 (PR #1207): the Z.ai (`zai::`) direct provider gets its own route + # paragraph (plan-selected endpoint, low/high/max projection, 1113 billing) plus two + # settings rows; the base sat 95 bytes under the previous budget, no older text to displace. + "docs/architecture/07-configuration.md": 38400, # 18947 -> 19287: CI failure collection now documents diagnostic desktop builds while release remains gated. # 19287 -> 20560 (#1215): three contracts the chapter had no older text for — the # ONE reusable browser lane and the two triggers that share it (the unfiltered diff --git a/tests/test_room_knowledge_correction.py b/tests/test_room_knowledge_correction.py index 69664e507..b99660364 100644 --- a/tests/test_room_knowledge_correction.py +++ b/tests/test_room_knowledge_correction.py @@ -162,5 +162,6 @@ def test_unread_correction_failure_remains_visible_after_dialogue_publication(tm assert stored["knowledge_writes"][0]["reason"] == "revision_required" state = json.loads(meta.read_text(encoding="utf-8")) assert state["last_consolidated_offset"] == 100 - assert state["last_unpublished_nominations"]["failed"] == 1 + assert len(state["pending_knowledge_nominations"]) == 1 + assert state["pending_knowledge_nominations"][0]["reason"] == "revision_required" assert k.read_knowledge_note(address).raw == original.raw diff --git a/tests/test_telegram_quiz_answers.py b/tests/test_telegram_quiz_answers.py index 0cd6a1cb7..5755ba665 100644 --- a/tests/test_telegram_quiz_answers.py +++ b/tests/test_telegram_quiz_answers.py @@ -363,3 +363,280 @@ def test_recommended_option_is_starred_in_the_button_caption(tmp_path, monkeypat assert "1. sqlite\n2. ★ postgres" in record["text"] keyboard = plugin.telegram_quiz.quiz_keyboard(token, record["options"]) assert [row[0]["text"] for row in keyboard] == ["1. sqlite", "2. ★ postgres"] + + +# --- Full card: project, host facts, option details, localization, overflow --- + +_HINT_EN = "Tap an option, or reply to this message with your own answer." +_HINT_RU = "Нажмите вариант или ответьте на это сообщение своим текстом." +_FULL_EVENT = { + **_EVENT, + "project_name": "Alpha site", + "host_facts": "Asked by task task-1, started by your message of 2026-09-25 00:21.", + "stake": "Whether the data survives a restart.", + "options": [{"label": "sqlite", "detail": "one file, nothing to run"}, + {"label": "postgres", "detail": "scales, needs a server", "recommended": True}], +} + + +def _recording_client(plugin): + """The REAL TelegramClient with only the HTTP call recorded, so the + production chunker and keyboard encoding are what the test observes.""" + calls: list = [] + + class Recording(plugin.TelegramClient): + async def call(self, method, *, data=None, files=None, timeout=30): + calls.append((method, dict(data or {}))) + return {"ok": True, "result": {"message_id": 700 + len(calls)}} + + return Recording, calls + + +def _send_full_card(plugin, tmp_path, monkeypatch, event, **settings): + _settings(tmp_path, **settings) + recording, calls = _recording_client(plugin) + monkeypatch.setattr(plugin, "TelegramClient", recording) + api = Api(tmp_path) + asyncio.run(plugin._make_quiz(api)(event)) + assert api.logs == [] + state = json.loads((tmp_path / "quiz_state.json").read_text(encoding="utf-8")) + (token, record), = state["quizzes"].items() + return api, calls, token, record + + +def _keyboard_calls(calls): + return [(m, d) for m, d in calls if m == "sendMessage" and "reply_markup" in d] + + +def _plain_calls(calls): + return [(m, d) for m, d in calls if m == "sendMessage" and "reply_markup" not in d] + + +@pytest.mark.parametrize("lang,expected_body,hint_text", [ + ("en", + "Project: Alpha site\n" + "Asked by task task-1, started by your message of 2026-09-25 00:21.\n" + "Question: Which db?\n" + "At stake: Whether the data survives a restart.\n" + "1. sqlite — one file, nothing to run\n" + "2. ★ postgres — scales, needs a server\n" + "Continuing meanwhile: sqlite meanwhile", + _HINT_EN), + ("ru", + "Проект: Alpha site\n" + "Asked by task task-1, started by your message of 2026-09-25 00:21.\n" + "Вопрос: Which db?\n" + "Что на кону: Whether the data survives a restart.\n" + "1. sqlite — one file, nothing to run\n" + "2. ★ postgres — scales, needs a server\n" + "Пока продолжаю так: sqlite meanwhile", + _HINT_RU), +], ids=["en", "ru"]) +def test_short_card_is_one_message_with_project_facts_details_and_star( + tmp_path, monkeypatch, lang, expected_body, hint_text): + plugin = _load_plugin() + _api, calls, token, record = _send_full_card( + plugin, tmp_path, monkeypatch, _FULL_EVENT, TELEGRAM_LANGUAGE=lang) + assert len(calls) == 1 and calls[0][0] == "sendMessage" + data = calls[0][1] + assert data["text"] == f"{expected_body}\n{hint_text}" + assert "parse_mode" not in data, "the authored text is sent verbatim" + buttons = json.loads(data["reply_markup"])["inline_keyboard"] + assert [row[0]["text"] for row in buttons] == ["1. sqlite", "2. ★ postgres"] + assert [row[0]["callback_data"] for row in buttons] == [f"qz:{token}:0", f"qz:{token}:1"] + assert record["message_id"] == 701 + assert record["options"] == ["sqlite", "★ postgres"] + assert record["text"] == expected_body + + +@pytest.mark.parametrize("lang,waiting", [ + ("en", "Waiting for your answer; Stop and the task deadline still apply."), + ("ru", "Жду вашего ответа; Stop и срок задачи по-прежнему действуют."), +], ids=["en", "ru"]) +def test_waiting_line_is_localized_and_dropped_from_the_settled_text(tmp_path, monkeypatch, lang, waiting): + plugin = _load_plugin() + _api, calls, _token, record = _send_full_card( + plugin, tmp_path, monkeypatch, {**_FULL_EVENT, "wait_for_answer": True}, TELEGRAM_LANGUAGE=lang) + assert waiting in calls[0][1]["text"] + assert waiting not in record["text"] + assert record["text"].splitlines()[-1] == "2. ★ postgres — scales, needs a server" + + +def test_old_event_without_project_facts_or_details_keeps_the_plain_card(tmp_path, monkeypatch): + plugin = _load_plugin() + _api, calls, _token, record = _send_full_card(plugin, tmp_path, monkeypatch, dict(_EVENT)) + assert calls[0][1]["text"] == ( + "Question: Which db?\n1. sqlite\n2. postgres\nContinuing meanwhile: sqlite meanwhile\n" + _HINT_EN) + assert "Project" not in record["text"] and " — " not in record["text"] + + +def _long_question(): + # Multi-paragraph, with astral characters that count twice in UTF-16. + paragraph = ("Why this matters 🧭: " + "the migration keeps every row intact. " * 20).rstrip() + return "\n\n".join(f"{index}. {paragraph}" for index in range(8)) + + +def test_long_card_is_sent_in_ordered_plain_parts_then_one_keyboard_message(tmp_path, monkeypatch): + plugin = _load_plugin() + quiz = plugin.telegram_quiz + event = {**_FULL_EVENT, "question": _long_question()} + _api, calls, token, record = _send_full_card(plugin, tmp_path, monkeypatch, event) + body = quiz.render_quiz_text( + event["question"], ["sqlite", "postgres"], event["stake"], event["assumption"], + project_name="Alpha site", host_facts=event["host_facts"], + option_details=["one file, nothing to run", "scales, needs a server"], recommended_index=1) + assert quiz._u16len(f"{body}\n{_HINT_EN}") > quiz._TELEGRAM_TEXT_LIMIT + + plain = [d["text"] for _m, d in _plain_calls(calls)] + keyboards = _keyboard_calls(calls) + assert len(plain) >= 2 and len(keyboards) == 1 + assert calls[-1] == keyboards[0], "the keyboard message comes last" + assert all(quiz._u16len(part) <= quiz._TELEGRAM_TEXT_LIMIT for part in plain) + assert all("parse_mode" not in d for _m, d in calls) + assert "\n".join(plain) == body, "no authored character is lost across parts" + + compact = "Project: Alpha site\n1. sqlite\n2. ★ postgres" + assert keyboards[0][1]["text"] == f"{compact}\n{_HINT_EN}" + assert record["message_id"] == 700 + len(calls) + assert record["text"] == compact + assert record["options"] == ["sqlite", "★ postgres"] + buttons = json.loads(keyboards[0][1]["reply_markup"])["inline_keyboard"] + assert [row[0]["callback_data"] for row in buttons] == [f"qz:{token}:0", f"qz:{token}:1"] + + +def test_single_long_line_question_keeps_every_authored_character(tmp_path, monkeypatch): + plugin = _load_plugin() + quiz = plugin.telegram_quiz + question = " ".join(f"word{index}🧭" for index in range(700)) # ~5 600 UTF-16 units, one line + event = {**_FULL_EVENT, "question": question} + _api, calls, _token, _record = _send_full_card(plugin, tmp_path, monkeypatch, event) + plain = [d["text"] for _m, d in _plain_calls(calls)] + assert len(plain) >= 2 and len(_keyboard_calls(calls)) == 1 + assert all(quiz._u16len(part) <= quiz._TELEGRAM_TEXT_LIMIT for part in plain) + # The chunker breaks a single long line at a space and the message boundary + # stands in for that space; every non-whitespace character survives in order. + sent = "".join("".join(part.split()) for part in plain) + assert question.replace(" ", "") in sent + assert "".join(sent.split()) == "".join(quiz.render_quiz_text( + question, ["sqlite", "postgres"], event["stake"], event["assumption"], + project_name="Alpha site", host_facts=event["host_facts"], + option_details=["one file, nothing to run", "scales, needs a server"], + recommended_index=1).split()) + + +def test_overflowing_card_answers_through_its_keyboard_message(tmp_path, monkeypatch): + plugin = _load_plugin() + event = {**_FULL_EVENT, "question": _long_question()} + api, calls, token, record = _send_full_card(plugin, tmp_path, monkeypatch, event) + keyboard_id = record["message_id"] + first_part_id = 701 + assert keyboard_id != first_part_id + + # A tap resolves to the card identity and settles the keyboard message. + Client.updates = [{"update_id": 21, "callback_query": { + "id": "cb", "data": f"qz:{token}:1", "from": {"id": 42}, + "message": {"message_id": keyboard_id, "chat": {"id": 42, "type": "private"}}, + }}] + posts = [] + _run_poller(plugin, api, monkeypatch, posts, + reply=(200, {"ok": True, "state": "answered", "answered_index": 1})) + assert posts == [("/chat/decision", { + "request_id": "tg:21", "decision_id": "quiz:task-1:q1", "option_index": 1, + })] + assert _LAST_CLIENT[-1].edits == [(42, keyboard_id, + "Project: Alpha site\n1. sqlite\n2. ★ postgres\nAnswered: 2. ★ postgres", + [])] + + # A reply to the keyboard message is the owner's own answer ... + Client.updates = [{"update_id": 22, "message": { + "message_id": 900, "chat": {"id": 42, "type": "private"}, "from": {"id": 42}, + "text": "postgres, but later", "reply_to_message": {"message_id": keyboard_id}, + }}] + posts = [] + assert _run_poller(plugin, api, monkeypatch, posts) == [] + assert posts == [("/chat/decision", { + "request_id": "tg:22", "decision_id": "quiz:task-1:q1", "comment": "postgres, but later", + })] + + # ... while a reply to an earlier explanation part is an ordinary owner message. + Client.updates = [{"update_id": 23, "message": { + "message_id": 901, "chat": {"id": 42, "type": "private"}, "from": {"id": 42}, + "text": "about paragraph 3", "reply_to_message": {"message_id": first_part_id}, + }}] + posts = [] + injected = _run_poller(plugin, api, monkeypatch, posts) + assert posts == [] + assert [row["text"] for row in injected] == ["about paragraph 3"] + + +def test_record_without_details_from_an_older_skill_still_answers(tmp_path, monkeypatch): + plugin = _load_plugin() + _settings(tmp_path) + api = Api(tmp_path) + token = plugin.telegram_quiz.mint_token("task-1", "q1") + plugin.telegram_quiz.remember_quiz(api, token, { + "task_id": "task-1", "quiz_id": "q1", "chat_id": 42, "message_id": 555, + "options": ["sqlite", "★ postgres"], "text": "Question: Which db?\n1. sqlite\n2. ★ postgres", + }) + Client.updates = [{"update_id": 24, "callback_query": { + "id": "cb", "data": f"qz:{token}:0", "from": {"id": 42}, + "message": {"message_id": 555, "chat": {"id": 42, "type": "private"}}, + }}] + posts = [] + _run_poller(plugin, api, monkeypatch, posts, + reply=(200, {"ok": True, "state": "answered", "answered_index": 0})) + assert posts[0][1]["option_index"] == 0 + assert _LAST_CLIENT[-1].edits[0][2].endswith("\nAnswered: 1. sqlite") + + +class _CardClient: + """Records what send_quiz_card sends; message ids count up from 500.""" + + def __init__(self): + self.sent, self.next_id = [], 500 + + async def send_message(self, chat_id, text, parse_mode="HTML"): + self.sent.append(("plain", text)); self.next_id += 1 + return self.next_id + + async def send_message_with_inline_keyboard(self, chat_id, text, keyboard, parse_mode="HTML"): + self.sent.append(("keyboard", text)); self.next_id += 1 + return self.next_id + + +def _send_card_direct(body, hint="tap"): + import asyncio + from skills.telegram.lib import telegram_quiz as quiz + + client = _CardClient() + message_id, overflowed = asyncio.run(quiz.send_quiz_card( + client, 42, body=body, compact="1. a\n2. b", hint_text=hint, keyboard=[[{"text": "1. a", "callback_data": "x"}]])) + return client, message_id, overflowed + + +def test_card_that_fits_only_without_its_answered_edit_goes_out_as_parts(): + """Both directions: a card leaves room for "Answered: ", or it is split (Opus L1).""" + from skills.telegram.lib import telegram_quiz as quiz + + fits = "q" * (quiz._TELEGRAM_TEXT_LIMIT - quiz._ANSWERED_EDIT_RESERVE - len("\ntap")) + client, message_id, overflowed = _send_card_direct(fits) + assert not overflowed and [kind for kind, _ in client.sent] == ["keyboard"] + assert quiz._u16len(client.sent[0][1]) + quiz._ANSWERED_EDIT_RESERVE <= quiz._TELEGRAM_TEXT_LIMIT + + barely = fits + "q" * 8 # fits the message limit alone, not with the answered edit + assert quiz._u16len(f"{barely}\ntap") <= quiz._TELEGRAM_TEXT_LIMIT + client, message_id, overflowed = _send_card_direct(barely) + assert overflowed and [kind for kind, _ in client.sent] == ["plain", "keyboard"] + assert client.sent[0][1] == barely and message_id == 502 + + +def test_card_length_is_measured_in_utf16_units_not_code_points(): + """Astral characters count double in Telegram's limit; len() would let this card through (Opus L2).""" + from skills.telegram.lib import telegram_quiz as quiz + + budget = quiz._TELEGRAM_TEXT_LIMIT - quiz._ANSWERED_EDIT_RESERVE - len("\ntap") + body = "\U0001F600" * (budget // 2 + 4) # fewer code points than the budget, more UTF-16 units + assert len(body) < budget < quiz._u16len(body) + client, _message_id, overflowed = _send_card_direct(body) + assert overflowed and [kind for kind, _ in client.sent] == ["plain", "keyboard"] + assert client.sent[0][1] == body diff --git a/tests/test_zai_provider.py b/tests/test_zai_provider.py new file mode 100644 index 000000000..a72caace5 --- /dev/null +++ b/tests/test_zai_provider.py @@ -0,0 +1,321 @@ +"""Z.ai (GLM) direct provider: registry, plan-selected endpoint, the effort +projection at the send boundary, and the 429/1113 billing classification. + +Facts pinned here come from the contributor's live probe (PR #1207, 2026-09-21, +Coding Plan key, glm-5.3) and docs.z.ai: the provider accepts exactly +``low``/``high``/``max``, an ABSENT ``reasoning_effort`` is served at max, +thinking cannot be disabled (HTTP 400 code 1210), forced tool_choice works with +thinking on, and plan exhaustion arrives as HTTP 429 code 1113. +""" +import os + +import pytest + +from ouroboros import provider_models +from ouroboros.llm import LLMClient +from ouroboros.provider_models import ( + DIRECT_PROVIDER_DEFAULTS, + DIRECT_PROVIDER_REVIEW_ROLES, + DIRECT_PROVIDER_SCOPE_DEFAULTS, + ZAI_DIRECT_DEFAULTS, + ZAI_PLAN_ENDPOINTS, + ZAI_REASONING_EFFORT_ALIASES, + migrate_model_value, + normalize_model_identity, + normalize_zai_reasoning_effort, + provider_for_model, + provider_has_credentials, + resolve_zai_base_url, +) + +_PROVIDER_ENV_KEYS = ( + "OPENROUTER_API_KEY", "OPENAI_API_KEY", "OPENAI_BASE_URL", + "OPENAI_COMPATIBLE_API_KEY", "OPENAI_COMPATIBLE_BASE_URL", + "ANTHROPIC_API_KEY", "MINIMAX_API_KEY", "DEEPSEEK_API_KEY", + "ZAI_API_KEY", "ZAI_PLAN", + "CLOUDRU_FOUNDATION_MODELS_API_KEY", "GIGACHAT_CREDENTIALS", + "GIGACHAT_USER", "GIGACHAT_PASSWORD", "USE_LOCAL_MAIN", +) + + +def _clear_provider_env(monkeypatch): + for key in _PROVIDER_ENV_KEYS: + monkeypatch.delenv(key, raising=False) + + +def _zai_target(model="glm-5.3"): + return { + "provider": "zai", + "resolved_model": model, + "usage_model": f"zai/{model}", + "api_key": "sk-x", + "base_url": ZAI_PLAN_ENDPOINTS["payg"], + "supports_openrouter_extensions": False, + } + + +def _build(target, effort, tool_choice="auto", tools=None): + client = LLMClient() + kwargs = client._build_remote_kwargs( + target, [{"role": "user", "content": "hi"}], effort, 256, tool_choice, None, tools, + ) + return kwargs, client._pop_effort_clamp_disclosure() + + +class TestRegistry: + def test_prefix_routes_direct(self): + assert provider_for_model("zai::glm-5.3") == "zai" + + def test_slash_form_stays_openrouter(self): + from ouroboros.pricing import infer_api_key_type + + assert provider_for_model("zai/glm-5.3") == "openrouter" + assert infer_api_key_type("zai/glm-5.3") == "openrouter" + assert infer_api_key_type("zai::glm-5.3") == "zai" + + def test_credentials_mapping(self, monkeypatch): + _clear_provider_env(monkeypatch) + assert provider_has_credentials("zai") is False + monkeypatch.setenv("ZAI_API_KEY", "sk-x") + assert provider_has_credentials("zai") is True + + def test_direct_defaults_registered(self): + assert DIRECT_PROVIDER_DEFAULTS["zai"] is ZAI_DIRECT_DEFAULTS + assert ZAI_DIRECT_DEFAULTS["main"] == "zai::glm-5.3" + assert ZAI_DIRECT_DEFAULTS["light"] == "zai::glm-5.3-flash" + assert DIRECT_PROVIDER_REVIEW_ROLES["zai"] == ("main", "main", "main") + assert DIRECT_PROVIDER_SCOPE_DEFAULTS["zai"] == "zai::glm-5.3" + + def test_migrate_and_normalize_round_trip(self): + assert migrate_model_value("zai", "zai/glm-5.3") == "zai::glm-5.3" + assert migrate_model_value("zai", "zai::glm-5.3") == "zai::glm-5.3" + assert normalize_model_identity("zai::glm-5.3") == "zai/glm-5.3" + + +class TestPlanSwitch: + @pytest.mark.parametrize("plan", [None, "", "payg", " PAYG ", "unknown-plan"]) + def test_payg_is_the_default_and_the_fallback(self, plan): + assert resolve_zai_base_url(plan) == ZAI_PLAN_ENDPOINTS["payg"] + + def test_coding_plan_endpoint(self): + assert resolve_zai_base_url("coding") == ZAI_PLAN_ENDPOINTS["coding"] + assert ZAI_PLAN_ENDPOINTS["payg"].startswith("https://api.z.ai/") + assert ZAI_PLAN_ENDPOINTS["coding"].startswith("https://api.z.ai/") + + def test_resolve_target_uses_plan(self, monkeypatch): + _clear_provider_env(monkeypatch) + monkeypatch.setenv("ZAI_API_KEY", "sk-x") + monkeypatch.setenv("ZAI_PLAN", "coding") + monkeypatch.setattr( + "ouroboros.llm_routing.runtime_setting", + lambda key, default="": os.environ.get(key, default), + ) + target = LLMClient()._resolve_remote_target("zai::glm-5.3") + assert target["provider"] == "zai" + assert target["base_url"] == ZAI_PLAN_ENDPOINTS["coding"] + assert target["api_key"] == "sk-x" + assert target["usage_model"] == "zai/glm-5.3" + + def test_route_readers_follow_the_plan(self, monkeypatch): + # The Capability Evidence route identity (main route + reviewer route) + # must name the plan's endpoint, exactly as MiniMax's follows its region. + from ouroboros.gateway.settings import _active_main_route + from ouroboros.reviewer_window import reviewer_route + + monkeypatch.setattr("ouroboros.config.runtime_settings", lambda: {"ZAI_PLAN": "coding"}) + assert reviewer_route("zai::glm-5.3") == ("zai", ZAI_PLAN_ENDPOINTS["coding"]) + route = _active_main_route({"OUROBOROS_MODEL": "zai::glm-5.3", "ZAI_PLAN": "coding"}) + assert (route["provider"], route["base_url"]) == ("zai", ZAI_PLAN_ENDPOINTS["coding"]) + assert _active_main_route({"OUROBOROS_MODEL": "zai::glm-5.3"})["base_url"] == ZAI_PLAN_ENDPOINTS["payg"] + + def test_provider_test_rejects_an_unknown_plan(self, monkeypatch): + from ouroboros.gateway import models as provider_api + + monkeypatch.setattr(provider_api, "load_settings", lambda: {}) + monkeypatch.setattr( + provider_api, "_run_provider_test_with_settings", + lambda *_args: (_ for _ in ()).throw(AssertionError("must not probe")), + ) + body = provider_api._run_provider_test("zai", {"ZAI_API_KEY": "x", "ZAI_PLAN": "codign"}) + assert body == {"error": "unknown Z.ai plan", "_http_status": 400} + + def test_plan_alone_is_not_a_provider(self): + # A plan is a transport choice, not a credential: a draft carrying only + # ZAI_PLAN must be refused exactly like a MiniMax region without a key, + # while the same draft with the key is accepted. + from ouroboros.settings_setup_contract import validate_setup_payload + + plan_only = {"ZAI_PLAN": "coding", "OUROBOROS_MODEL": "zai::glm-5.3", + "OUROBOROS_MODEL_LIGHT": "zai::glm-5.3-flash", "OUROBOROS_MODEL_FALLBACKS": "zai::glm-5.3-flash"} + _prepared, error = validate_setup_payload(plan_only, {}) + assert error + _prepared, error = validate_setup_payload({**plan_only, "ZAI_API_KEY": "sk-zai-key-1234567890"}, {}) + assert not error + + +class TestEffortCarriage: + """The canonical scale is always projected onto Z.ai's low/high/max enum: + an absent tier would be served (and billed) at max.""" + + @pytest.mark.parametrize( + ("requested", "wire"), + [ + ("none", "low"), + ("minimal", "low"), + ("low", "low"), + ("medium", "high"), + ("high", "high"), + ("xhigh", "max"), + ("max", "max"), + ("ultra", "max"), + ], + ) + def test_projection_reaches_the_wire(self, requested, wire): + kwargs, note = _build(_zai_target(), requested) + assert kwargs["reasoning_effort"] == wire + assert "thinking" not in (kwargs.get("extra_body") or {}) + if requested == wire: + assert note is None + else: + assert note == { + "requested": requested, "applied": wire, + "reason": "provider_wire_mapping", "model": "glm-5.3", + } + + def test_table_stays_inside_the_provider_enum(self): + assert set(ZAI_REASONING_EFFORT_ALIASES.values()) == {"low", "high", "max"} + assert normalize_zai_reasoning_effort("not-a-tier") == "low" + + def test_forced_tool_choice_keeps_thinking_on(self): + tools = [{"type": "function", "function": {"name": "f", "parameters": {"type": "object"}}}] + kwargs, _ = _build(_zai_target(), "high", tool_choice="required", tools=tools) + assert kwargs["reasoning_effort"] == "high" + assert "thinking" not in (kwargs.get("extra_body") or {}) + + def test_generic_compatible_lane_is_untouched(self): + # The same GLM model id on an owner's OpenAI-compatible endpoint keeps + # today's behavior: the projection is keyed on the zai provider id, + # never on the model name. + target = { + "provider": "openai-compatible", "resolved_model": "glm-5.3", + "usage_model": "openai-compatible::glm-5.3", "api_key": "", + "base_url": "http://127.0.0.1:11434/v1", "supports_openrouter_extensions": False, + } + kwargs, note = _build(target, "medium") + assert "reasoning_effort" not in kwargs + assert note is None + + +class TestSingleProviderIndependence: + def test_exclusive_direct_env_detection(self, monkeypatch): + _clear_provider_env(monkeypatch) + monkeypatch.setenv("ZAI_API_KEY", "sk-x") + from ouroboros.config import _exclusive_direct_remote_provider_env + + assert _exclusive_direct_remote_provider_env() == "zai" + + def test_startup_gate_accepts_zai_only(self): + from ouroboros.server_runtime import ( + _exclusive_direct_remote_provider, + has_remote_provider, + has_startup_ready_provider, + ) + + settings = {"ZAI_API_KEY": "sk-x"} + assert has_remote_provider(settings) is True + assert has_startup_ready_provider(settings) is True + assert _exclusive_direct_remote_provider(settings) == "zai" + + def test_review_fallback_compiles_for_zai(self, monkeypatch): + _clear_provider_env(monkeypatch) + monkeypatch.setenv("ZAI_API_KEY", "sk-x") + monkeypatch.setenv("OUROBOROS_MODEL", "zai::glm-5.3") + monkeypatch.setenv("OUROBOROS_MODEL_LIGHT", "zai::glm-5.3-flash") + monkeypatch.setattr( + "ouroboros.review_model_routes.runtime_setting", + lambda key, default="": os.environ.get(key, default), + ) + from ouroboros.config import get_review_models + + assert get_review_models() == ["zai::glm-5.3"] * 3 + + def test_local_only_review_route_sees_zai(self, monkeypatch): + _clear_provider_env(monkeypatch) + monkeypatch.setenv("USE_LOCAL_MAIN", "1") + monkeypatch.setenv("ZAI_API_KEY", "sk-x") + assert provider_models.local_only_review_route_env() is False + + +class TestSecretSurfaces: + def test_forbidden_for_skills_and_masked(self): + from ouroboros.contracts.plugin_api import FORBIDDEN_SKILL_SETTINGS + from ouroboros.secret_masking import MASKED_SECRET_SETTING_KEYS + + assert "ZAI_API_KEY" in FORBIDDEN_SKILL_SETTINGS + assert "ZAI_API_KEY" in MASKED_SECRET_SETTING_KEYS + + def test_settings_defaults(self): + from ouroboros.config import SETTINGS_DEFAULTS + + assert SETTINGS_DEFAULTS["ZAI_API_KEY"] == "" + assert SETTINGS_DEFAULTS["ZAI_PLAN"] == "" + + +class TestSafetyRouting: + def test_zai_only_install_reaches_the_real_safety_check(self, monkeypatch): + """A zai-only install must reach the remote safety check, not fail open.""" + from ouroboros import safety + + _clear_provider_env(monkeypatch) + assert safety._any_remote_provider_configured() is False + monkeypatch.setenv("ZAI_API_KEY", "sk-x") + assert safety._any_remote_provider_configured() is True + assert safety._PROVIDER_KEY_ENV["zai"] == "ZAI_API_KEY" + + def test_light_model_reaches_its_provider_key(self, monkeypatch): + from ouroboros import safety + from ouroboros.pricing import infer_api_key_type + + _clear_provider_env(monkeypatch) + monkeypatch.setenv("ZAI_API_KEY", "sk-x") + assert safety._PROVIDER_KEY_ENV.get(infer_api_key_type("zai::glm-5.3-flash")) == "ZAI_API_KEY" + + +class TestProbeBilling: + """HTTP 429 code 1113 "Insufficient balance" is billing, not rate limiting.""" + + def test_1113_maps_to_no_credits(self): + from ouroboros.llm_probe import controlled_probe_error + + class Exhausted(Exception): + status_code = 429 + code = "1113" + type = "" + + result = controlled_probe_error(Exhausted("Insufficient balance")) + assert result["error"] == "No credits" + assert result["status_code"] == 429 + + def test_task_loop_does_not_retry_an_exhausted_plan(self): + from ouroboros.loop_llm_call import classify_llm_exception + + class Exhausted(Exception): + status_code = 429 + code = "1113" + body = {"error": {"code": "1113", "message": "Insufficient balance"}} + + exhausted = classify_llm_exception(Exhausted("Insufficient balance")) + assert exhausted.kind == "quota_exhausted" + assert exhausted.retry_same_request is False + # An ordinary 429 keeps its transient, retryable classification. + assert classify_llm_exception(RuntimeError("Error code: 429 - too many requests")).retry_same_request is True + + def test_plain_429_stays_rate_limited(self): + from ouroboros.llm_probe import controlled_probe_error + + class Plain(Exception): + status_code = 429 + code = "" + type = "" + + assert controlled_probe_error(Plain("too many requests"))["error"] == "Rate limited" diff --git a/web/modules/api_types.js b/web/modules/api_types.js index 321e6eb55..6e7f3f642 100644 --- a/web/modules/api_types.js +++ b/web/modules/api_types.js @@ -297,6 +297,7 @@ * @property {number=} recommended_index * @property {number=} answered_index * @property {string=} comment + * @property {string=} host_facts * @property {"chat"} type * @property {"user"|"assistant"|"system"} role * @property {string} content @@ -569,6 +570,7 @@ * @property {string} ts * @property {number=} answered_index * @property {string=} comment + * @property {string=} host_facts * @property {number=} chat_id * @property {string=} task_id * @property {boolean=} project_thread diff --git a/web/modules/chat_decision.js b/web/modules/chat_decision.js index ec5aa5638..2402882df 100644 --- a/web/modules/chat_decision.js +++ b/web/modules/chat_decision.js @@ -156,7 +156,7 @@ export function createChatDecision({ const MIRROR_SETTLE_MS = 5000; // Display fields a narrower delivery (the census, a lifecycle frame) may lack: an empty value // there never blanks what a complete row already carried. - const MIRROR_FIELDS = ['question', 'options', 'option_details', 'stake', 'project_name', 'assumption', 'recommended_index']; + const MIRROR_FIELDS = ['question', 'options', 'option_details', 'stake', 'project_name', 'assumption', 'recommended_index', 'host_facts']; const MIRROR_SIGNATURE = ['quiz_state', ...MIRROR_FIELDS, 'answered_index', 'comment', ...WAIT_FIELDS]; // The pointer row in the shape of the Project's quiz row, so one normalizer reads both. const mirrorQuiz = (row) => ({ ...row, type: 'quiz', role: 'assistant', state: row.quiz_state }); @@ -373,6 +373,8 @@ export function createChatDecision({ options, stake: String(src.stake || ''), assumption: String(src.assumption || ''), + // The host's sentence (asking task, run start, the owner's last message): plain text. + hostFacts: String(src.host_facts || ''), // The wait facts the header and the signature line read (waitFacts): the // original required flag, the closed bound, and the task's wait record when // history or a detail read attached it. @@ -392,6 +394,13 @@ export function createChatDecision({ }; } + function hostFactsLine(text) { + const line = document.createElement('div'); + line.className = 'chat-quiz-host-facts'; + line.textContent = text; + return line; + } + function appendRecommendedBadge(button) { // The asker's recommendation (the "A" option) is a badge on that option, every surface alike. if (button.querySelector('.chat-quiz-option-recommended')) return; @@ -588,6 +597,10 @@ export function createChatDecision({ const wait = waitFacts(current); const existing = quizViews.get(key); if (existing) { + // A narrower first delivery may have lacked the host's sentence; a later one adds it. + if (quiz.hostFacts && !existing.querySelector('.chat-quiz-host-facts')) { + existing.querySelector('.chat-quiz-question')?.nextElementSibling?.before(hostFactsLine(quiz.hostFacts)); + } if (quiz.comment) existing.dataset.ownerComment = quiz.comment; else if (Object.hasOwn(current, 'comment')) delete existing.dataset.ownerComment; if (!quiz.detailsUnavailable) { @@ -648,6 +661,7 @@ export function createChatDecision({ if (mountMarkdown) mountMarkdown(question, questionText); else question.textContent = questionText; card.append(question); + if (quiz.hostFacts) card.append(hostFactsLine(quiz.hostFacts)); if (quiz.stake) { const stake = document.createElement('div'); diff --git a/web/modules/onboarding_wizard.js b/web/modules/onboarding_wizard.js index bbe1b5ce6..4cec5c82f 100644 --- a/web/modules/onboarding_wizard.js +++ b/web/modules/onboarding_wizard.js @@ -75,10 +75,8 @@ import { accountRowFacts } from './harness_accounts.js'; subagentsOpen: false, reviewersOpen: false, localSourceOpen: Boolean(INITIAL_STATE.localSource), - moreProvidersOpen: Boolean( - INITIAL_STATE.cloudruKey || INITIAL_STATE.minimaxKey || INITIAL_STATE.deepseekKey - || INITIAL_STATE.compatibleBaseUrl || INITIAL_STATE.compatibleApiKey, - ), + moreProvidersOpen: Boolean(INITIAL_STATE.cloudruKey || INITIAL_STATE.minimaxKey || INITIAL_STATE.deepseekKey + || INITIAL_STATE.zaiKey || INITIAL_STATE.compatibleBaseUrl || INITIAL_STATE.compatibleApiKey), localStatusText: 'Status: Offline', localStatusTone: 'muted', localTestResult: '', @@ -135,7 +133,7 @@ import { accountRowFacts } from './harness_accounts.js'; } function hasApiAccess() { - return PROVIDER_FIELDS.some((field) => !['MINIMAX_REGION', 'OPENAI_COMPATIBLE_API_KEY'].includes(field.settingKey) + return PROVIDER_FIELDS.some((field) => !['MINIMAX_REGION', 'ZAI_PLAN', 'OPENAI_COMPATIBLE_API_KEY'].includes(field.settingKey) && trim(state[field.stateKey])); } @@ -266,7 +264,7 @@ import { accountRowFacts } from './harness_accounts.js'; ['OPENAI_API_KEY', 'openai'], ['CLOUDRU_FOUNDATION_MODELS_API_KEY', 'cloudru'], ['MINIMAX_API_KEY', 'minimax'], - ['DEEPSEEK_API_KEY', 'deepseek'], + ['DEEPSEEK_API_KEY', 'deepseek'], ['ZAI_API_KEY', 'zai'], ['ANTHROPIC_API_KEY', 'anthropic'], ].filter(([settingKey]) => configured[settingKey]); if (hasOpenrouter) return 'openrouter'; @@ -413,7 +411,7 @@ import { accountRowFacts } from './harness_accounts.js'; // one value the wizard can never replace. const shortKey = keyValues.find(([field, value]) => value && (field.inputType || 'password') === 'password' && value.length < 10 && value !== trim(INITIAL_STATE[field.stateKey])); if (shortKey) return `${shortKey[0].label.replace(' API Key', '')} API key looks too short.`; - const hasRemote = keyValues.some(([field, value]) => value && !['OPENAI_COMPATIBLE_API_KEY', 'MINIMAX_REGION'].includes(field.settingKey)); + const hasRemote = keyValues.some(([field, value]) => value && !['OPENAI_COMPATIBLE_API_KEY', 'MINIMAX_REGION', 'ZAI_PLAN'].includes(field.settingKey)); if (!hasRemote && !localSource && !hasModelSubscription()) { return state.agentsConnected.length ? 'A Main model source has not been confirmed. Retry discovery, add an API key, or choose a local model.' @@ -613,6 +611,7 @@ import { accountRowFacts } from './harness_accounts.js'; if (trim(state.cloudruKey)) rows.splice(1, 0, ['Cloud.ru', 'configured']); if (trim(state.minimaxKey)) rows.splice(1, 0, ['MiniMax', 'configured']); if (trim(state.deepseekKey)) rows.splice(1, 0, ['DeepSeek', 'configured']); + if (trim(state.zaiKey)) rows.splice(1, 0, ['Z.ai (GLM)', trim(state.zaiPlan) === 'coding' ? 'configured · coding plan' : 'configured']); if (trim(state.anthropicKey)) rows.splice(1, 0, ['Anthropic', 'configured']); if (hasLocalModel()) { rows.splice( diff --git a/web/modules/route_editor_primitives.js b/web/modules/route_editor_primitives.js index 5945485a5..e29bc3a48 100644 --- a/web/modules/route_editor_primitives.js +++ b/web/modules/route_editor_primitives.js @@ -84,13 +84,13 @@ export function composeModelSource(source, model) { // Owner-facing order of the direct API providers. OpenRouter first because an // unprefixed model id routes through it; the rest follow the settings order. export const API_PROVIDER_ORDER = ['openrouter', 'openai', 'anthropic', 'deepseek', - 'minimax', 'cloudru', 'gigachat', 'openai-compatible']; + 'zai', 'minimax', 'cloudru', 'gigachat', 'openai-compatible']; // Fallback names for providers the setup contract does not describe (GigaChat // has no profile spec). The contract's label wins whenever it exists. const API_PROVIDER_LABELS = { openrouter: 'OpenRouter', openai: 'OpenAI', anthropic: 'Anthropic', deepseek: 'DeepSeek', - minimax: 'MiniMax', cloudru: 'Cloud.ru Foundation Models', gigachat: 'GigaChat', + zai: 'Z.ai (GLM)', minimax: 'MiniMax', cloudru: 'Cloud.ru Foundation Models', gigachat: 'GigaChat', 'openai-compatible': 'OpenAI-compatible endpoint', }; @@ -103,6 +103,7 @@ const API_PROVIDER_CREDENTIALS = { openai: [['OPENAI_API_KEY']], anthropic: [['ANTHROPIC_API_KEY']], deepseek: [['DEEPSEEK_API_KEY']], + zai: [['ZAI_API_KEY']], minimax: [['MINIMAX_API_KEY']], cloudru: [['CLOUDRU_FOUNDATION_MODELS_API_KEY']], gigachat: [['GIGACHAT_CREDENTIALS'], ['GIGACHAT_USER', 'GIGACHAT_PASSWORD']], diff --git a/web/modules/settings.js b/web/modules/settings.js index 55aee3b16..5bb5b2a5f 100644 --- a/web/modules/settings.js +++ b/web/modules/settings.js @@ -38,6 +38,7 @@ const INPUT_FIELDS = [ ['s-openai-base-url', 'OPENAI_BASE_URL'], ['s-openai-compatible-base-url', 'OPENAI_COMPATIBLE_BASE_URL'], ['s-cloudru-base-url', 'CLOUDRU_FOUNDATION_MODELS_BASE_URL'], ['s-gigachat-scope', 'GIGACHAT_SCOPE'], ['s-gigachat-user', 'GIGACHAT_USER'], ['s-gigachat-base-url', 'GIGACHAT_BASE_URL'], ['s-gigachat-verify-ssl', 'GIGACHAT_VERIFY_SSL_CERTS'], ['s-minimax-region', 'MINIMAX_REGION'], + ['s-zai-plan', 'ZAI_PLAN'], ['s-server-host', 'OUROBOROS_SERVER_HOST', '127.0.0.1'], // 6.1: OUROBOROS_REVIEW_MODELS / OUROBOROS_SCOPE_REVIEW_MODELS are no // longer authored here — the Review lanes section composes the ONE @@ -401,12 +402,13 @@ function collectSecretValue(id, body) { * Exported for dependency-free node tests. */ export function moreProvidersCredentialConfigured({ - cloudruKey = '', minimaxKey = '', deepseekKey = '', gigachatCredentials = '', gigachatUser = '', gigachatPassword = '', + cloudruKey = '', minimaxKey = '', deepseekKey = '', zaiKey = '', gigachatCredentials = '', gigachatUser = '', gigachatPassword = '', } = {}) { const has = (v) => Boolean(String(v ?? '').trim()); return has(cloudruKey) || has(minimaxKey) || has(deepseekKey) + || has(zaiKey) || has(gigachatCredentials) || (has(gigachatUser) && has(gigachatPassword)); } @@ -757,6 +759,7 @@ export function initSettings({ state, setBeforePageLeave, ws } = {}) { cloudruKey: value('s-cloudru-key'), minimaxKey: value('s-minimax-key'), deepseekKey: value('s-deepseek-key'), + zaiKey: value('s-zai-key'), gigachatCredentials: value('s-gigachat-credentials'), gigachatUser: value('s-gigachat-user'), gigachatPassword: value('s-gigachat-password'), diff --git a/web/modules/settings_ui.js b/web/modules/settings_ui.js index cc028947a..59011822a 100644 --- a/web/modules/settings_ui.js +++ b/web/modules/settings_ui.js @@ -135,6 +135,16 @@ const PROVIDER_CARDS = [ testInputs: { 's-deepseek-key': 'DEEPSEEK_API_KEY' }, note: 'Pick DeepSeek as the source in Models or Agents, then choose deepseek-v4-pro or deepseek-v4-flash.', }, + { + id: 'zai', title: 'Z.ai (GLM)', icon: '', hint: 'Direct OpenAI-compatible runtime', advanced: true, + fields: [ + { id: 's-zai-key', settingKey: 'ZAI_API_KEY', label: 'API Key', placeholder: 'sk-...' }, + { id: 's-zai-plan', label: 'Plan', placeholder: 'payg or coding' }, + ], + testProvider: 'zai', + testInputs: { 's-zai-key': 'ZAI_API_KEY', 's-zai-plan': 'ZAI_PLAN' }, + note: 'Pick Z.ai as the source in Models or Agents, then choose glm-5.3 or glm-5.3-flash. Coding Plan subscribers: set Plan to coding; the default payg is pay-as-you-go, and a Coding Plan key tested there reports No credits.', + }, { id: 'gigachat', title: 'GigaChat', icon: '/static/providers/gigachat.svg', hint: 'Sber GigaChat via the gigachat library', advanced: true, fields: [ @@ -230,6 +240,7 @@ export const SECRET_KEYS = [ ['ANTHROPIC_API_KEY', 'Anthropic API Key', 'sk-ant-...'], ['MINIMAX_API_KEY', 'MiniMax API Key', 'MiniMax key'], ['DEEPSEEK_API_KEY', 'DeepSeek API Key', 'sk-...'], + ['ZAI_API_KEY', 'Z.ai API Key (GLM)', 'Z.ai key'], ['GITHUB_TOKEN', 'GitHub Token', 'ghp_...'], ['OUROBOROS_NETWORK_PASSWORD', 'Network Password', 'Required for LAN/Docker binds'], ]; diff --git a/web/style.css b/web/style.css index 2e699b059..4d416ae75 100644 --- a/web/style.css +++ b/web/style.css @@ -7219,6 +7219,14 @@ textarea.chat-input { min-inline-size: 0; } .chat-quiz-stake .inline-code { font-size: inherit; } +/* The host's own sentence under the question (who asks, how the run started, the owner's last message). */ +.chat-quiz-host-facts { + font-size: var(--type-meta); + line-height: var(--line-meta, 1.35); + color: var(--text-meta); + min-inline-size: 0; + overflow-wrap: anywhere; +} .chat-quiz-options { display: flex; flex-direction: column; diff --git a/web/tests/api_provider_choices.test.js b/web/tests/api_provider_choices.test.js index 71042c53b..e703650f3 100644 --- a/web/tests/api_provider_choices.test.js +++ b/web/tests/api_provider_choices.test.js @@ -27,7 +27,8 @@ test('a provider is offered only while its credential is stored, in one owner-fa OPENAI_COMPATIBLE_BASE_URL: 'http://localhost:11434/v1', GIGACHAT_USER: 'owner', GIGACHAT_PASSWORD: '***set***', CLOUDRU_FOUNDATION_MODELS_API_KEY: '***set***', MINIMAX_API_KEY: 'mm', - DEEPSEEK_API_KEY: 'ds', ANTHROPIC_API_KEY: 'sk-ant', OPENAI_API_KEY: 'sk', + DEEPSEEK_API_KEY: 'ds', ZAI_API_KEY: 'zai', + ANTHROPIC_API_KEY: 'sk-ant', OPENAI_API_KEY: 'sk', OPENROUTER_API_KEY: 'sk-or', }); assert.deepEqual(every.map((provider) => provider.id), API_PROVIDER_ORDER); diff --git a/web/tests/chat_decision.test.js b/web/tests/chat_decision.test.js index e66cb0c7f..8c40501eb 100644 --- a/web/tests/chat_decision.test.js +++ b/web/tests/chat_decision.test.js @@ -805,3 +805,49 @@ test('a late answer says where it went', async () => { } finally { fx.restore(); } } }); + +test('the host facts line sits under the question when the card carries it and is absent otherwise', () => { + const fx = fixture(); + const facts = 'Asked by task t-1, started by your message of 2026-09-25 00:21 UTC; ' + + 'your last message in this chat: 2026-09-25 00:21 UTC (47 minutes before this question).'; + try { + const card = fx.decision.buildQuizCard({ ...WS_MSG, quiz_id: 'qz-facts', host_facts: facts }); + const line = card.querySelector('.chat-quiz-host-facts'); + assert.equal(line.textContent, facts); + // Plain text directly under the question, never through the markdown pipeline. + assert.equal(line.innerHTML, undefined); + const question = card.querySelector('.chat-quiz-question'); + assert.equal(question.nextElementSibling, line); + // A stored history row nests the quiz; the sentence rides the nested block. + const replay = fx.decision.buildQuizCard({ task_id: 't-1', text: WS_MSG.question, + quiz: { ...WS_MSG, quiz_id: 'qz-facts-replay', host_facts: facts } }); + assert.equal(replay.querySelector('.chat-quiz-host-facts').textContent, facts); + const bare = fx.decision.buildQuizCard({ ...WS_MSG, quiz_id: 'qz-bare' }); + assert.equal(bare.querySelector('.chat-quiz-host-facts'), null); + const empty = fx.decision.buildQuizCard({ ...WS_MSG, quiz_id: 'qz-empty', host_facts: '' }); + assert.equal(empty.querySelector('.chat-quiz-host-facts'), null); + // A later, richer delivery of an already rendered card adds the line once. + assert.equal(fx.decision.buildQuizCard({ ...WS_MSG, quiz_id: 'qz-bare', host_facts: facts }), null); + assert.equal(bare.querySelectorAll('.chat-quiz-host-facts').length, 1); + assert.equal(bare.querySelector('.chat-quiz-question').nextElementSibling.textContent, facts); + fx.decision.buildQuizCard({ ...WS_MSG, quiz_id: 'qz-bare', host_facts: facts }); + assert.equal(bare.querySelectorAll('.chat-quiz-host-facts').length, 1); + } finally { fx.restore(); } +}); + +test('a Main mirror of a Project question carries the host facts line from its pointer row', () => { + const column = new NodeStub(); + const fx = fixture({ isMain: true, + frameNode: (_msg, card) => { const bubble = new NodeStub(); bubble.append(card); return bubble; }, + insertMessageNode: (node) => { column.append(node); return true; } }); + const row = { role: 'system', system_type: 'project_question_pointer', task_id: 't-9', quiz_id: 'qz-9', + project_id: 'p1', project_chat_id: 23, project_name: 'Storage', ts: '2026-09-25T00:00:00+00:00', + quiz_state: 'open', question: 'Merge now?', options: ['Yes', 'No'] }; + try { + assert.ok(fx.decision.appendQuestionPointer({ ...row, host_facts: 'Asked by task t-9, origin unknown.' })); + const card = column.children[0].children[0]; + assert.equal(card.querySelector('.chat-quiz-host-facts').textContent, 'Asked by task t-9, origin unknown.'); + assert.ok(fx.decision.appendQuestionPointer({ ...row, task_id: 't-10' })); + assert.equal(column.children[1].children[0].querySelector('.chat-quiz-host-facts'), null); + } finally { fx.restore(); } +}); diff --git a/web/tests/fixtures/onboarding_bootstrap.json b/web/tests/fixtures/onboarding_bootstrap.json index cbedd49c3..666f88ca6 100644 --- a/web/tests/fixtures/onboarding_bootstrap.json +++ b/web/tests/fixtures/onboarding_bootstrap.json @@ -181,6 +181,28 @@ "settingsInputId": "s-deepseek-key", "stateKey": "deepseekKey" }, + { + "group": "more", + "id": "zai-key", + "inputType": "password", + "label": "Z.ai API Key (GLM)", + "note": "Optional. If this is the only remote key, the next step prefills Z.ai's own GLM model ids.", + "placeholder": "...", + "settingKey": "ZAI_API_KEY", + "settingsInputId": "s-zai-key", + "stateKey": "zaiKey" + }, + { + "group": "more", + "id": "zai-plan", + "inputType": "text", + "label": "Z.ai Plan", + "note": "Choose payg (pay-as-you-go, default) or coding (Coding Plan endpoint; officially intended for supported coding tools only).", + "placeholder": "payg or coding", + "settingKey": "ZAI_PLAN", + "settingsInputId": "s-zai-plan", + "stateKey": "zaiPlan" + }, { "group": "primary", "id": "anthropic-key", @@ -260,6 +282,11 @@ "label": "OpenRouter", "modelCopy": "OpenRouter-style routing remains active. OpenRouter stays the source for ids such as openai/gpt-5.6-terra or anthropic/claude-sonnet-5.", "providerCopy": "OpenRouter is present, so the next step keeps router-style defaults while still saving any extra direct keys you paste here." + }, + "zai": { + "label": "Z.ai (GLM)", + "modelCopy": "Z.ai-only setup detected. These defaults use glm-5.3 for main work and glm-5.3-flash for the light lane. The plan setting selects the pay-as-you-go or Coding Plan endpoint.", + "providerCopy": "Z.ai is present, so the next step prefills Z.ai's own GLM model ids." } }, "reviewModes": [ @@ -394,7 +421,9 @@ "runtimeMode": "advanced", "skillsRepoPath": "", "totalBudget": 200.0, - "visionModel": "" + "visionModel": "", + "zaiKey": "", + "zaiPlan": "" }, "localPresets": { "qwen25-7b": { @@ -478,6 +507,13 @@ "light": "openai/gpt-5.6-luna", "main": "google/gemini-3.8-flash", "vision": "" + }, + "zai": { + "consciousness": "", + "fallback": "zai::glm-5.3-flash", + "light": "zai::glm-5.3-flash", + "main": "zai::glm-5.3", + "vision": "" } }, "modelSuggestions": [ @@ -497,6 +533,8 @@ "deepseek/deepseek-v4-pro", "deepseek::deepseek-v4-pro", "deepseek::deepseek-v4-flash", + "zai::glm-5.3", + "zai::glm-5.3-flash", "openai-compatible::meta-llama/compatible", "cloudru::zai-org/GLM-4.7", "minimax::MiniMax-M3", diff --git a/web/tests/provider_test.test.js b/web/tests/provider_test.test.js index 8eb5678b1..c4addc8d5 100644 --- a/web/tests/provider_test.test.js +++ b/web/tests/provider_test.test.js @@ -45,7 +45,7 @@ test('provider test responses apply only to the exact draft generation', () => { test('every provider test button warns that one charged request is sent', () => { const html = renderSettingsPage(); const buttons = [...html.matchAll(/data-provider-test="[^"]+"[^>]*>/g)]; - assert.equal(buttons.length, 8); + assert.equal(buttons.length, 9); for (const [button] of buttons) { assert.match( button, @@ -58,10 +58,10 @@ test('provider actions use the shared status-first action row contract', () => { const html = renderSettingsPage(); const rows = [...html.matchAll(/
/g)] .map(([row]) => row); - // Eight provider probes plus the catalog action. The Claude-runtime + // Nine provider probes plus the catalog action. The Claude-runtime // status/Repair panel is retired with its product surface (the advisory // pre-review runs on a configured routed model or agent session now). - assert.equal(rows.length, 9, 'eight provider probes plus the catalog action'); + assert.equal(rows.length, 10, 'nine provider probes plus the catalog action'); assert.doesNotMatch(html, /settings-claude-code/); assert.doesNotMatch(html, /settings-ghost-btn/); for (const row of rows) {