diff --git a/docs/DOMAIN_MAP.md b/docs/DOMAIN_MAP.md index 4591a7e13..62092f377 100644 --- a/docs/DOMAIN_MAP.md +++ b/docs/DOMAIN_MAP.md @@ -14,7 +14,7 @@ The manifest is the SSOT of the module→domain assignment (1:1, complete over t | D04 | Tool execution: registry, access & typed results | 20 | 0 | | D05 | Tool surfaces: files, code, shell, media, external | 28 | 0 | | D06 | Review stack | 66 | 0 | -| D07 | Delegation, subagents & Claudexor | 50 | 0 | +| D07 | Delegation, subagents & Claudexor | 51 | 0 | | D08 | Supervisor: queue, workers, events & runtime control | 46 | 0 | | D09 | Cancellation, owner control & process custody | 13 | 0 | | D10 | Git, update & release machinery | 28 | 0 | @@ -28,7 +28,7 @@ The manifest is the SSOT of the module→domain assignment (1:1, complete over t | D18 | Launcher, packaging, platform & shared substrate | 14 | 0 | | D19 | Frozen contracts (ABI) | 10 | 0 | | D20 | Presence | 9 | 0 | -| **total** | | **550** | **0** | +| **total** | | **551** | **0** | ## Dependency direction matrix (strict, pinned) @@ -422,6 +422,7 @@ No function body (≥ 10 normalized lines) is shared verbatim across domains. Ne - `ouroboros/nanny_pacing.py` - `ouroboros/subagent_bootstrap.py` - `ouroboros/subagent_dispatch_notes.py` +- `ouroboros/subagent_history.py` - `ouroboros/subagent_messages.py` - `ouroboros/subagent_route_health.py` - `ouroboros/subagent_runtime.py` diff --git a/docs/inventories/FACADE_INVENTORY.md b/docs/inventories/FACADE_INVENTORY.md index aba9d99f9..79677940b 100644 --- a/docs/inventories/FACADE_INVENTORY.md +++ b/docs/inventories/FACADE_INVENTORY.md @@ -2,7 +2,7 @@ AST-derived inventory of compatibility facades, regenerated by `python scripts/regenerate_inventories.py`. Do not edit. A facade row is any runtime module whose top-level `from import ...` statements carry the `noqa: F401` re-export marker — the codebase's declared "this binding exists for its binding, not for this module's own use" convention (reference FACADE_CONSUMERS method). Leaf domains come from `ouroboros/domains.toml`; a leaf outside the facade's domain is marked ✗ (that edge also appears in the manifest's pinned direction matrix). `tests/test_generated_inventories.py` pins byte-identity, so any re-export surface change must regenerate this file. -- facade modules: **58**; marked re-export bindings: **2357**; cross-domain facade→leaf pairs: **130** +- facade modules: **58**; marked re-export bindings: **2360**; cross-domain facade→leaf pairs: **130** | facade | domain | bindings | leaves | |---|---|---:|---| @@ -33,7 +33,7 @@ AST-derived inventory of compatibility facades, regenerated by `python scripts/r | `ouroboros/review_substrate.py` | D06 | 74 | `ouroboros/_outcome_receipts.py` (1 ✗D01)
`ouroboros/observability.py` (3 ✗D16)
`ouroboros/outcomes.py` (3 ✗D01)
`ouroboros/provider_models.py` (1 ✗D02)
`ouroboros/review_dispatch.py` (8)
`ouroboros/review_execution.py` (15)
`ouroboros/review_execution_projection.py` (3)
`ouroboros/review_projection.py` (8)
`ouroboros/review_records.py` (8)
`ouroboros/review_verdict.py` (20)
`ouroboros/reviewer_slot_config.py` (3)
`ouroboros/task_results.py` (1 ✗D17) | | `ouroboros/skill_review.py` | D14 | 61 | `ouroboros/skill_review_cycles.py` (8)
`ouroboros/skill_review_history.py` (5)
`ouroboros/skill_review_output.py` (4)
`ouroboros/skill_review_packs.py` (8)
`ouroboros/skill_review_prompt.py` (9)
`ouroboros/skill_review_rebuttals.py` (4)
`ouroboros/skill_review_status.py` (8)
`ouroboros/tools/review_helpers.py` (8 ✗D06)
`ouroboros/triad_review.py` (3 ✗D06)
`ouroboros/utils.py` (4 ✗D18) | | `ouroboros/subagent_dispatch_notes.py` | D07 | 4 | `ouroboros/agent_dispatch.py` (2 ✗D01)
`ouroboros/subagents.py` (2) | -| `ouroboros/subagents.py` | D07 | 4 | `ouroboros/subagent_route_health.py` (4) | +| `ouroboros/subagents.py` | D07 | 7 | `ouroboros/subagent_history.py` (3)
`ouroboros/subagent_route_health.py` (4) | | `ouroboros/task_pacing.py` | D01 | 2 | `ouroboros/task_results.py` (2 ✗D17) | | `ouroboros/task_results.py` | D17 | 8 | `ouroboros/task_result_schema.py` (8) | | `ouroboros/tool_access.py` | D04 | 43 | `ouroboros/contracts/task_constraint.py` (2 ✗D19)
`ouroboros/tool_access_paths.py` (10)
`ouroboros/tool_access_roots.py` (9)
`ouroboros/tool_access_types.py` (15)
`ouroboros/tool_access_user_files.py` (5)
`ouroboros/tool_capabilities.py` (2) | diff --git a/ouroboros/agent_task_pipeline.py b/ouroboros/agent_task_pipeline.py index 8206e035e..a1c2babff 100644 --- a/ouroboros/agent_task_pipeline.py +++ b/ouroboros/agent_task_pipeline.py @@ -1060,6 +1060,8 @@ def _store_task_result(env: Any, task: Dict[str, Any], text: str, **({"swarm_efficiency": swarm_efficiency} if swarm_efficiency else {}), ts=utc_now_iso(), ) + from ouroboros.subagent_history import record_task_execution + record_task_execution(task, usage, drive_root=task.get("budget_drive_root") or env.drive_root) except Exception as e: log.warning("Failed to store task result: %s", e) diff --git a/ouroboros/context_runtime_facts.py b/ouroboros/context_runtime_facts.py index f0c8ecbf3..e032b81ed 100644 --- a/ouroboros/context_runtime_facts.py +++ b/ouroboros/context_runtime_facts.py @@ -217,7 +217,13 @@ def _delegation_capability_fact() -> Optional[Dict[str, Any]]: last_fact["applied_profile"] = str(last["applied_profile"]) if last.get("selected_subagent_id"): last_fact["selected_subagent_id"] = str(last["selected_subagent_id"]) + for key in ("outcome", "failure_code", "reset_at", "occurred_at", "observed_at"): + if key in last: + last_fact[key] = last[key] delegation["subagent_last_delegation"] = last_fact + rows = last.get("latest_by_subagent") + if isinstance(rows, dict) and rows: + delegation["subagents_last_executions"] = list(rows.values()) if len(delegation) == 1: return None return delegation diff --git a/ouroboros/delegate_custody.py b/ouroboros/delegate_custody.py index d9d523a2e..249e6a3cb 100644 --- a/ouroboros/delegate_custody.py +++ b/ouroboros/delegate_custody.py @@ -229,12 +229,19 @@ def emit(drive_root: Any, kind: str, payload: Dict[str, Any]) -> bool: answer. """ try: - written = bool(append_jsonl(event_log_path(drive_root), {"ts": utc_now_iso(), "type": kind, **payload})) + event = {"ts": utc_now_iso(), "type": kind, **payload} + written = bool(append_jsonl(event_log_path(drive_root), event)) except Exception: log.warning("delegate custody row could not be written (%s)", kind, exc_info=True) return False if not written: log.warning("delegate custody row was rejected by the event log (%s)", kind) + elif kind == START_FAILED: + from ouroboros.subagent_history import record_session_start_failure + try: + record_session_start_failure(drive_root, event) + except Exception: + log.debug("Start history unavailable", exc_info=True) return written def daemon_says_absent(exc: Any) -> bool: @@ -1018,6 +1025,11 @@ def settle_run(drive_root: Any, gateway: Any, custody: RunCustody, detail: Dict[ if custody.settled: _retire_project_locked(drive_root, gateway, custody) if custody.settled: + from ouroboros.subagent_history import record_session_execution + try: + record_session_execution(drive_root, custody, detail, observed) + except Exception: + log.debug("Session history unavailable", exc_info=True) resolve_containment_fault(drive_root, custody, "settled_terminal") # CONSUMPTION BEFORE SETTLEMENT is a fact, not a gate; asking before staging # now would answer "no omission" for every first settlement (the render- diff --git a/ouroboros/domains.toml b/ouroboros/domains.toml index b93c7882b..c93242a9d 100644 --- a/ouroboros/domains.toml +++ b/ouroboros/domains.toml @@ -390,6 +390,7 @@ D20 = "Presence" "ouroboros/skill_token.py" = "D14" "ouroboros/subagent_bootstrap.py" = "D07" "ouroboros/subagent_dispatch_notes.py" = "D07" +"ouroboros/subagent_history.py" = "D07" "ouroboros/subagent_messages.py" = "D07" "ouroboros/subagent_runtime.py" = "D07" "ouroboros/subagent_work_order.py" = "D07" diff --git a/ouroboros/gateway/contracts.py b/ouroboros/gateway/contracts.py index a8e61e2e6..9998cd5df 100644 --- a/ouroboros/gateway/contracts.py +++ b/ouroboros/gateway/contracts.py @@ -824,6 +824,7 @@ class SettingsMeta(SettingsNetworkMeta, total=False): setup_contract: Dict[str, Any] available_subagents: AvailableSubagentsSettingsMeta policy_state: SettingsPolicyState + restart_state: Dict[str, Any] class SettingsSaveResponse(TypedDict, total=False): @@ -831,6 +832,7 @@ class SettingsSaveResponse(TypedDict, total=False): no_changes: bool restart_required: bool restart_keys: list[str] + restart_state: Dict[str, Any] immediate_changed: bool next_task_changed: bool warnings: list[str] @@ -1002,6 +1004,7 @@ class LocalModelStatusResponse(TypedDict, total=False): port: int message: str error: str + settings_application: Dict[str, Any] class McpStatusResponse(TypedDict, total=False): diff --git a/ouroboros/gateway/models.py b/ouroboros/gateway/models.py index 9a3aec679..ee0572e3c 100644 --- a/ouroboros/gateway/models.py +++ b/ouroboros/gateway/models.py @@ -576,7 +576,8 @@ async def api_local_model_start(request: Request) -> JSONResponse: # Download can be slow, run in thread to not block the async event loop model_path = await asyncio.to_thread(mgr.download_model, source, filename) - mgr.start_server(model_path, port=port, n_gpu_layers=n_gpu_layers, n_ctx=n_ctx, chat_format=chat_format) + mgr.start_server(model_path, port=port, n_gpu_layers=n_gpu_layers, n_ctx=n_ctx, + chat_format=chat_format, source=source, filename=filename) return JSONResponse({"status": "starting", "model_path": model_path}) except Exception as e: return json_exception(e) @@ -600,7 +601,7 @@ async def api_local_model_status(request: Request) -> JSONResponse: # on the very first poll — before the user clicks Start. if mgr._runtime_status == "unknown" and mgr.get_status() == "offline": await asyncio.to_thread(mgr.check_runtime) - return JSONResponse(mgr.status_dict()) + return JSONResponse({**mgr.status_dict(), "settings_application": mgr.settings_application(load_settings())}) except Exception as e: return JSONResponse({"status": "error", "error": str(e)}) diff --git a/ouroboros/gateway/settings.py b/ouroboros/gateway/settings.py index 2a202c3eb..31e34dd23 100644 --- a/ouroboros/gateway/settings.py +++ b/ouroboros/gateway/settings.py @@ -209,6 +209,36 @@ def _build_policy_state(settings: Dict[str, Any]) -> dict: } +def _build_restart_state(settings: Dict[str, Any]) -> dict: + """Compare saved intent with component-owned inputs, never os.environ.""" + from ouroboros.config import get_runtime_mode, normalize_runtime_mode + from ouroboros.local_model import get_manager, local_model_settings + from ouroboros.server_process import applied_restart_settings + + applied = applied_restart_settings() + desired = {key: settings.get(key, _SETTINGS_DEFAULTS.get(key, "")) + for key in _RESTART_REQUIRED_KEYS if key not in local_model_settings({})} + applied["OUROBOROS_RUNTIME_MODE"] = get_runtime_mode() + desired["OUROBOROS_RUNTIME_MODE"] = normalize_runtime_mode(settings.get("OUROBOROS_RUNTIME_MODE")) + pending = [] + for key, value in desired.items(): + if key not in applied: + continue + actual = applied[key] + if key == "OUROBOROS_SKILLS_REPO_PATH": + value = str(pathlib.Path(str(value).strip()).expanduser()) if str(value).strip() else "" + actual = str(pathlib.Path(str(actual).strip()).expanduser()) if str(actual).strip() else "" + if str(value).strip() != str(actual).strip(): + pending.append(key) + unknown = sorted(set(desired) - set(applied)) + local = get_manager().settings_application(settings) + summary = f"Restart Ouroboros to apply {len(pending)} saved setting(s)." if pending else "" + if unknown: + summary += f" Application state is not reported for {len(unknown)} runtime setting(s)." + return {"restart_required": bool(pending), "restart_keys": sorted(pending), + "unknown_keys": unknown, "local_model": local, "summary": summary.strip()} + + def _rehydrate_mcp_servers_payload(incoming: Any, current: Any) -> list: if not isinstance(incoming, list): return [] @@ -246,17 +276,6 @@ from ouroboros.settings_scales import ( ) -def _classify_settings_changes( - old: Dict[str, Any], - new: Dict[str, Any], -) -> list: - """Return changed keys requiring process restart; others hot-reload next task.""" - return [ - k for k in _RESTART_REQUIRED_KEYS - if str(new.get(k, "") or "") != str(old.get(k, "") or "") - ] - - def _effect_buckets(all_changed: list) -> tuple: """Split changed keys into the honest effect buckets for the save response. @@ -1003,6 +1022,7 @@ async def api_settings_get(request: Request) -> JSONResponse: # not a second policy store. try: meta["policy_state"] = _build_policy_state(settings) + meta["restart_state"] = _build_restart_state(settings) except Exception: # A settings read must stay available even if an optional projection # helper is unavailable during startup. The persisted values remain @@ -1370,10 +1390,8 @@ def _api_settings_post_locked(request: Request, body: Any) -> JSONResponse: k for k in current if str(current.get(k, "") or "") != str(old_effective_settings.get(k, "") or "") ] - restart_keys = _classify_settings_changes(old_effective_settings, current) if runtime_changed: all_changed.append("OUROBOROS_RUNTIME_MODE") - restart_keys.append("OUROBOROS_RUNTIME_MODE") # Snapshot BEFORE the save lands: only a task already started at that # moment keeps the previous configuration. Measuring after the write @@ -1483,9 +1501,11 @@ def _api_settings_post_locked(request: Request, body: Any) -> JSONResponse: resp["agent_task_running"] = True if not all_changed: resp["no_changes"] = True - if restart_keys: + restart_state = _build_restart_state(settings_to_save) + resp["restart_state"] = restart_state + if restart_state["restart_required"]: resp["restart_required"] = True - resp["restart_keys"] = restart_keys + resp["restart_keys"] = restart_state["restart_keys"] if immediate_changed: resp["immediate_changed"] = True if next_task_changed: diff --git a/ouroboros/local_model.py b/ouroboros/local_model.py index 73b6f5c3a..539c14360 100644 --- a/ouroboros/local_model.py +++ b/ouroboros/local_model.py @@ -20,6 +20,19 @@ log = logging.getLogger(__name__) _LOCAL_MODEL_DEFAULT_PORT = 8766 + +def local_model_settings(values: dict) -> dict: + """Effective launch values, shared by execution and saved/applied comparison.""" + context = int(values.get("LOCAL_MODEL_CONTEXT_LENGTH", 16384)) + return { + "LOCAL_MODEL_SOURCE": str(values.get("LOCAL_MODEL_SOURCE") or "").strip(), + "LOCAL_MODEL_FILENAME": str(values.get("LOCAL_MODEL_FILENAME") or "").strip(), + "LOCAL_MODEL_PORT": int(values.get("LOCAL_MODEL_PORT", _LOCAL_MODEL_DEFAULT_PORT)), + "LOCAL_MODEL_N_GPU_LAYERS": int(values.get("LOCAL_MODEL_N_GPU_LAYERS", 0)), + "LOCAL_MODEL_CONTEXT_LENGTH": context if context > 0 else 16384, + "LOCAL_MODEL_CHAT_FORMAT": str(values.get("LOCAL_MODEL_CHAT_FORMAT") or "").strip(), + } + def _get_install_command() -> list: """Return the llama-cpp-python pip install command.""" from ouroboros.platform_layer import pip_install_target_args @@ -82,6 +95,8 @@ class LocalModelManager: self._serving_context_length: int = 0 self._measurement_route: bool = False self._model_name: str = "" + self._launch_settings: dict = {} + self._applied_settings: dict = {} self._download_progress: float = 0.0 self._stderr_buf: bytes = b"" # Runtime (llama-cpp-python) install state @@ -119,6 +134,25 @@ class LocalModelManager: "runtime_install_log": self._runtime_install_log[-500:] if self._runtime_install_log else "", } + def settings_application(self, settings: dict) -> dict: + """A ready owned process applies its captured inputs; other states do not.""" + desired = local_model_settings(settings) + with self._lock: + status = self.get_status() + applied = dict(self._applied_settings) if status == "ready" else {} + pending = sorted(key for key in desired if key in applied and desired[key] != applied[key]) + unknown = sorted(set(desired) - set(applied)) if status == "ready" else [] + if pending: + action = "Stop, then Start" if desired["LOCAL_MODEL_SOURCE"] else "Stop" + summary = f"Saved local model settings differ from the running model. Use {action} to apply them." + elif status != "ready" and desired["LOCAL_MODEL_SOURCE"]: + summary = "Local model settings are not applied to a ready model. Use the local model controls to start it." + elif unknown: + summary = "Some running local model settings were not reported." + else: + summary = "" + return {"status": status, "pending_keys": pending, "unknown_keys": unknown, "summary": summary} + def check_runtime(self) -> bool: """Check llama-cpp-python importability and update runtime status.""" try: @@ -400,6 +434,8 @@ class LocalModelManager: n_gpu_layers: int = -1, n_ctx: int = 0, chat_format: str = "", + source: str | None = None, + filename: str | None = None, ) -> None: """Start the server; rechecks runtime as a safety net before Popen.""" with self._lock: @@ -410,6 +446,15 @@ class LocalModelManager: self._port = port self._status = "loading" self._error = None + launch = local_model_settings({"LOCAL_MODEL_SOURCE": source, "LOCAL_MODEL_FILENAME": filename, + "LOCAL_MODEL_PORT": port, "LOCAL_MODEL_N_GPU_LAYERS": n_gpu_layers, + "LOCAL_MODEL_CONTEXT_LENGTH": n_ctx, "LOCAL_MODEL_CHAT_FORMAT": chat_format}) + self._launch_settings = {key: value for key, value in launch.items() + if not ((key == "LOCAL_MODEL_SOURCE" and source is None) + or (key == "LOCAL_MODEL_FILENAME" and filename is None))} + self._applied_settings = {} + self._port = port = launch["LOCAL_MODEL_PORT"] + n_gpu_layers, chat_format = launch["LOCAL_MODEL_N_GPU_LAYERS"], launch["LOCAL_MODEL_CHAT_FORMAT"] python = sys.executable cmd = [ @@ -420,7 +465,7 @@ class LocalModelManager: ] if chat_format: cmd.extend(["--chat_format", chat_format]) - effective_ctx = n_ctx if n_ctx > 0 else 16384 + effective_ctx = launch["LOCAL_MODEL_CONTEXT_LENGTH"] self._context_length = effective_ctx self._serving_context_length = effective_ctx @@ -510,6 +555,8 @@ class LocalModelManager: proc = self._proc start = time.time() while time.time() - start < timeout: + if self._proc is not proc: + return if self._proc is None or self._proc.poll() is not None: self._status = "error" rc = self._proc.returncode if self._proc else "?" @@ -542,6 +589,7 @@ class LocalModelManager: self._service_binding = None log.warning("Local model service binding unavailable; health remains ready", exc_info=True) self._status = "ready" + self._applied_settings = dict(self._launch_settings) self._context_length = health.get("context_length", 0) self._model_name = health.get("model_name", "") log.info( @@ -553,6 +601,8 @@ class LocalModelManager: pass time.sleep(2.0) + if self._proc is not proc: + return self._status = "error" self._error = f"Server failed to become healthy within {timeout}s" log.error(self._error) @@ -569,6 +619,7 @@ class LocalModelManager: self._proc = None self._install_proc = None self._status = "offline" + self._applied_settings = {} self._error = None self._context_length = 0 diff --git a/ouroboros/local_model_autostart.py b/ouroboros/local_model_autostart.py index f1f1c9f8a..963b5ddc1 100644 --- a/ouroboros/local_model_autostart.py +++ b/ouroboros/local_model_autostart.py @@ -44,6 +44,7 @@ def auto_start_local_model(settings: dict) -> None: n_gpu_layers=n_gpu_layers, n_ctx=n_ctx, chat_format=chat_format, + source=source, filename=filename, ) log.info("Local model auto-started successfully") except Exception as exc: diff --git a/ouroboros/loop_llm_call.py b/ouroboros/loop_llm_call.py index 79513e5e6..b99b3d8cc 100644 --- a/ouroboros/loop_llm_call.py +++ b/ouroboros/loop_llm_call.py @@ -288,6 +288,7 @@ def _record_and_emit_empty_response( "_last_llm_error": _short_error_text(log_msg), "execution_status": status, "reason_code": reason, "_last_llm_error_kind": kind, }) + accumulated_usage.get("_last_llm_call_meta", {}).update(failure_code=kind) return event_type, is_provider_glitch, permanent_body_error @@ -827,7 +828,7 @@ def _remember_llm_call( reported_model: Any = None, use_local: Optional[bool] = None, ) -> Dict[str, Any]: - call_meta = { + call_meta = {"ts": utc_now_iso(), "llm_call_id": llm_call_id, "execution_id": execution_id, "round_id": round_id, @@ -943,10 +944,8 @@ def _record_llm_call_error( "route": dict(getattr(error, "route", {}) or {}), "outcome": "unknown", "request_ref": ctx.request_ref.get("manifest_ref") if ctx.request_ref else None, } - # The caller chooses this existing repeat rail. Ordinary managed tasks - # set it to zero: their transport episode requires upstream recovery - # before a marked NEW attempt. Other callers retain their bounded rail; - # grant before logging, and uncount a grant refused before dispatch. + # Ordinary managed tasks require upstream recovery before a new attempt; + # other callers retain their bounded repeat rail. if ( is_retryable_transport_death(error) and repeats < ctx.transport_death_retries and ctx.attempt < ctx.transient_budget - 1 @@ -973,10 +972,7 @@ def _record_llm_call_error( "llm_call_id": ctx.llm_call_id, "round": ctx.round_idx, "attempt": ctx.attempt + 1, "model": ctx.model, } - # ONE error row (#355): a successful append's registered sink owns live - # delivery. Without that path, send the SAME evidence through the queue, - # preserving its identity for live/backfill dedupe. No llm_round_error - # sibling here; Background Consciousness keeps its own separate producer. + # The append sink owns delivery; queue fallback preserves the same identity. error_event = { "ts": utc_now_iso(), "type": "llm_api_error", **identity, "error": display_error, "error_kind": classification.kind, "retry_same_request": will_retry, @@ -988,11 +984,15 @@ def _record_llm_call_error( } if not append_jsonl(ctx.drive_logs / "events.jsonl", error_event) or not has_log_sink(): emit_log_event(ctx.event_queue, error_event, log_label="LLM call error") + ctx.accumulated_usage.setdefault("llm_call_refs", []).append({ + **{key: error_event[key] for key in ("ts", "llm_call_id", "model")}, + "failure_code": classification.kind, "reset_at": classification.reset_at, + }) ctx.accumulated_usage.update(_last_llm_error=_short_error_text(display_error), _last_llm_error_kind=classification.kind, _last_llm_retry_same_request=will_retry) if classification.retry_after_sec is not None: - ctx.accumulated_usage["_last_llm_retry_after_sec"] = classification.retry_after_sec - ctx.accumulated_usage["_last_llm_reset_at"] = classification.reset_at + ctx.accumulated_usage.update(_last_llm_retry_after_sec=classification.retry_after_sec, + _last_llm_reset_at=classification.reset_at) else: ctx.accumulated_usage.pop("_last_llm_retry_after_sec", None) ctx.accumulated_usage.pop("_last_llm_reset_at", None) diff --git a/ouroboros/review.py b/ouroboros/review.py index e3e22b9ce..b9cf8608a 100644 --- a/ouroboros/review.py +++ b/ouroboros/review.py @@ -750,10 +750,11 @@ def _validate_manifest_candidate( *, manifest_path: str, include_staged: bool, + inventory: SizeRatchetInventory | None = None, ) -> list[str]: """Shared exactness + merge-aware transition core for live and in-memory candidates.""" current = parse_size_ratchet_manifest(current_text) - inventory = collect_size_ratchet_inventory(root) + inventory = inventory if inventory is not None else collect_size_ratchet_inventory(root) errors = _manifest_inventory_errors(current, inventory) previous_text = resolve_committed_manifest_text(root, manifest_path=manifest_path) @@ -794,6 +795,7 @@ def validate_size_ratchet( repo_dir: pathlib.Path, *, manifest_path: str = SIZE_RATCHET_MANIFEST_PATH, + inventory: SizeRatchetInventory | None = None, ) -> list[str]: """Validate live and staged candidates against the merge-aware committed authority. @@ -809,7 +811,8 @@ def validate_size_ratchet( root = pathlib.Path(repo_dir).resolve() current_path = root.joinpath(*pathlib.PurePosixPath(manifest_path).parts) current_text = current_path.read_text(encoding="utf-8") - return _validate_manifest_candidate(root, current_text, manifest_path=manifest_path, include_staged=True) + return _validate_manifest_candidate(root, current_text, manifest_path=manifest_path, + include_staged=True, inventory=inventory) def validate_size_ratchet_candidate( @@ -957,9 +960,53 @@ def _metrics_from_inventory(inventory: SizeRatchetInventory) -> Dict[str, Any]: } -def compute_repo_complexity_metrics(repo_dir: pathlib.Path) -> Dict[str, Any]: +def compute_repo_complexity_metrics( + repo_dir: pathlib.Path, *, inventory: SizeRatchetInventory | None = None, +) -> Dict[str, Any]: """Compute health metrics from the same production inventory as the hard gate.""" - return _metrics_from_inventory(collect_size_ratchet_inventory(repo_dir)) + return _metrics_from_inventory(inventory if inventory is not None else collect_size_ratchet_inventory(repo_dir)) + + +def size_headroom_lines( + inventory: SizeRatchetInventory, *, paths: Iterable[str] | None = None, limit: int = 5, +) -> list[str]: + """Informational capacity from the same inventory as validation, never a gate. + + Show touched paths when supplied; otherwise show the closest ordinary + boundaries before registered debt, so giant legacy files cannot hide a + nearly-full ordinary module. Bounds are presentation only and disclosed. + """ + selected = set(paths) if paths is not None else None + functions = len(inventory.functions) + lines = [f"Runtime functions: {functions}/{MAX_TOTAL_FUNCTIONS}; " + f"{MAX_TOTAL_FUNCTIONS - functions} remaining."] + modules = sorted( + (m for m in inventory.modules if selected is None or m.path in selected), + key=lambda m: (m.path in GIANT_PATHS or m.path in _CHECKED_IN_MANIFEST.byte_debt, + min((MAX_MODULE_LINES - m.line_count) / MAX_MODULE_LINES, + (MAX_MODULE_BYTES - m.utf8_bytes) / MAX_MODULE_BYTES), m.path), + ) + for module in modules[:limit]: + line_note = ("registered line debt" if module.path in GIANT_PATHS + else f"{MAX_MODULE_LINES - module.line_count} remaining") + byte_note = ("registered byte debt" if module.path in _CHECKED_IN_MANIFEST.byte_debt + else f"{MAX_MODULE_BYTES - module.utf8_bytes} remaining") + lines.append(f"{module.path}: {module.line_count}/{MAX_MODULE_LINES} lines ({line_note}); " + f"{module.utf8_bytes}/{MAX_MODULE_BYTES} UTF-8 bytes ({byte_note}).") + if len(modules) > limit: + lines.append(f"{len(modules) - limit} more modules omitted; codebase_health provides the overview.") + functions_by_size = sorted( + (f for f in inventory.functions if selected is None or f.path in selected), + key=lambda f: ((f.path, f.qualname) in FUNCTION_DEBT, -f.line_count, f.path, f.qualname), + ) + for function in functions_by_size[:limit]: + note = ("registered function debt" if (function.path, function.qualname) in FUNCTION_DEBT + else f"{MAX_FUNCTION_LINES - function.line_count} remaining") + lines.append(f"{function.path}:{function.line_start} {function.qualname}: " + f"{function.line_count}/{MAX_FUNCTION_LINES} lines ({note}).") + if len(functions_by_size) > limit: + lines.append(f"{len(functions_by_size) - limit} more functions omitted.") + return lines def compute_complexity_metrics(sections: List[Tuple[str, str]]) -> Dict[str, Any]: diff --git a/ouroboros/server_entrypoint.py b/ouroboros/server_entrypoint.py index 9a6820d45..982e3fc26 100644 --- a/ouroboros/server_entrypoint.py +++ b/ouroboros/server_entrypoint.py @@ -73,7 +73,7 @@ def bound_service_socket(drive_root: pathlib.Path, service: str, host: str, port The existing port selector chooses the port. Uvicorn accepts this socket on Linux, macOS and Windows; no second probe/rebind race or process authority. """ - from ouroboros.server_process import clear_service_binding, record_service_binding + from ouroboros.server_process import clear_service_binding, record_service_binding, record_applied_restart_settings family = socket.AF_INET6 if ":" in host else socket.AF_INET sock = socket.socket(family, socket.SOCK_STREAM) @@ -82,6 +82,11 @@ def bound_service_socket(drive_root: pathlib.Path, service: str, host: str, port sock.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) sock.bind((host, port)) address = sock.getsockname() + if service in {"main", "host_service"}: + record_applied_restart_settings({ + "OUROBOROS_SERVER_HOST" if service == "main" else "OUROBOROS_HOST_SERVICE_PORT": + host if service == "main" else address[1], + }) try: binding = record_service_binding(drive_root, service, address[0], address[1], pid=os.getpid()) except Exception: diff --git a/ouroboros/server_process.py b/ouroboros/server_process.py index 6dd9ab426..782fc9443 100644 --- a/ouroboros/server_process.py +++ b/ouroboros/server_process.py @@ -38,6 +38,26 @@ _supervisor_stop = threading.Event() # re-exec needs to decide whether the runtime-mode ratchet pin rides along. _owner_restart_requested = threading.Event() +# Confirmed inputs of the components this process started. Settings saves may +# replace os.environ, so it is not an applied-state baseline. Never persisted. +_applied_restart_settings: dict = {} +_applied_settings_lock = threading.Lock() + + +def record_applied_restart_settings(values: dict) -> None: + """Publish only known startup inputs, after their component starts.""" + from ouroboros.settings_scales import RESTART_REQUIRED_SETTINGS + + with _applied_settings_lock: + _applied_restart_settings.update({key: value for key, value in values.items() + if key in RESTART_REQUIRED_SETTINGS}) + + +def applied_restart_settings() -> dict: + """Return process facts without deriving them from mutable saved intent.""" + with _applied_settings_lock: + return dict(_applied_restart_settings) + def _request_restart_exit(owner: bool = False) -> None: """Signal server shutdown with restart exit code. diff --git a/ouroboros/size_ratchet_manifest.py b/ouroboros/size_ratchet_manifest.py index 35e1c15c6..24a8b1f26 100644 --- a/ouroboros/size_ratchet_manifest.py +++ b/ouroboros/size_ratchet_manifest.py @@ -227,7 +227,6 @@ BAND_PATHS = { "web/modules/chat_activity.js": "Existing task activity renderer consumes the shared quota/auth wait state; no parallel task card or lifecycle.", "web/modules/harness_accounts.js": None, "web/modules/log_events.js": None, - "web/modules/settings.js": None, "web/modules/skills.js": "One installed-skill page controller owns independently settling primary/optional reads and current-generation menu, identity and badge updates; domain lifecycle, cards, hub truth and shared interactions remain separate owners.", "web/tests/chat_instance_dom.test.js": "Entered the band from 1000 lines with the alias-free subagent cost pin (stage-2 fix wave): that regression reproduces only through the real createChatInstance card path, and this file owns the DOM harness that drives it; split when the next createChatInstance face lands.", "web/tests/harness_login_cards.test.js": "Login-card suite grew past 1000 lines with the name-the-account face cases (agy pickup, issue #232); split when the next face lands.", diff --git a/ouroboros/subagent_bootstrap.py b/ouroboros/subagent_bootstrap.py index 0ce722f1c..9e46abf45 100644 --- a/ouroboros/subagent_bootstrap.py +++ b/ouroboros/subagent_bootstrap.py @@ -89,6 +89,7 @@ def _record_startup_refusal( """Stash the typed unrun refusal for the caller's zero-spend terminal.""" from ouroboros.subagent_runtime import current_subagent_alternatives + from ouroboros.utils import utc_now_iso snapshot = task.get("configured_subagent") if isinstance(task.get("configured_subagent"), dict) else {} alternatives = current_subagent_alternatives( @@ -102,6 +103,7 @@ def _record_startup_refusal( availability = dict(task.get("subagent_availability") or {}) if isinstance( task.get("subagent_availability"), dict) else {} availability.update({ + "observed_at": utc_now_iso(), "status": "unavailable", "reason": str(reason or "configured_session_unavailable"), "reset_at": str(reset_at or ""), diff --git a/ouroboros/subagent_history.py b/ouroboros/subagent_history.py new file mode 100644 index 000000000..df87e44cc --- /dev/null +++ b/ouroboros/subagent_history.py @@ -0,0 +1,190 @@ +"""Dated execution disclosure in the existing subagent receipt, never routing policy.""" + +from __future__ import annotations + +import json +import logging +import pathlib +from typing import Any, Mapping + +from ouroboros.utils import utc_now_iso, write_text_atomic + +log = logging.getLogger(__name__) +LAST_DELEGATION_FILENAME = "subagent_last_delegation.json" + + +def _last_delegation_path(drive_root=None): + from ouroboros.config import DATA_DIR + + return pathlib.Path(drive_root or DATA_DIR) / "state" / LAST_DELEGATION_FILENAME + + +def subagent_last_delegation(drive_root=None) -> dict[str, Any]: + """Read old single receipts and their additive per-actor history alike.""" + try: + data = json.loads(_last_delegation_path(drive_root).read_text(encoding="utf-8")) + return data if isinstance(data, dict) else {} + except (OSError, ValueError): + return {} + + +def execution_identity(snapshot: Mapping[str, Any]) -> dict[str, str]: + """Only execution-affecting choices, not prose, order or list fingerprints.""" + route = snapshot.get("route") or {} + return {"kind": str(route.get("kind") or ""), + "target_id": str(route.get("target_id") or ""), + "credential_profile_id": str(route.get("credential_profile_id") or ""), + "effort": str(snapshot.get("effort") or ""), + "processing_preference": str(snapshot.get("processing_preference") or "")} + + +def record_last_delegation(*, route: str, requested_model: str, applied_model: str, + run_id: str, selected_subagent_id: str = "", + requested_profile: str = "", applied_profile: str = "", + drive_root=None, occurred_at: str = "", outcome: str = "unknown", + failure_code: str = "", reset_at: str = "", + identity: Mapping[str, Any] | None = None, task_id: str = "", + invocation_id: str = "", attempt_id: str = "", fallback=None) -> None: + """Keep the latest fact per actor plus the old top-level receipt interface. + + Occurrence and observation differ when recovery collects an old run. Replays + neither refresh its date nor displace a newer fact. The existing actor-count + bound keeps this a compact projection; the event log retains full history. + """ + from ouroboros.configured_subagents import MAX_CONFIGURED_SUBAGENTS + from ouroboros.platform_layer import acquire_exclusive_file_lock, release_exclusive_file_lock + + path = _last_delegation_path(drive_root) + try: + path.parent.mkdir(parents=True, exist_ok=True) + lock_path = path.with_suffix(".lock") + lock = acquire_exclusive_file_lock(lock_path, timeout_sec=2.0, stale_sec=30.0) + if lock is None: + return + try: + old = subagent_last_delegation(drive_root) + rows = dict(old.get("latest_by_subagent") or {}) + old_actor = rows.get(selected_subagent_id) or {} + previous = old_actor if old_actor.get("run_id") == run_id else old if old.get("run_id") == run_id else {} + if run_id and previous and (not occurred_at or ( + previous.get("outcome") == outcome and previous.get("failure_code", "") == failure_code)): + return + observed = str(previous.get("observed_at") or utc_now_iso()) + row = {"ts": occurred_at or observed, "observed_at": observed, "occurred_at": occurred_at, + "route": str(route or ""), "requested_model": str(requested_model or ""), + "applied_model": str(applied_model or ""), + "requested_profile": str(requested_profile or ""), + "applied_profile": str(applied_profile or ""), + "selected_subagent_id": str(selected_subagent_id or ""), + "run_id": str(run_id or ""), "outcome": outcome} + if identity: + row["identity"] = dict(identity) + if fallback: + row["fallback"] = dict(fallback) + for key, value in (("failure_code", failure_code), ("reset_at", reset_at), + ("task_id", task_id), ("invocation_id", invocation_id), ("attempt_id", attempt_id)): + if value: + row[key] = str(value) + if selected_subagent_id and (not old_actor or (occurred_at and row["ts"] >= str(old_actor.get("ts") or ""))): + rows[selected_subagent_id] = row + rows = dict(sorted(rows.items(), key=lambda item: str(item[1].get("ts") or ""), + reverse=True)[:MAX_CONFIGURED_SUBAGENTS]) + latest = row if not old or (occurred_at and row["ts"] >= str(old.get("ts") or "")) else old + write_text_atomic(path, json.dumps({**latest, "latest_by_subagent": rows}, ensure_ascii=False, indent=1)) + finally: + release_exclusive_file_lock(lock_path, lock) + except Exception: + log.debug("subagent history projection write failed", exc_info=True) + + +def record_task_execution(task: Mapping[str, Any], usage: Mapping[str, Any], *, drive_root) -> None: + """Project configured attempt/start facts, never task correctness.""" + snapshot = task.get("configured_subagent") or {} + identity = execution_identity(snapshot) + availability = task.get("subagent_availability") or {} + if identity["kind"] == "agent_session" and availability.get("status") == "unavailable" and availability.get("reason"): + route, _, model = identity["target_id"].partition("=") + record_last_delegation(route=route, requested_model=model, applied_model="", + run_id="task:" + str(task.get("id") or ""), task_id=str(task.get("id") or ""), + selected_subagent_id=str(snapshot.get("selected_subagent_id") or ""), + requested_profile=identity["credential_profile_id"], identity=identity, drive_root=drive_root, + occurred_at=str(availability.get("observed_at") or ""), outcome="not_started", + failure_code=str(availability["reason"]), reset_at=str(availability.get("reset_at") or "")) + return + if identity["kind"] != "api_model": + return + target = identity["target_id"] + model = target[:-8].strip() if target.endswith(" (local)") else target + calls = usage.get("llm_call_refs") or [] + call = next((row for row in reversed(calls) if isinstance(row, dict) + and row.get("model") == model + and (row.get("failure_code") or row.get("usable_solve_response"))), {}) + if call: + failure = str(call.get("failure_code") or "") + outcome = "unknown" if failure == "provider_outcome_unknown" else "failed" if failure else "succeeded" + when = str(call.get("ts") or "") + elif availability.get("status") not in (None, "", "ready"): + failure, outcome = str(availability.get("reason") or availability["status"]), "not_started" + when = str(availability.get("observed_at") or "") + else: + return # No observation is not success or failure. + other = next((row for row in reversed(calls) if isinstance(row, dict) + and row.get("usable_solve_response") and row.get("model") != model), {}) if failure else {} + fallback = {key: other[key] for key in ("model", "llm_call_id", "ts") if key in other} + record_last_delegation( + route="api_model", requested_model=target, + applied_model=str(call.get("reported_model") or call.get("model") or "") if not failure else "", + run_id=str(call.get("llm_call_id") or task.get("id") or ""), + selected_subagent_id=str(snapshot.get("selected_subagent_id") or ""), + drive_root=drive_root, occurred_at=when, outcome=outcome, + failure_code=failure, reset_at=str(call.get("reset_at") or availability.get("reset_at") or ""), + identity=identity, task_id=str(task.get("id") or ""), + attempt_id=str(call.get("llm_call_id") or ""), fallback=fallback) + + +def record_session_execution(drive_root, custody, detail: Mapping[str, Any], observed: Mapping[str, Any]) -> None: + """Common foreground/recovery settlement projection; reviewers keep their own history.""" + if custody.review_owned: + return + from ouroboros.delegate_custody import invocation_record, summary_of + + invocation = invocation_record(drive_root, custody.invocation_id) or {} + request = invocation.get("request") or {} + summary = summary_of(detail) + failure = summary.get("failure") if isinstance(summary.get("failure"), dict) else {} + identity = {"kind": "agent_session", + "target_id": custody.route_id + ("=" + custody.model if custody.model else ""), + "credential_profile_id": custody.profile_id, + "effort": str(request.get("effort") or ""), + "processing_preference": str((invocation.get("processing") or {}).get("requested") or "")} + record_last_delegation( + route=custody.route_id, requested_model=custody.model, + applied_model=str(observed.get("model") or ""), run_id=custody.run_id, + selected_subagent_id=custody.selected_subagent_id, + requested_profile=custody.profile_id, applied_profile=str(observed.get("profile_id") or ""), + drive_root=drive_root, occurred_at=str(summary.get("finishedAt") or ""), + outcome=str(summary.get("state") or "unknown"), + failure_code=str(failure.get("code") or ""), reset_at=str(failure.get("resetsAt") or ""), identity=identity, + task_id=custody.task_id, invocation_id=custody.invocation_id, + attempt_id=str(observed.get("attempt_id") or "")) + + +def record_session_start_failure(drive_root, event: Mapping[str, Any]) -> None: + from ouroboros.delegate_custody import invocation_record + + invocation = invocation_record(drive_root, str(event.get("invocation_id") or "")) or {} + actor = str(invocation.get("selected_subagent_id") or "") + if not actor: + return + request = invocation.get("request") or {} + route, model = str(invocation.get("route") or ""), str(request.get("model") or "") + pin = str(request.get("credentialProfileId") or "") + record_last_delegation( + route=route, requested_model=model, applied_model="", run_id=str(event.get("invocation_id") or ""), + selected_subagent_id=actor, requested_profile=pin, drive_root=drive_root, + task_id=str(invocation.get("task_id") or ""), invocation_id=str(event.get("invocation_id") or ""), + occurred_at=str(event.get("ts") or ""), outcome="not_started" if event.get("definite") else "unknown", + failure_code=str(event.get("reason") or ""), identity={ + "kind": "agent_session", "target_id": route + ("=" + model if model else ""), + "credential_profile_id": pin, "effort": str(request.get("effort") or ""), + "processing_preference": str((invocation.get("processing") or {}).get("requested") or "")}) diff --git a/ouroboros/subagents.py b/ouroboros/subagents.py index 0637ba575..242c3a804 100644 --- a/ouroboros/subagents.py +++ b/ouroboros/subagents.py @@ -274,68 +274,9 @@ def get_subagent_harness() -> DelegationRoute | None: # only — nothing routes off it, and absence is shown as absence. # --------------------------------------------------------------------------- -LAST_DELEGATION_FILENAME = "subagent_last_delegation.json" - - -def _last_delegation_path(): - import pathlib - - from ouroboros.config import DATA_DIR - - return pathlib.Path(DATA_DIR) / "state" / LAST_DELEGATION_FILENAME - - -def record_last_delegation(*, route: str, requested_model: str, - applied_model: str, run_id: str, - selected_subagent_id: str = "", - requested_profile: str = "", - applied_profile: str = "") -> None: - """Record the last delegated run's route + requested/applied model + account. - - Best-effort and atomic, in the CANONICAL data plane beside the saved - settings (the reviewer-slot projection's own rule): this is UI state, not - per-task forensics — those live in the custody event log and the ledger. - ``applied_model`` and ``applied_profile`` come from the same final attempt - in the engine's telemetry, '' when that attempt disclosed no such fact. - Neither the requested model nor a prior attempt supplies missing evidence; - ``requested_profile`` is the pin the request carried ('' = rotation) — the - two stay separate so a requested-vs-ran mismatch is disclosable, never - rewritten. - """ - import json - - from ouroboros.utils import utc_now_iso, write_text_atomic - - try: - path = _last_delegation_path() - path.parent.mkdir(parents=True, exist_ok=True) - # Idempotent per run: a re-read of an ALREADY-terminal run must not - # re-stamp `ts`, or the "N ago" line would call an old run fresh. - if subagent_last_delegation().get("run_id") == str(run_id or ""): - return - write_text_atomic(path, json.dumps({ - "ts": utc_now_iso(), - "route": str(route or ""), - "requested_model": str(requested_model or ""), - "applied_model": str(applied_model or ""), - "requested_profile": str(requested_profile or ""), - "applied_profile": str(applied_profile or ""), - "selected_subagent_id": str(selected_subagent_id or ""), - "run_id": str(run_id or ""), - }, ensure_ascii=False, indent=1)) - except Exception: - log.debug("subagent last-delegation projection write failed", exc_info=True) - - -def subagent_last_delegation() -> Dict[str, Any]: - """Read the projection ({} on any read problem — disclosure only).""" - import json - - try: - data = json.loads(_last_delegation_path().read_text(encoding="utf-8")) - return data if isinstance(data, dict) else {} - except (OSError, ValueError): - return {} +from ouroboros.subagent_history import ( # noqa: F401 + LAST_DELEGATION_FILENAME, record_last_delegation, subagent_last_delegation, +) @dataclass(frozen=True) diff --git a/ouroboros/tools/claude_advisory_review.py b/ouroboros/tools/claude_advisory_review.py index e79dfc460..052038e12 100644 --- a/ouroboros/tools/claude_advisory_review.py +++ b/ouroboros/tools/claude_advisory_review.py @@ -966,7 +966,10 @@ def _advisory_pre_sdk_gate( state = load_state(drive_root) # Readiness gate first: reject clean worktree before fresh-run shortcut. - readiness_warnings = check_worktree_readiness(repo_dir, paths=paths) + readiness_information: List[str] = [] + readiness_warnings = check_worktree_readiness(repo_dir, paths=paths, information=readiness_information) + if readiness_information: + ctx.emit_progress_fn("Size headroom (information):\n" + "\n".join(readiness_information)) if readiness_warnings and any("no uncommitted changes" in w.lower() for w in readiness_warnings): ctx.emit_progress_fn(f"⚠️ Advisory readiness gate: {'; '.join(readiness_warnings)}") return readiness_warnings, "", _json_response({ diff --git a/ouroboros/tools/delegate.py b/ouroboros/tools/delegate.py index 2500ff144..c388842be 100644 --- a/ouroboros/tools/delegate.py +++ b/ouroboros/tools/delegate.py @@ -972,7 +972,6 @@ def _delegate_wait(ctx: ToolContext, run_id: str, wait_sec: Optional[int] = None if breach: return _halt_breached_run(ctx, gateway, entry, breach) if state in _TERMINAL_STATES: - was_settled = bool(entry.settled) settlement = custody.settle_run(custody.custody_root(ctx), gateway, entry, detail) payload = _delivered_terminal_payload(ctx, rid, detail, authority, entry, gateway) payload["settlement"] = settlement @@ -984,23 +983,6 @@ def _delegate_wait(ctx: ToolContext, run_id: str, wait_sec: Optional[int] = None {"gateway": gateway} if entry.resource_ref.get("workspace_kind") == "directory" else {})) if capture is not None: payload["workspace_capture"] = capture - # The «last delegated run» settings receipt (Subagents section): - # requested vs applied model, written ONLY when THIS call performed - # a SUCCESSFUL settlement — a later wait re-reading an already-settled - # run must not re-date it (or replace a newer run as "last"), and a - # settlement whose durable obligations failed must not mint a receipt - # it would re-mint on every retry. The delegated REVIEW sessions never - # pass here — they have their own receipt store - # (reviewer_slot_last_execution.json). - if not was_settled and bool(settlement.get("settled")): - from ouroboros.subagents import record_last_delegation - record_last_delegation( - route=entry.route_id, requested_model=entry.model, - applied_model=str(payload.get("model") or ""), run_id=rid, - selected_subagent_id=entry.selected_subagent_id, - # Applied = the same final attempt as the model; requested replays off STARTED. - requested_profile=entry.profile_id, - applied_profile=str((payload.get("observed_attempt") or {}).get("profile_id") or "")) # D7 made load-bearing: settlement is where "paid for and never read" # becomes permanent, so the parent is told in WORDS here — not left to # infer it from `output_delivery.consumed`. Re-settling an already diff --git a/ouroboros/tools/health.py b/ouroboros/tools/health.py index d59a7eebb..eb5c4d53e 100644 --- a/ouroboros/tools/health.py +++ b/ouroboros/tools/health.py @@ -13,10 +13,11 @@ log = logging.getLogger(__name__) def _codebase_health(ctx: ToolContext) -> str: """Compute and format codebase health report.""" try: - from ouroboros.review import compute_repo_complexity_metrics + from ouroboros.review import collect_size_ratchet_inventory, compute_repo_complexity_metrics, size_headroom_lines repo_dir = pathlib.Path(ctx.repo_dir) - metrics = compute_repo_complexity_metrics(repo_dir) + inventory = collect_size_ratchet_inventory(repo_dir) + metrics = compute_repo_complexity_metrics(repo_dir, inventory=inventory) stats = { "files": metrics["total_files"], "chars": metrics["total_bytes"], @@ -42,6 +43,8 @@ def _codebase_health(ctx: ToolContext) -> str: lines.append(f"**Functions:** {metrics['total_functions']}") lines.append(f"**Avg function length:** {metrics['avg_function_length']} lines") lines.append(f"**Max function length:** {metrics['max_function_length']} lines") + lines.append("\n### Size Headroom (information; official CI enforces the limits)") + lines.extend(size_headroom_lines(inventory)) from ouroboros.review import ( MAX_FUNCTION_LINES, @@ -131,7 +134,7 @@ def _codebase_health(ctx: ToolContext) -> str: try: from ouroboros.review import validate_size_ratchet - ratchet_findings = validate_size_ratchet(repo_dir) + ratchet_findings = validate_size_ratchet(repo_dir, inventory=inventory) except Exception as ratchet_exc: lines.append(f"\n### Size-Ratchet Findings\n ⚠️ validator unavailable: {ratchet_exc}") else: diff --git a/ouroboros/tools/review_helpers.py b/ouroboros/tools/review_helpers.py index dd329b8c1..bbbb748ea 100644 --- a/ouroboros/tools/review_helpers.py +++ b/ouroboros/tools/review_helpers.py @@ -695,6 +695,7 @@ def get_advisory_runtime_diagnostics(model: str, prompt_chars: int, def check_worktree_readiness( repo_dir: "Path", paths: "list[str] | None" = None, + *, information: "list[str] | None" = None, ) -> "list[str]": """Run cheap deterministic pre-advisory checks; never crash.""" repo_dir = Path(repo_dir) @@ -772,9 +773,13 @@ def check_worktree_readiness( # line rejects the same finding. Cheap since the history replay retired # (one live inventory plus a couple of git object reads). try: - from ouroboros.review import validate_size_ratchet # local import: ouroboros.review imports this module + from ouroboros.review import collect_size_ratchet_inventory, size_headroom_lines, validate_size_ratchet - for finding in validate_size_ratchet(repo_dir): + inventory = collect_size_ratchet_inventory(repo_dir) if information is not None else None + if information is not None and inventory is not None: + touched = parse_changed_paths_from_porcelain(status_result.stdout or "") if status_result is not None else [] + information.extend(size_headroom_lines(inventory, paths=touched)) + for finding in validate_size_ratchet(repo_dir, inventory=inventory): warnings.append(f"official CI will enforce: {finding}") except Exception as exc: # A broken validator must not silently disable the only local surface — diff --git a/server.py b/server.py index 0ea1cc737..6026dacd1 100644 --- a/server.py +++ b/server.py @@ -685,6 +685,9 @@ def _run_supervisor(settings: dict) -> None: restored_pending = restore_pending_from_snapshot(terminalized=interrupted_running) kill_workers(preserve_pending=True) spawn_workers(max_workers) + from ouroboros.server_process import record_applied_restart_settings + record_applied_restart_settings({"OUROBOROS_MAX_WORKERS": max_workers, + "OUROBOROS_SKILLS_REPO_PATH": settings.get("OUROBOROS_SKILLS_REPO_PATH", "")}) persist_queue_snapshot(reason="startup") try: from ouroboros.delegate_recovery import pre_adopt_planned_handoffs @@ -877,17 +880,13 @@ def _run_supervisor(settings: dict) -> None: except Exception as exc: if _supervisor_stop.is_set() or _restart_requested.is_set(): - # The shutdown/restart tore the bus down under this tick (a - # Manager proxy raising BrokenPipe/EOF): not a crash, no alarm. + # A shutdown-torn Manager proxy is not a supervisor crash. log.info("Supervisor loop exiting on shutdown: %s", exc) break crash_count += 1 log.error("Supervisor loop crash #%d: %s", crash_count, exc, exc_info=True) if crash_count >= 3: - # Visible death: previously the loop returned with - # _supervisor_ready still set and no _supervisor_error, so - # tasks silently stopped being assigned with a healthy-looking - # /api/state. Record the failure and tell the owner. + # Clear readiness and notify: a dead loop must not look healthy. _supervisor_error = f"Supervisor loop died after 3 consecutive crashes: {exc}" _supervisor_ready.clear() log.critical("Supervisor exceeded max retries: %s", _supervisor_error) diff --git a/tests/test_delegated_run_accounting.py b/tests/test_delegated_run_accounting.py index c837e94fa..f5b46071c 100644 --- a/tests/test_delegated_run_accounting.py +++ b/tests/test_delegated_run_accounting.py @@ -194,7 +194,7 @@ def test_d29_absent_authroute_records_empty_never_invented(tmp_path, monkeypatch def test_final_attempt_identity_survives_settlement_and_parent_delivery(tmp_path, monkeypatch, observed): from ouroboros.subagents import subagent_last_delegation - monkeypatch.setattr("ouroboros.config.DATA_DIR", tmp_path / "canonical-data") + monkeypatch.setattr("ouroboros.config.DATA_DIR", tmp_path) payload, ledger, event = _settled_run(tmp_path, monkeypatch, { "state": "succeeded", "spendUsd": 0.0, "model": "request-echo", "authRoute": {"profileId": "old-profile"}, "effectiveAccess": "readonly", @@ -263,7 +263,7 @@ def _settled_run(tmp_path, monkeypatch, summary, observed=None): monkeypatch.setattr(gw, "ClaudexorGateway", lambda *a, **k: _Stub()) delegate._CUSTODY.clear() delegate._CUSTODY["run-1"] = delegate._RunCustody( - task_id="t-a", route_id="r", model="m", project_id="p", project_owned=True) + run_id="run-1", task_id="t-a", route_id="r", model="m", project_id="p", project_owned=True) ctx = ToolContext(repo_dir=tmp_path, drive_root=tmp_path) ctx.task_id = "t-a" ctx.task_metadata = {"root_task_id": "t-a"} @@ -298,7 +298,7 @@ def _waited_run(tmp_path, monkeypatch, summary, requested_model="m", observed=No monkeypatch.setattr(gw, "ClaudexorGateway", lambda *a, **k: _Stub()) delegate._CUSTODY.clear() delegate._CUSTODY["run-1"] = delegate._RunCustody( - task_id="t-a", route_id="r", model=requested_model, + run_id="run-1", task_id="t-a", route_id="r", model=requested_model, project_id="p", project_owned=False) ctx = ToolContext(repo_dir=tmp_path, drive_root=tmp_path) ctx.task_id = "t-a" @@ -352,7 +352,7 @@ def test_the_last_delegation_projection_is_written_at_the_settle_seam(tmp_path, # Isolated data plane: the projection is keyed off config.DATA_DIR, which # xdist workers would otherwise share (and the sibling test writes it too). - monkeypatch.setattr("ouroboros.config.DATA_DIR", tmp_path / "proj-data") + monkeypatch.setattr("ouroboros.config.DATA_DIR", tmp_path / "proj") _waited_run(tmp_path / "proj", monkeypatch, {"state": "succeeded", "spendUsd": 0.0, "model": "claude-opus-5"}, requested_model="sonnet") @@ -386,29 +386,24 @@ def test_no_receipt_on_a_failed_settlement_and_one_after_the_successful_retry(tm import ouroboros.delegate_custody as custody_mod from ouroboros.subagents import subagent_last_delegation - monkeypatch.setattr("ouroboros.config.DATA_DIR", tmp_path / "receipt-data") + monkeypatch.setattr("ouroboros.config.DATA_DIR", tmp_path / "w1") - real_settle = custody_mod.settle_run + real_emit = custody_mod.emit outcomes = iter([False, True]) - def _flaky_settle(drive_root, gateway, custody, detail): - ok = next(outcomes) - result = real_settle(drive_root, gateway, custody, detail) - if not ok: - custody.settled = False - result = dict(result) - result["settled"] = False - return result + def flaky_emit(drive_root, kind, payload): + if kind == custody_mod.SETTLED and not next(outcomes): + return False + return real_emit(drive_root, kind, payload) - monkeypatch.setattr(custody_mod, "settle_run", _flaky_settle) - monkeypatch.setattr("ouroboros.tools.delegate.custody.settle_run", _flaky_settle, raising=False) + monkeypatch.setattr(custody_mod, "emit", flaky_emit) _waited_run(tmp_path / "w1", monkeypatch, {"state": "succeeded", "spendUsd": 0.0, "model": "claude-opus-5"}, requested_model="sonnet") assert subagent_last_delegation() == {}, "receipt minted on a FAILED settlement" - _waited_run(tmp_path / "w2", monkeypatch, + _waited_run(tmp_path / "w1", monkeypatch, {"state": "succeeded", "spendUsd": 0.0, "model": "claude-opus-5"}, requested_model="sonnet") record = subagent_last_delegation() diff --git a/tests/test_delegation_account_pin.py b/tests/test_delegation_account_pin.py index 3f3cb502a..572ac313f 100644 --- a/tests/test_delegation_account_pin.py +++ b/tests/test_delegation_account_pin.py @@ -114,7 +114,7 @@ def _waited_run(tmp_path, monkeypatch, summary, requested_model="m", monkeypatch.setattr(gw, "ClaudexorGateway", lambda *a, **k: _Stub()) delegate._CUSTODY.clear() delegate._CUSTODY["run-1"] = delegate._RunCustody( - task_id="t-a", route_id="r", model=requested_model, + run_id="run-1", task_id="t-a", route_id="r", model=requested_model, profile_id=requested_profile, selected_subagent_id=selected_subagent_id, project_id="p", project_owned=False) ctx = ToolContext(repo_dir=tmp_path, drive_root=tmp_path) @@ -134,7 +134,7 @@ def test_the_receipt_carries_the_requested_and_applied_account(tmp_path, monkeyp writes '', never the request dressed up as the applied account.""" from ouroboros.subagents import subagent_last_delegation - monkeypatch.setattr("ouroboros.config.DATA_DIR", tmp_path / "acct-data") + monkeypatch.setattr("ouroboros.config.DATA_DIR", tmp_path / "acct") _waited_run(tmp_path / "acct", monkeypatch, {"state": "succeeded", "spendUsd": 0.0, "model": "m", "authRoute": {"profileId": "previous-profile"}}, @@ -155,7 +155,7 @@ def test_the_receipt_carries_the_requested_and_applied_account(tmp_path, monkeyp assert subagent_last_delegation() == record # A summary echo cannot replace the missing final-attempt receipt. - monkeypatch.setattr("ouroboros.config.DATA_DIR", tmp_path / "acct-data-2") + monkeypatch.setattr("ouroboros.config.DATA_DIR", tmp_path / "acct2") _waited_run(tmp_path / "acct2", monkeypatch, {"state": "succeeded", "spendUsd": 0.0, "model": "m", "authRoute": {"profileId": "previous-profile"}}, diff --git a/tests/test_gateway_parity.py b/tests/test_gateway_parity.py index ba4deb184..74380062f 100644 --- a/tests/test_gateway_parity.py +++ b/tests/test_gateway_parity.py @@ -173,7 +173,7 @@ def test_gateway_contract_endpoint_index_matches_router_and_types(tmp_path): version = (pathlib.Path(__file__).resolve().parent.parent / "VERSION").read_text(encoding="utf-8").strip() assert f"GATEWAY_CONTRACT_VERSION = '{version}'" in text settings_meta_fields = { - "custom_secret_keys", "setup_contract", "available_subagents", "policy_state", + "custom_secret_keys", "setup_contract", "available_subagents", "policy_state", "restart_state", } assert settings_meta_fields <= set(SettingsMeta.__annotations__) assert _js_typedef_fields(text, "SettingsMeta") == settings_meta_fields diff --git a/tests/test_owner_settings_write_seam.py b/tests/test_owner_settings_write_seam.py index cb961598e..285986f3f 100644 --- a/tests/test_owner_settings_write_seam.py +++ b/tests/test_owner_settings_write_seam.py @@ -325,7 +325,7 @@ def test_a_pre_commit_failure_is_still_reported_as_unsaved(monkeypatch, isolated from ouroboros.gateway import settings as settings_mod app = _settings_app(monkeypatch, isolated_settings) - monkeypatch.setattr(settings_mod, "_classify_settings_changes", + monkeypatch.setattr(settings_mod, "_merge_settings_payload", lambda *_a, **_k: (_ for _ in ()).throw(RuntimeError("before the write"))) resp = TestClient(app).post("/api/settings", json={"TOTAL_BUDGET": "25"}) diff --git a/tests/test_settings_honesty.py b/tests/test_settings_honesty.py index 295ba64d2..267e624fc 100644 --- a/tests/test_settings_honesty.py +++ b/tests/test_settings_honesty.py @@ -29,6 +29,11 @@ def isolated_settings(tmp_path, monkeypatch): monkeypatch.setattr(cfg, "DATA_DIR", data_dir, raising=True) monkeypatch.setattr(cfg, "SETTINGS_PATH", settings_path, raising=True) cfg.reset_runtime_mode_baseline_for_tests() + from ouroboros import server_process + monkeypatch.setattr(server_process, "_applied_restart_settings", { + key: cfg.SETTINGS_DEFAULTS[key] for key in ( + "OUROBOROS_MAX_WORKERS", "OUROBOROS_SERVER_HOST", "OUROBOROS_HOST_SERVICE_PORT", + "OUROBOROS_SKILLS_REPO_PATH")}) yield settings_path cfg.reset_runtime_mode_baseline_for_tests() diff --git a/tests/test_settings_restart_application.py b/tests/test_settings_restart_application.py new file mode 100644 index 000000000..e902ae3e5 --- /dev/null +++ b/tests/test_settings_restart_application.py @@ -0,0 +1,173 @@ +"""Saved settings are compared with actual component startup/health facts.""" + +import asyncio +import json +import subprocess +import sys +import threading +from types import SimpleNamespace + +import pytest +from starlette.requests import Request + +from ouroboros import config, local_model, server_process +from ouroboros.gateway import models, settings +from ouroboros.server_entrypoint import bound_service_socket + + +@pytest.fixture +def applied(monkeypatch): + monkeypatch.setattr(server_process, "_applied_restart_settings", {}) + monkeypatch.setattr(config, "get_runtime_mode", lambda: "advanced") + manager = local_model.LocalModelManager() + monkeypatch.setattr(local_model, "_manager", manager) + saved = dict(config.SETTINGS_DEFAULTS) + server_process.record_applied_restart_settings({key: saved[key] for key in ( + "OUROBOROS_MAX_WORKERS", "OUROBOROS_SERVER_HOST", "OUROBOROS_HOST_SERVICE_PORT", + "OUROBOROS_SKILLS_REPO_PATH", + )}) + return saved, manager + + +def test_server_pending_truth_survives_reload_revert_and_environment_save(applied, monkeypatch): + saved, _manager = applied + for key, value in { + "OUROBOROS_RUNTIME_MODE": "pro", "OUROBOROS_MAX_WORKERS": 19, + "OUROBOROS_SERVER_HOST": "0.0.0.0", "OUROBOROS_HOST_SERVICE_PORT": 9123, + "OUROBOROS_SKILLS_REPO_PATH": "/example/skills", + }.items(): + changed = {**saved, key: value} + monkeypatch.setenv(key, str(value)) # a Save is not application + state = settings._build_restart_state(changed) + assert state["restart_keys"] == [key] + assert settings._build_restart_state({**changed, "TOTAL_BUDGET": 300}) == state + assert settings._build_restart_state(saved)["restart_required"] is False + + +def test_missing_component_facts_stay_unknown(applied, monkeypatch): + saved, _manager = applied + monkeypatch.setattr(server_process, "_applied_restart_settings", {}) + state = settings._build_restart_state(saved) + assert len(state["unknown_keys"]) == 4 + assert "not reported" in state["summary"] + + +@pytest.mark.serial +def test_socket_owners_publish_only_successfully_bound_inputs(tmp_path, monkeypatch): + monkeypatch.setattr(server_process, "_applied_restart_settings", {}) + with bound_service_socket(tmp_path, "host_service", "127.0.0.1", 0) as sock: + port = sock.getsockname()[1] + assert server_process.applied_restart_settings()["OUROBOROS_HOST_SERVICE_PORT"] == port + with pytest.raises(OSError): + with bound_service_socket(tmp_path, "main", "127.0.0.1", port): + pass + assert "OUROBOROS_SERVER_HOST" not in server_process.applied_restart_settings() + + +def test_every_local_setting_is_compared_only_after_confirmed_health(applied): + saved, manager = applied + saved.update(LOCAL_MODEL_SOURCE="fixture/model", LOCAL_MODEL_FILENAME="fixture.gguf") + manager._launch_settings = local_model.local_model_settings(saved) + manager._status = "loading" + assert manager.settings_application(saved)["pending_keys"] == [] + assert "not applied" in manager.settings_application(saved)["summary"] + manager._status = "ready" + manager._applied_settings = dict(manager._launch_settings) + for key, value in { + "LOCAL_MODEL_SOURCE": "another/model", "LOCAL_MODEL_FILENAME": "other.gguf", + "LOCAL_MODEL_PORT": 9345, "LOCAL_MODEL_N_GPU_LAYERS": 17, + "LOCAL_MODEL_CONTEXT_LENGTH": 32768, "LOCAL_MODEL_CHAT_FORMAT": "other-format", + }.items(): + state = settings._build_restart_state({**saved, key: value}) + assert state["restart_required"] is False + assert state["local_model"]["pending_keys"] == [key] + assert manager.settings_application({**saved, "LOCAL_MODEL_CONTEXT_LENGTH": 0})["pending_keys"] == [] + manager._status = "error" + assert "not applied" in manager.settings_application(saved)["summary"] + assert manager.settings_application(saved)["status"] == "error" + + +def _request(body): + async def receive(): + return {"type": "http.request", "body": json.dumps(body).encode(), "more_body": False} + return Request({"type": "http", "method": "POST", "path": "/api/local-model/start", + "headers": [], "app": SimpleNamespace(state=SimpleNamespace())}, receive) + + +@pytest.fixture +def healthy_model_process(tmp_path, monkeypatch): + """The real manager starts a lightweight /v1/models child, never downloads a model.""" + manager = local_model.LocalModelManager() + monkeypatch.setattr(local_model, "_manager", manager) + monkeypatch.setattr(config, "DATA_DIR", tmp_path / "data") + script = tmp_path / "model_server.py" + script.write_text( + 'import http.server,json,sys\n' + 'class Handler(http.server.BaseHTTPRequestHandler):\n' + ' def do_GET(self):\n' + ' body=json.dumps({"data":[{"id":"fixture","context_window":9999}]}).encode()\n' + ' self.send_response(200);self.send_header("Content-Length",str(len(body)));self.end_headers();self.wfile.write(body)\n' + ' def log_message(self,*args):pass\n' + 'http.server.HTTPServer(("127.0.0.1",int(sys.argv[1])),Handler).serve_forever()\n', encoding="utf-8") + original_popen, original_run = subprocess.Popen, subprocess.run + commands = [] + + def popen(command, **kwargs): + if "ouroboros.local_model_server" in command: + commands.append(list(command)) + command = [sys.executable, str(script), str(command[command.index("--port") + 1])] + return original_popen(command, **kwargs) + + def run(command, **kwargs): + if command[-1] == "import llama_cpp": + return subprocess.CompletedProcess(command, 0, "", "") + return original_run(command, **kwargs) + + settled = threading.Event() + original_health = manager._wait_for_healthy + + def health(): + try: + original_health(timeout=8) + finally: + settled.set() + + monkeypatch.setattr(local_model.subprocess, "Popen", popen) + monkeypatch.setattr(local_model.subprocess, "run", run) + monkeypatch.setattr(manager, "_wait_for_healthy", health) + monkeypatch.setattr(manager, "download_model", lambda source, filename: str(tmp_path / filename)) + try: + yield manager, settled, commands + finally: + manager.stop_server() + + +@pytest.mark.serial +def test_actual_local_start_health_and_restart_clear_saved_mismatch(healthy_model_process, monkeypatch, free_tcp_port): + manager, settled, commands = healthy_model_process + saved = {**config.SETTINGS_DEFAULTS, "LOCAL_MODEL_SOURCE": "fixture/model", + "LOCAL_MODEL_FILENAME": "fixture.gguf", "LOCAL_MODEL_PORT": free_tcp_port, + "LOCAL_MODEL_CONTEXT_LENGTH": 0, "LOCAL_MODEL_N_GPU_LAYERS": 3} + monkeypatch.setattr(models, "load_settings", lambda: saved) + body = {"source": "fixture/model", "filename": "fixture.gguf", "port": free_tcp_port, + "n_ctx": 0, "n_gpu_layers": 3, "chat_format": ""} + response = asyncio.run(models.api_local_model_start(_request(body))) + assert response.status_code == 200 + assert settled.wait(10) + assert manager.is_running, manager.status_dict() + assert commands[-1][commands[-1].index("--n_ctx") + 1] == "16384" + assert manager.settings_application(saved)["pending_keys"] == [] + saved["LOCAL_MODEL_CONTEXT_LENGTH"] = 8192 + assert manager.settings_application(saved)["pending_keys"] == ["LOCAL_MODEL_CONTEXT_LENGTH"] + asyncio.run(models.api_local_model_stop(_request({}))) + assert manager.settings_application(saved)["status"] == "offline" + settled.clear() + response = asyncio.run(models.api_local_model_start(_request({**body, "n_ctx": 8192}))) + assert response.status_code == 200 + assert settled.wait(10) + status = json.loads(asyncio.run(models.api_local_model_status(_request({}))).body) + assert status["status"] == "ready" + assert status["settings_application"]["pending_keys"] == [] + # Provider training metadata does not replace the effective launch window. + assert status["context_length"] == 9999 + assert manager._applied_settings["LOCAL_MODEL_CONTEXT_LENGTH"] == 8192 diff --git a/tests/test_settings_restart_browser.py b/tests/test_settings_restart_browser.py new file mode 100644 index 000000000..5a9ebdcfa --- /dev/null +++ b/tests/test_settings_restart_browser.py @@ -0,0 +1,168 @@ +"""Production Settings against an isolated server and a tiny local-model HTTP child.""" + +import os +import json +import pathlib +import sys +import textwrap + +import pytest + +from tests import test_ui_smoke_playwright as smoke +from tests.ui_chat_viewport_smoke import _CAPTURE_TEST_SOCKET, _SETTLE_RESTORE_FRAMES + +pytest_plugins = ("tests.test_ui_smoke_playwright",) + + +@pytest.fixture +def settings_server(request, tmp_path, monkeypatch): + if os.name == "nt": + pytest.skip("POSIX test launcher; production browser behavior is shared") + bootstrap = tmp_path / "bootstrap.py" + fake_model = tmp_path / "fake_model.py" + fake_model.write_text(textwrap.dedent('''\ + import http.server, json, sys + class Handler(http.server.BaseHTTPRequestHandler): + def do_GET(self): + body = json.dumps({"data": [{"id": "fixture", "context_window": 9999}]}).encode() + self.send_response(200) + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + def log_message(self, *args): pass + http.server.HTTPServer(("127.0.0.1", int(sys.argv[1])), Handler).serve_forever() + '''), encoding="utf-8") + bootstrap.write_text(textwrap.dedent(f'''\ + import pathlib, runpy, subprocess, sys + sys.path.insert(0, {smoke.REPO_ROOT!r}) + from ouroboros.local_model import LocalModelManager + LocalModelManager.download_model = lambda self, source, filename: str(pathlib.Path({str(tmp_path)!r}) / filename) + original_popen, original_run = subprocess.Popen, subprocess.run + class Popen(original_popen): + def __init__(self, command, **kwargs): + if isinstance(command, list) and "ouroboros.local_model_server" in command: + command = [sys.executable, {str(fake_model)!r}, str(command[command.index("--port") + 1])] + super().__init__(command, **kwargs) + def run(command, **kwargs): + if isinstance(command, list) and command[-1] == "import llama_cpp": + return subprocess.CompletedProcess(command, 0, "", "") + return original_run(command, **kwargs) + subprocess.Popen, subprocess.run = Popen, run + sys.argv = sys.argv[1:] + runpy.run_path({str(pathlib.Path(smoke.REPO_ROOT) / 'server.py')!r}, run_name="__main__") + '''), encoding="utf-8") + launcher = tmp_path / "python-fixture" + launcher.write_text(f'#!/bin/sh\nexec "{sys.executable}" "{bootstrap}" "$@"\n', encoding="utf-8") + launcher.chmod(0o700) + monkeypatch.setattr(smoke, "_fixture_python", lambda: str(launcher)) + return request.getfixturevalue("direct_server_with_data") + + +@pytest.mark.serial +@pytest.mark.ui_browser +@pytest.mark.parametrize("engine,width", [("chromium", 1400), ("webkit", 390)]) +def test_pending_survives_reconnect_draft_and_restart_request(settings_server, engine, width, free_tcp_port): + from playwright.sync_api import expect, sync_playwright + + with sync_playwright() as pw: + browser = getattr(pw, engine).launch(headless=True) + page = browser.new_page(viewport={"width": width, "height": 1000}) + errors = [] + page.on('pageerror', lambda error: errors.append(str(error))) + page.add_init_script(f"({_CAPTURE_TEST_SOCKET})()") + page.add_init_script("""const originalSend = WebSocket.prototype.send; + WebSocket.prototype.send = function(data) { + if (JSON.parse(data).cmd === '/restart') { window.restartSent = true; return; } + return originalSend.call(this, data); + };""") + try: + def open_settings(): + page.goto(settings_server["url"], wait_until="domcontentloaded") + if width < 980: + page.locator('#page-chat [data-mobile-nav-toggle]').click() + page.locator('[data-nav-page="settings"]').click() + expect(page.locator('#btn-save-settings')).to_be_enabled(timeout=30_000) + page.evaluate(_SETTLE_RESTORE_FRAMES) + + def fill_value(selector, value): + page.locator(selector).evaluate("(node, value) => {node.value=value; node.dispatchEvent(new Event('input',{bubbles:true}));}", str(value)) + + def save(): + page.locator('#btn-save-settings').click() + page.wait_for_function("!document.querySelector('#btn-save-settings').disabled || document.querySelector('[data-confirm-ok]')") + if page.locator('[data-confirm-ok]').is_visible(): + page.locator('[data-confirm-ok]').click() + expect(page.locator('#btn-save-settings')).to_be_enabled(timeout=30_000) + expect(page.locator('#settings-status')).to_contain_text('saved', timeout=30_000) + + open_settings() + fill_value('#s-workers', 2) + save() + expect(page.locator('#btn-restart-now')).to_be_visible() + open_settings() # a completely new page, no local latch + expect(page.locator('#btn-restart-now')).to_be_visible() + page.locator('#btn-restart-now').click() + page.locator('[data-confirm-ok]').click() + page.wait_for_function('window.restartSent === true') + expect(page.locator('#btn-restart-now')).to_be_visible() + fill_value('#s-workers', 3) # preserve a draft over reconnect metadata + page.evaluate("window.__testSockets[0].close()") + page.wait_for_function("window.__testSockets.some(socket => socket.readyState === WebSocket.OPEN)") + expect(page.locator('#s-workers')).to_have_value('3') + expect(page.locator('#btn-restart-now')).to_be_visible() + fill_value('#s-workers', 1) + save() + expect(page.locator('#btn-restart-now')).to_be_hidden() + + fill_value('#s-local-source', 'fixture/model') + fill_value('#s-local-filename', 'fixture.gguf') + fill_value('#s-local-port', free_tcp_port) + fill_value('#s-local-ctx', 0) + save() + page.locator('[data-settings-tab="advanced"]').click() + page.locator('#btn-local-start').click() + expect(page.locator('#local-model-status')).to_contain_text('Ready', timeout=30_000) + fill_value('#s-local-ctx', 8192) + save() + expect(page.locator('#settings-restart-status')).to_contain_text('Stop, then Start') + expect(page.locator('#btn-restart-now')).to_be_hidden() + evidence = pathlib.Path(os.environ.get('OUROBOROS_UI_EVIDENCE_DIR', str(settings_server['data_dir'].parent))) + evidence.mkdir(parents=True, exist_ok=True) + page.locator('#settings-restart-status').scroll_into_view_if_needed() + page.screenshot(path=str(evidence / f'settings-applied-pending-{engine}.png')) + page.locator('#btn-local-stop').click() + expect(page.locator('#local-model-status')).to_contain_text('Offline', timeout=30_000) + page.locator('#btn-local-start').click() + expect(page.locator('#local-model-status')).to_contain_text('Ready', timeout=30_000) + expect(page.locator('#settings-restart-status')).to_be_hidden(timeout=30_000) + page.screenshot(path=str(evidence / f'settings-applied-ready-{engine}.png')) + # The same actual receipt consumer on the owner-facing Agents card. + actor = {"subagent_id": "fixture-helper", "recommended_use": "Fixture helper", + "route": {"kind": "api_model", "target_id": "openai-compatible::mock-model"}, + "effort": "high", "processing_preference": "standard"} + from ouroboros.subagent_history import record_last_delegation, execution_identity + record_last_delegation(route="api_model", requested_model=actor["route"]["target_id"], + applied_model="", run_id="browser-failure", selected_subagent_id="fixture-helper", + drive_root=settings_server['data_dir'], occurred_at="2026-09-18T12:00:00Z", + outcome="failed", failure_code="quota_exhausted", identity=execution_identity(actor)) + response = page.request.post(settings_server['url'] + '/api/settings', data={ + "OUROBOROS_SUBAGENTS": json.dumps({"enabled": True, "items": [actor]})}) + assert response.ok, response.text() + open_settings() + page.locator('[data-settings-tab="agents"]').click() + meta = page.locator('[data-subagent-meta]').first + expect(meta).to_contain_text('failed (quota_exhausted)', timeout=30_000) + expect(meta).to_contain_text('2026-09-18T12:00:00Z') + assert meta.evaluate("node => getComputedStyle(node).whiteSpace") == 'normal' + assert meta.evaluate("node => node.scrollWidth <= node.clientWidth") + meta.scroll_into_view_if_needed() + page.screenshot(path=str(evidence / f'subagent-history-{engine}.png')) + assert errors == [] + except Exception: + evidence = pathlib.Path(os.environ.get('OUROBOROS_UI_EVIDENCE_DIR', str(settings_server['data_dir'].parent))) + evidence.mkdir(parents=True, exist_ok=True) + page.screenshot(path=str(evidence / f'settings-failure-{engine}.png')) + print(page.locator('body').inner_text()[-4000:]) + raise + finally: + browser.close() diff --git a/tests/test_size_headroom.py b/tests/test_size_headroom.py new file mode 100644 index 000000000..b44ebbc54 --- /dev/null +++ b/tests/test_size_headroom.py @@ -0,0 +1,98 @@ +"""Capacity is information at the real health/readiness consumers, not a gate.""" + +import subprocess +from types import SimpleNamespace + +from ouroboros import review +from ouroboros.tools import health, review_helpers + + +def test_near_limit_module_is_not_hidden_by_registered_giants(monkeypatch): + giants = tuple(review.GatedModule(f"old{i}.py", 2000 + i, 5000) for i in range(12)) + monkeypatch.setattr(review, "GIANT_PATHS", frozenset(m.path for m in giants)) + inventory = review.SizeRatchetInventory( + modules=(*giants, review.GatedModule("near.py", review.MAX_MODULE_LINES - 1, 199999)), + functions=(), giant_paths=frozenset(), function_debt=frozenset(), + band_paths=frozenset(), byte_debt={}, + ) + + lines = review.size_headroom_lines(inventory) + + assert lines[1].startswith("near.py:") + assert "1 remaining" in lines[1] + assert any("registered line debt" in line for line in lines) + assert any("8 more modules omitted" in line for line in lines) + assert review.size_headroom_lines(inventory, paths=["near.py"])[1:] == [lines[1]] + + +def test_headroom_uses_live_limits_and_retains_zero_and_negative(monkeypatch): + monkeypatch.setattr(review, "MAX_TOTAL_FUNCTIONS", 1) + module = review.GatedModule("sample.py", review.MAX_MODULE_LINES, review.MAX_MODULE_BYTES + 1) + functions = (review.GatedFunction("sample.py", "a", 1, review.MAX_FUNCTION_LINES), + review.GatedFunction("sample.py", "b", 2, review.MAX_FUNCTION_LINES + 1)) + inventory = review.SizeRatchetInventory((module,), functions, frozenset(), frozenset(), frozenset(), {}) + + lines = review.size_headroom_lines(inventory) + + assert "2/1; -1 remaining" in lines[0] + assert "lines (0 remaining)" in lines[1] + assert "UTF-8 bytes (-1 remaining)" in lines[1] + assert "b:" in lines[2] and "-1 remaining" in lines[2] + assert "a:" in lines[3] and "0 remaining" in lines[3] + + +def test_headroom_counts_normalized_utf8_source(tmp_path): + source = "# Snowman: ☃\r\ndef sample():\r\n return 1\r\n" + (tmp_path / "sample.py").write_bytes(source.encode("utf-8")) + inventory = review.collect_size_ratchet_inventory(tmp_path) + + line = review.size_headroom_lines(inventory)[1] + + assert "sample.py: 3/" in line + assert f"{len(source.replace(chr(13), '').encode('utf-8'))}/" in line + + +def _repo(tmp_path): + root = tmp_path / "repo" + root.mkdir() + (root / "ouroboros").mkdir() + (root / "ouroboros/size_ratchet_manifest.py").write_text( + 'BASELINE_SOURCE_SHA = "' + "0" * 40 + '"\n' + 'GIANT_PATHS = ()\nFUNCTION_DEBT = ()\nBAND_BASELINE_PATHS = ()\n' + 'BAND_PATHS = {}\nBYTE_BASELINE_DEBT = {}\nBYTE_DEBT = {}\n', encoding="utf-8") + (root / "sample.py").write_text("def sample():\n return 1\n", encoding="utf-8") + subprocess.run(["git", "init", "-q"], cwd=root, check=True) + subprocess.run(["git", "add", "."], cwd=root, check=True) + subprocess.run(["git", "-c", "user.name=Test", "-c", "user.email=test@example.invalid", + "-c", "commit.gpgsign=false", "commit", "-qm", "fixture"], cwd=root, check=True) + return root + + +def test_real_readiness_keeps_information_out_of_warning_and_reuses_inventory(tmp_path, monkeypatch): + root = _repo(tmp_path) + information = [] + assert review_helpers.check_worktree_readiness(root, information=information) == [ + "No uncommitted changes detected — nothing to review."] + assert information == [] + (root / "sample.py").write_text("# useful capacity fact\n" * (review.MAX_MODULE_LINES - 1), encoding="utf-8") + real_collect = review.collect_size_ratchet_inventory + calls = [] + + def collect(*args, **kwargs): + calls.append(args[0]) + return real_collect(*args, **kwargs) + + monkeypatch.setattr(review, "collect_size_ratchet_inventory", collect) + assert review_helpers.check_worktree_readiness(root, information=information) == [] + assert len(calls) == 1 + assert any("sample.py:" in line for line in information) + assert any("lines (1 remaining)" in line for line in information) + assert not any("size_ratchet_manifest.py:" in line for line in information) + calls.clear() + + result = health._codebase_health(SimpleNamespace(repo_dir=root)) + + assert len(calls) == 1 + assert "Size Headroom (information; official CI enforces the limits)" in result + assert "lines (1 remaining)" in result + assert "manifest is exact" in result diff --git a/tests/test_subagent_execution_history.py b/tests/test_subagent_execution_history.py new file mode 100644 index 000000000..09f4f82fb --- /dev/null +++ b/tests/test_subagent_execution_history.py @@ -0,0 +1,120 @@ +"""Real terminal producers feed dated disclosure without becoming admission policy.""" + +import json +from types import SimpleNamespace + +from ouroboros import delegate_custody as custody +from ouroboros.context_runtime_facts import _delegation_capability_fact +from ouroboros.agent_task_pipeline import _store_task_result +from ouroboros.loop_llm_call import call_llm_with_retry +from ouroboros.subagent_history import record_task_execution, record_last_delegation, subagent_last_delegation +from ouroboros.subagent_runtime import resolve_configured_actor_dispatch + + +def _task(kind="api_model", target="openai::fixture"): + return {"id": "task-history", "type": "task", "configured_subagent": { + "schema": 1, "config_fingerprint": "irrelevant-list-hash", "selected_subagent_id": "worker", + "route": {"kind": kind, "target_id": target, "credential_profile_id": ""}, + "effort": "high", "processing_preference": "standard"}} + + +def test_api_failure_fallback_and_retry_reach_next_task_context(tmp_path, monkeypatch): + monkeypatch.setattr("ouroboros.config.DATA_DIR", tmp_path) + task = _task() + usage = {} + + class Provider: + broken = True + + def chat(self, **_kwargs): + if self.broken: + raise RuntimeError("insufficient_quota") + return {"content": "Useful reply"}, {"prompt_tokens": 1, "completion_tokens": 1, "cost": 0.0} + + provider = Provider() + args = (provider, [{"role": "user", "content": "work"}], "openai::fixture", None, "high", 0, + tmp_path / "logs", task["id"], 1, None, usage) + assert call_llm_with_retry(*args)[0] is None + failure = usage["llm_call_refs"][-1] + assert failure["failure_code"] == "quota_exhausted" + # Another model's usable response does not erase this route's own incident. + usage["llm_call_refs"].append({"model": "openai::fallback", "usable_solve_response": True}) + _store_task_result(SimpleNamespace(drive_root=tmp_path), task, "Fallback answered", usage, {"tool_calls": []}) + assert (tmp_path / "task_results" / "task-history.json").is_file() + row = _delegation_capability_fact()["subagents_last_executions"][0] + assert row["outcome"] == "failed" and row["occurred_at"] == failure["ts"] + assert row["task_id"] == task["id"] and row["fallback"]["model"] == "openai::fallback" + assert row["applied_model"] == "" + monkeypatch.setattr("ouroboros.provider_models.model_has_credentials", lambda *_: True) + assert resolve_configured_actor_dispatch(task, task_type="task").executor == "native" + provider.broken = False + assert call_llm_with_retry(*args)[0]["content"] == "Useful reply" + usage["_last_llm_call_meta"]["usable_solve_response"] = True + usage["execution_status"] = "failed" # a later code/test failure is not a provider failure + record_task_execution(task, usage, drive_root=tmp_path) + row = _delegation_capability_fact()["subagents_last_executions"][0] + assert row["outcome"] == "succeeded" and "failure_code" not in row + before = subagent_last_delegation(tmp_path) + record_task_execution(task, usage, drive_root=tmp_path) + assert subagent_last_delegation(tmp_path) == before + + +def test_recovery_settlement_and_refused_start_share_history(tmp_path, monkeypatch): + monkeypatch.setattr("ouroboros.config.DATA_DIR", tmp_path) + request = {"model": "fixture", "effort": "high", "credentialProfileId": "account-a"} + custody.record_start_requested(tmp_path, invocation_id="invoke", selected_subagent_id="worker", + route="codex", request=request, task_id="task-history") + custody.emit(tmp_path, custody.START_FAILED, {"invocation_id": "invoke", "definite": False, + "reason": "transport_unavailable"}) + uncertain = subagent_last_delegation(tmp_path) + assert uncertain["outcome"] == "unknown" + custody.emit(tmp_path, custody.START_FAILED, {"invocation_id": "invoke", "definite": True, + "reason": "quota_exhausted"}) + row = subagent_last_delegation(tmp_path) + assert row["outcome"] == "not_started" and row["requested_profile"] == "account-a" + assert row["observed_at"] == uncertain["observed_at"] + entry = custody.RunCustody(run_id="run-new", task_id="task-history", route_id="codex", model="fixture", + selected_subagent_id="worker", profile_id="account-a") + custody.record_started(tmp_path, entry) + detail = {"summary": {"state": "succeeded", "spendUsd": 0, + "finishedAt": "2099-09-18T12:00:00Z"}} + assert custody.settle_run(tmp_path, SimpleNamespace(), entry, detail)["settled"] + row = subagent_last_delegation(tmp_path) + assert row["outcome"] == "succeeded" and row["applied_model"] == "" + assert row["occurred_at"] == "2099-09-18T12:00:00Z" + assert custody.settle_run(tmp_path, SimpleNamespace(), entry, detail)["settled"] + assert subagent_last_delegation(tmp_path) == row + old = custody.RunCustody(run_id="old", task_id="old-task", route_id="codex", model="fixture", + selected_subagent_id="worker") + custody.record_started(tmp_path, old) + detail = {"summary": {"state": "failed", "spendUsd": 0, + "finishedAt": "2001-09-18T12:00:00Z", "failure": {"code": "quota_exhausted"}}} + custody.settle_run(tmp_path, SimpleNamespace(), old, detail) + assert subagent_last_delegation(tmp_path) == row + + +def test_old_corrupt_missing_and_unknown_time_are_not_health(tmp_path): + assert subagent_last_delegation(tmp_path) == {} + path = tmp_path / "state" / "subagent_last_delegation.json" + path.parent.mkdir() + path.write_text('{"run_id":"legacy","ts":"2001-01-01T00:00:00Z"}', encoding="utf-8") + assert subagent_last_delegation(tmp_path)["run_id"] == "legacy" + path.write_text("corrupt", encoding="utf-8") + record_last_delegation(route="codex", requested_model="fixture", applied_model="", run_id="unknown", + selected_subagent_id="worker", drive_root=tmp_path) + row = json.loads(path.read_text(encoding="utf-8")) + assert row["outcome"] == "unknown" and row["occurred_at"] == "" + assert row["observed_at"] and row["latest_by_subagent"]["worker"]["occurred_at"] == "" + + +def test_pre_invocation_session_refusal_is_dated_at_its_existing_producer(tmp_path, monkeypatch): + from ouroboros.subagent_bootstrap import _record_startup_refusal + monkeypatch.setattr("ouroboros.subagent_runtime.current_subagent_alternatives", lambda *_: []) + task = _task("agent_session", "codex=fixture") + ctx = SimpleNamespace() + _record_startup_refusal(ctx, task, reason="route_disabled") + _store_task_result(SimpleNamespace(drive_root=tmp_path), task, "Could not start", {}, {"tool_calls": []}) + row = subagent_last_delegation(tmp_path) + assert row["outcome"] == "not_started" and row["failure_code"] == "route_disabled" + assert row["occurred_at"] == task["subagent_availability"]["observed_at"] + assert row["task_id"] == task["id"] diff --git a/web/modules/api_types.js b/web/modules/api_types.js index 59bb13cf2..5d09c44a3 100644 --- a/web/modules/api_types.js +++ b/web/modules/api_types.js @@ -157,6 +157,7 @@ * @property {Object=} setup_contract * @property {AvailableSubagentsSettingsMeta=} available_subagents * @property {SettingsPolicyState=} policy_state + * @property {Object=} restart_state Component-owned pending restart/application facts. */ /** @@ -1171,6 +1172,17 @@ * @property {string=} applied_profile * @property {string=} run_id * @property {string=} ts + * @property {string=} occurred_at + * @property {string=} observed_at + * @property {string=} outcome + * @property {string=} failure_code + * @property {string=} reset_at + * @property {Object=} identity + * @property {Object=} latest_by_subagent + * @property {string=} task_id + * @property {string=} invocation_id + * @property {string=} attempt_id + * @property {Object=} fallback */ /** diff --git a/web/modules/route_editor_primitives.js b/web/modules/route_editor_primitives.js index cf71ac0a2..0005ba6d3 100644 --- a/web/modules/route_editor_primitives.js +++ b/web/modules/route_editor_primitives.js @@ -538,7 +538,7 @@ export function describeExecutionEvidence(entry) { if ('requested_model' in entry || 'applied_model' in entry) { const parts = []; const route = String(entry.route || ''); - if (route) parts.push(`${route} session`); + if (route) parts.push(route === 'api_model' ? 'API model' : `${route} session`); // Last-actual evidence is APPLIED telemetry only. Older receipts may // retain the requested route while omitting what the harness actually // served; never dress that requested value up as execution truth. @@ -551,6 +551,9 @@ export function describeExecutionEvidence(entry) { const processing = processingExecutionText(entry.processing); if (processing) parts.push(processing); if (when) parts.push(when); + if (entry.outcome) parts.push(`${entry.outcome}${entry.failure_code ? ` (${entry.failure_code})` : ''}`); + if (entry.fallback?.model) parts.push(`fallback replied: ${entry.fallback.model}`); + if ('occurred_at' in entry) parts.push(entry.occurred_at || `observed ${entry.observed_at || entry.ts}; occurrence time unknown`); return parts.join(' · '); } const effective = entry.effective || entry; diff --git a/web/modules/settings.js b/web/modules/settings.js index a75aaaae0..176c268a2 100644 --- a/web/modules/settings.js +++ b/web/modules/settings.js @@ -455,7 +455,8 @@ export function initSettings({ state, setBeforePageLeave, ws } = {}) { const disposeSettingsTabs = bindSettingsTabs(page, { state }); bindSecretInputs(page); bindEffortSegments(page); - const disposeLocalModel = bindLocalModelControls({ state }); + const disposeLocalModel = bindLocalModelControls({ state, + onApplication: (local) => syncRestartState({ ...restartState, local_model: local }) }); // Best-effort About version from /api/health. apiFetch('/api/health') .then((r) => (r.ok ? r.json() : Promise.reject(new Error(`HTTP ${r.status}`)))) @@ -471,6 +472,8 @@ export function initSettings({ state, setBeforePageLeave, ws } = {}) { let settingsDirty = false; let draftRevision = 0; let loadSequence = 0; + let restartReadSequence = 0; + let restartState = { restart_required: false }; let settingsSaving = false; let saveOutcomeUnknown = false; let validationAttempted = false; @@ -505,6 +508,24 @@ export function initSettings({ state, setBeforePageLeave, ws } = {}) { setReviewerProcessingPreference(settings[PROCESSING_PREFERENCE_KEY], modelRoleMap(settings[MODEL_PROCESSING_PREFERENCES_KEY])); } + function syncRestartState(value) { + if (!value || typeof value.restart_required !== 'boolean') return; + restartState = value; + byId('btn-restart-now').hidden = !value.restart_required; + const text = [value.summary, value.local_model?.summary].filter(Boolean).join(' '); + const target = byId('settings-restart-status'); + target.hidden = !text; + setInlineStatus(target, text, value.restart_required || value.local_model?.pending_keys?.length ? 'warn' : 'muted'); + } + + async function refreshRestartState() { + const sequence = ++restartReadSequence; + try { + const data = await apiClient.settings(); + if (sequence === restartReadSequence) syncRestartState(data?._meta?.restart_state); + } catch { /* An unavailable read cannot clear a known pending change. */ } + } + function syncSettingsLoadState() { const saveBtn = byId('btn-save-settings'); if (saveBtn) { @@ -758,6 +779,7 @@ export function initSettings({ state, setBeforePageLeave, ws } = {}) { async function loadSettings() { const sequence = ++loadSequence; + const restartSequence = ++restartReadSequence; const revision = draftRevision; const [data, extData] = await Promise.all([ apiClient.settings(), @@ -766,6 +788,7 @@ export function initSettings({ state, setBeforePageLeave, ws } = {}) { if (!data || typeof data !== 'object' || Array.isArray(data) || data.error) { throw new Error(data?.error || 'The server did not return a settings document.'); } + if (restartSequence === restartReadSequence) syncRestartState(data._meta?.restart_state); const sections = Array.isArray(extData?.live?.settings_sections) ? extData.live.settings_sections : []; @@ -832,6 +855,7 @@ export function initSettings({ state, setBeforePageLeave, ws } = {}) { async function refreshSettingsAfterExtensionChange(reason = 'skills changed') { if (extensionRefreshPending || settingsSaving || saveOutcomeUnknown) return; if (settingsDirty) { + await refreshRestartState(); setStatus(`Settings changed externally (${reason}). Reload after saving or discarding your draft.`, 'warn'); return; } @@ -1165,6 +1189,7 @@ export function initSettings({ state, setBeforePageLeave, ws } = {}) { refreshSettingsAfterExtensionChange(action); }); } + const disposeRestartReconnect = ws?.on?.('open', refreshRestartState); window.addEventListener('ouro:page-shown', (event) => { if (event.detail?.page === 'settings') refreshSettingsAfterExtensionChange('settings page shown'); @@ -1181,6 +1206,8 @@ export function initSettings({ state, setBeforePageLeave, ws } = {}) { disposeSettingsTabs(); window.removeEventListener('beforeunload', beforeUnload); disposeLocalModel(); + disposeRestartReconnect?.(); + restartReadSequence += 1; baselineSettleDisposer?.(); modelRoles.destroy(); document.removeEventListener('settings-model-catalog:updated', onModelCatalog); @@ -1264,10 +1291,6 @@ export function initSettings({ state, setBeforePageLeave, ws } = {}) { await reloadSettingsWithFeedback(); }); - // #285: true from a restart-required save until the restart command is - // actually sent — keeps the Restart now affordance across later saves. - let restartPending = false; - byId('btn-save-settings').addEventListener('click', async () => { if (settingsSaving || saveOutcomeUnknown) return; if (!settingsLoaded) { @@ -1306,9 +1329,6 @@ export function initSettings({ state, setBeforePageLeave, ws } = {}) { settingsSaving = true; setButtonBusy(saveButton, true); setStatus('Saving…', 'muted'); - // A pending restart LATCHES: a later save that needs no restart must - // not hide the button while the process still runs the old config. - if (!restartPending) byId('btn-restart-now')?.setAttribute('hidden', ''); let saved = false; try { const data = await apiClient.saveSettings(body); @@ -1371,6 +1391,8 @@ export function initSettings({ state, setBeforePageLeave, ws } = {}) { } else if (data.restart_required) { statusMsg = 'Settings saved. Some changes require a restart to take effect'; statusType = 'warn'; + } else if (data.restart_state?.local_model?.summary) { + statusMsg = 'Settings saved'; } else if (data.immediate_changed && data.next_task_changed) { statusMsg = 'Settings saved. Some changes took effect immediately; others apply on the next task'; } else if (data.immediate_changed) { @@ -1431,10 +1453,8 @@ export function initSettings({ state, setBeforePageLeave, ws } = {}) { statusType = 'warn'; } setStatus(statusMsg, statusType); - if (data.restart_required || runtimeModeResult?.restart_required) { - restartPending = true; - } - if (restartPending) byId('btn-restart-now')?.removeAttribute('hidden'); + syncRestartState(data.restart_state); + await refreshRestartState(); window.dispatchEvent(new CustomEvent('ouro:settings-updated', { detail: { reason: 'settings saved', source: 'settings' } })); } catch (e) { const receipt = e?.body || e?.payload; @@ -1455,8 +1475,6 @@ export function initSettings({ state, setBeforePageLeave, ws } = {}) { byId('btn-restart-now')?.addEventListener('click', async () => { const outcome = await confirmAndSendRestart({ openConfirmDialog, ws }); if (outcome === 'sent') { - restartPending = false; - byId('btn-restart-now')?.setAttribute('hidden', ''); setStatus('Restart requested. If the agent refuses, the reason appears in the main chat.', 'muted'); } else if (outcome === 'not_connected') { setStatus('Not connected — the restart command was not sent.', 'warn'); diff --git a/web/modules/settings_local_model.js b/web/modules/settings_local_model.js index 46d127a4b..08202dcfc 100644 --- a/web/modules/settings_local_model.js +++ b/web/modules/settings_local_model.js @@ -45,15 +45,18 @@ function setProgressBar(fraction) { bar.setAttribute('aria-valuenow', Math.round(fraction * 100)); } -export function bindLocalModelControls({ state }) { +export function bindLocalModelControls({ state, onApplication } = {}) { let destroyed = false; let stopPending = false; let ready = false; + let statusSequence = 0; async function updateLocalStatus() { if (destroyed || state.activePage !== 'settings') return; + const sequence = ++statusSequence; try { const d = await fetchJson('/api/local-model/status', { cache: 'no-store' }); - if (destroyed) return; + if (destroyed || sequence !== statusSequence) return; + if (d.settings_application && onApplication) onApplication(d.settings_application); const el = document.getElementById('local-model-status'); if (!el) return; const isReady = d.status === 'ready'; @@ -75,6 +78,7 @@ export function bindLocalModelControls({ state }) { if (d.runtime_status === 'install_ok') text += ' — Runtime installed ✓'; if (d.runtime_status === 'install_error') text += ' — Runtime install failed'; if (d.error && !isInstalling) text += ' — ' + d.error; + if (d.settings_application?.pending_keys?.length) text += ' — Saved changes need Stop, then Start.'; el.textContent = text; el.dataset.tone = isReady ? 'ok' : (d.status === 'error' || d.runtime_status === 'install_error' ? 'error' : 'muted'); @@ -119,6 +123,7 @@ export function bindLocalModelControls({ state }) { } setTestResult(''); setProgressBar(null); + setInlineStatus(document.getElementById('local-model-action-status'), '', 'muted'); try { const resp = await apiFetch('/api/local-model/start', { method: 'POST', diff --git a/web/modules/settings_ui.js b/web/modules/settings_ui.js index 72b03f062..cbc7e508b 100644 --- a/web/modules/settings_ui.js +++ b/web/modules/settings_ui.js @@ -909,6 +909,7 @@ export function renderSettingsPage() { diff --git a/web/modules/subagent_status_primitives.js b/web/modules/subagent_status_primitives.js index af2e94f35..6ae81b32c 100644 --- a/web/modules/subagent_status_primitives.js +++ b/web/modules/subagent_status_primitives.js @@ -221,7 +221,8 @@ export function rowStatus(row, state) { const ROUTE_HINT = 'Choose how this subagent runs: an API model or an agent session.'; function executionFor(snapshot, subagentId) { - const receipt = snapshot?.subagent_last_delegation; + const history = snapshot?.subagent_last_delegation; + const receipt = history?.latest_by_subagent?.[subagentId] || history; if (!receipt || typeof receipt !== 'object') return null; return String(receipt.selected_subagent_id || '') === String(subagentId || '') ? receipt : null; @@ -238,13 +239,21 @@ export function rowMeta(row, state, errors) { // An empty draft (`openai::` with no model yet) is still an invitation. if (!String(row.route?.target_id || '').trim() || (!session && !routeModelFields(row.route).model.trim())) return { text: ROUTE_HINT, tone: '' }; - const evidence = describeExecutionEvidence(executionFor(state.snapshot, row.subagent_id)); + const receipt = executionFor(state.snapshot, row.subagent_id); + const evidence = describeExecutionEvidence(receipt); + const identity = receipt?.identity; + const sameRoute = identity && identity.kind === row.route.kind + && identity.target_id === row.route.target_id + && identity.credential_profile_id === String(routePin(row.route) || '') + && identity.effort === String(row.effort || '') + && identity.processing_preference === String(row.processing_preference || state.processingPreference || ''); + const historyLabel = identity ? (sameRoute ? 'Last run' : 'Earlier settings') : 'Last actual run'; // The exact stored spelling is disclosed here, where it informs, and never // in a placeholder, where it would instruct (docs/DESIGN.md §7). A session // target already reads as harness plus model in its own controls. const saved = session ? '' : `stored as ${String(row.route.target_id).trim()}`; return { - text: [saved, evidence ? `Last actual run: ${evidence}` : ''].filter(Boolean).join(' · '), - tone: '', + text: [saved, evidence ? `${historyLabel}: ${evidence}` : ''].filter(Boolean).join(' · '), + tone: '', ...(evidence ? { history: true } : {}), }; } diff --git a/web/modules/subagents_settings.js b/web/modules/subagents_settings.js index 6e4bc0d4a..02f5b8410 100644 --- a/web/modules/subagents_settings.js +++ b/web/modules/subagents_settings.js @@ -366,7 +366,7 @@ export function availableSubagentRowMarkup(row, state, index = 0) { ${effortSelectHtml(`data-subagent-field="effort" aria-label="Reasoning effort for Subagent ${ordinal}"`, row.effort || '', 'route default')} ${processingDetailsHtml(`data-subagent-field="processing_preference" aria-label="Processing for Subagent ${ordinal}"`, row.processing_preference, state.processingPreference)} -
${escapeHtml(meta.text)}
+
${escapeHtml(meta.text)}
`; } @@ -498,6 +498,7 @@ export function createAvailableSubagentsEditor({ const metaEl = el.querySelector('[data-subagent-meta]'); if (!metaEl) return; Object.assign(metaEl, { hidden: !meta.text, textContent: meta.text, title: meta.text }); + metaEl.toggleAttribute('data-run-history', Boolean(meta.history)); if (meta.tone) metaEl.dataset.tone = meta.tone; else delete metaEl.dataset.tone; }); @@ -507,7 +508,6 @@ export function createAvailableSubagentsEditor({ if (state.saveAttempted) onJudged(!shown.length); } - // Judge existing rows on Save/Finish; later new rows remain fresh. function noteSaveAttempt() { state.saveAttempted = true; state.setting.items.forEach((row) => { row._uiAttempted = true; }); diff --git a/web/onboarding.css b/web/onboarding.css index 39e89b3c0..1773b64ca 100644 --- a/web/onboarding.css +++ b/web/onboarding.css @@ -940,6 +940,11 @@ textarea::placeholder { text-overflow: ellipsis; } +.available-subagent-meta[data-run-history] { + white-space: normal; + overflow-wrap: anywhere; +} + .available-subagent-actions .btn-default { min-height: 34px; padding: 0 11px; diff --git a/web/settings.css b/web/settings.css index 3d4ce73fc..02ccd6c16 100644 --- a/web/settings.css +++ b/web/settings.css @@ -576,6 +576,11 @@ text-overflow: ellipsis; } +.available-subagent-meta[data-run-history] { + white-space: normal; + overflow-wrap: anywhere; +} + .available-subagent-actions .btn-default { min-height: 34px; padding: 0 11px; diff --git a/web/tests/subagents_settings.test.js b/web/tests/subagents_settings.test.js index e85d5b3ff..1d330f09f 100644 --- a/web/tests/subagents_settings.test.js +++ b/web/tests/subagents_settings.test.js @@ -599,6 +599,29 @@ test('preview replaces only a clean generated baseline', () => { assert.equal(editor.setting.items[0].subagent_id, 'codex_builder'); }); +test('dated API failures stay informational and bind to the exact execution choices', () => { + const row = apiRow({ processing_preference: 'standard' }); + const state = { snapshot: { subagent_last_delegation: { latest_by_subagent: { + api_scout: { selected_subagent_id: 'api_scout', route: 'api_model', + requested_model: row.route.target_id, applied_model: '', outcome: 'failed', + failure_code: 'quota_exhausted', ts: '2026-09-18T12:00:00Z', occurred_at: '2026-09-18T12:00:00Z', + identity: { ...row.route, credential_profile_id: '', effort: 'high', processing_preference: 'standard' } }, + } } } }; + const meta = rowMeta(row, state, []); + assert.equal(meta.tone, ''); + assert.match(meta.text, /Last run: API model.*failed \(quota_exhausted\).*2026-09-18/); + assert.equal(rowMeta({ ...row, recommended_use: 'Changed description' }, state, []).text, meta.text); + for (const changed of [ + { ...row, effort: 'low' }, + { ...row, processing_preference: 'flex' }, + { ...row, route: { ...row.route, target_id: 'another-model' } }, + { ...row, route: { ...row.route, credential_profile_id: 'another-account' } }, + ]) assert.match(rowMeta(changed, state, []).text, /Earlier settings:/); + const oldStatus = rowStatus(row, state); + delete state.snapshot.subagent_last_delegation; + assert.deepEqual(rowStatus(row, state), oldStatus, 'history never changes live admission/status'); +}); + test('a typed preview refusal stays typed and cannot become an empty fictional draft', () => { const editor = createAvailableSubagentsEditor({ doc: null, win: null }); editor.setPreviewFailure({