diff --git a/docs/v7next/LEDGER_CORRECTIONS.md b/docs/v7next/LEDGER_CORRECTIONS.md index be14c438b..7abc447a9 100644 --- a/docs/v7next/LEDGER_CORRECTIONS.md +++ b/docs/v7next/LEDGER_CORRECTIONS.md @@ -39,3 +39,58 @@ with evidence, found lane by lane. Applied to the campaign's carried ledger at F ouroboros/usage_accounting.py (1600 lines, exactly at the hard cap; IMPORT_REL at :60, the four defs at :1374-:1600). The extraction was performed by this lane from tip bytes. + +## From the D03 lane (base f61ea3c2, 2026-08-30) +7. MIGRATION rows 3943-3946 (`ouroboros/context.py::{_project_room_fact, + _runtime_budget_info,_promoted_task_toolset,_delegation_capability_fact}` -> + `ouroboros/context_runtime_facts.py`, "pending upstream transfer") — + RE-CONFIRMED pending at this tip (context.py 1590 lines, the four defs at + :325-:544); the extraction was performed by this lane from tip bytes. The + reference leaf is BYTE-FALSIFIED as a copy source for ONE of the four + symbols: upstream b14ba397 ("expose available subagents in runtime + context") rewrote `_delegation_capability_fact` (docstring collapsed to a + one-line summary, `configured_route` dropped from the returned fact, + requested/applied profile evidence and `selected_subagent_id` added, plus + an all-absent -> None guard). Drift-probe `--check` of the reference leaf + against tip bytes: 3/4 spans ast=tokens=bytes=True, this span + ast=False/tokens=False; re-emitting from tip bytes was proof-green on the + first round. Copying the reference leaf verbatim would have silently + reverted the upstream subagent-profile feature. +8. MIGRATION row 3960 (`tests/test_context.py:: + test_delegation_fact_carries_configured_route_and_historical_rows` -> + `tests/test_context_runtime_section.py::`) — SOURCE SYMBOL FALSIFIED + by the same upstream train: b14ba397 replaced the test with + `test_delegation_fact_carries_historical_rows_and_profile_evidence` + (asserts `"configured_route" not in delegation`). The upstream successor + was moved to the row's destination as an identity continuation (tip + bytes); the carried ledger must rename the row at F5. +9. MIGRATION row 1641 (`tests/test_context.py:: + test_runtime_section_includes_improvement_backlog_digest` -> + `tests/test_context_runtime_section.py::`) — SOURCE SYMBOL FALSIFIED: + upstream 1b7f9497 replaced the test with + `test_improvement_backlog_digest_is_actor_scoped` (the digest is now + asserted ABSENT for ordinary/main/project/subagent tasks and present only + for evolution/deep_self_review). Moved to the row's destination as an + identity continuation (tip bytes); rename at F5. +10. S7a rows 1614-1640/1642-1648 — RE-CONFIRMED against tip bytes: every other + moved symbol of the tests/test_context.py split is byte-identical between + the tip monolith and the reference siblings (the D15-carried + tests/test_context_memory.py re-derived from tip bytes came out identical + — the carry was NOT stale), except row 1623's span + (`test_force_plan_metadata_adds_structured_notice_without_rewriting_user_text`), + which upstream drifted ADDITIVELY (rc-phaseC execution-shape assertions) — + tip bytes transplanted. Note: between the D15 pilot and this lane the 15 + memory tests existed in BOTH tests/test_context.py and + tests/test_context_memory.py on the integration branch (ran twice); this + lane completed the split and deduplicated. +11. NO-ROW upstream additions (candidate rows for the carried ledger): 3459dd12 + added 8 recent-chat/archive-generation tests to tests/test_context.py + (filters_archives_before_recent_bound, retention_proof_cross_thread, + reads_only_bounded_generation_suffix, materializes_a_bounded_row_suffix, + malformed_gap_even_when_search_matches_nothing, + resumes_unconsolidated_archived_generation, + archive_only_chat_chain_is_complete, missing_cursor_generation_hot_path). + They have no MIGRATION rows, so this lane left them in the remainder + tests/test_context.py (612 lines) rather than deciding their theme-home + unilaterally; by the memory-file theme they are candidates for + tests/test_context_memory.py at F5. diff --git a/ouroboros/context.py b/ouroboros/context.py index 152f65e6c..bc1f0c0bd 100644 --- a/ouroboros/context.py +++ b/ouroboros/context.py @@ -322,226 +322,16 @@ _OWNER_CLIENT_NOTE = ( ) -def _project_room_fact(task: Dict[str, Any]) -> Optional[Dict[str, Any]]: - """The project-room working-folder FACT for a room turn, or None. - - Extracted verbatim from ``build_runtime_section`` (v6.90.x submarine unwind) - to keep that builder under the hard method gate; the resolution and the - stated rule are unchanged. - """ - # v6.58.0 (2.2): a conversation/decision turn in a project ROOM sees the room's - # working folder as a structural FACT — it can promote work into that folder - # without ITSELF becoming a workspace task (decision turns deliberately keep the - # promote/steer/route toolset, which workspace profiles exclude). The default - # transport: promote_chat_to_task from this room inherits working_dir unless - # workspace='none'. Registry read is anchored at the canonical DATA_DIR. - # v6.61.3 room lens: the rule now states the REAL chat-lane affordances (reads + - # default shell cwd resolve to the folder; writes go through promoted tasks) — - # the robot-room incident was exactly a fact/affordance split. A set-but-broken - # working_dir is disclosed loudly instead of a silent system-repo fallback. - try: - _room_pid = str(task.get("project_id") or "").strip() - if _room_pid and not str(task.get("workspace_root") or "").strip(): - from ouroboros.config import DATA_DIR as _DATA_DIR - from ouroboros.projects_registry import get_project as _get_project - from ouroboros.workspace_admission import room_chat_lens_dir as _room_lens - - _room = _get_project(_DATA_DIR, _room_pid) or {} - _room_wd = str(_room.get("working_dir") or "").strip() - if _room_wd: - # Same resolver the agent uses for the tool lens, so the stated rule - # and the actual tool surface cannot diverge (the robot incident). - _lens_dir, _room_note = _room_lens(_DATA_DIR, _room_pid) - _lens_active = bool(task.get("_is_direct_chat")) and bool(_lens_dir) - fact = { - "project_id": _room_pid, - "working_dir": _room_wd, - "rule": ( - ( - "This room's chat lane LOOKS AT the project folder: read_file/" - "list_files/search_code/query_code with root=active_workspace and " - "the DEFAULT shell cwd resolve to working_dir. The Ouroboros " - "system repo needs explicit root=\"system_repo\" (reads) or an " - "explicit cwd (shell). File WRITES here go through " - "promote_chat_to_task — the promoted task inherits this folder as " - "its workspace (workspace='none' opts out)." - ) - if _lens_active - else ( - "This project has a working folder. Tasks promoted from this room " - "run with it as their active workspace by default; pass " - "workspace='none' to promote a folder-less task." - ) - ), - } - if _room_note: - fact["working_dir_warning"] = _room_note - return fact - except Exception: - log.debug("Failed to inject project_room working_dir fact", exc_info=True) - return None - - -def _runtime_budget_info(env: Any, task: Dict[str, Any]) -> Dict[str, Any]: - """Start-of-task budget block: global projection + the STATIC per-task tree cap, - written once at task start so the cached prefix stays byte-stable (DEVELOPMENT - cache_friendliness item 22); live tree spend rides only the cache-breaking - surfaces (checkpoint/pacing/milestones).""" - try: - from ouroboros.usage_accounting import usage_projection - - total_usd = float(os.environ.get("TOTAL_BUDGET", "1")) - budget_root = pathlib.Path(task.get("budget_drive_root") or env.drive_root) - projection = usage_projection(budget_root, global_limit_usd=total_usd) - spent_usd = float(projection.get("accounted_usd") or 0.0) - budget_info = { - "status": "available", "total_usd": total_usd, - "spent_usd": spent_usd, "remaining_usd": total_usd - spent_usd, - "reserved_usd": float(projection.get("reserved_usd") or 0.0), - "unresolved_upper_bound_usd": float(projection.get("unresolved_upper_bound_usd") or 0.0), - "unknown_unmetered": int(projection.get("unknown_unmetered") or 0), - } - except Exception: - log.error("Budget authority unavailable for runtime context", exc_info=True) - budget_info = {"status": "unavailable"} - try: - root_cap = float(os.environ.get("OUROBOROS_PER_TASK_COST_USD", "0") or 0) - except (TypeError, ValueError): - root_cap = 0.0 - if root_cap > 0: - budget_info["per_task_tree_cap_usd"] = root_cap - budget_info["per_task_tree_cap_rule"] = ( - "Hard cap for THIS task's WHOLE tree (own model calls + all subagents), enforced " - "by the physical-attempt ledger: dispatches are refused once the tree's accounted " - "spend reaches it and the task is force-stopped. Budget checkpoints during the task report the live tree number." - ) - return budget_info - - -def _promoted_task_toolset(env: Any) -> Dict[str, Any]: - """The LIVE built-in toolset available to an ordinary promoted task. - - Workspace focus changes the default target, not the top-level principal's - tool names. The projection therefore asks the real registry once and keeps - credential omissions typed instead of maintaining a second static catalog. - Dynamic extension/MCP availability remains task-time state. - """ - from types import SimpleNamespace - - from ouroboros.tools.registry import ToolRegistry, _builtin_tool_availability - - registry = ToolRegistry(pathlib.Path(env.repo_dir), pathlib.Path(getattr(env, "drive_root", "."))) - - probe = SimpleNamespace( - task_id="promote_toolset_probe", - task_metadata={}, - task_contract={}, - task_constraint=None, - is_workspace_mode=lambda: False, - is_ephemeral_turn=False, - ) - registry.set_context(probe) - top_level_tools = set(registry.available_tools()) - # Typed omissions: registered built-ins that live availability removes right - # now (credential gates). Named with their reason so the router can tell - # "does not exist" from "exists but currently unavailable". - unavailable = {} - for name in registry._entries: - available, reason, detail = _builtin_tool_availability(name, probe) - if not available: - unavailable[name] = f"{reason}: {detail}" if detail else reason - return { - "top_level_tools": sorted(top_level_tools), - **({"unavailable_builtin_tools": dict(sorted(unavailable.items()))} if unavailable else {}), - "rule": ( - "LIVE built-in tool availability, evaluated by the real tool " - "registry at promote time. Project focus changes the default root, " - "not this ordinary top-level toolset. unavailable_builtin_tools " - "exist but are currently unusable (e.g. missing credentials) — do " - "not demand them. Dynamic extension/MCP tools are NOT listed (their " - "availability is unknowable at promote time). If an objective/" - "expected_output demands specific BUILT-IN tools, demand only names " - "listed here." - ), - } - - -def _delegation_capability_fact() -> Optional[Dict[str, Any]]: - """B4-lite: honestly-labeled HISTORICAL delegation observations. - - Deliberately NOT live health — receipts prove what the last execution did, - not what a lane can do now; live lane facts arrive from plan-review wave - rows and typed delegate refusals. Pure bounded file reads over the existing - receipt projections: no daemon probes, no new health authority. Absent - receipt files mean absent observations, never "healthy". Fail-soft on its - own (None on any failure) so a problem here never drops the surrounding - capabilities digest. - """ - try: - from ouroboros.reviewer_slot_config import reviewer_slot_last_executions - from ouroboros.subagents import subagent_last_delegation - - def _observed_label(ts: Any) -> str: - # Timestamp only: the verbatim "historical, not live health" disclaimer - # lives ONCE in the note below, never repeated per row. - return f"last observed at {str(ts or '').strip() or 'unknown time'}" - - delegation: Dict[str, Any] = { - "note": ( - "Every row here is historical, not live health (the last " - "recorded execution per reviewer slot / delegated run): " - "live lane facts arrive from plan-review wave rows and typed " - "delegate refusals. A missing row means no observation on " - "record — never healthy." - ), - } - slot_rows: List[Dict[str, Any]] = [] - for slot_id, row in sorted(reviewer_slot_last_executions().items()): - if not isinstance(row, dict): - continue - status = str(row.get("status") or "").strip() - fact: Dict[str, Any] = { - "slot": str(slot_id), - "outcome": (("ok" if status == "ok" else "failed") if status - else "unknown"), - "observed": _observed_label(row.get("ts")), - } - requested = row.get("requested") if isinstance(row.get("requested"), dict) else {} - effective = row.get("effective") if isinstance(row.get("effective"), dict) else {} - if requested.get("profile_id"): - fact["requested_profile"] = str(requested["profile_id"]) - if effective.get("profile_id"): - fact["applied_profile"] = str(effective["profile_id"]) - # B1's typed failure facts, forwarded only when recorded (a dated - # window carries reset_at without a code and an undated one the - # code without a reset — read both independently). - for key in ("failure_code", "reset_at"): - if row.get(key): - fact[key] = row[key] - slot_rows.append(fact) - if slot_rows: - delegation["reviewer_slots_last"] = slot_rows - last = subagent_last_delegation() - if isinstance(last, dict) and last: - last_fact = { - "route": str(last.get("route") or ""), - "requested_model": str(last.get("requested_model") or ""), - "applied_model": str(last.get("applied_model") or ""), - "observed": _observed_label(last.get("ts")), - } - if last.get("requested_profile"): - last_fact["requested_profile"] = str(last["requested_profile"]) - if last.get("applied_profile"): - last_fact["applied_profile"] = str(last["applied_profile"]) - if last.get("selected_subagent_id"): - last_fact["selected_subagent_id"] = str(last["selected_subagent_id"]) - delegation["subagent_last_delegation"] = last_fact - if len(delegation) == 1: - return None - return delegation - except Exception: - log.debug("Failed to build delegation capability fact", exc_info=True) - return None +# The runtime section's fact builders live in ouroboros/context_runtime_facts.py +# (extracted at this module's size ceiling); re-exported here because the section +# builder below and the tests that monkeypatch these names address them on THIS +# surface. +from ouroboros.context_runtime_facts import ( # noqa: E402,F401 — re-exported public surface + _delegation_capability_fact, + _project_room_fact, + _promoted_task_toolset, + _runtime_budget_info, +) def _task_authority_projection(env: Any, task: Dict[str, Any]) -> Dict[str, Any]: diff --git a/ouroboros/context_runtime_facts.py b/ouroboros/context_runtime_facts.py new file mode 100644 index 000000000..68b75f442 --- /dev/null +++ b/ouroboros/context_runtime_facts.py @@ -0,0 +1,241 @@ +"""The runtime section's FACT builders: what the host can honestly say it knows. + +Extracted whole from ``context.py`` at its module ceiling (v7 leaf) so the four +facts the runtime section renders keep one home: the project room a task sits in, +the budget rails it runs under, the toolset a promoted task materialized, and the +configured delegation route with its honestly-labeled historical observations. +Each returns a plain projection and reads no context state, so nothing here can +change what the section MEANS — only what it reports. ``context`` re-exports every +name, so historical imports and monkeypatch targets keep working unchanged. +""" + +from __future__ import annotations + +import logging +import os +import pathlib +from typing import Any, Dict, List, Optional + +log = logging.getLogger(__name__) + + +def _project_room_fact(task: Dict[str, Any]) -> Optional[Dict[str, Any]]: + """The project-room working-folder FACT for a room turn, or None. + + Extracted verbatim from ``build_runtime_section`` (v6.90.x submarine unwind) + to keep that builder under the hard method gate; the resolution and the + stated rule are unchanged. + """ + # v6.58.0 (2.2): a conversation/decision turn in a project ROOM sees the room's + # working folder as a structural FACT — it can promote work into that folder + # without ITSELF becoming a workspace task (decision turns deliberately keep the + # promote/steer/route toolset, which workspace profiles exclude). The default + # transport: promote_chat_to_task from this room inherits working_dir unless + # workspace='none'. Registry read is anchored at the canonical DATA_DIR. + # v6.61.3 room lens: the rule now states the REAL chat-lane affordances (reads + + # default shell cwd resolve to the folder; writes go through promoted tasks) — + # the robot-room incident was exactly a fact/affordance split. A set-but-broken + # working_dir is disclosed loudly instead of a silent system-repo fallback. + try: + _room_pid = str(task.get("project_id") or "").strip() + if _room_pid and not str(task.get("workspace_root") or "").strip(): + from ouroboros.config import DATA_DIR as _DATA_DIR + from ouroboros.projects_registry import get_project as _get_project + from ouroboros.workspace_admission import room_chat_lens_dir as _room_lens + + _room = _get_project(_DATA_DIR, _room_pid) or {} + _room_wd = str(_room.get("working_dir") or "").strip() + if _room_wd: + # Same resolver the agent uses for the tool lens, so the stated rule + # and the actual tool surface cannot diverge (the robot incident). + _lens_dir, _room_note = _room_lens(_DATA_DIR, _room_pid) + _lens_active = bool(task.get("_is_direct_chat")) and bool(_lens_dir) + fact = { + "project_id": _room_pid, + "working_dir": _room_wd, + "rule": ( + ( + "This room's chat lane LOOKS AT the project folder: read_file/" + "list_files/search_code/query_code with root=active_workspace and " + "the DEFAULT shell cwd resolve to working_dir. The Ouroboros " + "system repo needs explicit root=\"system_repo\" (reads) or an " + "explicit cwd (shell). File WRITES here go through " + "promote_chat_to_task — the promoted task inherits this folder as " + "its workspace (workspace='none' opts out)." + ) + if _lens_active + else ( + "This project has a working folder. Tasks promoted from this room " + "run with it as their active workspace by default; pass " + "workspace='none' to promote a folder-less task." + ) + ), + } + if _room_note: + fact["working_dir_warning"] = _room_note + return fact + except Exception: + log.debug("Failed to inject project_room working_dir fact", exc_info=True) + return None + + +def _runtime_budget_info(env: Any, task: Dict[str, Any]) -> Dict[str, Any]: + """Start-of-task budget block: global projection + the STATIC per-task tree cap, + written once at task start so the cached prefix stays byte-stable (DEVELOPMENT + cache_friendliness item 22); live tree spend rides only the cache-breaking + surfaces (checkpoint/pacing/milestones).""" + try: + from ouroboros.usage_accounting import usage_projection + + total_usd = float(os.environ.get("TOTAL_BUDGET", "1")) + budget_root = pathlib.Path(task.get("budget_drive_root") or env.drive_root) + projection = usage_projection(budget_root, global_limit_usd=total_usd) + spent_usd = float(projection.get("accounted_usd") or 0.0) + budget_info = { + "status": "available", "total_usd": total_usd, + "spent_usd": spent_usd, "remaining_usd": total_usd - spent_usd, + "reserved_usd": float(projection.get("reserved_usd") or 0.0), + "unresolved_upper_bound_usd": float(projection.get("unresolved_upper_bound_usd") or 0.0), + "unknown_unmetered": int(projection.get("unknown_unmetered") or 0), + } + except Exception: + log.error("Budget authority unavailable for runtime context", exc_info=True) + budget_info = {"status": "unavailable"} + try: + root_cap = float(os.environ.get("OUROBOROS_PER_TASK_COST_USD", "0") or 0) + except (TypeError, ValueError): + root_cap = 0.0 + if root_cap > 0: + budget_info["per_task_tree_cap_usd"] = root_cap + budget_info["per_task_tree_cap_rule"] = ( + "Hard cap for THIS task's WHOLE tree (own model calls + all subagents), enforced " + "by the physical-attempt ledger: dispatches are refused once the tree's accounted " + "spend reaches it and the task is force-stopped. Budget checkpoints during the task report the live tree number." + ) + return budget_info + + +def _promoted_task_toolset(env: Any) -> Dict[str, Any]: + """The LIVE built-in toolset available to an ordinary promoted task. + + Workspace focus changes the default target, not the top-level principal's + tool names. The projection therefore asks the real registry once and keeps + credential omissions typed instead of maintaining a second static catalog. + Dynamic extension/MCP availability remains task-time state. + """ + from types import SimpleNamespace + + from ouroboros.tools.registry import ToolRegistry, _builtin_tool_availability + + registry = ToolRegistry(pathlib.Path(env.repo_dir), pathlib.Path(getattr(env, "drive_root", "."))) + + probe = SimpleNamespace( + task_id="promote_toolset_probe", + task_metadata={}, + task_contract={}, + task_constraint=None, + is_workspace_mode=lambda: False, + is_ephemeral_turn=False, + ) + registry.set_context(probe) + top_level_tools = set(registry.available_tools()) + # Typed omissions: registered built-ins that live availability removes right + # now (credential gates). Named with their reason so the router can tell + # "does not exist" from "exists but currently unavailable". + unavailable = {} + for name in registry._entries: + available, reason, detail = _builtin_tool_availability(name, probe) + if not available: + unavailable[name] = f"{reason}: {detail}" if detail else reason + return { + "top_level_tools": sorted(top_level_tools), + **({"unavailable_builtin_tools": dict(sorted(unavailable.items()))} if unavailable else {}), + "rule": ( + "LIVE built-in tool availability, evaluated by the real tool " + "registry at promote time. Project focus changes the default root, " + "not this ordinary top-level toolset. unavailable_builtin_tools " + "exist but are currently unusable (e.g. missing credentials) — do " + "not demand them. Dynamic extension/MCP tools are NOT listed (their " + "availability is unknowable at promote time). If an objective/" + "expected_output demands specific BUILT-IN tools, demand only names " + "listed here." + ), + } + + +def _delegation_capability_fact() -> Optional[Dict[str, Any]]: + """B4-lite: honestly-labeled HISTORICAL delegation observations. + + Deliberately NOT live health — receipts prove what the last execution did, + not what a lane can do now; live lane facts arrive from plan-review wave + rows and typed delegate refusals. Pure bounded file reads over the existing + receipt projections: no daemon probes, no new health authority. Absent + receipt files mean absent observations, never "healthy". Fail-soft on its + own (None on any failure) so a problem here never drops the surrounding + capabilities digest. + """ + try: + from ouroboros.reviewer_slot_config import reviewer_slot_last_executions + from ouroboros.subagents import subagent_last_delegation + + def _observed_label(ts: Any) -> str: + # Timestamp only: the verbatim "historical, not live health" disclaimer + # lives ONCE in the note below, never repeated per row. + return f"last observed at {str(ts or '').strip() or 'unknown time'}" + + delegation: Dict[str, Any] = { + "note": ( + "Every row here is historical, not live health (the last " + "recorded execution per reviewer slot / delegated run): " + "live lane facts arrive from plan-review wave rows and typed " + "delegate refusals. A missing row means no observation on " + "record — never healthy." + ), + } + slot_rows: List[Dict[str, Any]] = [] + for slot_id, row in sorted(reviewer_slot_last_executions().items()): + if not isinstance(row, dict): + continue + status = str(row.get("status") or "").strip() + fact: Dict[str, Any] = { + "slot": str(slot_id), + "outcome": (("ok" if status == "ok" else "failed") if status + else "unknown"), + "observed": _observed_label(row.get("ts")), + } + requested = row.get("requested") if isinstance(row.get("requested"), dict) else {} + effective = row.get("effective") if isinstance(row.get("effective"), dict) else {} + if requested.get("profile_id"): + fact["requested_profile"] = str(requested["profile_id"]) + if effective.get("profile_id"): + fact["applied_profile"] = str(effective["profile_id"]) + # B1's typed failure facts, forwarded only when recorded (a dated + # window carries reset_at without a code and an undated one the + # code without a reset — read both independently). + for key in ("failure_code", "reset_at"): + if row.get(key): + fact[key] = row[key] + slot_rows.append(fact) + if slot_rows: + delegation["reviewer_slots_last"] = slot_rows + last = subagent_last_delegation() + if isinstance(last, dict) and last: + last_fact = { + "route": str(last.get("route") or ""), + "requested_model": str(last.get("requested_model") or ""), + "applied_model": str(last.get("applied_model") or ""), + "observed": _observed_label(last.get("ts")), + } + if last.get("requested_profile"): + last_fact["requested_profile"] = str(last["requested_profile"]) + if last.get("applied_profile"): + last_fact["applied_profile"] = str(last["applied_profile"]) + if last.get("selected_subagent_id"): + last_fact["selected_subagent_id"] = str(last["selected_subagent_id"]) + delegation["subagent_last_delegation"] = last_fact + if len(delegation) == 1: + return None + return delegation + except Exception: + log.debug("Failed to build delegation capability fact", exc_info=True) + return None diff --git a/ouroboros/size_ratchet_manifest.py b/ouroboros/size_ratchet_manifest.py index 25d5bdb85..99d09dd15 100644 --- a/ouroboros/size_ratchet_manifest.py +++ b/ouroboros/size_ratchet_manifest.py @@ -22,7 +22,6 @@ GIANT_PATHS = ( "tests/test_agent_task_pipeline.py", "tests/test_cancel_intents_phase_a.py", "tests/test_claudexor_owned_daemon.py", - "tests/test_context.py", "tests/test_delegated_subagent_transport.py", "tests/test_delivery_forced_finalization.py", "tests/test_devtools_benchmarks.py", @@ -139,6 +138,7 @@ BAND_PATHS = { "ouroboros/cancel_intents.py": "Entered the band from 929 lines: reciprocal timeout-retry lineage validation and physical-leaf/logical-root aliasing stay with the durable cancel-intent mutation authority so Stop-now hardens the same request across retry races.", "ouroboros/capability_evidence.py": "Grew INTO the band by the #284 fix: a fresh exact-model density witness may honestly undercut the cold floor \u2014 evidence logic belongs beside the witness store it reads.", "ouroboros/consciousness.py": "Durable Background Consciousness observation inbox and bounded truthful replay", + "ouroboros/context.py": "Entered the band from the 1501-1600 zone (1590 lines) by the v7 D03 extraction of the runtime-section fact builders into ouroboros/context_runtime_facts.py; shrink-only residue of the split, not new growth.", "ouroboros/extension_process_runner.py": None, "ouroboros/gateway/control.py": "Entered the band from 966 lines: the update-flow redesign added the shared stash-first prologue (_stash_local_work_fenced/_unwind_stashed_update) and the review-wave affordability floor to the update apply orchestration (update-flow-redesign sprint, Q9/Q10 owner decisions).", "ouroboros/gateway/history.py": None, diff --git a/tests/_context_shared.py b/tests/_context_shared.py new file mode 100644 index 000000000..4010efac3 --- /dev/null +++ b/tests/_context_shared.py @@ -0,0 +1,49 @@ +"""The health environment builder shared by the context suites. + +Split out of ``tests/test_context.py`` when that module was divided by theme; the +builder is verbatim, so every sibling suite keeps the exact drive layout and state it +was written against. +""" + +from __future__ import annotations + + + + + +def _make_health_env(tmp_path, events_lines=None): + class FakeEnv: + def drive_path(self, p): + return tmp_path / p + + def repo_path(self, p): + return tmp_path / "repo" / p + + @property + def repo_dir(self): + return tmp_path / "repo" + + @property + def drive_root(self): + return tmp_path + + (tmp_path / "state").mkdir(parents=True, exist_ok=True) + (tmp_path / "logs").mkdir(parents=True, exist_ok=True) + (tmp_path / "memory").mkdir(parents=True, exist_ok=True) + (tmp_path / "repo" / "docs").mkdir(parents=True, exist_ok=True) + (tmp_path / "repo" / "prompts").mkdir(parents=True, exist_ok=True) + (tmp_path / "archive" / "rescue").mkdir(parents=True, exist_ok=True) + (tmp_path / "repo" / "VERSION").write_text("1.2.3", encoding="utf-8") + (tmp_path / "repo" / "pyproject.toml").write_text('version = "1.2.3"', encoding="utf-8") + (tmp_path / "repo" / "web").mkdir(parents=True, exist_ok=True) + (tmp_path / "repo" / "web" / "package.json").write_text('{"version": "1.2.3"}', encoding="utf-8") + (tmp_path / "repo" / "README.md").write_text('version-1.2.3', encoding="utf-8") + (tmp_path / "repo" / "docs" / "ARCHITECTURE.md").write_text('# Ouroboros v1.2.3', encoding="utf-8") + (tmp_path / "repo" / "docs" / "DEVELOPMENT.md").write_text('# Dev', encoding="utf-8") + (tmp_path / "repo" / "prompts" / "CONSCIOUSNESS.md").write_text('Prompt text', encoding="utf-8") + (tmp_path / "state" / "state.json").write_text('{"spent_usd": 0, "budget_drift_alert": false}', encoding="utf-8") + (tmp_path / "memory" / "identity.md").write_text('x' * 300, encoding="utf-8") + (tmp_path / "memory" / "scratchpad.md").write_text('x' * 300, encoding="utf-8") + event_lines = events_lines or [] + (tmp_path / "logs" / "events.jsonl").write_text("\n".join(event_lines) + ("\n" if event_lines else ""), encoding="utf-8") + return FakeEnv() diff --git a/tests/test_context.py b/tests/test_context.py index 70a23bfe4..a203ad433 100644 --- a/tests/test_context.py +++ b/tests/test_context.py @@ -1,68 +1,25 @@ -"""Tests for ouroboros.context health invariants.""" +"""The health invariants ouroboros.context builds, and where they must appear. + +This module owns the cache hit-rate invariant, the remote context overflow it reports, +the hot-store growth it watches, the rest of the invariant coverage, and the rule that +the invariants come first in both the dynamic and the background-consciousness context. + +The runtime section, the advisory review status, the memory/consolidation sections and +the drive-state projection were split verbatim into +``tests/test_context_runtime_section.py``, ``tests/test_context_advisory_review.py``, +``tests/test_context_memory.py`` and ``tests/test_context_drive_state.py``; the health +environment builder they share lives in ``tests/_context_shared.py``. +""" from __future__ import annotations -import inspect import json -import pytest -from ouroboros.context import build_health_invariants, build_runtime_section, build_user_content +from ouroboros.context import build_health_invariants +from tests._context_shared import _make_health_env -def test_build_llm_messages_has_no_recorder_only_soft_cap_chain(): - from ouroboros import context as context_module - from ouroboros.context import build_llm_messages - - assert "soft_cap_tokens" not in inspect.signature(build_llm_messages).parameters - assert not hasattr(context_module, "apply_message_token_soft_cap") - source = inspect.getsource(build_llm_messages) - assert "estimated_tokens_before" not in source - assert "trimmed_sections" not in source - assert "context_fit" in source - - -@pytest.mark.parametrize("enforcement", ["blocking", "advisory"]) -def test_force_plan_metadata_adds_structured_notice_without_rewriting_user_text( - monkeypatch, enforcement, -): - monkeypatch.setenv("OUROBOROS_REVIEW_ENFORCEMENT", enforcement) - content = build_user_content( - { - "text": "Fix the marketplace retry flow.", - "metadata": {"force_plan": True, "force_plan_source": "swarm"}, - } - ) - - assert content.startswith("[SWARM_INITIATIVE]") - assert "Source: swarm." in content - assert f"Resolved review enforcement: {enforcement}." in content - assert "Under blocking" in content - assert "non-mutating preparation" in content - assert "begin implementation only after review closes" in content - # Fan-out integration mechanics (owner-approved, 2026-08-05): parallel - # children cannot see each other's edits, so a plan gives them disjoint - # write regions or plans the parent synthesis for the expected overlap. - assert "cannot see each other's edits" in content - assert "disjoint write regions" in content - # Owner intent as prose, not schema (rc-phaseC, P5): the plan must name its - # execution shape so reviewers can judge a silent zero-fan-out. - assert "State the chosen execution shape explicitly" in content - assert "delegation required, optional, or intentionally not used" in content - assert content.rstrip().endswith("Fix the marketplace retry flow.") - - -def test_ephemeral_force_plan_is_routing_only_and_transfers_work(): - content = build_user_content({ - "text": "Fix the marketplace retry flow.", - "_ephemeral_turn": True, - "metadata": {"force_plan": True, "force_plan_source": "swarm"}, - }) - - assert content.startswith("[SWARM_ROUTING_INTENT]") - assert "exactly one NEW managed root" in content - assert "do not execute it" in content - assert content.rstrip().endswith("Fix the marketplace retry flow.") class TestCacheHitRateInvariant: @@ -110,126 +67,6 @@ class TestCacheHitRateInvariant: assert "LOW CACHE HIT RATE" in result -def _make_health_env(tmp_path, events_lines=None): - class FakeEnv: - def drive_path(self, p): - return tmp_path / p - - def repo_path(self, p): - return tmp_path / "repo" / p - - @property - def repo_dir(self): - return tmp_path / "repo" - - @property - def drive_root(self): - return tmp_path - - (tmp_path / "state").mkdir(parents=True, exist_ok=True) - (tmp_path / "logs").mkdir(parents=True, exist_ok=True) - (tmp_path / "memory").mkdir(parents=True, exist_ok=True) - (tmp_path / "repo" / "docs").mkdir(parents=True, exist_ok=True) - (tmp_path / "repo" / "prompts").mkdir(parents=True, exist_ok=True) - (tmp_path / "archive" / "rescue").mkdir(parents=True, exist_ok=True) - (tmp_path / "repo" / "VERSION").write_text("1.2.3", encoding="utf-8") - (tmp_path / "repo" / "pyproject.toml").write_text('version = "1.2.3"', encoding="utf-8") - (tmp_path / "repo" / "web").mkdir(parents=True, exist_ok=True) - (tmp_path / "repo" / "web" / "package.json").write_text('{"version": "1.2.3"}', encoding="utf-8") - (tmp_path / "repo" / "README.md").write_text('version-1.2.3', encoding="utf-8") - (tmp_path / "repo" / "docs" / "ARCHITECTURE.md").write_text('# Ouroboros v1.2.3', encoding="utf-8") - (tmp_path / "repo" / "docs" / "DEVELOPMENT.md").write_text('# Dev', encoding="utf-8") - (tmp_path / "repo" / "prompts" / "CONSCIOUSNESS.md").write_text('Prompt text', encoding="utf-8") - (tmp_path / "state" / "state.json").write_text('{"spent_usd": 0, "budget_drift_alert": false}', encoding="utf-8") - (tmp_path / "memory" / "identity.md").write_text('x' * 300, encoding="utf-8") - (tmp_path / "memory" / "scratchpad.md").write_text('x' * 300, encoding="utf-8") - event_lines = events_lines or [] - (tmp_path / "logs" / "events.jsonl").write_text("\n".join(event_lines) + ("\n" if event_lines else ""), encoding="utf-8") - return FakeEnv() - - -def test_runtime_section_includes_light_runtime_mode_rule(tmp_path, monkeypatch): - env = _make_health_env(tmp_path) - monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "light") - section = build_runtime_section(env, {"id": "task-1", "type": "task"}) - payload = json.loads(section.split("\n\n", 1)[1]) - - assert payload["runtime_mode"] == "light" - assert "forbids Ouroboros repo mutation" in payload["runtime_mode_rule"] - assert "user_files" in payload["runtime_mode_rule"] - assert "artifact_store" in payload["runtime_mode_rule"] - assert "explicit scoped skill-payload work/repair" in payload["runtime_mode_rule"] - assert "runtime_data/uploads" in payload["runtime_mode_rule"] - - -def test_runtime_section_includes_filesystem_affordances_with_ctx(tmp_path, monkeypatch): - from ouroboros.tools.registry import ToolContext - - env = _make_health_env(tmp_path) - monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "light") - ctx = ToolContext(repo_dir=tmp_path / "repo", drive_root=tmp_path) - - section = build_runtime_section(env, {"id": "task-1", "type": "task"}, ctx=ctx) - payload = json.loads(section.split("\n\n", 1)[1]) - fs = payload["capabilities"]["filesystem"] - - assert fs["profile"] == "self_modification" - assert "runtime_data" in fs["searchable_roots"] - assert "task_drive" not in fs["searchable_roots"] - assert "task_drive" in fs["allowed_shell_cwd_roots"] - assert "status" in fs["git_readonly_subcommands"] - assert "active_workspace" in fs["light_gated_roots"] - - -def test_runtime_section_external_workspace_includes_user_files_shell_affordance(tmp_path, monkeypatch): - from ouroboros.tools.registry import ToolContext - - env = _make_health_env(tmp_path) - monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "advanced") - drive = tmp_path / "data" - repo = tmp_path / "repo" - workspace = tmp_path / "workspace" - drive.mkdir() - repo.mkdir(exist_ok=True) - workspace.mkdir(exist_ok=True) - ctx = ToolContext( - repo_dir=repo, - drive_root=drive, - workspace_root=workspace, - workspace_mode="external", - ) - - section = build_runtime_section(env, {"id": "task-1", "type": "task"}, ctx=ctx) - payload = json.loads(section.split("\n\n", 1)[1]) - fs = payload["capabilities"]["filesystem"] - - assert fs["profile"] == "external_workspace_task" - assert "user_files" in fs["allowed_shell_cwd_roots"] - - -def test_runtime_section_workspace_rule_preserves_system_review_commit_authority(tmp_path, monkeypatch): - env = _make_health_env(tmp_path) - monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "advanced") - workspace = tmp_path / "workspace" - workspace.mkdir() - section = build_runtime_section( - env, - { - "id": "task-1", - "type": "task", - "workspace_root": str(workspace), - "workspace_mode": "external", - "memory_mode": "forked", - }, - ) - rule = json.loads(section.split("\n\n", 1)[1])["active_workspace"]["rule"] - - assert "default to the active workspace" in rule - assert "explicit typed root/cwd" in rule - assert "self-review/commit tools remain available" in rule - assert "self-review/commit tools are unavailable" not in rule - - def test_health_invariants_reports_remote_context_overflow(tmp_path): env = _make_health_env( tmp_path, @@ -242,67 +79,6 @@ def test_health_invariants_reports_remote_context_overflow(tmp_path): assert "provider/model x1" in result -def test_runtime_section_omits_light_rule_for_advanced(tmp_path, monkeypatch): - env = _make_health_env(tmp_path) - monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "advanced") - section = build_runtime_section(env, {"id": "task-1", "type": "task"}) - payload = json.loads(section.split("\n\n", 1)[1]) - - assert payload["runtime_mode"] == "advanced" - assert "runtime_mode_rule" not in payload - - -def test_runtime_section_includes_non_workspace_memory_boundary(tmp_path, monkeypatch): - env = _make_health_env(tmp_path) - monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "advanced") - section = build_runtime_section( - env, - { - "id": "task-1", - "type": "task", - "memory_mode": "forked", - "drive_root": str(tmp_path / "child"), - "child_drive_root": str(tmp_path / "child"), - "budget_drive_root": str(tmp_path / "data"), - }, - ) - payload = json.loads(section.split("\n\n", 1)[1]) - assert payload["task"]["memory_mode"] == "forked" - assert payload["task"]["child_drive_root"].endswith("child") - assert payload["task"]["budget_drive_root"].endswith("data") - - -def test_runtime_section_exposes_host_routing_manifest_and_manual_contract(tmp_path, monkeypatch): - env = _make_health_env(tmp_path) - monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "advanced") - task = { - "id": "decision-1", - "type": "task", - "metadata": { - "current_chat": { - "chat_id": 1, - "running_tasks": [], - "addressable_root_tasks": [{"task_id": "pending-1", "status": "pending"}], - }, - "main_routing_manifest": { - "projects": [{"project_id": "racer", "name": "Racer"}], - "root_tasks": [{"task_id": "pending-1", "status": "pending"}], - }, - "routing_contract": { - "source_lane": "main", - "on_uncertain_or_invalid_target": "needs_manual_target", - "manual_options": [{"task_id": "pending-1"}], - }, - }, - } - - payload = json.loads(build_runtime_section(env, task).split("\n\n", 1)[1]) - - assert payload["current_chat"]["addressable_root_tasks"][0]["task_id"] == "pending-1" - assert payload["main_routing_manifest"]["projects"][0]["project_id"] == "racer" - assert payload["routing_contract"]["on_uncertain_or_invalid_target"] == "needs_manual_target" - - class TestAdditionalHealthInvariantCoverage: def test_version_desync_warning(self, tmp_path): env = _make_health_env(tmp_path) @@ -507,295 +283,7 @@ class TestHotStoreGrowthInvariant: assert "HOT STORE GROWTH" not in result -class TestAdvisoryReviewStatusInContext: - """Tests that advisory review status appears in LLM context when runs exist.""" - - def _make_env(self, tmp_path): - class FakeEnv: - def drive_path(self, p): - return tmp_path / p - def repo_path(self, p): - return tmp_path / "repo" / p - @property - def repo_dir(self): - return tmp_path / "repo" - @property - def drive_root(self): - return tmp_path - - (tmp_path / "state").mkdir(parents=True, exist_ok=True) - (tmp_path / "logs").mkdir(parents=True, exist_ok=True) - (tmp_path / "memory").mkdir(parents=True, exist_ok=True) - (tmp_path / "repo" / "docs").mkdir(parents=True, exist_ok=True) - (tmp_path / "repo" / "VERSION").write_text("1.2.3", encoding="utf-8") - (tmp_path / "repo" / "pyproject.toml").write_text('version = "1.2.3"', encoding="utf-8") - (tmp_path / "repo" / "README.md").write_text('version-1.2.3', encoding="utf-8") - (tmp_path / "repo" / "docs" / "ARCHITECTURE.md").write_text('# Ouroboros v1.2.3', encoding="utf-8") - (tmp_path / "repo" / "docs" / "DEVELOPMENT.md").write_text('# Dev', encoding="utf-8") - (tmp_path / "state" / "state.json").write_text('{"spent_usd": 0, "budget_drift_alert": false}', encoding="utf-8") - (tmp_path / "memory" / "identity.md").write_text('x' * 300, encoding="utf-8") - (tmp_path / "memory" / "scratchpad.md").write_text('x' * 300, encoding="utf-8") - return FakeEnv() - - def test_advisory_status_in_build_llm_messages(self, tmp_path): - """format_status_section returns non-empty string when runs exist.""" - from ouroboros.review_state import ( - AdvisoryReviewState, AdvisoryRunRecord, save_state, format_status_section - ) - state = AdvisoryReviewState() - state.add_run(AdvisoryRunRecord( - snapshot_hash="abc123", - commit_message="test commit", - status="fresh", - ts="2026-01-01T00:00:00", - items=[{"item": "bible_compliance", "verdict": "PASS", "severity": "critical", "reason": "ok"}], - )) - save_state(tmp_path, state) - - loaded = __import__("ouroboros.review_state", fromlist=["load_state"]).load_state(tmp_path) - section = format_status_section(loaded) - assert "Advisory Pre-Review Status" in section - assert "FRESH" in section - assert "abc123" in section - - def test_advisory_status_empty_when_no_runs(self, tmp_path): - """format_status_section returns 'No advisory runs' when state is empty.""" - from ouroboros.review_state import AdvisoryReviewState, format_status_section - state = AdvisoryReviewState() - section = format_status_section(state) - assert "No advisory runs" in section - - def test_review_continuity_context_surfaces_live_gate_and_continuation(self, tmp_path): - from ouroboros.agent_task_pipeline import build_review_context - from ouroboros.context import build_llm_messages - from ouroboros.memory import Memory - from ouroboros.review_state import ( - AdvisoryReviewState, - AdvisoryRunRecord, - CommitAttemptRecord, - compute_snapshot_hash, - make_repo_key, - save_state, - ) - from ouroboros.task_continuation import ReviewContinuation, save_review_continuation - from ouroboros.task_results import STATUS_COMPLETED, write_task_result - - env = self._make_env(tmp_path) - (tmp_path / "repo" / ".git").mkdir(parents=True, exist_ok=True) - (tmp_path / "repo" / "prompts").mkdir(parents=True, exist_ok=True) - (tmp_path / "repo" / "prompts" / "SYSTEM.md").write_text("System", encoding="utf-8") - (tmp_path / "repo" / "BIBLE.md").write_text("Bible", encoding="utf-8") - (tmp_path / "repo" / "docs" / "CHECKLISTS.md").write_text("Checklist", encoding="utf-8") - (tmp_path / "repo" / "tracked.py").write_text("print('hi')\n", encoding="utf-8") - - repo_key = make_repo_key(tmp_path / "repo") - snapshot_hash = compute_snapshot_hash(tmp_path / "repo") - state = AdvisoryReviewState() - state.add_run(AdvisoryRunRecord( - snapshot_hash=snapshot_hash, - commit_message="test commit", - status="bypassed", - ts="2026-04-07T09:59:00+00:00", - repo_key=repo_key, - bypass_reason="manual audit override", - )) - state.advisory_runs[-1].status = "stale" - state.last_stale_from_edit_ts = "2026-04-07T10:00:00+00:00" - state.last_stale_reason = "edit_text mutated tracked.py" - state.last_stale_repo_key = repo_key - state.record_attempt(CommitAttemptRecord( - ts="2026-04-07T10:01:00+00:00", - commit_message="blocked commit", - status="blocked", - repo_key=repo_key, - tool_name="commit_reviewed", - task_id="task-old", - attempt=1, - critical_findings=[{ - "item": "tests_affected", - "reason": "Fix the failing test before commit", - "severity": "critical", - "verdict": "FAIL", - }], - readiness_warnings=["Review was blocked and needs follow-up."], - )) - save_state(tmp_path, state) - - save_review_continuation( - tmp_path, - ReviewContinuation( - task_id="task-old", - source="blocked_review", - stage="blocking_review", - repo_key=repo_key, - tool_name="commit_reviewed", - attempt=1, - block_reason="critical_findings", - critical_findings=[{ - "item": "tests_affected", - "reason": "Fix the failing test before commit", - "severity": "critical", - "verdict": "FAIL", - }], - readiness_warnings=["Review was blocked and needs follow-up."], - ), - expect_task_id="task-old", - ) - write_task_result( - tmp_path, - "task-old", - STATUS_COMPLETED, - result="Commit blocked by review.", - ) - - messages, _ = build_llm_messages( - env=env, - memory=Memory(drive_root=tmp_path), - task={"id": "task-new", "type": "task", "text": "continue"}, - review_context_builder=lambda: build_review_context(env), - ) - dynamic_text = messages[0]["content"][2]["text"] - - assert "## Review Continuity" in dynamic_text - assert "repo_commit_ready=no" in dynamic_text - assert "retry_anchor=commit_readiness_debt" in dynamic_text - assert "Commit-readiness debt" in dynamic_text - assert "bypass_reason=manual audit override" in dynamic_text - assert "stale_marker=2026-04-07T10:00:00" in dynamic_text - assert "### Open review continuations" in dynamic_text - assert "critical_finding=tests_affected: Fix the failing test before commit" in dynamic_text - assert "### Historical review ledger" in dynamic_text - assert "## Scratchpad" in dynamic_text - assert dynamic_text.index("## Scratchpad") < dynamic_text.index("## Drive state") - assert dynamic_text.index("## Runtime context") < dynamic_text.index("## Review Continuity") - - def test_review_continuity_context_ignores_foreign_repo_obligations(self, tmp_path): - from ouroboros.agent_task_pipeline import build_review_context - from ouroboros.review_state import ( - AdvisoryReviewState, - AdvisoryRunRecord, - CommitAttemptRecord, - compute_snapshot_hash, - make_repo_key, - save_state, - ) - - env = self._make_env(tmp_path) - repo_a = tmp_path / "repo" - repo_b = tmp_path / "repo-other" - (repo_a / ".git").mkdir(parents=True, exist_ok=True) - (repo_b / ".git").mkdir(parents=True, exist_ok=True) - (repo_a / "tracked.py").write_text("print('repo a')\n", encoding="utf-8") - (repo_b / "tracked.py").write_text("print('repo b')\n", encoding="utf-8") - - repo_a_key = make_repo_key(repo_a) - repo_b_key = make_repo_key(repo_b) - state = AdvisoryReviewState() - state.add_run(AdvisoryRunRecord( - snapshot_hash=compute_snapshot_hash(repo_a), - commit_message="repo a ready", - status="fresh", - ts="2026-04-07T10:00:00+00:00", - repo_key=repo_a_key, - )) - state.record_attempt(CommitAttemptRecord( - ts="2026-04-07T10:01:00+00:00", - commit_message="repo b blocked", - status="blocked", - repo_key=repo_b_key, - tool_name="commit_reviewed", - task_id="task-b", - attempt=1, - block_reason="critical_findings", - critical_findings=[{ - "item": "foreign_issue", - "reason": "other repo only", - "severity": "critical", - "verdict": "FAIL", - }], - )) - save_state(tmp_path, state) - - dynamic_text = build_review_context(env) - assert "repo_commit_ready=yes" in dynamic_text - assert "foreign_issue" not in dynamic_text - assert "repo b blocked" not in dynamic_text - - def test_review_continuity_context_keeps_open_obligations_without_runs(self, tmp_path): - from ouroboros.agent_task_pipeline import build_review_context - from ouroboros.review_state import ( - AdvisoryReviewState, - ObligationItem, - make_repo_key, - save_state, - ) - - env = self._make_env(tmp_path) - (tmp_path / "repo" / ".git").mkdir(parents=True, exist_ok=True) - (tmp_path / "repo" / "tracked.py").write_text("print('hi')\n", encoding="utf-8") - - repo_key = make_repo_key(tmp_path / "repo") - state = AdvisoryReviewState( - open_obligations=[ - ObligationItem( - obligation_id="obl-0001", - item="tests_affected", - severity="critical", - reason="Coverage still missing", - source_attempt_ts="2026-04-07T10:00:00+00:00", - source_attempt_msg="blocked commit", - repo_key=repo_key, - fingerprint="finding:tests_affected:abc123", - ) - ] - ) - save_state(tmp_path, state) - - dynamic_text = build_review_context(env) - assert "## Review Continuity" in dynamic_text - assert "open_obligations=1" in dynamic_text - assert "[obl-0001] tests_affected: Coverage still missing" in dynamic_text - - def test_review_continuity_context_keeps_all_debt_evidence(self, tmp_path): - from ouroboros.agent_task_pipeline import build_review_context - from ouroboros.review_state import ( - AdvisoryReviewState, - CommitReadinessDebtItem, - make_repo_key, - save_state, - ) - - env = self._make_env(tmp_path) - (tmp_path / "repo" / ".git").mkdir(parents=True, exist_ok=True) - (tmp_path / "repo" / "tracked.py").write_text("print('hi')\n", encoding="utf-8") - - repo_key = make_repo_key(tmp_path / "repo") - state = AdvisoryReviewState( - commit_readiness_debts=[ - CommitReadinessDebtItem( - debt_id="debt-0001", - category="repeated_obligation", - title="Commit readiness debt", - summary="Repeated tests blocker", - repo_key=repo_key, - source_obligation_ids=["obl-0001"], - evidence=[ - "first evidence", - "second evidence", - "third evidence", - ], - ) - ] - ) - save_state(tmp_path, state) - - dynamic_text = build_review_context(env) - assert "first evidence" in dynamic_text - assert "second evidence" in dynamic_text - assert "third evidence" in dynamic_text - - -def test_improvement_backlog_digest_is_actor_scoped(tmp_path): +def test_health_invariants_come_first_in_dynamic_context(tmp_path): from ouroboros.context import build_llm_messages from ouroboros.memory import Memory @@ -816,259 +304,76 @@ def test_improvement_backlog_digest_is_actor_scoped(tmp_path): (tmp_path / "repo" / "prompts").mkdir(parents=True, exist_ok=True) (tmp_path / "repo" / "docs").mkdir(parents=True, exist_ok=True) - (tmp_path / "memory" / "knowledge").mkdir(parents=True, exist_ok=True) + (tmp_path / "memory").mkdir(parents=True, exist_ok=True) (tmp_path / "logs").mkdir(parents=True, exist_ok=True) (tmp_path / "state").mkdir(parents=True, exist_ok=True) (tmp_path / "repo" / "prompts" / "SYSTEM.md").write_text("System prompt", encoding="utf-8") (tmp_path / "repo" / "BIBLE.md").write_text("Bible", encoding="utf-8") (tmp_path / "repo" / "README.md").write_text("README", encoding="utf-8") - (tmp_path / "repo" / "docs" / "ARCHITECTURE.md").write_text('# Ouroboros v1.2.3', encoding="utf-8") - (tmp_path / "repo" / "docs" / "DEVELOPMENT.md").write_text('# Dev', encoding="utf-8") - (tmp_path / "repo" / "docs" / "CHECKLISTS.md").write_text('Checklist', encoding="utf-8") + (tmp_path / "repo" / "docs" / "ARCHITECTURE.md").write_text("# Ouroboros v1.2.3", encoding="utf-8") + (tmp_path / "repo" / "docs" / "DEVELOPMENT.md").write_text( + "### File Size Budgets\n| Path | Budget chars |\n|------|--------------|\n| memory/identity.md | 1000 |\n", + encoding="utf-8", + ) + (tmp_path / "repo" / "docs" / "CHECKLISTS.md").write_text("Checklist", encoding="utf-8") (tmp_path / "repo" / "VERSION").write_text("1.2.3", encoding="utf-8") (tmp_path / "repo" / "pyproject.toml").write_text('version = "1.2.3"', encoding="utf-8") - (tmp_path / "state" / "state.json").write_text('{"spent_usd": 0}', encoding="utf-8") - (tmp_path / "memory" / "identity.md").write_text("I am Ouroboros", encoding="utf-8") + (tmp_path / "state" / "state.json").write_text('{"spent_usd": 0, "budget_drift_alert": false}', encoding="utf-8") + (tmp_path / "memory" / "identity.md").write_text("x" * 950, encoding="utf-8") (tmp_path / "memory" / "scratchpad.md").write_text("scratchpad", encoding="utf-8") - (tmp_path / "memory" / "knowledge" / "improvement-backlog.md").write_text( - "# Improvement Backlog\n\n### ibl-1\n- status: open\n- created_at: 2026-04-14T09:00:00+00:00\n- source: execution_reflection\n- category: process\n- task_id: task-1\n- requires_plan_review: yes\n- fingerprint: fp-1\n- summary: Reduce recurring task friction around REVIEW_BLOCKED\n", - encoding="utf-8", + + messages, _cap_info = build_llm_messages( + env=FakeEnv(), + memory=Memory(drive_root=tmp_path), + task={"id": "task-a", "type": "task", "text": "hello"}, ) - ordinary = ({"id": "main", "type": "task", "_is_direct_chat": True}, - {"id": "project", "type": "task", "project_id": "p1"}, - {"id": "managed", "type": "task"}, - {"id": "child", "type": "task", "delegation_role": "subagent"}) - for task in ordinary: - task["text"] = "hello" - messages, _ = build_llm_messages( - env=FakeEnv(), memory=Memory(drive_root=tmp_path), task=task, - ) - assert "## Improvement Backlog" not in messages[0]["content"][2]["text"] - - for task_type in ("evolution", "deep_self_review"): - messages, _ = build_llm_messages( - env=FakeEnv(), memory=Memory(drive_root=tmp_path), - task={"id": task_type, "type": task_type, "text": "improve"}) - dynamic_text = messages[0]["content"][2]["text"] - assert "## Improvement Backlog" in dynamic_text - assert "Reduce recurring task friction around REVIEW_BLOCKED" in dynamic_text + dynamic_text = messages[0]["content"][2]["text"] + assert dynamic_text.startswith("## Health Invariants") + assert dynamic_text.index("## Health Invariants") < dynamic_text.index("## Drive state") -class TestRuntimeEnvSection: - """build_runtime_section: runtime_env carries presentation + platform, and - the per-message owner_client fact renders beside it (is_desktop retired).""" +def test_health_invariants_come_first_in_background_consciousness_context(tmp_path): + from ouroboros.consciousness import BackgroundConsciousness - def _make_env(self, tmp_path): - class FakeEnv: - repo_dir = tmp_path / "repo" - drive_root = tmp_path + repo_dir = tmp_path / "repo" + drive_root = tmp_path / "drive" + (repo_dir / "prompts").mkdir(parents=True, exist_ok=True) + (repo_dir / "docs").mkdir(parents=True, exist_ok=True) + (drive_root / "memory" / "knowledge").mkdir(parents=True, exist_ok=True) + (drive_root / "logs").mkdir(parents=True, exist_ok=True) + (drive_root / "state").mkdir(parents=True, exist_ok=True) - def drive_path(self, p): - return tmp_path / p - - (tmp_path / "state").mkdir(parents=True, exist_ok=True) - (tmp_path / "state" / "state.json").write_text( - '{"spent_usd": 0}', encoding="utf-8" - ) - return FakeEnv() - - def test_runtime_env_presentation_absent_means_web(self, tmp_path, monkeypatch): - from ouroboros.context import build_runtime_section - - monkeypatch.delenv("OUROBOROS_PRESENTATION", raising=False) - env = self._make_env(tmp_path) - section = build_runtime_section(env, {"id": "t1", "type": "task"}) - data = json.loads(section.split("## Runtime context\n\n", 1)[1]) - assert "runtime_env" in data - assert "platform" in data["runtime_env"] - assert isinstance(data["runtime_env"]["platform"], str) - assert data["runtime_env"]["presentation"] == "web" - # The dead is_desktop flag is retired; presentation replaced it. - assert "is_desktop" not in data["runtime_env"] - - def test_runtime_env_presentation_from_launcher_export(self, tmp_path, monkeypatch): - from ouroboros.context import build_runtime_section - - for value in ("desktop_window", "browser_fallback"): - monkeypatch.setenv("OUROBOROS_PRESENTATION", value) - env = self._make_env(tmp_path) - section = build_runtime_section(env, {"id": "t2", "type": "task"}) - data = json.loads(section.split("## Runtime context\n\n", 1)[1]) - assert data["runtime_env"]["presentation"] == value - - def test_owner_client_rendered_from_metadata(self, tmp_path, monkeypatch): - from ouroboros.context import build_runtime_section - - monkeypatch.delenv("OUROBOROS_PRESENTATION", raising=False) - env = self._make_env(tmp_path) - fact = {"pywebview": True, "ua": "TestShell/1.0", "viewport": {"w": 1200, "h": 800}} - section = build_runtime_section( - env, {"id": "t3", "type": "task", "metadata": {"client_surface": fact}} - ) - data = json.loads(section.split("## Runtime context\n\n", 1)[1]) - assert data["owner_client"] == fact - assert "SENT" in data["owner_client_note"] - - def test_owner_client_absent_is_a_gap_not_a_default(self, tmp_path, monkeypatch): - from ouroboros.context import build_runtime_section - - env = self._make_env(tmp_path) - section = build_runtime_section(env, {"id": "t4", "type": "task"}) - data = json.loads(section.split("## Runtime context\n\n", 1)[1]) - assert "owner_client" not in data - assert "owner_client_note" not in data - - def test_owner_client_channel_fact_stamped_by_external_admission(self, tmp_path): - from ouroboros.context import build_runtime_section - - env = self._make_env(tmp_path) - # /api/tasks and CLI STAMP the channel fact at admission; the renderer - # reads only the producer-assembled fact. - section = build_runtime_section( - env, {"id": "t5", "type": "task", "metadata": {"client_surface": {"channel": "cli"}}} - ) - data = json.loads(section.split("## Runtime context\n\n", 1)[1]) - assert data["owner_client"] == {"channel": "cli"} - - def test_owner_client_never_inferred_from_metadata_source(self, tmp_path): - from ouroboros.context import build_runtime_section - - env = self._make_env(tmp_path) - # metadata.source is OVERLOADED (scheduler writes scheduled_task / - # skill_scheduled_task): the renderer must never dress it up as an - # owner surface — no producer stamp, no fact (codex scope round 2 N1). - for source in ("cli", "scheduled_task", "skill_scheduled_task", "web"): - section = build_runtime_section( - env, {"id": "t6", "type": "task", "metadata": {"source": source}} - ) - data = json.loads(section.split("## Runtime context\n\n", 1)[1]) - assert "owner_client" not in data, f"source={source!r} must not render" - # Internal producers use top-level task["source"], never rendered. - section = build_runtime_section( - env, {"id": "t7", "type": "task", "source": "promote_chat_to_task"} - ) - data = json.loads(section.split("## Runtime context\n\n", 1)[1]) - assert "owner_client" not in data - - -# =========================================================================== -# Memory / consolidation offset behavior (merged from former -# test_context_memory_overhaul.py). Inspect-only `limit=50` / `limit=1000` -# source-string pins were dropped — behavioral coverage below already -# exercises the offset path. test_no_identity_truncation_in_consolidator_ -# prompts was also dropped (inspect-only); identity-truncation is covered -# behaviorally by consolidator tests. -# =========================================================================== - - -def test_recent_chat_starts_after_consolidated_offset(tmp_path): - from ouroboros.context import build_recent_sections - from ouroboros.memory import Memory - - logs_dir = tmp_path / "logs" - memory_dir = tmp_path / "memory" - logs_dir.mkdir(parents=True, exist_ok=True) - memory_dir.mkdir(parents=True, exist_ok=True) - entries = [ - {"ts": f"2026-03-19T16:{i:02d}:00Z", "direction": "in", "username": "User", "text": f"msg-{i}"} - for i in range(5) - ] - (logs_dir / "chat.jsonl").write_text( - "\n".join(json.dumps(entry) for entry in entries) + "\n", + (repo_dir / "prompts" / "CONSCIOUSNESS.md").write_text("Consciousness prompt", encoding="utf-8") + (repo_dir / "BIBLE.md").write_text("Bible", encoding="utf-8") + (repo_dir / "VERSION").write_text("1.2.3", encoding="utf-8") + (repo_dir / "pyproject.toml").write_text('version = "1.2.3"', encoding="utf-8") + (repo_dir / "README.md").write_text("README", encoding="utf-8") + (repo_dir / "docs" / "ARCHITECTURE.md").write_text("# Ouroboros v1.2.3", encoding="utf-8") + (repo_dir / "docs" / "DEVELOPMENT.md").write_text( + "### File Size Budgets\n| Path | Budget chars |\n|------|--------------|\n| memory/identity.md | 1000 |\n", encoding="utf-8", ) - memory = Memory(drive_root=tmp_path) - (memory_dir / "dialogue_meta.json").write_text( - json.dumps({ - "last_consolidated_offset": 3, - "chat_log_signature": memory.jsonl_generation_signature("chat.jsonl"), - }), - encoding="utf-8", + (drive_root / "state" / "state.json").write_text('{"spent_usd": 0, "budget_drift_alert": false}', encoding="utf-8") + (drive_root / "memory" / "identity.md").write_text("x" * 950, encoding="utf-8") + (drive_root / "memory" / "scratchpad.md").write_text("scratchpad", encoding="utf-8") + (drive_root / "logs" / "chat.jsonl").write_text("", encoding="utf-8") + (drive_root / "logs" / "progress.jsonl").write_text("", encoding="utf-8") + (drive_root / "logs" / "tools.jsonl").write_text("", encoding="utf-8") + (drive_root / "logs" / "events.jsonl").write_text("", encoding="utf-8") + (drive_root / "logs" / "supervisor.jsonl").write_text("", encoding="utf-8") + (drive_root / "logs" / "task_reflections.jsonl").write_text("", encoding="utf-8") + + bg = BackgroundConsciousness( + drive_root=drive_root, + repo_dir=repo_dir, + event_queue=None, + owner_chat_id_fn=lambda: None, ) - sections = build_recent_sections(memory, env=None) - combined = "\n\n".join(sections) - - assert "msg-0" not in combined - assert "msg-1" not in combined - assert "msg-2" not in combined - assert "msg-3" in combined - assert "msg-4" in combined - - -def test_recent_chat_main_includes_all_threads_full_awareness(tmp_path): - """Full project awareness (v6.32.0): the one identity's main/global context - sees its WHOLE conversation — main + project threads alike (BIBLE P1, one - awareness across direct chat, project rooms, and consciousness). Project chat - is part of the one mind's memory, NOT partitioned out; only A2A virtual - transport is excluded (covered elsewhere).""" - from ouroboros.context import build_recent_sections - from ouroboros.memory import Memory - from ouroboros.projects_registry import create_project - - logs_dir = tmp_path / "logs" - logs_dir.mkdir(parents=True, exist_ok=True) - - project = create_project(tmp_path, "racer") - project_chat = int(project["chat_id"]) - transport_chat = 555000111 # large NON-project id (e.g. a Telegram mirror) - - entries = [ - {"chat_id": 1, "direction": "in", "username": "User", "text": "main-keep"}, - {"chat_id": project_chat, "direction": "in", "username": "User", "text": "project-visible"}, - {"chat_id": transport_chat, "direction": "in", "username": "User", "text": "transport-keep"}, - {"direction": "in", "username": "User", "text": "legacy-keep"}, # no chat_id -> main - ] - (logs_dir / "chat.jsonl").write_text( - "\n".join(json.dumps(entry) for entry in entries) + "\n", - encoding="utf-8", - ) - - combined = "\n\n".join(build_recent_sections(Memory(drive_root=tmp_path), env=None)) - - assert "main-keep" in combined - assert "legacy-keep" in combined - assert "transport-keep" in combined - assert "project-visible" in combined # full awareness: the one mind sees project chat - - -def test_recent_chat_for_project_thread_shows_only_its_own_thread(tmp_path): - """A project TASK gets a FOCUSED working view of its own thread (full - awareness, v6.32.0): its "## Recent chat" is its own project thread, not the - штаб's main chat nor a sibling project's chat, so cross-project noise does not - bloat its working context. This is focus, not memory isolation — the one mind - still sees everything via the main/background path. Pins that thread_chat_id - selects the project's own raw tail rather than the main consolidation stream.""" - from ouroboros.context import build_recent_sections - from ouroboros.memory import Memory - from ouroboros.projects_registry import create_project - - logs_dir = tmp_path / "logs" - logs_dir.mkdir(parents=True, exist_ok=True) - - proj_a = create_project(tmp_path, "racer") - proj_b = create_project(tmp_path, "research") - chat_a = int(proj_a["chat_id"]) - chat_b = int(proj_b["chat_id"]) - - entries = [ - {"chat_id": 1, "direction": "in", "username": "User", "text": "main-stab-chat"}, - {"chat_id": chat_a, "direction": "in", "username": "User", "text": "project-a-own-thread"}, - {"chat_id": chat_b, "direction": "in", "username": "User", "text": "project-b-sibling"}, - ] - (logs_dir / "chat.jsonl").write_text( - "\n".join(json.dumps(entry) for entry in entries) + "\n", - encoding="utf-8", - ) - - combined = "\n\n".join(build_recent_sections( - Memory(drive_root=tmp_path), env=None, thread_chat_id=chat_a)) - - assert "project-a-own-thread" in combined # its own thread is visible - assert "project-b-sibling" not in combined # sibling project not in focused view - assert "main-stab-chat" not in combined # main chat not in focused project view + text = bg._build_context() + assert text.index("## Health Invariants") < text.index("## Drive state") def test_project_recent_chat_filters_archives_before_recent_bound(tmp_path, monkeypatch): @@ -1305,776 +610,3 @@ def test_missing_cursor_generation_hot_path_never_replays_full_archive(tmp_path, assert coverage["omitted_matching_rows_unknown"] is True assert coverage["gaps"][0]["kind"] == "consolidation_cursor_generation_missing" assert coverage["reader"] == "chat_history(count, offset, search)" - - -def test_project_workpad_and_journal_not_silently_sliced(tmp_path, monkeypatch): - """BIBLE P1 (no silent truncation): project cognitive artifacts are not - prefix-sliced into context. The workpad rides in FULL; journal milestones show - full text (no per-row [:N]) with a visible journal_read pointer for older.""" - import types - - monkeypatch.setattr("ouroboros.config.DATA_DIR", tmp_path) - from ouroboros.context import build_knowledge_sections - from ouroboros.project_facts import project_journal_path, project_workpad_path - from ouroboros.utils import append_jsonl - - pid = "builder" - wp = project_workpad_path(pid) - wp.parent.mkdir(parents=True, exist_ok=True) - tail = "WORKPAD_TAIL_MARKER" - wp.write_text("A" * 20_000 + tail, encoding="utf-8") # > old 12_000 slice - append_jsonl(project_journal_path(pid), { - "ts": "2026-06-14T00:00:00Z", "kind": "checkpoint", "text": "M" * 600, # > old 200 slice - }) - - env = types.SimpleNamespace(drive_path=lambda rel: tmp_path / rel) - combined = "\n\n".join(build_knowledge_sections(env, project_id=pid)) - - assert tail in combined # full workpad, not prefix-sliced to 12_000 - assert ("M" * 600) in combined # full journal milestone, not sliced to 200 - - -def test_append_journal_milestone_bounds_over_limit_with_pointer(tmp_path, monkeypatch): - """An AUTOMATIC completion milestone honors the journal's durable per-row cap: - over-limit text is bounded with a VISIBLE pointer (recorded, never silently - sliced nor dropped) — same _MAX_TEXT_CHARS contract as the journal_write tool, - so emit_task_results cannot append a raw unbounded row.""" - monkeypatch.setattr("ouroboros.config.DATA_DIR", tmp_path) - from ouroboros.project_facts import project_journal_path - from ouroboros.tools.project_journal import _MAX_TEXT_CHARS, append_journal_milestone - from ouroboros.utils import iter_jsonl_objects - - pid = "lh" - append_journal_milestone(pid, "done", "Z" * (_MAX_TEXT_CHARS + 500), task_id="t1") - rows = [r for r in iter_jsonl_objects(project_journal_path(pid)) if isinstance(r, dict)] - assert len(rows) == 1 # recorded (not dropped/rejected) - txt = rows[0]["text"] - assert len(txt) <= _MAX_TEXT_CHARS # honors the durable per-row contract - assert "task_results" in txt # VISIBLE pointer to the full text - - -def test_low_mode_preserves_full_unconsolidated_dialogue_suffix(tmp_path, monkeypatch): - from ouroboros.context import build_recent_sections - from ouroboros.memory import Memory - - logs_dir = tmp_path / "logs" - memory_dir = tmp_path / "memory" - logs_dir.mkdir(parents=True, exist_ok=True) - memory_dir.mkdir(parents=True, exist_ok=True) - fresh_count = 305 - entries = [ - {"chat_id": 1, "direction": "in", "username": "User", "text": f"consolidated-{i}"} - for i in range(3) - ] + [ - {"chat_id": 1, "direction": "in", "username": "User", "text": f"fresh-{i}"} - for i in range(fresh_count) - ] - (logs_dir / "chat.jsonl").write_text( - "\n".join(json.dumps(entry) for entry in entries) + "\n", - encoding="utf-8", - ) - memory = Memory(drive_root=tmp_path) - (memory_dir / "dialogue_meta.json").write_text( - json.dumps({ - "last_consolidated_offset": 3, - "chat_log_signature": memory.jsonl_generation_signature("chat.jsonl"), - }), - encoding="utf-8", - ) - monkeypatch.setenv("OUROBOROS_CONTEXT_MODE", "low") - - combined = "\n\n".join(build_recent_sections(memory, env=None)) - - assert "consolidated-0" not in combined - assert "fresh-0" in combined - assert f"fresh-{fresh_count - 1}" in combined - - -def test_low_mode_without_consolidation_keeps_max_raw_dialogue_tail(tmp_path, monkeypatch): - from ouroboros.context import build_recent_sections - from ouroboros.context_budget import MAX_RECENT_CHAT_TAIL - from ouroboros.memory import Memory - - logs_dir = tmp_path / "logs" - logs_dir.mkdir(parents=True, exist_ok=True) - fresh_count = 305 - entries = [ - {"chat_id": 1, "direction": "in", "username": "User", "text": f"fresh-{i}"} - for i in range(fresh_count) - ] - (logs_dir / "chat.jsonl").write_text( - "\n".join(json.dumps(entry) for entry in entries) + "\n", - encoding="utf-8", - ) - monkeypatch.setenv("OUROBOROS_CONTEXT_MODE", "low") - - combined = "\n\n".join(build_recent_sections(Memory(drive_root=tmp_path), env=None)) - - assert fresh_count < MAX_RECENT_CHAT_TAIL - assert "fresh-0" in combined - assert f"fresh-{fresh_count - 1}" in combined - - -def test_recent_chat_offset_uses_filtered_dialogue_entries(tmp_path): - from ouroboros.context import build_recent_sections - from ouroboros.memory import Memory - - logs_dir = tmp_path / "logs" - memory_dir = tmp_path / "memory" - logs_dir.mkdir(parents=True, exist_ok=True) - memory_dir.mkdir(parents=True, exist_ok=True) - entries = [ - {"chat_id": 1, "direction": "in", "username": "User", "text": "consolidated-0"}, - {"chat_id": -1, "direction": "in", "username": "Agent", "text": "a2a-noise"}, - {"chat_id": 1, "direction": "in", "username": "User", "text": "consolidated-1"}, - {"chat_id": 1, "direction": "in", "username": "User", "text": "fresh"}, - ] - (logs_dir / "chat.jsonl").write_text( - "\n".join(json.dumps(entry) for entry in entries) + "\n", - encoding="utf-8", - ) - memory = Memory(drive_root=tmp_path) - (memory_dir / "dialogue_meta.json").write_text( - json.dumps({ - "last_consolidated_offset": 2, - "chat_log_signature": memory.jsonl_generation_signature("chat.jsonl"), - }), - encoding="utf-8", - ) - - combined = "\n\n".join(build_recent_sections(memory, env=None)) - - assert "consolidated-0" not in combined - assert "consolidated-1" not in combined - assert "a2a-noise" not in combined - assert "fresh" in combined - - -def test_recent_chat_ignores_stale_consolidation_offset_after_rotation(tmp_path): - from ouroboros.context import build_recent_sections - from ouroboros.memory import Memory - - logs_dir = tmp_path / "logs" - memory_dir = tmp_path / "memory" - logs_dir.mkdir(parents=True, exist_ok=True) - memory_dir.mkdir(parents=True, exist_ok=True) - initial = [ - {"chat_id": 1, "direction": "in", "username": "User", "text": f"early-{i}"} - for i in range(3) - ] - (logs_dir / "chat.jsonl").write_text( - "\n".join(json.dumps(entry) for entry in initial) + "\n", - encoding="utf-8", - ) - memory = Memory(drive_root=tmp_path) - stale_signature = memory.jsonl_generation_signature("chat.jsonl") - (memory_dir / "dialogue_meta.json").write_text( - json.dumps({ - "last_consolidated_offset": 3, - "chat_log_signature": stale_signature, - }), - encoding="utf-8", - ) - - rotated = [ - {"chat_id": 1, "direction": "in", "username": "User", "text": f"post-rotate-{i}"} - for i in range(2) - ] - (logs_dir / "chat.jsonl").write_text( - "\n".join(json.dumps(entry) for entry in rotated) + "\n", - encoding="utf-8", - ) - - combined = "\n\n".join(build_recent_sections(memory, env=None)) - - # Rotation invalidates the stale offset; rotated entries appear. - assert "post-rotate-0" in combined - assert "post-rotate-1" in combined - - -def test_recent_chat_keeps_offset_when_same_log_gets_appended(tmp_path): - from ouroboros.context import build_recent_sections - from ouroboros.memory import Memory - - logs_dir = tmp_path / "logs" - memory_dir = tmp_path / "memory" - logs_dir.mkdir(parents=True, exist_ok=True) - memory_dir.mkdir(parents=True, exist_ok=True) - initial = [ - {"chat_id": 1, "direction": "in", "username": "User", "text": f"old-{i}"} - for i in range(3) - ] - (logs_dir / "chat.jsonl").write_text( - "\n".join(json.dumps(entry) for entry in initial) + "\n", - encoding="utf-8", - ) - memory = Memory(drive_root=tmp_path) - (memory_dir / "dialogue_meta.json").write_text( - json.dumps({ - "last_consolidated_offset": 3, - "chat_log_signature": memory.jsonl_generation_signature("chat.jsonl"), - }), - encoding="utf-8", - ) - - with open(logs_dir / "chat.jsonl", "a", encoding="utf-8") as handle: - handle.write(json.dumps({"chat_id": 1, "direction": "in", "username": "User", "text": "new"}) + "\n") - - combined = "\n\n".join(build_recent_sections(memory, env=None)) - - assert "old-0" not in combined - assert "new" in combined - - -def test_world_profile_is_loaded_with_stable_memory(tmp_path): - from ouroboros.context import build_memory_sections - from ouroboros.memory import Memory - - (tmp_path / "memory").mkdir(parents=True, exist_ok=True) - (tmp_path / "memory" / "WORLD.md").write_text("world-profile-data", encoding="utf-8") - memory = Memory(drive_root=tmp_path) - - sections = build_memory_sections(memory) - combined = "\n\n".join(sections) - - assert "world-profile-data" in combined - - -def test_retired_dialogue_summary_remains_visible_when_blocks_exist(tmp_path): - from ouroboros.context import build_memory_sections - from ouroboros.memory import Memory - - memory_dir = tmp_path / "memory" - memory_dir.mkdir(parents=True, exist_ok=True) - (memory_dir / "dialogue_summary.md").write_text("legacy dialogue", encoding="utf-8") - (memory_dir / "dialogue_blocks.json").write_text( - json.dumps([{"content": "new dialogue block"}]), - encoding="utf-8", - ) - memory = Memory(drive_root=tmp_path) - - combined = "\n\n".join(build_memory_sections(memory, partition="volatile")) - - assert "## Dialogue History" in combined - assert "new dialogue block" in combined - assert "## Legacy Dialogue Summary (retired flat format, read-only fallback)" in combined - assert "legacy dialogue" in combined - - -def test_retired_dialogue_summary_fallback_preserves_continuity_without_blocks(tmp_path): - from ouroboros.context import build_memory_sections - from ouroboros.memory import Memory - - memory_dir = tmp_path / "memory" - memory_dir.mkdir(parents=True, exist_ok=True) - (memory_dir / "dialogue_summary.md").write_text("legacy dialogue only", encoding="utf-8") - memory = Memory(drive_root=tmp_path) - - combined = "\n\n".join(build_memory_sections(memory, partition="volatile")) - - assert "## Legacy Dialogue Summary (retired flat format, read-only fallback)" in combined - assert "legacy dialogue only" in combined - - -def test_recent_sections_filter_process_logs_by_task_id(tmp_path): - from ouroboros.context import build_recent_sections - from ouroboros.memory import Memory - - logs_dir = tmp_path / "logs" - logs_dir.mkdir(parents=True, exist_ok=True) - (logs_dir / "progress.jsonl").write_text( - "\n".join([ - json.dumps({"task_id": "task-a", "text": "in-scope"}), - json.dumps({"task_id": "task-b", "text": "out-of-scope"}), - ]) + "\n", - encoding="utf-8", - ) - (logs_dir / "tools.jsonl").write_text( - "\n".join([ - json.dumps({"task_id": "task-a", "tool": "shell"}), - json.dumps({"task_id": "task-b", "tool": "shell"}), - ]) + "\n", - encoding="utf-8", - ) - - memory = Memory(drive_root=tmp_path) - sections = build_recent_sections(memory, env=None, task_id="task-a") - combined = "\n\n".join(sections) - assert "in-scope" in combined - assert "out-of-scope" not in combined - - -def test_installed_skills_section_includes_warnings_verdict(tmp_path, monkeypatch): - from ouroboros.context import _build_installed_skills_section - - class FakeEnv: - drive_root = tmp_path - - monkeypatch.setattr( - "ouroboros.skill_loader.summarize_skills", - lambda _root: { - "skills": [ - { - "name": "weather", - "type": "script", - "enabled": True, - "review_status": "warnings", - "executable_review": True, - "review_stale": False, - "description": "Weather helper", - } - ] - }, - ) - - section = _build_installed_skills_section(FakeEnv()) - - assert "## Installed Skills" in section - assert "weather" in section - assert "warnings" in section - - -def test_health_invariants_come_first_in_dynamic_context(tmp_path): - from ouroboros.context import build_llm_messages - from ouroboros.memory import Memory - - class FakeEnv: - def drive_path(self, p): - return tmp_path / p - - def repo_path(self, p): - return tmp_path / "repo" / p - - @property - def repo_dir(self): - return tmp_path / "repo" - - @property - def drive_root(self): - return tmp_path - - (tmp_path / "repo" / "prompts").mkdir(parents=True, exist_ok=True) - (tmp_path / "repo" / "docs").mkdir(parents=True, exist_ok=True) - (tmp_path / "memory").mkdir(parents=True, exist_ok=True) - (tmp_path / "logs").mkdir(parents=True, exist_ok=True) - (tmp_path / "state").mkdir(parents=True, exist_ok=True) - - (tmp_path / "repo" / "prompts" / "SYSTEM.md").write_text("System prompt", encoding="utf-8") - (tmp_path / "repo" / "BIBLE.md").write_text("Bible", encoding="utf-8") - (tmp_path / "repo" / "README.md").write_text("README", encoding="utf-8") - (tmp_path / "repo" / "docs" / "ARCHITECTURE.md").write_text("# Ouroboros v1.2.3", encoding="utf-8") - (tmp_path / "repo" / "docs" / "DEVELOPMENT.md").write_text( - "### File Size Budgets\n| Path | Budget chars |\n|------|--------------|\n| memory/identity.md | 1000 |\n", - encoding="utf-8", - ) - (tmp_path / "repo" / "docs" / "CHECKLISTS.md").write_text("Checklist", encoding="utf-8") - (tmp_path / "repo" / "VERSION").write_text("1.2.3", encoding="utf-8") - (tmp_path / "repo" / "pyproject.toml").write_text('version = "1.2.3"', encoding="utf-8") - (tmp_path / "state" / "state.json").write_text('{"spent_usd": 0, "budget_drift_alert": false}', encoding="utf-8") - (tmp_path / "memory" / "identity.md").write_text("x" * 950, encoding="utf-8") - (tmp_path / "memory" / "scratchpad.md").write_text("scratchpad", encoding="utf-8") - - messages, _cap_info = build_llm_messages( - env=FakeEnv(), - memory=Memory(drive_root=tmp_path), - task={"id": "task-a", "type": "task", "text": "hello"}, - ) - - dynamic_text = messages[0]["content"][2]["text"] - assert dynamic_text.startswith("## Health Invariants") - assert dynamic_text.index("## Health Invariants") < dynamic_text.index("## Drive state") - - -def test_health_invariants_come_first_in_background_consciousness_context(tmp_path): - from ouroboros.consciousness import BackgroundConsciousness - - repo_dir = tmp_path / "repo" - drive_root = tmp_path / "drive" - (repo_dir / "prompts").mkdir(parents=True, exist_ok=True) - (repo_dir / "docs").mkdir(parents=True, exist_ok=True) - (drive_root / "memory" / "knowledge").mkdir(parents=True, exist_ok=True) - (drive_root / "logs").mkdir(parents=True, exist_ok=True) - (drive_root / "state").mkdir(parents=True, exist_ok=True) - - (repo_dir / "prompts" / "CONSCIOUSNESS.md").write_text("Consciousness prompt", encoding="utf-8") - (repo_dir / "BIBLE.md").write_text("Bible", encoding="utf-8") - (repo_dir / "VERSION").write_text("1.2.3", encoding="utf-8") - (repo_dir / "pyproject.toml").write_text('version = "1.2.3"', encoding="utf-8") - (repo_dir / "README.md").write_text("README", encoding="utf-8") - (repo_dir / "docs" / "ARCHITECTURE.md").write_text("# Ouroboros v1.2.3", encoding="utf-8") - (repo_dir / "docs" / "DEVELOPMENT.md").write_text( - "### File Size Budgets\n| Path | Budget chars |\n|------|--------------|\n| memory/identity.md | 1000 |\n", - encoding="utf-8", - ) - (drive_root / "state" / "state.json").write_text('{"spent_usd": 0, "budget_drift_alert": false}', encoding="utf-8") - (drive_root / "memory" / "identity.md").write_text("x" * 950, encoding="utf-8") - (drive_root / "memory" / "scratchpad.md").write_text("scratchpad", encoding="utf-8") - (drive_root / "logs" / "chat.jsonl").write_text("", encoding="utf-8") - (drive_root / "logs" / "progress.jsonl").write_text("", encoding="utf-8") - (drive_root / "logs" / "tools.jsonl").write_text("", encoding="utf-8") - (drive_root / "logs" / "events.jsonl").write_text("", encoding="utf-8") - (drive_root / "logs" / "supervisor.jsonl").write_text("", encoding="utf-8") - (drive_root / "logs" / "task_reflections.jsonl").write_text("", encoding="utf-8") - - bg = BackgroundConsciousness( - drive_root=drive_root, - repo_dir=repo_dir, - event_queue=None, - owner_chat_id_fn=lambda: None, - ) - - text = bg._build_context() - assert text.index("## Health Invariants") < text.index("## Drive state") - - -def test_drive_state_section_is_typed_projection_with_pointer(tmp_path): - """W3 adjacent (a): the Drive state section projects the fields the agent - reasons about and NAMES the omitted internal caches with an on-demand - pointer (P1: disclosed omission) instead of dumping state.json wholesale — - the budget narrative stays with the usage-accounting authority in the - Runtime section.""" - import json - - from ouroboros.context import _drive_state_section - - class FakeEnv: - def drive_path(self, p): - return tmp_path / p - - (tmp_path / "state").mkdir(parents=True, exist_ok=True) - (tmp_path / "state" / "state.json").write_text(json.dumps({ - "session_id": "abc123", - "current_branch": "ouroboros", - "evolution_mode_enabled": False, - "budget_drift_alert": True, - "budget_drift_pct": 48.05, - "spent_usd": 1699.3, - "managed_update_cache": {"latest_sha": "x" * 40, "latest_message": "big"}, - "usage_accounting": {"settled_usd": 1633.1}, - "openrouter_last_check_call": 5750, - }), encoding="utf-8") - - section = _drive_state_section(FakeEnv()) - - assert section.startswith("## Drive state") - assert '"session_id": "abc123"' in section - assert '"budget_drift_alert": true' in section - # Internal caches / duplicated spend narrative are OMITTED but NAMED. - assert '"managed_update_cache"' not in section - assert '"usage_accounting"' not in section - assert '"spent_usd"' not in section - for named in ("managed_update_cache", "usage_accounting", "spent_usd", "openrouter_last_check_call"): - assert named in section # named in the omission note - assert "read_file(root='runtime_data', path='state/state.json')" in section - - # Missing/empty file: still a valid section, no omission note needed. - (tmp_path / "state" / "state.json").unlink() - empty = _drive_state_section(FakeEnv()) - assert empty.startswith("## Drive state") - assert "read_file" not in empty - - -def test_review_ledger_caps_runs_and_attempts_with_omission_notes(tmp_path): - """W3 adjacent (b): the historical review ledger rides into EVERY task's - context — cap runs/attempts at the 5 most recent with EXPLICIT omission - notes (the continuation pattern) and truncate commit messages; the full - ledger stays behind review_status.""" - from ouroboros.review_state import ( - AdvisoryReviewState, - AdvisoryRunRecord, - CommitAttemptRecord, - format_status_section, - ) - - state = AdvisoryReviewState() - long_msg = "feat: " + ("y" * 2000) - for i in range(8): - state.add_run(AdvisoryRunRecord( - snapshot_hash=f"hash{i:04d}00000000", - commit_message=long_msg if i == 7 else f"commit {i}", - status="fresh", - ts=f"2026-01-0{i + 1}T00:00:00", - )) - for i in range(8): - state.record_attempt(CommitAttemptRecord( - status="succeeded", - commit_message=f"attempt commit {i}", - ts=f"2026-01-0{i + 1}T01:00:00", - attempt=i + 1, - )) - - section = format_status_section(state) - - assert "3 older advisory run(s) omitted" in section - assert "3 older attempt(s) omitted" in section - assert "review_status" in section - assert "hash0007" in section # newest kept - assert "hash0000" not in section # oldest omitted - assert "attempt commit 7" in section - assert "attempt commit 0" not in section - # The 2000-char commit message is display-truncated with the explicit notice. - assert "y" * 2000 not in section - assert "truncated at 300 chars" in section - - -def test_settled_continuations_retire_after_age_window(tmp_path): - """W3 adjacent (b): a continuation whose owning task SETTLED and that sat - un-resumed past the age window is archived (durable move, never deleted); - fresh settled records stay — they are the designed cross-task resume - pointer.""" - from ouroboros.task_continuation import ( - ReviewContinuation, - archived_continuation_dir, - continuation_path, - list_review_continuations, - retire_settled_continuations, - save_review_continuation, - ) - - old = save_review_continuation(tmp_path, ReviewContinuation( - task_id="oldtask", source="commit_blocked", stage="review")) - # Age the record past the window (rewrite the stored timestamps). - import json as _json - path = continuation_path(tmp_path, "oldtask") - data = _json.loads(path.read_text(encoding="utf-8")) - data["created_ts"] = data["updated_ts"] = "2026-01-01T00:00:00+00:00" - path.write_text(_json.dumps(data), encoding="utf-8") - - save_review_continuation(tmp_path, ReviewContinuation( - task_id="freshtask", source="commit_blocked", stage="review")) - - settled = {"oldtask": True, "freshtask": True, "runningtask": False} - retired = retire_settled_continuations(tmp_path, is_settled=lambda tid: settled.get(tid, False)) - - assert retired == ["oldtask"] - assert not continuation_path(tmp_path, "oldtask").exists() - assert (archived_continuation_dir(tmp_path) / "oldtask.json").exists() # durable, not deleted - remaining, _corrupt = list_review_continuations(tmp_path) - assert [c.task_id for c in remaining] == ["freshtask"] - - # An old continuation of a NON-settled task stays put. - save_review_continuation(tmp_path, ReviewContinuation( - task_id="runningtask", source="commit_blocked", stage="review")) - path = continuation_path(tmp_path, "runningtask") - data = _json.loads(path.read_text(encoding="utf-8")) - data["created_ts"] = data["updated_ts"] = "2026-01-01T00:00:00+00:00" - path.write_text(_json.dumps(data), encoding="utf-8") - assert retire_settled_continuations(tmp_path, is_settled=lambda tid: settled.get(tid, False)) == [] - assert continuation_path(tmp_path, "runningtask").exists() - assert old.task_id == "oldtask" - - -def test_settled_continuation_with_open_obligations_survives_age_retirement(tmp_path): - """A settled FAILED task whose continuation records obligations that are - STILL open in the review ledger is genuinely unresolved review work: age - must not archive it out of context (P1/P3). A same-age settled sibling with - no open markers still retires — the noise-reduction path stays.""" - import json as _json - - from ouroboros.agent_task_pipeline import build_review_context - from ouroboros.review_state import ( - AdvisoryReviewState, - ObligationItem, - make_repo_key, - save_state, - ) - from ouroboros.task_continuation import ( - ReviewContinuation, - archived_continuation_dir, - continuation_path, - save_review_continuation, - ) - - class FakeEnv: - def drive_path(self, p): - return tmp_path / p - - def repo_path(self, p): - return tmp_path / "repo" / p - - @property - def repo_dir(self): - return tmp_path / "repo" - - @property - def drive_root(self): - return tmp_path - - env = FakeEnv() - (tmp_path / "repo" / ".git").mkdir(parents=True, exist_ok=True) - (tmp_path / "repo" / "tracked.py").write_text("print('hi')\n", encoding="utf-8") - repo_key = make_repo_key(tmp_path / "repo") - - def _aged_continuation(task_id, obligation_ids): - save_review_continuation(tmp_path, ReviewContinuation( - task_id=task_id, source="commit_blocked", stage="review", - block_reason="critical_findings", obligation_ids=obligation_ids)) - path = continuation_path(tmp_path, task_id) - data = _json.loads(path.read_text(encoding="utf-8")) - data["created_ts"] = data["updated_ts"] = "2026-01-01T00:00:00+00:00" - path.write_text(_json.dumps(data), encoding="utf-8") - - _aged_continuation("unresolvedtask", ["obl-open-1"]) - _aged_continuation("closedtask", ["obl-long-gone"]) - task_results = tmp_path / "task_results" - task_results.mkdir(parents=True, exist_ok=True) - for tid in ("unresolvedtask", "closedtask"): - (task_results / f"{tid}.json").write_text( - _json.dumps({"id": tid, "status": "failed"}), encoding="utf-8") - - state = AdvisoryReviewState(open_obligations=[ - ObligationItem( - obligation_id="obl-open-1", - item="tests_affected", - severity="critical", - reason="Coverage still missing", - source_attempt_ts="2026-01-01T00:00:00+00:00", - source_attempt_msg="blocked commit", - repo_key=repo_key, - fingerprint="finding:tests_affected:abc123", - ) - ]) - save_state(tmp_path, state) - - dynamic_text = build_review_context(env) - - # Unresolved work survives the age window and stays in cognitive context. - assert continuation_path(tmp_path, "unresolvedtask").exists() - assert "task=unresolvedtask" in dynamic_text - # The provably-closed sibling still rides the age path (durable, disclosed). - assert not continuation_path(tmp_path, "closedtask").exists() - assert (archived_continuation_dir(tmp_path) / "closedtask.json").exists() - assert "closedtask" in dynamic_text # transient archive disclosure line - - -# --------------------------------------------------------------------------- -# B4-lite: capabilities["delegation"] — honestly-labeled HISTORICAL -# observations only (never saved intent or live health). -# --------------------------------------------------------------------------- - - -def _delegation_data_root(tmp_path, monkeypatch): - root = tmp_path / "delegation_data_root" - (root / "state").mkdir(parents=True, exist_ok=True) - monkeypatch.setattr("ouroboros.config.DATA_DIR", root) - return root - - -def _delegation_fact(tmp_path, monkeypatch): - env = _make_health_env(tmp_path) - monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "advanced") - section = build_runtime_section(env, {"id": "task-1", "type": "task"}) - payload = json.loads(section.split("\n\n", 1)[1]) - return payload["capabilities"] - - -def test_delegation_fact_carries_historical_rows_and_profile_evidence(tmp_path, monkeypatch): - root = _delegation_data_root(tmp_path, monkeypatch) - monkeypatch.setenv("OUROBOROS_SUBAGENT_HARNESS", "claudexor=opus-5:high") - (root / "state" / "reviewer_slot_last_execution.json").write_text(json.dumps({ - "triad_1": { - "ts": "2026-08-18T01:02:03+00:00", - "surface": "triad", - "status": "ok", - "requested": {"profile_id": "requested-review-profile"}, - "effective": { - "route": "agent_session:claudexor", - "model": "opus-5", - "profile_id": "applied-review-profile", - }, - }, - "triad_2": { - "ts": "2026-08-18T01:02:04+00:00", - "surface": "triad", - "status": "error", - # B1 typed facts: a dated window carries reset_at, an undated one - # only the code — both must surface independently. - "failure_code": "subscription_window_exhausted", - "reset_at": "2026-08-18T09:20:00+00:00", - }, - }), encoding="utf-8") - (root / "state" / "subagent_last_delegation.json").write_text(json.dumps({ - "ts": "2026-08-18T02:00:00+00:00", - "route": "claudexor", - "requested_model": "opus-5", - "applied_model": "claude-opus-5", - "requested_profile": "requested-delegate-profile", - "applied_profile": "applied-delegate-profile", - "selected_subagent_id": "builder", - "run_id": "run-1", - }), encoding="utf-8") - - capabilities = _delegation_fact(tmp_path, monkeypatch) - delegation = capabilities["delegation"] - - assert "configured_route" not in delegation - rows = {row["slot"]: row for row in delegation["reviewer_slots_last"]} - assert rows["triad_1"]["outcome"] == "ok" - assert rows["triad_1"]["requested_profile"] == "requested-review-profile" - assert rows["triad_1"]["applied_profile"] == "applied-review-profile" - assert "failure_code" not in rows["triad_1"] - assert rows["triad_2"]["outcome"] == "failed" - assert rows["triad_2"]["failure_code"] == "subscription_window_exhausted" - assert rows["triad_2"]["reset_at"] == "2026-08-18T09:20:00+00:00" - # Per-row label is the timestamp only; the verbatim historical disclaimer - # lives ONCE in the note (review fix 12), never repeated per row. - assert rows["triad_1"]["observed"] == "last observed at 2026-08-18T01:02:03+00:00" - last = delegation["subagent_last_delegation"] - assert last["route"] == "claudexor" - assert last["applied_model"] == "claude-opus-5" - assert last["requested_profile"] == "requested-delegate-profile" - assert last["applied_profile"] == "applied-delegate-profile" - assert last["selected_subagent_id"] == "builder" - assert last["observed"] == "last observed at 2026-08-18T02:00:00+00:00" - assert "historical" not in rows["triad_1"]["observed"] - # The prompt-visible note teaches the semantics ONCE: rows are history, live - # facts come from plan-review waves and typed delegate refusals. - assert "historical, not live health" in delegation["note"] - assert "plan-review wave rows" in delegation["note"] - assert "typed" in delegation["note"] and "refusal" in delegation["note"] - assert "never healthy" in delegation["note"] - - -def test_delegation_fact_undated_window_code_surfaces_without_reset(tmp_path, monkeypatch): - root = _delegation_data_root(tmp_path, monkeypatch) - monkeypatch.delenv("OUROBOROS_SUBAGENT_HARNESS", raising=False) - (root / "state" / "reviewer_slot_last_execution.json").write_text(json.dumps({ - "scope": { - "ts": "2026-08-18T03:00:00+00:00", - "status": "error", - "failure_code": "credential_pool_exhausted", - }, - }), encoding="utf-8") - - delegation = _delegation_fact(tmp_path, monkeypatch)["delegation"] - - (row,) = delegation["reviewer_slots_last"] - assert row["failure_code"] == "credential_pool_exhausted" - assert "reset_at" not in row - assert row["outcome"] == "failed" - - -def test_delegation_fact_absent_files_mean_absent_observations_not_health(tmp_path, monkeypatch): - _delegation_data_root(tmp_path, monkeypatch) - monkeypatch.delenv("OUROBOROS_SUBAGENT_HARNESS", raising=False) - - capabilities = _delegation_fact(tmp_path, monkeypatch) - - assert "delegation" not in capabilities - - -def test_delegation_fact_failure_never_drops_capability_digest(tmp_path, monkeypatch): - _delegation_data_root(tmp_path, monkeypatch) - - def _boom(): - raise RuntimeError("reader exploded") - - monkeypatch.setattr( - "ouroboros.reviewer_slot_config.reviewer_slot_last_executions", _boom) - - capabilities = _delegation_fact(tmp_path, monkeypatch) - - assert "delegation" not in capabilities - # The surrounding digest survives intact. - assert "allow_mutative_subagents" in capabilities - assert "write_surfaces" in capabilities diff --git a/tests/test_context_advisory_review.py b/tests/test_context_advisory_review.py new file mode 100644 index 000000000..d1fb89bc0 --- /dev/null +++ b/tests/test_context_advisory_review.py @@ -0,0 +1,300 @@ +"""How an advisory review status is presented inside the context. + +Split verbatim out of ``tests/test_context.py`` by theme. This module owns the advisory +review status block the context carries and everything it must and must not claim about +a run. +""" + +from __future__ import annotations + + + + + + +class TestAdvisoryReviewStatusInContext: + """Tests that advisory review status appears in LLM context when runs exist.""" + + def _make_env(self, tmp_path): + class FakeEnv: + def drive_path(self, p): + return tmp_path / p + def repo_path(self, p): + return tmp_path / "repo" / p + @property + def repo_dir(self): + return tmp_path / "repo" + @property + def drive_root(self): + return tmp_path + + (tmp_path / "state").mkdir(parents=True, exist_ok=True) + (tmp_path / "logs").mkdir(parents=True, exist_ok=True) + (tmp_path / "memory").mkdir(parents=True, exist_ok=True) + (tmp_path / "repo" / "docs").mkdir(parents=True, exist_ok=True) + (tmp_path / "repo" / "VERSION").write_text("1.2.3", encoding="utf-8") + (tmp_path / "repo" / "pyproject.toml").write_text('version = "1.2.3"', encoding="utf-8") + (tmp_path / "repo" / "README.md").write_text('version-1.2.3', encoding="utf-8") + (tmp_path / "repo" / "docs" / "ARCHITECTURE.md").write_text('# Ouroboros v1.2.3', encoding="utf-8") + (tmp_path / "repo" / "docs" / "DEVELOPMENT.md").write_text('# Dev', encoding="utf-8") + (tmp_path / "state" / "state.json").write_text('{"spent_usd": 0, "budget_drift_alert": false}', encoding="utf-8") + (tmp_path / "memory" / "identity.md").write_text('x' * 300, encoding="utf-8") + (tmp_path / "memory" / "scratchpad.md").write_text('x' * 300, encoding="utf-8") + return FakeEnv() + + def test_advisory_status_in_build_llm_messages(self, tmp_path): + """format_status_section returns non-empty string when runs exist.""" + from ouroboros.review_state import ( + AdvisoryReviewState, AdvisoryRunRecord, save_state, format_status_section + ) + state = AdvisoryReviewState() + state.add_run(AdvisoryRunRecord( + snapshot_hash="abc123", + commit_message="test commit", + status="fresh", + ts="2026-01-01T00:00:00", + items=[{"item": "bible_compliance", "verdict": "PASS", "severity": "critical", "reason": "ok"}], + )) + save_state(tmp_path, state) + + loaded = __import__("ouroboros.review_state", fromlist=["load_state"]).load_state(tmp_path) + section = format_status_section(loaded) + assert "Advisory Pre-Review Status" in section + assert "FRESH" in section + assert "abc123" in section + + def test_advisory_status_empty_when_no_runs(self, tmp_path): + """format_status_section returns 'No advisory runs' when state is empty.""" + from ouroboros.review_state import AdvisoryReviewState, format_status_section + state = AdvisoryReviewState() + section = format_status_section(state) + assert "No advisory runs" in section + + def test_review_continuity_context_surfaces_live_gate_and_continuation(self, tmp_path): + from ouroboros.agent_task_pipeline import build_review_context + from ouroboros.context import build_llm_messages + from ouroboros.memory import Memory + from ouroboros.review_state import ( + AdvisoryReviewState, + AdvisoryRunRecord, + CommitAttemptRecord, + compute_snapshot_hash, + make_repo_key, + save_state, + ) + from ouroboros.task_continuation import ReviewContinuation, save_review_continuation + from ouroboros.task_results import STATUS_COMPLETED, write_task_result + + env = self._make_env(tmp_path) + (tmp_path / "repo" / ".git").mkdir(parents=True, exist_ok=True) + (tmp_path / "repo" / "prompts").mkdir(parents=True, exist_ok=True) + (tmp_path / "repo" / "prompts" / "SYSTEM.md").write_text("System", encoding="utf-8") + (tmp_path / "repo" / "BIBLE.md").write_text("Bible", encoding="utf-8") + (tmp_path / "repo" / "docs" / "CHECKLISTS.md").write_text("Checklist", encoding="utf-8") + (tmp_path / "repo" / "tracked.py").write_text("print('hi')\n", encoding="utf-8") + + repo_key = make_repo_key(tmp_path / "repo") + snapshot_hash = compute_snapshot_hash(tmp_path / "repo") + state = AdvisoryReviewState() + state.add_run(AdvisoryRunRecord( + snapshot_hash=snapshot_hash, + commit_message="test commit", + status="bypassed", + ts="2026-04-07T09:59:00+00:00", + repo_key=repo_key, + bypass_reason="manual audit override", + )) + state.advisory_runs[-1].status = "stale" + state.last_stale_from_edit_ts = "2026-04-07T10:00:00+00:00" + state.last_stale_reason = "edit_text mutated tracked.py" + state.last_stale_repo_key = repo_key + state.record_attempt(CommitAttemptRecord( + ts="2026-04-07T10:01:00+00:00", + commit_message="blocked commit", + status="blocked", + repo_key=repo_key, + tool_name="commit_reviewed", + task_id="task-old", + attempt=1, + critical_findings=[{ + "item": "tests_affected", + "reason": "Fix the failing test before commit", + "severity": "critical", + "verdict": "FAIL", + }], + readiness_warnings=["Review was blocked and needs follow-up."], + )) + save_state(tmp_path, state) + + save_review_continuation( + tmp_path, + ReviewContinuation( + task_id="task-old", + source="blocked_review", + stage="blocking_review", + repo_key=repo_key, + tool_name="commit_reviewed", + attempt=1, + block_reason="critical_findings", + critical_findings=[{ + "item": "tests_affected", + "reason": "Fix the failing test before commit", + "severity": "critical", + "verdict": "FAIL", + }], + readiness_warnings=["Review was blocked and needs follow-up."], + ), + expect_task_id="task-old", + ) + write_task_result( + tmp_path, + "task-old", + STATUS_COMPLETED, + result="Commit blocked by review.", + ) + + messages, _ = build_llm_messages( + env=env, + memory=Memory(drive_root=tmp_path), + task={"id": "task-new", "type": "task", "text": "continue"}, + review_context_builder=lambda: build_review_context(env), + ) + dynamic_text = messages[0]["content"][2]["text"] + + assert "## Review Continuity" in dynamic_text + assert "repo_commit_ready=no" in dynamic_text + assert "retry_anchor=commit_readiness_debt" in dynamic_text + assert "Commit-readiness debt" in dynamic_text + assert "bypass_reason=manual audit override" in dynamic_text + assert "stale_marker=2026-04-07T10:00:00" in dynamic_text + assert "### Open review continuations" in dynamic_text + assert "critical_finding=tests_affected: Fix the failing test before commit" in dynamic_text + assert "### Historical review ledger" in dynamic_text + assert "## Scratchpad" in dynamic_text + assert dynamic_text.index("## Scratchpad") < dynamic_text.index("## Drive state") + assert dynamic_text.index("## Runtime context") < dynamic_text.index("## Review Continuity") + + def test_review_continuity_context_ignores_foreign_repo_obligations(self, tmp_path): + from ouroboros.agent_task_pipeline import build_review_context + from ouroboros.review_state import ( + AdvisoryReviewState, + AdvisoryRunRecord, + CommitAttemptRecord, + compute_snapshot_hash, + make_repo_key, + save_state, + ) + + env = self._make_env(tmp_path) + repo_a = tmp_path / "repo" + repo_b = tmp_path / "repo-other" + (repo_a / ".git").mkdir(parents=True, exist_ok=True) + (repo_b / ".git").mkdir(parents=True, exist_ok=True) + (repo_a / "tracked.py").write_text("print('repo a')\n", encoding="utf-8") + (repo_b / "tracked.py").write_text("print('repo b')\n", encoding="utf-8") + + repo_a_key = make_repo_key(repo_a) + repo_b_key = make_repo_key(repo_b) + state = AdvisoryReviewState() + state.add_run(AdvisoryRunRecord( + snapshot_hash=compute_snapshot_hash(repo_a), + commit_message="repo a ready", + status="fresh", + ts="2026-04-07T10:00:00+00:00", + repo_key=repo_a_key, + )) + state.record_attempt(CommitAttemptRecord( + ts="2026-04-07T10:01:00+00:00", + commit_message="repo b blocked", + status="blocked", + repo_key=repo_b_key, + tool_name="commit_reviewed", + task_id="task-b", + attempt=1, + block_reason="critical_findings", + critical_findings=[{ + "item": "foreign_issue", + "reason": "other repo only", + "severity": "critical", + "verdict": "FAIL", + }], + )) + save_state(tmp_path, state) + + dynamic_text = build_review_context(env) + assert "repo_commit_ready=yes" in dynamic_text + assert "foreign_issue" not in dynamic_text + assert "repo b blocked" not in dynamic_text + + def test_review_continuity_context_keeps_open_obligations_without_runs(self, tmp_path): + from ouroboros.agent_task_pipeline import build_review_context + from ouroboros.review_state import ( + AdvisoryReviewState, + ObligationItem, + make_repo_key, + save_state, + ) + + env = self._make_env(tmp_path) + (tmp_path / "repo" / ".git").mkdir(parents=True, exist_ok=True) + (tmp_path / "repo" / "tracked.py").write_text("print('hi')\n", encoding="utf-8") + + repo_key = make_repo_key(tmp_path / "repo") + state = AdvisoryReviewState( + open_obligations=[ + ObligationItem( + obligation_id="obl-0001", + item="tests_affected", + severity="critical", + reason="Coverage still missing", + source_attempt_ts="2026-04-07T10:00:00+00:00", + source_attempt_msg="blocked commit", + repo_key=repo_key, + fingerprint="finding:tests_affected:abc123", + ) + ] + ) + save_state(tmp_path, state) + + dynamic_text = build_review_context(env) + assert "## Review Continuity" in dynamic_text + assert "open_obligations=1" in dynamic_text + assert "[obl-0001] tests_affected: Coverage still missing" in dynamic_text + + def test_review_continuity_context_keeps_all_debt_evidence(self, tmp_path): + from ouroboros.agent_task_pipeline import build_review_context + from ouroboros.review_state import ( + AdvisoryReviewState, + CommitReadinessDebtItem, + make_repo_key, + save_state, + ) + + env = self._make_env(tmp_path) + (tmp_path / "repo" / ".git").mkdir(parents=True, exist_ok=True) + (tmp_path / "repo" / "tracked.py").write_text("print('hi')\n", encoding="utf-8") + + repo_key = make_repo_key(tmp_path / "repo") + state = AdvisoryReviewState( + commit_readiness_debts=[ + CommitReadinessDebtItem( + debt_id="debt-0001", + category="repeated_obligation", + title="Commit readiness debt", + summary="Repeated tests blocker", + repo_key=repo_key, + source_obligation_ids=["obl-0001"], + evidence=[ + "first evidence", + "second evidence", + "third evidence", + ], + ) + ] + ) + save_state(tmp_path, state) + + dynamic_text = build_review_context(env) + assert "first evidence" in dynamic_text + assert "second evidence" in dynamic_text + assert "third evidence" in dynamic_text diff --git a/tests/test_context_drive_state.py b/tests/test_context_drive_state.py new file mode 100644 index 000000000..98c3e51f1 --- /dev/null +++ b/tests/test_context_drive_state.py @@ -0,0 +1,233 @@ +"""The drive-state projection, the review ledger and the settled continuations. + +Split verbatim out of ``tests/test_context.py`` by theme. This module owns the typed +drive-state section and its pointer, the review ledger that caps runs and attempts with +omission notes, the continuations that retire after their age window, and the one with +open obligations that must survive that retirement. +""" + +from __future__ import annotations + + + + + + +def test_drive_state_section_is_typed_projection_with_pointer(tmp_path): + """W3 adjacent (a): the Drive state section projects the fields the agent + reasons about and NAMES the omitted internal caches with an on-demand + pointer (P1: disclosed omission) instead of dumping state.json wholesale — + the budget narrative stays with the usage-accounting authority in the + Runtime section.""" + import json + + from ouroboros.context import _drive_state_section + + class FakeEnv: + def drive_path(self, p): + return tmp_path / p + + (tmp_path / "state").mkdir(parents=True, exist_ok=True) + (tmp_path / "state" / "state.json").write_text(json.dumps({ + "session_id": "abc123", + "current_branch": "ouroboros", + "evolution_mode_enabled": False, + "budget_drift_alert": True, + "budget_drift_pct": 48.05, + "spent_usd": 1699.3, + "managed_update_cache": {"latest_sha": "x" * 40, "latest_message": "big"}, + "usage_accounting": {"settled_usd": 1633.1}, + "openrouter_last_check_call": 5750, + }), encoding="utf-8") + + section = _drive_state_section(FakeEnv()) + + assert section.startswith("## Drive state") + assert '"session_id": "abc123"' in section + assert '"budget_drift_alert": true' in section + # Internal caches / duplicated spend narrative are OMITTED but NAMED. + assert '"managed_update_cache"' not in section + assert '"usage_accounting"' not in section + assert '"spent_usd"' not in section + for named in ("managed_update_cache", "usage_accounting", "spent_usd", "openrouter_last_check_call"): + assert named in section # named in the omission note + assert "read_file(root='runtime_data', path='state/state.json')" in section + + # Missing/empty file: still a valid section, no omission note needed. + (tmp_path / "state" / "state.json").unlink() + empty = _drive_state_section(FakeEnv()) + assert empty.startswith("## Drive state") + assert "read_file" not in empty + + +def test_review_ledger_caps_runs_and_attempts_with_omission_notes(tmp_path): + """W3 adjacent (b): the historical review ledger rides into EVERY task's + context — cap runs/attempts at the 5 most recent with EXPLICIT omission + notes (the continuation pattern) and truncate commit messages; the full + ledger stays behind review_status.""" + from ouroboros.review_state import ( + AdvisoryReviewState, + AdvisoryRunRecord, + CommitAttemptRecord, + format_status_section, + ) + + state = AdvisoryReviewState() + long_msg = "feat: " + ("y" * 2000) + for i in range(8): + state.add_run(AdvisoryRunRecord( + snapshot_hash=f"hash{i:04d}00000000", + commit_message=long_msg if i == 7 else f"commit {i}", + status="fresh", + ts=f"2026-01-0{i + 1}T00:00:00", + )) + for i in range(8): + state.record_attempt(CommitAttemptRecord( + status="succeeded", + commit_message=f"attempt commit {i}", + ts=f"2026-01-0{i + 1}T01:00:00", + attempt=i + 1, + )) + + section = format_status_section(state) + + assert "3 older advisory run(s) omitted" in section + assert "3 older attempt(s) omitted" in section + assert "review_status" in section + assert "hash0007" in section # newest kept + assert "hash0000" not in section # oldest omitted + assert "attempt commit 7" in section + assert "attempt commit 0" not in section + # The 2000-char commit message is display-truncated with the explicit notice. + assert "y" * 2000 not in section + assert "truncated at 300 chars" in section + + +def test_settled_continuations_retire_after_age_window(tmp_path): + """W3 adjacent (b): a continuation whose owning task SETTLED and that sat + un-resumed past the age window is archived (durable move, never deleted); + fresh settled records stay — they are the designed cross-task resume + pointer.""" + from ouroboros.task_continuation import ( + ReviewContinuation, + archived_continuation_dir, + continuation_path, + list_review_continuations, + retire_settled_continuations, + save_review_continuation, + ) + + old = save_review_continuation(tmp_path, ReviewContinuation( + task_id="oldtask", source="commit_blocked", stage="review")) + # Age the record past the window (rewrite the stored timestamps). + import json as _json + path = continuation_path(tmp_path, "oldtask") + data = _json.loads(path.read_text(encoding="utf-8")) + data["created_ts"] = data["updated_ts"] = "2026-01-01T00:00:00+00:00" + path.write_text(_json.dumps(data), encoding="utf-8") + + save_review_continuation(tmp_path, ReviewContinuation( + task_id="freshtask", source="commit_blocked", stage="review")) + + settled = {"oldtask": True, "freshtask": True, "runningtask": False} + retired = retire_settled_continuations(tmp_path, is_settled=lambda tid: settled.get(tid, False)) + + assert retired == ["oldtask"] + assert not continuation_path(tmp_path, "oldtask").exists() + assert (archived_continuation_dir(tmp_path) / "oldtask.json").exists() # durable, not deleted + remaining, _corrupt = list_review_continuations(tmp_path) + assert [c.task_id for c in remaining] == ["freshtask"] + + # An old continuation of a NON-settled task stays put. + save_review_continuation(tmp_path, ReviewContinuation( + task_id="runningtask", source="commit_blocked", stage="review")) + path = continuation_path(tmp_path, "runningtask") + data = _json.loads(path.read_text(encoding="utf-8")) + data["created_ts"] = data["updated_ts"] = "2026-01-01T00:00:00+00:00" + path.write_text(_json.dumps(data), encoding="utf-8") + assert retire_settled_continuations(tmp_path, is_settled=lambda tid: settled.get(tid, False)) == [] + assert continuation_path(tmp_path, "runningtask").exists() + assert old.task_id == "oldtask" + + +def test_settled_continuation_with_open_obligations_survives_age_retirement(tmp_path): + """A settled FAILED task whose continuation records obligations that are + STILL open in the review ledger is genuinely unresolved review work: age + must not archive it out of context (P1/P3). A same-age settled sibling with + no open markers still retires — the noise-reduction path stays.""" + import json as _json + + from ouroboros.agent_task_pipeline import build_review_context + from ouroboros.review_state import ( + AdvisoryReviewState, + ObligationItem, + make_repo_key, + save_state, + ) + from ouroboros.task_continuation import ( + ReviewContinuation, + archived_continuation_dir, + continuation_path, + save_review_continuation, + ) + + class FakeEnv: + def drive_path(self, p): + return tmp_path / p + + def repo_path(self, p): + return tmp_path / "repo" / p + + @property + def repo_dir(self): + return tmp_path / "repo" + + @property + def drive_root(self): + return tmp_path + + env = FakeEnv() + (tmp_path / "repo" / ".git").mkdir(parents=True, exist_ok=True) + (tmp_path / "repo" / "tracked.py").write_text("print('hi')\n", encoding="utf-8") + repo_key = make_repo_key(tmp_path / "repo") + + def _aged_continuation(task_id, obligation_ids): + save_review_continuation(tmp_path, ReviewContinuation( + task_id=task_id, source="commit_blocked", stage="review", + block_reason="critical_findings", obligation_ids=obligation_ids)) + path = continuation_path(tmp_path, task_id) + data = _json.loads(path.read_text(encoding="utf-8")) + data["created_ts"] = data["updated_ts"] = "2026-01-01T00:00:00+00:00" + path.write_text(_json.dumps(data), encoding="utf-8") + + _aged_continuation("unresolvedtask", ["obl-open-1"]) + _aged_continuation("closedtask", ["obl-long-gone"]) + task_results = tmp_path / "task_results" + task_results.mkdir(parents=True, exist_ok=True) + for tid in ("unresolvedtask", "closedtask"): + (task_results / f"{tid}.json").write_text( + _json.dumps({"id": tid, "status": "failed"}), encoding="utf-8") + + state = AdvisoryReviewState(open_obligations=[ + ObligationItem( + obligation_id="obl-open-1", + item="tests_affected", + severity="critical", + reason="Coverage still missing", + source_attempt_ts="2026-01-01T00:00:00+00:00", + source_attempt_msg="blocked commit", + repo_key=repo_key, + fingerprint="finding:tests_affected:abc123", + ) + ]) + save_state(tmp_path, state) + + dynamic_text = build_review_context(env) + + # Unresolved work survives the age window and stays in cognitive context. + assert continuation_path(tmp_path, "unresolvedtask").exists() + assert "task=unresolvedtask" in dynamic_text + # The provably-closed sibling still rides the age path (durable, disclosed). + assert not continuation_path(tmp_path, "closedtask").exists() + assert (archived_continuation_dir(tmp_path) / "closedtask.json").exists() + assert "closedtask" in dynamic_text # transient archive disclosure line diff --git a/tests/test_context_runtime_section.py b/tests/test_context_runtime_section.py new file mode 100644 index 000000000..876461de5 --- /dev/null +++ b/tests/test_context_runtime_section.py @@ -0,0 +1,502 @@ +"""The runtime section and the user content the context builder emits. + +Split verbatim out of ``tests/test_context.py`` by theme. This module owns the +force-plan notice that must not rewrite the user's text, the ephemeral force plan that +only routes, the light-mode rule and filesystem affordances the runtime section states, +the workspace rules that preserve system review/commit authority, the host routing +manifest and manual contract, the improvement backlog digest, and the runtime_env +block. +""" + +from __future__ import annotations + +import inspect +import json + +import pytest + +from ouroboros.context import build_runtime_section, build_user_content + +from tests._context_shared import _make_health_env + +def test_build_llm_messages_has_no_recorder_only_soft_cap_chain(): + from ouroboros import context as context_module + from ouroboros.context import build_llm_messages + + assert "soft_cap_tokens" not in inspect.signature(build_llm_messages).parameters + assert not hasattr(context_module, "apply_message_token_soft_cap") + source = inspect.getsource(build_llm_messages) + assert "estimated_tokens_before" not in source + assert "trimmed_sections" not in source + assert "context_fit" in source + + +@pytest.mark.parametrize("enforcement", ["blocking", "advisory"]) +def test_force_plan_metadata_adds_structured_notice_without_rewriting_user_text( + monkeypatch, enforcement, +): + monkeypatch.setenv("OUROBOROS_REVIEW_ENFORCEMENT", enforcement) + content = build_user_content( + { + "text": "Fix the marketplace retry flow.", + "metadata": {"force_plan": True, "force_plan_source": "swarm"}, + } + ) + + assert content.startswith("[SWARM_INITIATIVE]") + assert "Source: swarm." in content + assert f"Resolved review enforcement: {enforcement}." in content + assert "Under blocking" in content + assert "non-mutating preparation" in content + assert "begin implementation only after review closes" in content + # Fan-out integration mechanics (owner-approved, 2026-08-05): parallel + # children cannot see each other's edits, so a plan gives them disjoint + # write regions or plans the parent synthesis for the expected overlap. + assert "cannot see each other's edits" in content + assert "disjoint write regions" in content + # Owner intent as prose, not schema (rc-phaseC, P5): the plan must name its + # execution shape so reviewers can judge a silent zero-fan-out. + assert "State the chosen execution shape explicitly" in content + assert "delegation required, optional, or intentionally not used" in content + assert content.rstrip().endswith("Fix the marketplace retry flow.") + + +def test_ephemeral_force_plan_is_routing_only_and_transfers_work(): + content = build_user_content({ + "text": "Fix the marketplace retry flow.", + "_ephemeral_turn": True, + "metadata": {"force_plan": True, "force_plan_source": "swarm"}, + }) + + assert content.startswith("[SWARM_ROUTING_INTENT]") + assert "exactly one NEW managed root" in content + assert "do not execute it" in content + assert content.rstrip().endswith("Fix the marketplace retry flow.") + + +def test_runtime_section_includes_light_runtime_mode_rule(tmp_path, monkeypatch): + env = _make_health_env(tmp_path) + monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "light") + section = build_runtime_section(env, {"id": "task-1", "type": "task"}) + payload = json.loads(section.split("\n\n", 1)[1]) + + assert payload["runtime_mode"] == "light" + assert "forbids Ouroboros repo mutation" in payload["runtime_mode_rule"] + assert "user_files" in payload["runtime_mode_rule"] + assert "artifact_store" in payload["runtime_mode_rule"] + assert "explicit scoped skill-payload work/repair" in payload["runtime_mode_rule"] + assert "runtime_data/uploads" in payload["runtime_mode_rule"] + + +def test_runtime_section_includes_filesystem_affordances_with_ctx(tmp_path, monkeypatch): + from ouroboros.tools.registry import ToolContext + + env = _make_health_env(tmp_path) + monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "light") + ctx = ToolContext(repo_dir=tmp_path / "repo", drive_root=tmp_path) + + section = build_runtime_section(env, {"id": "task-1", "type": "task"}, ctx=ctx) + payload = json.loads(section.split("\n\n", 1)[1]) + fs = payload["capabilities"]["filesystem"] + + assert fs["profile"] == "self_modification" + assert "runtime_data" in fs["searchable_roots"] + assert "task_drive" not in fs["searchable_roots"] + assert "task_drive" in fs["allowed_shell_cwd_roots"] + assert "status" in fs["git_readonly_subcommands"] + assert "active_workspace" in fs["light_gated_roots"] + + +def test_runtime_section_external_workspace_includes_user_files_shell_affordance(tmp_path, monkeypatch): + from ouroboros.tools.registry import ToolContext + + env = _make_health_env(tmp_path) + monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "advanced") + drive = tmp_path / "data" + repo = tmp_path / "repo" + workspace = tmp_path / "workspace" + drive.mkdir() + repo.mkdir(exist_ok=True) + workspace.mkdir(exist_ok=True) + ctx = ToolContext( + repo_dir=repo, + drive_root=drive, + workspace_root=workspace, + workspace_mode="external", + ) + + section = build_runtime_section(env, {"id": "task-1", "type": "task"}, ctx=ctx) + payload = json.loads(section.split("\n\n", 1)[1]) + fs = payload["capabilities"]["filesystem"] + + assert fs["profile"] == "external_workspace_task" + assert "user_files" in fs["allowed_shell_cwd_roots"] + + +def test_runtime_section_workspace_rule_preserves_system_review_commit_authority(tmp_path, monkeypatch): + env = _make_health_env(tmp_path) + monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "advanced") + workspace = tmp_path / "workspace" + workspace.mkdir() + section = build_runtime_section( + env, + { + "id": "task-1", + "type": "task", + "workspace_root": str(workspace), + "workspace_mode": "external", + "memory_mode": "forked", + }, + ) + rule = json.loads(section.split("\n\n", 1)[1])["active_workspace"]["rule"] + + assert "default to the active workspace" in rule + assert "explicit typed root/cwd" in rule + assert "self-review/commit tools remain available" in rule + assert "self-review/commit tools are unavailable" not in rule + + +def test_runtime_section_omits_light_rule_for_advanced(tmp_path, monkeypatch): + env = _make_health_env(tmp_path) + monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "advanced") + section = build_runtime_section(env, {"id": "task-1", "type": "task"}) + payload = json.loads(section.split("\n\n", 1)[1]) + + assert payload["runtime_mode"] == "advanced" + assert "runtime_mode_rule" not in payload + + +def test_runtime_section_includes_non_workspace_memory_boundary(tmp_path, monkeypatch): + env = _make_health_env(tmp_path) + monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "advanced") + section = build_runtime_section( + env, + { + "id": "task-1", + "type": "task", + "memory_mode": "forked", + "drive_root": str(tmp_path / "child"), + "child_drive_root": str(tmp_path / "child"), + "budget_drive_root": str(tmp_path / "data"), + }, + ) + payload = json.loads(section.split("\n\n", 1)[1]) + assert payload["task"]["memory_mode"] == "forked" + assert payload["task"]["child_drive_root"].endswith("child") + assert payload["task"]["budget_drive_root"].endswith("data") + + +def test_runtime_section_exposes_host_routing_manifest_and_manual_contract(tmp_path, monkeypatch): + env = _make_health_env(tmp_path) + monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "advanced") + task = { + "id": "decision-1", + "type": "task", + "metadata": { + "current_chat": { + "chat_id": 1, + "running_tasks": [], + "addressable_root_tasks": [{"task_id": "pending-1", "status": "pending"}], + }, + "main_routing_manifest": { + "projects": [{"project_id": "racer", "name": "Racer"}], + "root_tasks": [{"task_id": "pending-1", "status": "pending"}], + }, + "routing_contract": { + "source_lane": "main", + "on_uncertain_or_invalid_target": "needs_manual_target", + "manual_options": [{"task_id": "pending-1"}], + }, + }, + } + + payload = json.loads(build_runtime_section(env, task).split("\n\n", 1)[1]) + + assert payload["current_chat"]["addressable_root_tasks"][0]["task_id"] == "pending-1" + assert payload["main_routing_manifest"]["projects"][0]["project_id"] == "racer" + assert payload["routing_contract"]["on_uncertain_or_invalid_target"] == "needs_manual_target" + + +def test_improvement_backlog_digest_is_actor_scoped(tmp_path): + from ouroboros.context import build_llm_messages + from ouroboros.memory import Memory + + class FakeEnv: + def drive_path(self, p): + return tmp_path / p + + def repo_path(self, p): + return tmp_path / "repo" / p + + @property + def repo_dir(self): + return tmp_path / "repo" + + @property + def drive_root(self): + return tmp_path + + (tmp_path / "repo" / "prompts").mkdir(parents=True, exist_ok=True) + (tmp_path / "repo" / "docs").mkdir(parents=True, exist_ok=True) + (tmp_path / "memory" / "knowledge").mkdir(parents=True, exist_ok=True) + (tmp_path / "logs").mkdir(parents=True, exist_ok=True) + (tmp_path / "state").mkdir(parents=True, exist_ok=True) + + (tmp_path / "repo" / "prompts" / "SYSTEM.md").write_text("System prompt", encoding="utf-8") + (tmp_path / "repo" / "BIBLE.md").write_text("Bible", encoding="utf-8") + (tmp_path / "repo" / "README.md").write_text("README", encoding="utf-8") + (tmp_path / "repo" / "docs" / "ARCHITECTURE.md").write_text('# Ouroboros v1.2.3', encoding="utf-8") + (tmp_path / "repo" / "docs" / "DEVELOPMENT.md").write_text('# Dev', encoding="utf-8") + (tmp_path / "repo" / "docs" / "CHECKLISTS.md").write_text('Checklist', encoding="utf-8") + (tmp_path / "repo" / "VERSION").write_text("1.2.3", encoding="utf-8") + (tmp_path / "repo" / "pyproject.toml").write_text('version = "1.2.3"', encoding="utf-8") + (tmp_path / "state" / "state.json").write_text('{"spent_usd": 0}', encoding="utf-8") + (tmp_path / "memory" / "identity.md").write_text("I am Ouroboros", encoding="utf-8") + (tmp_path / "memory" / "scratchpad.md").write_text("scratchpad", encoding="utf-8") + (tmp_path / "memory" / "knowledge" / "improvement-backlog.md").write_text( + "# Improvement Backlog\n\n### ibl-1\n- status: open\n- created_at: 2026-04-14T09:00:00+00:00\n- source: execution_reflection\n- category: process\n- task_id: task-1\n- requires_plan_review: yes\n- fingerprint: fp-1\n- summary: Reduce recurring task friction around REVIEW_BLOCKED\n", + encoding="utf-8", + ) + + ordinary = ({"id": "main", "type": "task", "_is_direct_chat": True}, + {"id": "project", "type": "task", "project_id": "p1"}, + {"id": "managed", "type": "task"}, + {"id": "child", "type": "task", "delegation_role": "subagent"}) + for task in ordinary: + task["text"] = "hello" + messages, _ = build_llm_messages( + env=FakeEnv(), memory=Memory(drive_root=tmp_path), task=task, + ) + assert "## Improvement Backlog" not in messages[0]["content"][2]["text"] + + for task_type in ("evolution", "deep_self_review"): + messages, _ = build_llm_messages( + env=FakeEnv(), memory=Memory(drive_root=tmp_path), + task={"id": task_type, "type": task_type, "text": "improve"}) + dynamic_text = messages[0]["content"][2]["text"] + assert "## Improvement Backlog" in dynamic_text + assert "Reduce recurring task friction around REVIEW_BLOCKED" in dynamic_text + + +class TestRuntimeEnvSection: + """build_runtime_section: runtime_env carries presentation + platform, and + the per-message owner_client fact renders beside it (is_desktop retired).""" + + def _make_env(self, tmp_path): + class FakeEnv: + repo_dir = tmp_path / "repo" + drive_root = tmp_path + + def drive_path(self, p): + return tmp_path / p + + (tmp_path / "state").mkdir(parents=True, exist_ok=True) + (tmp_path / "state" / "state.json").write_text( + '{"spent_usd": 0}', encoding="utf-8" + ) + return FakeEnv() + + def test_runtime_env_presentation_absent_means_web(self, tmp_path, monkeypatch): + from ouroboros.context import build_runtime_section + + monkeypatch.delenv("OUROBOROS_PRESENTATION", raising=False) + env = self._make_env(tmp_path) + section = build_runtime_section(env, {"id": "t1", "type": "task"}) + data = json.loads(section.split("## Runtime context\n\n", 1)[1]) + assert "runtime_env" in data + assert "platform" in data["runtime_env"] + assert isinstance(data["runtime_env"]["platform"], str) + assert data["runtime_env"]["presentation"] == "web" + # The dead is_desktop flag is retired; presentation replaced it. + assert "is_desktop" not in data["runtime_env"] + + def test_runtime_env_presentation_from_launcher_export(self, tmp_path, monkeypatch): + from ouroboros.context import build_runtime_section + + for value in ("desktop_window", "browser_fallback"): + monkeypatch.setenv("OUROBOROS_PRESENTATION", value) + env = self._make_env(tmp_path) + section = build_runtime_section(env, {"id": "t2", "type": "task"}) + data = json.loads(section.split("## Runtime context\n\n", 1)[1]) + assert data["runtime_env"]["presentation"] == value + + def test_owner_client_rendered_from_metadata(self, tmp_path, monkeypatch): + from ouroboros.context import build_runtime_section + + monkeypatch.delenv("OUROBOROS_PRESENTATION", raising=False) + env = self._make_env(tmp_path) + fact = {"pywebview": True, "ua": "TestShell/1.0", "viewport": {"w": 1200, "h": 800}} + section = build_runtime_section( + env, {"id": "t3", "type": "task", "metadata": {"client_surface": fact}} + ) + data = json.loads(section.split("## Runtime context\n\n", 1)[1]) + assert data["owner_client"] == fact + assert "SENT" in data["owner_client_note"] + + def test_owner_client_absent_is_a_gap_not_a_default(self, tmp_path, monkeypatch): + from ouroboros.context import build_runtime_section + + env = self._make_env(tmp_path) + section = build_runtime_section(env, {"id": "t4", "type": "task"}) + data = json.loads(section.split("## Runtime context\n\n", 1)[1]) + assert "owner_client" not in data + assert "owner_client_note" not in data + + def test_owner_client_channel_fact_stamped_by_external_admission(self, tmp_path): + from ouroboros.context import build_runtime_section + + env = self._make_env(tmp_path) + # /api/tasks and CLI STAMP the channel fact at admission; the renderer + # reads only the producer-assembled fact. + section = build_runtime_section( + env, {"id": "t5", "type": "task", "metadata": {"client_surface": {"channel": "cli"}}} + ) + data = json.loads(section.split("## Runtime context\n\n", 1)[1]) + assert data["owner_client"] == {"channel": "cli"} + + def test_owner_client_never_inferred_from_metadata_source(self, tmp_path): + from ouroboros.context import build_runtime_section + + env = self._make_env(tmp_path) + # metadata.source is OVERLOADED (scheduler writes scheduled_task / + # skill_scheduled_task): the renderer must never dress it up as an + # owner surface — no producer stamp, no fact (codex scope round 2 N1). + for source in ("cli", "scheduled_task", "skill_scheduled_task", "web"): + section = build_runtime_section( + env, {"id": "t6", "type": "task", "metadata": {"source": source}} + ) + data = json.loads(section.split("## Runtime context\n\n", 1)[1]) + assert "owner_client" not in data, f"source={source!r} must not render" + # Internal producers use top-level task["source"], never rendered. + section = build_runtime_section( + env, {"id": "t7", "type": "task", "source": "promote_chat_to_task"} + ) + data = json.loads(section.split("## Runtime context\n\n", 1)[1]) + assert "owner_client" not in data + + +def _delegation_data_root(tmp_path, monkeypatch): + root = tmp_path / "delegation_data_root" + (root / "state").mkdir(parents=True, exist_ok=True) + monkeypatch.setattr("ouroboros.config.DATA_DIR", root) + return root + + +def _delegation_fact(tmp_path, monkeypatch): + env = _make_health_env(tmp_path) + monkeypatch.setattr("ouroboros.config.get_runtime_mode", lambda: "advanced") + section = build_runtime_section(env, {"id": "task-1", "type": "task"}) + payload = json.loads(section.split("\n\n", 1)[1]) + return payload["capabilities"] + + +def test_delegation_fact_carries_historical_rows_and_profile_evidence(tmp_path, monkeypatch): + root = _delegation_data_root(tmp_path, monkeypatch) + monkeypatch.setenv("OUROBOROS_SUBAGENT_HARNESS", "claudexor=opus-5:high") + (root / "state" / "reviewer_slot_last_execution.json").write_text(json.dumps({ + "triad_1": { + "ts": "2026-08-18T01:02:03+00:00", + "surface": "triad", + "status": "ok", + "requested": {"profile_id": "requested-review-profile"}, + "effective": { + "route": "agent_session:claudexor", + "model": "opus-5", + "profile_id": "applied-review-profile", + }, + }, + "triad_2": { + "ts": "2026-08-18T01:02:04+00:00", + "surface": "triad", + "status": "error", + # B1 typed facts: a dated window carries reset_at, an undated one + # only the code — both must surface independently. + "failure_code": "subscription_window_exhausted", + "reset_at": "2026-08-18T09:20:00+00:00", + }, + }), encoding="utf-8") + (root / "state" / "subagent_last_delegation.json").write_text(json.dumps({ + "ts": "2026-08-18T02:00:00+00:00", + "route": "claudexor", + "requested_model": "opus-5", + "applied_model": "claude-opus-5", + "requested_profile": "requested-delegate-profile", + "applied_profile": "applied-delegate-profile", + "selected_subagent_id": "builder", + "run_id": "run-1", + }), encoding="utf-8") + + capabilities = _delegation_fact(tmp_path, monkeypatch) + delegation = capabilities["delegation"] + + assert "configured_route" not in delegation + rows = {row["slot"]: row for row in delegation["reviewer_slots_last"]} + assert rows["triad_1"]["outcome"] == "ok" + assert rows["triad_1"]["requested_profile"] == "requested-review-profile" + assert rows["triad_1"]["applied_profile"] == "applied-review-profile" + assert "failure_code" not in rows["triad_1"] + assert rows["triad_2"]["outcome"] == "failed" + assert rows["triad_2"]["failure_code"] == "subscription_window_exhausted" + assert rows["triad_2"]["reset_at"] == "2026-08-18T09:20:00+00:00" + # Per-row label is the timestamp only; the verbatim historical disclaimer + # lives ONCE in the note (review fix 12), never repeated per row. + assert rows["triad_1"]["observed"] == "last observed at 2026-08-18T01:02:03+00:00" + last = delegation["subagent_last_delegation"] + assert last["route"] == "claudexor" + assert last["applied_model"] == "claude-opus-5" + assert last["requested_profile"] == "requested-delegate-profile" + assert last["applied_profile"] == "applied-delegate-profile" + assert last["selected_subagent_id"] == "builder" + assert last["observed"] == "last observed at 2026-08-18T02:00:00+00:00" + assert "historical" not in rows["triad_1"]["observed"] + # The prompt-visible note teaches the semantics ONCE: rows are history, live + # facts come from plan-review waves and typed delegate refusals. + assert "historical, not live health" in delegation["note"] + assert "plan-review wave rows" in delegation["note"] + assert "typed" in delegation["note"] and "refusal" in delegation["note"] + assert "never healthy" in delegation["note"] + + +def test_delegation_fact_undated_window_code_surfaces_without_reset(tmp_path, monkeypatch): + root = _delegation_data_root(tmp_path, monkeypatch) + monkeypatch.delenv("OUROBOROS_SUBAGENT_HARNESS", raising=False) + (root / "state" / "reviewer_slot_last_execution.json").write_text(json.dumps({ + "scope": { + "ts": "2026-08-18T03:00:00+00:00", + "status": "error", + "failure_code": "credential_pool_exhausted", + }, + }), encoding="utf-8") + + delegation = _delegation_fact(tmp_path, monkeypatch)["delegation"] + + (row,) = delegation["reviewer_slots_last"] + assert row["failure_code"] == "credential_pool_exhausted" + assert "reset_at" not in row + assert row["outcome"] == "failed" + + +def test_delegation_fact_absent_files_mean_absent_observations_not_health(tmp_path, monkeypatch): + _delegation_data_root(tmp_path, monkeypatch) + monkeypatch.delenv("OUROBOROS_SUBAGENT_HARNESS", raising=False) + + capabilities = _delegation_fact(tmp_path, monkeypatch) + + assert "delegation" not in capabilities + + +def test_delegation_fact_failure_never_drops_capability_digest(tmp_path, monkeypatch): + _delegation_data_root(tmp_path, monkeypatch) + + def _boom(): + raise RuntimeError("reader exploded") + + monkeypatch.setattr( + "ouroboros.reviewer_slot_config.reviewer_slot_last_executions", _boom) + + capabilities = _delegation_fact(tmp_path, monkeypatch) + + assert "delegation" not in capabilities + # The surrounding digest survives intact. + assert "allow_mutative_subagents" in capabilities + assert "write_surfaces" in capabilities