From 98d80793cf9588b370281b133851f02c99306443 Mon Sep 17 00:00:00 2001 From: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com> Date: Sat, 26 Sep 2026 16:06:23 +0300 Subject: [PATCH] fix: count rescued deliverables through shared unmeasured listing Verify queued forwarding retains exact unread mail on cancellation before start. Keep unavailable stores unknown, exclude inputs and bookkeeping, preserve split actor stores and delivery identity. Document host factual speech. --- docs/DESIGN.md | 4 +- docs/architecture/06-agent-core.md | 2 +- ouroboros/task_finalization.py | 79 ++++++------ tests/test_cancel_terminal_delivery.py | 13 +- tests/test_task_summary.py | 160 +++++++++++++++++++++++-- tests/test_unread_mail_custody.py | 34 ++++++ 6 files changed, 237 insertions(+), 55 deletions(-) diff --git a/docs/DESIGN.md b/docs/DESIGN.md index 2d2f7af8a..b57d63309 100644 --- a/docs/DESIGN.md +++ b/docs/DESIGN.md @@ -298,7 +298,9 @@ lives in `log_events.js`). The routing receipt under an owner message is such a surface: a refused addressing act carries the host-composed `cause` sentence (`project_dialogue.routing_refusal_cause` — one host table for the receipt line, the System row and the picker toast), a landed act carries -none, and an unknown reason stays raw. A terminal whose preserved output was +none, and an unknown reason stays raw. Host text speaks only for the host's own +actions, its own counts and signed quotes; a source it could not read is +unknown, never zero. A terminal whose preserved output was never reviewed shows that output labelled rather than hidden: a short labelled excerpt beside the pointer to the full copy, so a `Failed` card over applied work is never a bare headline and never names preserved bytes without a way to reach diff --git a/docs/architecture/06-agent-core.md b/docs/architecture/06-agent-core.md index 394df0bf7..3ac5ff636 100644 --- a/docs/architecture/06-agent-core.md +++ b/docs/architecture/06-agent-core.md @@ -535,7 +535,7 @@ The advisory rows a reflection reads are ATTRIBUTED. Advisory runs are scoped by Only roots synthesize; `root_phase_checkpoint` makes paid synthesis at-most-once across restart, while children contribute evidence. Durable-result persistence owes the answer as `final::` in `supervisor/terminal_delivery.py`'s bounded outbox (§5; normal/cancel/reap). `send_message` delivers immediately; the retained buffered copy shares its ID for durable dedupe. Replays use bounded backoff. Exhaustion/eviction preserves full text on disk, emits `terminal_delivery_exhausted` and a chat notice; external delivery remains at-least-once. Buffered `task_done` stays last to retain the slot/child drive during synthesis; a hung-synthesis reap need not lose the delivered answer. Project roots keep early answers in Project. Their canonical row and deferred Main mirror use `terminal_projection.settle_terminal_projection` via task-done/checkpoint/startup/maintenance; §3 "Main rows and host-stamped card rows" owns readiness, retirement and limits. -Synthesis receives a sealed final package from the durable result — the submitted final text, its artifact manifest and completion_observations. Full redacted action observations live in the canonical artifact store (`task.budget_drive_root or drive_root`), in the write-once `source_handles/context_checkpoints` store with verified `task_source` refs, before compact publication and outside deliverables and inferred readiness; their native reader `get_task_result(include_completion_source=true)` returns complete length/hash first, then explicit `source_start_char`/`source_end_char` ranges (`artifacts.text_source_range_projection`, the shared work-order range contract), with bytes, kind, path containment and SHA checked before any excerpt. Packet-only reflection receives per-send-tool counts, each family's latest recorded return, and task-related skill readiness with coverage; full-source references are for later readers, not evidence the synthesizer has read. Positive observed facts correct error-trace impressions, while tool success does not prove owner receipt, empty material does not prove absence, and skill readiness does not attribute an owner's action to the task. Before context cleanup, `agent_task_pipeline.emit_task_results` also freezes `review_evidence.task_inputs` through `post_task_synthesis.capture_task_inputs`: `run_origin`, the existing task-local owner corpus, intact question/answer provenance and the canonical split-root verification-receipt union. Reflection receives the same complete redacted content through `reflection.task_inputs_prompt_section`, separate from bounded trace/review excerpts. A zero return code is positive evidence; an unrelated later pass cannot resolve another check's failure. Peer proposals stay attributed, and unavailable input is not evidence that approval or verification never existed. Recovery uses these stored observations and inputs, not a later conversation. A free `host_task_facts` row precedes paid stages (or follows result persistence when Stop skips them): no model call or narrative; its metrics, routing and cost serve history. `files_rescued` (TZ-2 C2) is a stat-only file count of the artifact stores: positive, zero or unknown if unreadable, `hash_computed: false`; the stop receipt repeats it so no salvageable text never means no files. +Synthesis receives a sealed final package from the durable result — the submitted final text, its artifact manifest and completion_observations. Full redacted action observations live in the canonical artifact store (`task.budget_drive_root or drive_root`), in the write-once `source_handles/context_checkpoints` store with verified `task_source` refs, before compact publication and outside deliverables and inferred readiness; their native reader `get_task_result(include_completion_source=true)` returns complete length/hash first, then explicit `source_start_char`/`source_end_char` ranges (`artifacts.text_source_range_projection`, the shared work-order range contract), with bytes, kind, path containment and SHA checked before any excerpt. Packet-only reflection receives per-send-tool counts, each family's latest recorded return, and task-related skill readiness with coverage; full-source references are for later readers, not evidence the synthesizer has read. Positive observed facts correct error-trace impressions, while tool success does not prove owner receipt, empty material does not prove absence, and skill readiness does not attribute an owner's action to the task. Before context cleanup, `agent_task_pipeline.emit_task_results` also freezes `review_evidence.task_inputs` through `post_task_synthesis.capture_task_inputs`: `run_origin`, the existing task-local owner corpus, intact question/answer provenance and the canonical split-root verification-receipt union. Reflection receives the same complete redacted content through `reflection.task_inputs_prompt_section`, separate from bounded trace/review excerpts. A zero return code is positive evidence; an unrelated later pass cannot resolve another check's failure. Peer proposals stay attributed, and unavailable input is not evidence that approval or verification never existed. Recovery uses these stored observations and inputs, not a later conversation. A free `host_task_facts` row precedes paid stages (or follows result persistence when Stop skips them): no model call or narrative; its metrics, routing and cost serve history. `files_rescued` (TZ-2 C2) counts files the shared unmeasured listing finds per store: positive, zero or unknown if unlistable, `hash_computed: false`; the stop receipt repeats it so no salvageable text never means no files. Pooled workers retain their slot until post-task work settles; early final-answer delivery is independent of that timing. Native work stays on its registered actor; `TaskModelWait` remains reachable through `POST_TASK_SYNTHESIS_INFLIGHT`, and detached work owns a separate live wait. Mailbox cleanup waits for the checkpoint. The solve-phase absolute ceiling does not cut a settled root's running post-work, but Stop, calendar deadline, monetary admission, per-call and idle rails still bind. The stage owner reads returned and raised failures alike, the reflection's nested Pattern Register write too: budget, a control or an unresolved attempt on any provider's chain (`transport_custody.outcome_unknown_on_chain`) skips later paid stages, degrading the checkpoint; an ordinary failure stays local (later stages run, completed reflection actions apply) yet reads `degraded` with no skipped list, never `completed`; a split-answered refusal (`resolution`) is history, not failure (TZ-2 C3). A running checkpoint after restart degrades without replaying a paid request. diff --git a/ouroboros/task_finalization.py b/ouroboros/task_finalization.py index bb2b352d1..7d7c14716 100644 --- a/ouroboros/task_finalization.py +++ b/ouroboros/task_finalization.py @@ -29,6 +29,7 @@ import json import logging import os import pathlib +import stat from typing import Any, Dict, List from ouroboros.utils import sanitize_tool_result_for_log, truncate_review_artifact @@ -600,13 +601,6 @@ def sealed_final_prompt_section(sealed_final: Dict[str, Any] | None) -> str: ) -# TZ-2 C2: a store's own bookkeeping is not a rescued file. Mirrors -# ``artifacts._ARTIFACT_MANIFEST`` and ``workspace_patch_capture.SCRATCH_MANIFEST_NAME`` -# (importing the D05 store owner here would invert the terminal-facts direction); -# tests/test_task_summary.py pins equality with the SSOT so the literals cannot drift. -RESCUED_FILES_BOOKKEEPING = frozenset({".artifact_manifest.json", ".scratch_manifest.json"}) - - def artifact_store_roots(canonical_root: Any, task_id: str, *, task: Any = None, child_root: Any = None) -> List[pathlib.Path]: """The task's artifact store directories: the canonical one and, for a split root, the child drive's. @@ -614,7 +608,7 @@ def artifact_store_roots(canonical_root: Any, task_id: str, *, task: Any = None, The child drive is the caller's when it knows it (the pipeline's ``env.drive_root``), else the task row's (``child_drive_root`` / ``drive_root``, the supervisor's own resolution of a running task's drive), else the durable result's; the same store - named twice is walked once. Fail-soft: an unreadable result adds no store. + named twice is listed once. Fail-soft: an unreadable result adds no store. """ from ouroboros.headless import ARTIFACTS_DIR @@ -638,21 +632,24 @@ def artifact_store_roots(canonical_root: Any, task_id: str, *, task: Any = None, def rescued_files_fact(task_id: str, stores: List[pathlib.Path]) -> Dict[str, Any]: - """How many files the task's artifact stores hold, by ``stat`` alone (TZ-2 C2). + """How many deliverable files the task's artifact stores hold (TZ-2 C2). - ``state`` is ``positive`` (files were found), ``zero`` (every store was walked and - holds none; a store never created is one nothing was written to) or ``unknown`` (a - store could not be read — ``count`` is then what the readable part held, a floor, - never a total). No file is opened and no hash is computed, and the fact says so - (``hash_computed``), so a reader never mistakes this occupancy count for the - per-file receipt the artifact manifest (``collect_task_artifact_records``) owns; - conversely an empty readable manifest alone never proves zero — only the walk does. - Never raises: a failure inside the walk is an unknown store. + Each distinct store is read by the shared unmeasured listing + (``artifacts.collect_task_artifact_records(measure=False, strict=True)``), so a + store's metadata, receipt stream, staged inputs and source handles are never + counted, a registration whose file is gone is not a file, and an unregistered + output is. The listing reads only the registration; no deliverable is opened, + hashed, copied or registered, and the fact says so (``hash_computed``). ``state`` is + ``positive``, ``zero`` (every store was listed and holds none; a store never + created is one nothing was written to) or ``unknown`` (a store could not be + listed — ``count`` is then what the listed stores held, a floor, never a total). + The count is physical files per store: the same name in two stores is two + listed files, never assumed to be one copy. Never raises. """ rows: List[Dict[str, Any]] = [] for store in stores: try: - count, readable = _stat_only_file_count(pathlib.Path(store)) + count, readable = _listed_file_count(task_id, pathlib.Path(store)), True except Exception: count, readable = 0, False rows.append({"store": str(store), "count": count, "readable": readable}) @@ -662,35 +659,37 @@ def rescued_files_fact(task_id: str, stores: List[pathlib.Path]) -> Dict[str, An return {"count": total, "state": state, "hash_computed": False, "stores": rows} -def _stat_only_file_count(store: pathlib.Path) -> tuple[int, bool]: - """(regular files below ``store`` minus bookkeeping, whether every directory was readable).""" - count, readable, pending = 0, True, [store] - while pending: - directory = pending.pop() - try: - with os.scandir(directory) as entries: - for entry in entries: - if entry.is_dir(follow_symlinks=False): - pending.append(pathlib.Path(entry.path)) - elif entry.is_file(follow_symlinks=False) and entry.name not in RESCUED_FILES_BOOKKEEPING: - count += 1 - except FileNotFoundError: - if directory != store: - readable = False # a subdirectory vanished mid-walk: the count is a floor - except OSError: - readable = False # permission denied, a file where the store should be, an I/O error - return count, readable +def _listed_file_count(task_id: str, store: pathlib.Path) -> int: + """Deliverables the shared listing finds in ``store``; raises when it cannot vouch for them. + + The listing reads a missing store as empty through ``exists()``, which also says + False for some unreadable paths and follows a link: the store's own ``lstat`` + decides first, so only a store that is not there counts as zero. + """ + from ouroboros.artifacts import collect_task_artifact_records, task_artifact_dir_path + + drive = store.parents[2] + if task_artifact_dir_path(drive, task_id) != store: + raise ValueError(f"{store} is not task {task_id}'s artifact store") + try: + mode = os.lstat(store).st_mode + except FileNotFoundError: + return 0 + if not stat.S_ISDIR(mode): # a file or a link where the store directory should be + raise NotADirectoryError(str(store)) + return len(collect_task_artifact_records(drive, task_id, measure=False, strict=True)) def rescued_files_sentence(fact: Dict[str, Any]) -> str: """ONE owner sentence for the stop receipt: the count, its state, and that no hash was computed.""" state, count = str(fact.get("state") or "unknown"), int(fact.get("count") or 0) + noun = "store" if len(fact.get("stores") or []) <= 1 else "stores" if state == "positive": - return f"Files rescued: {count} in the task's artifact store (counted by stat; hashes not computed)." + return f"Files rescued: {count} listed in the task's artifact {noun} (hashes not computed)." if state == "zero": - return "Files rescued: none — the task's artifact store was walked and holds no files (hashes not computed)." - seen = f"; {count} seen before the failure" if count else "" - return f"Files rescued: unknown — the task's artifact store could not be read{seen} (hashes not computed)." + return f"Files rescued: none — listing the task's artifact {noun} found no files (hashes not computed)." + seen = f"; {count} listed before the failure" if count else "" + return f"Files rescued: unknown — a task artifact store could not be listed{seen} (hashes not computed)." def model_execution_projection(usage: Dict[str, Any]) -> Dict[str, Any] | None: diff --git a/tests/test_cancel_terminal_delivery.py b/tests/test_cancel_terminal_delivery.py index 484f6b52a..b6d8b553c 100644 --- a/tests/test_cancel_terminal_delivery.py +++ b/tests/test_cancel_terminal_delivery.py @@ -645,10 +645,12 @@ def test_receipt_names_the_stop_cause_before_and_after_the_settle(tmp_path): def test_salvage_receipt_states_files_rescued_even_without_salvageable_text(tmp_path): """TZ-2 C2: "(no salvageable agent output ...)" must not read as "no files". The - receipt states the stat-only artifact-store count — positive, zero or unknown — and - that no hashes were computed; the typed fact rides ``cancel_receipt``. The count is a - mutable disclosure, never part of the content-derived delivery identity.""" + receipt states the artifact-store count from the shared unmeasured listing — positive, + zero or unknown — and that no hashes were computed; the typed fact rides + ``cancel_receipt``. The count is a mutable disclosure, never part of the + content-derived delivery identity, and staged inputs or receipts never raise it.""" from ouroboros.headless import task_artifacts_dir + from ouroboros.outcome_receipt_store import verification_receipts_path from supervisor import terminal_delivery as td def build(tid, task=None): @@ -668,6 +670,11 @@ def test_salvage_receipt_states_files_rescued_even_without_salvageable_text(tmp_ rebuilt = build("files-1") assert rebuilt["delivery_id"] == event["delivery_id"] and "Files rescued: 2 " in rebuilt["text"] assert load_task_result(tmp_path, "files-1")["cancel_receipt"]["files_rescued"]["count"] == 2 + (store / "attachments").mkdir() + (store / "attachments" / "brief.pdf").write_bytes(b"input") + verification_receipts_path(tmp_path, "files-1").write_text('{"check": "x"}\n', encoding="utf-8") + inputs_only = build("files-1") + assert inputs_only["delivery_id"] == event["delivery_id"] and "Files rescued: 2 " in inputs_only["text"] write_task_result(tmp_path, "files-0", STATUS_RUNNING, result="working") task_artifacts_dir(tmp_path, "files-0") diff --git a/tests/test_task_summary.py b/tests/test_task_summary.py index e7d608968..034598d63 100644 --- a/tests/test_task_summary.py +++ b/tests/test_task_summary.py @@ -248,12 +248,13 @@ def test_build_trace_summary_shows_structured_failure_facts(): assert "OMISSION NOTE" in pipeline.build_trace_summary(long_trace) -def test_facts_row_states_files_rescued_from_a_stat_only_walk(tmp_path, no_model_calls): +def test_facts_row_states_files_rescued_from_the_shared_unmeasured_listing(tmp_path, no_model_calls): """TZ-2 C2: at terminal the free facts row says how many files reached the task's - artifact store — a positive count, a confirmed zero, or unknown — from a stat-only - walk that discloses it computed no hashes. Store bookkeeping is not a rescued file, - an empty readable manifest alone never proves zero (the walk does), an unreadable - store is unknown (never zero), and a split root walks the child-drive store too.""" + artifact store — a positive count, a confirmed zero, or unknown — from the shared + unmeasured listing, disclosing that no hash was computed. Store bookkeeping is not a + rescued file, an empty readable manifest alone never proves zero (the listing does), + something other than a store directory is unknown (never zero), and a split root lists + the child-drive store too.""" from ouroboros.headless import task_artifacts_dir drive_logs = tmp_path / "logs" @@ -295,10 +296,149 @@ def test_facts_row_states_files_rescued_from_a_stat_only_walk(tmp_path, no_model {"store": str(task_artifacts_dir(child, "split-1", create=False)), "count": 1, "readable": True}]} -def test_rescued_files_walk_excludes_exactly_the_store_bookkeeping_names(): - """The bookkeeping names the walk skips are the SSOT literals, pinned so they cannot drift.""" +def test_rescued_files_count_listed_deliverables_never_store_metadata_inputs_or_receipts(tmp_path): + """The count is the shared listing's: registration metadata, the scratch manifest, the + receipt stream, staged inputs, chat media and source handles are not rescued files; an + unregistered output is one; a registration whose file is gone is not a file.""" from ouroboros.artifacts import _ARTIFACT_MANIFEST - from ouroboros.task_finalization import RESCUED_FILES_BOOKKEEPING - from ouroboros.workspace_patch_capture import SCRATCH_MANIFEST_NAME + from ouroboros.headless import SCRATCH_MANIFEST_NAME, task_artifacts_dir + from ouroboros.outcome_receipt_store import verification_receipts_path + from ouroboros.task_finalization import rescued_files_fact - assert RESCUED_FILES_BOOKKEEPING == frozenset({_ARTIFACT_MANIFEST, SCRATCH_MANIFEST_NAME}) + store = task_artifacts_dir(tmp_path, "mix-1") + (store / _ARTIFACT_MANIFEST).write_text(json.dumps({"schema_version": 1, "artifacts": { + "report.md": {"kind": "task_artifact", "name": "report.md", "path": str(store / "report.md")}, + "gone.md": {"kind": "task_artifact", "name": "gone.md", "path": str(store / "gone.md")}}}), encoding="utf-8") + (store / (_ARTIFACT_MANIFEST + ".lock")).write_text("", encoding="utf-8") + (store / SCRATCH_MANIFEST_NAME).write_text("{}", encoding="utf-8") + verification_receipts_path(tmp_path, "mix-1").write_text('{"check": "x"}\n', encoding="utf-8") + for rel in ("attachments/brief.pdf", "chat_media/photo.png", "source_handles/tool_results/r.txt"): + (store / rel).parent.mkdir(parents=True, exist_ok=True) + (store / rel).write_text("input", encoding="utf-8") + (store / "report.md").write_text("registered", encoding="utf-8") + (store / "unregistered.csv").write_text("1,2", encoding="utf-8") + (store / "nested").mkdir() + (store / "nested" / "notes.txt").write_text("n", encoding="utf-8") + + assert rescued_files_fact("mix-1", [store]) == { + "count": 3, "state": "positive", "hash_computed": False, + "stores": [{"store": str(store), "count": 3, "readable": True}]} + + only_inputs = task_artifacts_dir(tmp_path, "inputs-1") + (only_inputs / "attachments").mkdir() + (only_inputs / "attachments" / "brief.pdf").write_text("input", encoding="utf-8") + (only_inputs / _ARTIFACT_MANIFEST).write_text(json.dumps({"schema_version": 1, "artifacts": { + "gone.md": {"kind": "task_artifact", "name": "gone.md"}}}), encoding="utf-8") + assert rescued_files_fact("inputs-1", [only_inputs])["state"] == "zero" + + +def test_rescued_files_are_unknown_never_zero_when_the_listing_cannot_vouch(tmp_path, monkeypatch): + """A corrupt registration, a failed tree read, a link where the store should be and a + path that is not the task's store are each unknown with the readable stores' floor — + never a confirmed zero; a store never created is zero.""" + import os + + from ouroboros import artifacts + from ouroboros.headless import task_artifacts_dir + from ouroboros.task_finalization import rescued_files_fact, rescued_files_sentence + + good = task_artifacts_dir(tmp_path, "unk-2") + (good / "kept.txt").write_text("k", encoding="utf-8") + corrupt_drive = tmp_path / "corrupt-drive" + corrupt = task_artifacts_dir(corrupt_drive, "unk-2") + (corrupt / "out.txt").write_text("o", encoding="utf-8") + (corrupt / artifacts._ARTIFACT_MANIFEST).write_text('{"artifacts": {', encoding="utf-8") + fact = rescued_files_fact("unk-2", [good, corrupt]) + assert fact == {"count": 1, "state": "unknown", "hash_computed": False, + "stores": [{"store": str(good), "count": 1, "readable": True}, + {"store": str(corrupt), "count": 0, "readable": False}]} + assert rescued_files_sentence(fact) == ("Files rescued: unknown — a task artifact store could not be " + "listed; 1 listed before the failure (hashes not computed).") + + def unreadable_tree(_root): + raise PermissionError("tree read denied") + yield # pragma: no cover - a generator that fails on first read + + monkeypatch.setattr(artifacts, "iter_artifact_tree", unreadable_tree) + assert rescued_files_fact("unk-2", [good])["state"] == "unknown" + monkeypatch.undo() + + assert rescued_files_fact("unk-2", [good.parent / "someone-else"])["state"] == "unknown" + assert rescued_files_fact("unk-2", [tmp_path / "shallow"])["state"] == "unknown" + never = task_artifacts_dir(tmp_path, "never-2", create=False) + assert rescued_files_fact("never-2", [never]) == { + "count": 0, "state": "zero", "hash_computed": False, + "stores": [{"store": str(never), "count": 0, "readable": True}]} + + linked = task_artifacts_dir(tmp_path, "link-2", create=False) + try: + os.symlink(good, linked, target_is_directory=True) + except (OSError, NotImplementedError): + pytest.skip("directory symlinks are unavailable on this platform") + assert rescued_files_fact("link-2", [linked])["state"] == "unknown" + + +def test_rescued_files_count_physical_files_per_distinct_store_without_deduplicating(tmp_path): + """The canonical store and a custom actor drive's store are both listed; the same + relative path in each is two listed files (nothing proves one copy), and one store + named twice is listed once by ``artifact_store_roots``.""" + from ouroboros.headless import task_artifacts_dir + from ouroboros.task_finalization import artifact_store_roots, rescued_files_fact, rescued_files_sentence + + actor = tmp_path / "custom" / "actor-drive" + for drive in (tmp_path, actor): + (task_artifacts_dir(drive, "dup-1") / "out.txt").write_text("o", encoding="utf-8") + stores = artifact_store_roots(tmp_path, "dup-1", task={"child_drive_root": str(actor)}) + assert stores == [task_artifacts_dir(tmp_path, "dup-1", create=False), + task_artifacts_dir(actor, "dup-1", create=False)] + fact = rescued_files_fact("dup-1", stores) + assert (fact["count"], fact["state"], [row["count"] for row in fact["stores"]]) == (2, "positive", [1, 1]) + assert rescued_files_sentence(fact) == "Files rescued: 2 listed in the task's artifact stores (hashes not computed)." + assert artifact_store_roots(tmp_path, "dup-1", child_root=tmp_path) == [task_artifacts_dir(tmp_path, "dup-1", create=False)] + + +def test_rescued_files_fact_hashes_copies_registers_and_opens_nothing_but_the_registration(tmp_path, monkeypatch): + """Pure: no hash, no copy, no registration write, and the only file whose content is + read is the store's shared registration metadata.""" + import hashlib + import io + + from ouroboros import artifacts + from ouroboros.headless import task_artifacts_dir + from ouroboros.task_finalization import rescued_files_fact + + store = task_artifacts_dir(tmp_path, "pure-1") + (store / artifacts._ARTIFACT_MANIFEST).write_text(json.dumps({"schema_version": 1, "artifacts": { + "a.bin": {"kind": "task_artifact", "name": "a.bin", "immutable": True, "size": 2, "sha256": "0" * 64}}}), + encoding="utf-8") + (store / "a.bin").write_bytes(b"ab") + (store / "deep").mkdir() + (store / "deep" / "b.bin").write_bytes(b"cd") + + def snapshot(): + return sorted((str(p), p.stat().st_mtime_ns, p.read_bytes() if p.is_file() else b"") + for p in tmp_path.rglob("*")) + + before = snapshot() + opened = [] + real_open = io.open + + def spying_open(file, *args, **kwargs): + opened.append(str(file)) + return real_open(file, *args, **kwargs) + + def forbidden(*_args, **_kwargs): + raise AssertionError("the rescued-files fact hashes, measures, copies or registers nothing") + + monkeypatch.setattr(io, "open", spying_open) + monkeypatch.setattr(artifacts, "sha256", forbidden) + monkeypatch.setattr(hashlib, "sha256", forbidden) + for name in ("artifact_record", "stream_artifact_file", "_register_task_artifact_records", + "copy_file_to_task_artifacts", "update_json_locked"): + monkeypatch.setattr(artifacts, name, forbidden) + fact = rescued_files_fact("pure-1", [store]) + monkeypatch.undo() + + assert (fact["count"], fact["state"], fact["hash_computed"]) == (2, "positive", False) + assert set(opened) <= {str(store / artifacts._ARTIFACT_MANIFEST)}, opened + assert snapshot() == before diff --git a/tests/test_unread_mail_custody.py b/tests/test_unread_mail_custody.py index d473406bc..f214f0806 100644 --- a/tests/test_unread_mail_custody.py +++ b/tests/test_unread_mail_custody.py @@ -129,6 +129,40 @@ def test_forward_to_a_queued_task_is_read_when_it_starts_and_kept_if_it_never_do assert "start with the logs" in json.dumps(messages) and owner_mailbox.mail_read_state(drive, TASK, msg_id) is True +def test_forward_to_a_queued_task_cancelled_before_start_is_held_unread_as_its_exact_row(tmp_path): + """TZ-2 B5 acceptance: the real ``_forward_to_worker`` writes to a queued task and says + only that (queued; nothing has read it); the supervisor's pending drop then cancels the + task before it ever starts; the result reader shows the message as unread mail and the + authority carries the exact mailbox row. Nothing acknowledged it and nothing says "read".""" + from ouroboros.tools.control_task_results import _get_task_result + from ouroboros.tools.core import _forward_to_worker + + data = tmp_path / "data" + drive = headless.prepare_task_drive(data, TASK, "forked") + write_task_result(data, TASK, "scheduled", child_drive_root=str(drive), parent_task_id="parent1", + root_task_id="parent1", delegation_role="subagent") + ctx = SimpleNamespace(drive_root=data, task_id="parent1", task_metadata={}) + + receipt = _forward_to_worker(ctx, TASK, "start with the logs — then the config") + + assert f"({owner_mailbox.MAIL_QUEUED})" in receipt and "has not started, so nothing has read it" in receipt + assert "delivered" not in receipt and "next checkpoint" not in receipt + raw = owner_mailbox._mailbox_path(drive, TASK).read_text(encoding="utf-8") + assert raw.count("\n") == 1 and raw.endswith("\n") + row = raw[:-1] + msg_id = json.loads(row)["msg_id"] + + stored = write_task_result(data, TASK, "cancelled", strict_existing_dict=True, result="Cancelled before start.") + + assert stored["unread_mailbox"]["rows"] == [row] and stored["unread_mailbox"]["read_complete"] is True + text = _get_task_result(ctx, TASK) + assert "[UNREAD_MAILBOX] 1 message(s)" in text and "start with the logs — then the config" in text + authority = json.loads(_get_task_result(ctx, TASK, include_authority=True))["authority"] + assert authority["status"] == "cancelled" and authority["unread_mailbox"]["rows"] == [row] + assert owner_mailbox.mail_read_state(drive, TASK, msg_id) is False + assert owner_mailbox.acknowledged_task_message_ids(drive, TASK) == set() + + def test_mail_write_receipt_vocabulary(): assert owner_mailbox.mail_write_receipt("scheduled")["receipt"] == owner_mailbox.MAIL_QUEUED assert owner_mailbox.mail_write_receipt("running")["receipt"] == owner_mailbox.MAIL_DELIVERED