ouroboros/tests/test_reference_book_validation.py
Ouroboros 35b8151819 Split the two reference books into verbatim chapters
The books were one physical file each: ARCHITECTURE 2,386 lines and
DEVELOPMENT 3,886, with 13 and 14 `##` sections. `reference_books.py` had
shipped the chaptered reader — membership, authored introductions, exact
physical source views — with the migration still at zero, so every reader
took the legacy monolith branch and the validator had no production caller.

Each `##` section is now one chapter file under `docs/architecture/` or
`docs/development/`. The only new bytes per chapter are its prologue: the old
section title at H1 (numbering text kept, so every `ARCHITECTURE "8. Git
Branching, CI, and Build"` cross-reference still reads) and one authored
introductory paragraph saying what the chapter owns and why it exists.
Everything after that prologue is the old section body byte for byte, with
`###`/`####` levels untouched — so the residue rules keep reading the exact
subsection headings they exempt, and no section title was renamed.

The move is therefore INVERTIBLE, and `tests/test_reference_book_migration.py`
inverts it: drop each chapter's H1 line and its one introduction, re-prefix
`## `, concatenate in membership order, and require the recorded SHA-256 of
the old body — plus, whenever the base commit is reachable, byte equality with
`git show <base>:<path>`. `docs/reference-books-migration.md` is the operator
transfer table: every row a verbatim move, with its line range at the base, its
destination and an empty rename column. It lives directly under `docs/`, so it
is reviewable without becoming a book member.

`_preamble` now takes the FIRST paragraph under the H1 instead of demanding the
only one before the first H2. Most sections open with prose at the level they
already had, so the old rule could only be satisfied by promoting `###` to `##`
or inventing a sub-heading — either of which would rewrite what this migration
relocates verbatim. A source whose H1 is followed straight by a subsection
still has no introduction and is still refused, which is the property that
keeps an overview from quoting body prose as authored orientation.

`.gitattributes` pins `docs/**/*.md` to LF: chapter line ranges, byte spans and
SHA-256s are physical facts that a Windows checkout must not rewrite, and
`full-test` runs on windows-latest for every PR. The two ARCHITECTURE-derived
generated inventories are regenerated, because a chaptered section now carries
its physical provenance note.

Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com>
2026-09-15 19:20:44 +03:00

134 lines
5.9 KiB
Python

"""Candidate-exact book admission shared by CI and preflight callers."""
import hashlib
import pathlib
import subprocess
import pytest
from ouroboros.reference_books import (
BOOK_ENTRYPOINTS,
load_reference_book,
read_book_section,
validate_reference_books,
)
def _corpus():
corpus = {}
for book_id, entrypoint in BOOK_ENTRYPOINTS.items():
corpus[entrypoint] = (
f"# {book_id.title()}\n\nThe authored orientation.\n\n## Chapters\n\n"
f"- [Useful knowledge]({book_id}/a.md)\n"
).encode()
corpus[f"docs/{book_id}/a.md"] = (
"# Useful knowledge\n\nWHY this exists and where to look.\n\n"
"## Exact section\n\nComplete source.\n"
).encode()
return corpus
def _validate(corpus, *, tracked=None, require_chaptered=True):
return validate_reference_books(
pathlib.Path("/no-filesystem-source"),
tracked_paths=corpus if tracked is None else tracked,
require_chaptered=require_chaptered,
read_bytes=corpus.__getitem__,
)
def test_complete_candidate_and_exact_named_section_keep_physical_source():
corpus = _corpus()
assert _validate(corpus) == ()
book = load_reference_book(pathlib.Path("/unused"), "architecture", corpus.__getitem__)
view = read_book_section(book, "Exact section")
ref = view.sources[0]
assert ref.path == "docs/architecture/a.md"
raw = corpus[ref.path]
assert ref.sha256 == hashlib.sha256(raw).hexdigest()
assert raw[ref.span.start_byte:ref.span.end_byte].decode() == view.text
assert ref.span.start_line == 5
def test_exact_index_bytes_override_worktree_and_population(tmp_path):
corpus = _corpus()
for path in corpus:
target = tmp_path / path
target.parent.mkdir(parents=True, exist_ok=True)
target.write_text("Deliberately invalid working-tree copy")
assert validate_reference_books(
tmp_path, tracked_paths=corpus, require_chaptered=True,
read_bytes=corpus.__getitem__,
) == ()
assert validate_reference_books(tmp_path, tracked_paths=corpus, require_chaptered=True)
def test_orphan_and_untracked_members_remain_separate_missing_source_facts():
corpus = _corpus()
tracked = set(corpus) | {"docs/architecture/orphan.md"}
tracked.remove("docs/development/a.md")
problems = _validate(corpus, tracked=tracked)
assert any("unlisted chapter docs/architecture/orphan.md" in p for p in problems)
assert any("chapter absent from candidate tree docs/development/a.md" in p for p in problems)
assert not any("Complete source" in p for p in problems)
def test_short_entrypoint_without_membership_cannot_satisfy_final_admission():
corpus = {path: b"# Short entrypoint\n\nSome orientation only.\n" for path in BOOK_ENTRYPOINTS.values()}
assert _validate(corpus, require_chaptered=False) == ()
problems = _validate(corpus, require_chaptered=True)
assert len(problems) == 2
assert all("chaptered source required" in p for p in problems)
@pytest.mark.parametrize("edit,expected", [
(lambda c: c.pop("docs/architecture/a.md"), "docs/architecture/a.md"),
(lambda c: c.update({"docs/architecture/a.md": b""}), "nonempty H1"),
(lambda c: c.update({"docs/architecture/a.md": b"# Chapter\n\n## Details\n\nBody\n"}), "introductory paragraph"),
(lambda c: c.update({"docs/ARCHITECTURE.md": b"# Book\n\nOrientation.\n\n## Chapters\n\n"}), "nonempty, flat Markdown list"),
(lambda c: c.update({"docs/ARCHITECTURE.md": c["docs/ARCHITECTURE.md"] + b"- [Again](architecture/a.md)\n"}), "duplicate book chapter"),
(lambda c: c.update({"docs/architecture/a.md": b"---\nx: [unterminated\n---\n# Body\n"}), "expected"),
])
def test_invalid_candidate_reports_failure_instead_of_partial_success(edit, expected):
corpus = _corpus()
edit(corpus)
problems = _validate(corpus)
assert problems and any(expected in p for p in problems)
def test_duplicate_named_sections_are_ambiguous_instead_of_first_match():
corpus = _corpus()
corpus["docs/architecture/a.md"] += b"\n## Exact section\n\nDifferent source.\n"
book = load_reference_book(pathlib.Path("/unused"), "architecture", corpus.__getitem__)
with pytest.raises(ValueError, match="found 2"):
read_book_section(book, "Exact section")
def test_current_reference_books_pass_the_final_chaptered_admission():
"""The production caller of the validator: the tracked tree, chaptered.
This is the docs-lane check item 8 asks for -- a missing chapter, an
unlisted one, or a chapter whose authored introduction was lost fails here
rather than at the next review that assembles a book.
"""
root = pathlib.Path(__file__).resolve().parents[1]
tracked = subprocess.check_output(["git", "ls-files", "-z"], cwd=root).decode().split("\0")
assert validate_reference_books(root, tracked_paths=tracked, require_chaptered=True) == ()
def test_docs_sync_reads_full_chapters_and_never_loses_residue_between_files(monkeypatch):
import tests.test_docs_sync as docs
corpus = _corpus()
original_load = load_reference_book
monkeypatch.setattr(docs, "load_reference_book", lambda root, book_id: original_load(root, book_id, corpus.__getitem__))
assert "WHY this exists" in docs._read("docs/ARCHITECTURE.md")
assert "Complete source" in docs._architecture_section("Exact section")
corpus["docs/development/a.md"] += b"\n### Documentation contract\n\nThe word previously is quoted here.\n"
extra = "docs/development/b.md"
corpus[extra] = b"# Next chapter\n\nThis previously behaved differently.\n\n## Details\n\nBody.\n"
corpus["docs/DEVELOPMENT.md"] += b"- [Next](development/b.md)\n"
with pytest.raises(AssertionError, match="b.md.*residue grew"):
docs.test_resident_docs_residue_only_shrinks()
counts = docs.doc_residue_counts("docs/ARCHITECTURE.md", "# Chapter (v1.2)\n", is_entrypoint=False)
assert counts["(preamble)"]["version_stamp"] == 1