ouroboros/scripts/domain_report.py
Anton Razzhigaev 98ca8a13d5 Retire completed campaign artifacts and keep repository reports explicit
Remove completed campaign records, one-time adoption/transplant machinery and
incidental line floors. Keep current contracts and generated inventories with
their readers, and direct optional domain reports to stdout or an explicit file.
Document continuing-purpose review in the existing handbook and checklists.

No version bump; ordinary release gates and runtime behavior are preserved.

Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com>
2026-09-18 01:07:51 +03:00

263 lines
11 KiB
Python

#!/usr/bin/env python3
"""Report-only domain quotient graph for the current source tree.
Reads ``ouroboros/domains.toml`` (module -> domain, 1:1), computes the
import graph of THIS tree through the shared ``scripts/domain_graph.py`` core,
collapses modules to domain nodes and REPORTS:
- domain-level edges of the strict graph (unconditional module-level imports),
- cycles on the domain quotient, each with its exact module-edge witnesses,
- a dependency direction table,
- lazy / guarded / dynamic / TYPE_CHECKING imports, classified separately and
excluded from the strict graph.
This tool NEVER gates: the GATE over the same data is
``scripts/check_domains.py`` + ``tests/test_domain_manifest.py``, which pin
the manifest's generated baseline sections. The report stays the witness-level
companion: exit code is 0 whenever the report was produced, regardless of
findings. Exit 2 only when the report itself cannot be trusted
(unreadable/unparseable source, missing manifest) — a silent skip would
falsify the graph.
Output goes to stdout by default, or to an explicit ``--output PATH``.
Diagnostics go to stderr. The report header names the generator and input SHAs.
"""
from __future__ import annotations
import argparse
import hashlib
import pathlib
import subprocess
import sys
from collections import Counter, defaultdict
from datetime import datetime, timezone
REPO_ROOT = pathlib.Path(__file__).resolve().parent.parent
sys.path.insert(0, str(REPO_ROOT))
from scripts.domain_graph import ( # noqa: E402
DYNAMIC,
GUARDED,
LAZY,
MANIFEST_PATH,
STRICT,
TYPE_ONLY,
build_import_graph,
cycle_groups_of,
load_manifest,
tracked_population,
)
def main(argv: list[str] | None = None) -> int:
ap = argparse.ArgumentParser(description=__doc__.splitlines()[0])
ap.add_argument("--output", type=pathlib.Path,
help="write the report to this path instead of stdout")
args = ap.parse_args(argv)
if not MANIFEST_PATH.is_file():
print(f"missing manifest: {MANIFEST_PATH}", file=sys.stderr)
return 2
manifest = load_manifest()
modules = manifest.modules
domains = manifest.domains
proposed = manifest.proposed
bad_domains = sorted({d for d in modules.values() if d not in domains})
if bad_domains:
print(f"manifest names unknown domains: {bad_domains}", file=sys.stderr)
return 2
head = subprocess.run(
["git", "rev-parse", "HEAD"], cwd=REPO_ROOT, capture_output=True, text=True, check=True
).stdout.strip()
tracked_set = tracked_population(REPO_ROOT)
# Content fingerprint of the actual analyzed inputs (working-tree bytes,
# not the HEAD claim): the report is often generated pre-commit, so HEAD
# alone cannot bind it to a SHA. Verification against any later tree =
# regenerate there and compare fingerprints.
fp = hashlib.sha256()
for path in sorted(tracked_set):
f = REPO_ROOT / path
if f.is_file():
fp.update(path.encode())
fp.update(b"\x00")
fp.update(f.read_bytes())
fp.update(b"\x00")
content_fingerprint = fp.hexdigest()
drift_missing = sorted(tracked_set - set(modules)) # in tree, not in manifest
drift_stale = sorted(set(modules) - tracked_set) # in manifest, not in tree
try:
graph = build_import_graph(manifest, REPO_ROOT, population=tracked_set)
except (OSError, SyntaxError) as exc:
print(f"cannot parse: {exc}", file=sys.stderr)
return 2
edges = graph.edges
dom_of_path = graph.dom_of_path
dynamic_unresolved = graph.dynamic_unresolved
dom_edges = graph.quotient(STRICT)
cycles = cycle_groups_of(sorted(domains), dom_edges)
# ---- render ---------------------------------------------------------------
used_domains = sorted({d for pair in dom_edges for d in pair} | set(Counter(dom_of_path.values())))
manifest_sha = manifest.sha256
now = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M UTC")
dom_counts = Counter(dom_of_path.values())
L: list[str] = []
L.append("# Domain quotient report (report-only)")
L.append("")
L.append(f"Generated by `scripts/domain_report.py` on {now}. Do not edit.")
L.append("")
L.append(f"- generated from the WORKING TREE based on HEAD `{head}` "
"(uncommitted input changes are covered by the content fingerprint below)")
L.append(f"- analyzed-inputs content fingerprint: sha256 `{content_fingerprint}` "
"(sorted tracked runtime files, path+bytes; regenerate on any tree and "
"compare to verify freshness)")
L.append(f"- manifest: `ouroboros/domains.toml` sha256 `{manifest_sha}`")
L.append(f"- reference (domain vocabulary): {manifest.meta.get('reference_tree', 'unknown')}")
L.append("- discipline: this is a REPORT with witness-level detail; the GATE over the")
L.append(" same data is `scripts/check_domains.py` (+ `tests/test_domain_manifest.py`),")
L.append(" which pins the manifest's `[graph]`/`[duplicates]` baseline sections.")
L.append("")
L.append("## Population")
L.append("")
L.append(f"- modules mapped: **{len(modules)}** across **{len(dom_counts)}** domains"
f" ({len(proposed)} rows are `classification=proposed`)")
if drift_missing:
L.append(f"- **manifest drift** — tracked but unmapped: {', '.join(f'`{p}`' for p in drift_missing)}")
if drift_stale:
L.append(f"- **manifest drift** — mapped but not in tree: {', '.join(f'`{p}`' for p in drift_stale)}")
if not drift_missing and not drift_stale:
L.append("- manifest drift: none (manifest == tracked population)")
L.append("")
L.append("| domain | modules | proposed |")
L.append("|---|---:|---:|")
for d in sorted(dom_counts):
n_prop = sum(1 for p, dd in dom_of_path.items() if dd == d and p in proposed)
L.append(f"| {d} — {domains[d]} | {dom_counts[d]} | {n_prop} |")
L.append("")
n_strict = sum(len(v) for v in edges[STRICT].values())
L.append("## Strict import graph")
L.append("")
L.append(f"- module-level import edges (unconditional): **{len(edges[STRICT])}** unique, {n_strict} statements")
L.append(f"- cross-domain module edges: **{sum(len(v) for v in dom_edges.values())}**")
L.append(f"- domain-level edges (quotient): **{len(dom_edges)}**")
L.append(f"- quotient cycles (SCCs with >1 domain): **{len(cycles)}**")
L.append("")
L.append("## Quotient cycles")
L.append("")
if not cycles:
L.append("None. The strict domain quotient is acyclic.")
for i, comp in enumerate(cycles, 1):
comp_set = set(comp)
n_all_wit = sum(
len(wit) for (d1, d2), wit in dom_edges.items()
if d1 in comp_set and d2 in comp_set
)
L.append(f"### Cycle group {i}: {' ⇄ '.join(comp)}")
L.append("")
L.append(f"{len(comp)} domains form one strongly connected component with "
f"{n_all_wit} module-edge witnesses. Every edge below needs an owner "
"disposition (regroup / split / allowed-edge).")
L.append("")
for (d1, d2) in sorted(dom_edges):
if d1 in comp_set and d2 in comp_set:
wit = dom_edges[(d1, d2)]
L.append(f"- **{d1} → {d2}** ({len(wit)} module edges)")
for (src, dst) in wit:
lines = ",".join(str(n) for n in sorted(set(edges[STRICT][(src, dst)]))[:4])
L.append(f" - `{src}` → `{dst}` (line {lines})")
L.append("")
L.append("## Domain-level edges (strict)")
L.append("")
L.append("| from | to | module edges | in a cycle group |")
L.append("|---|---|---:|:---:|")
for (d1, d2) in sorted(dom_edges):
incyc = "yes" if any(d1 in c and d2 in c for c in map(set, cycles)) else ""
L.append(f"| {d1} | {d2} | {len(dom_edges[(d1, d2)])} | {incyc} |")
L.append("")
L.append("## Dependency direction table")
L.append("")
L.append("Rows import columns (module-edge counts, strict graph).")
L.append("")
header = "| ↓ imports → | " + " | ".join(used_domains) + " |"
L.append(header)
L.append("|---|" + "---|" * len(used_domains))
for d1 in used_domains:
row = [f"| **{d1}**"]
for d2 in used_domains:
n = len(dom_edges.get((d1, d2), ()))
row.append(str(n) if n else "·")
L.append(" | ".join(row) + " |")
L.append("")
for kind, title in ((LAZY, "Lazy imports (function-level)"),
(GUARDED, "Guarded imports (__main__-only / failure-swallowing try)"),
(TYPE_ONLY, "TYPE_CHECKING imports"),
(DYNAMIC, "Dynamic imports (importlib / __import__)")):
pairs = edges[kind]
dom_pairs: dict[tuple[str, str], list[tuple[str, str]]] = defaultdict(list)
for (src, dst) in sorted(pairs):
d1, d2 = dom_of_path[src], dom_of_path[dst]
if d1 != d2:
dom_pairs[(d1, d2)].append((src, dst))
L.append(f"## {title} — excluded from the strict graph")
L.append("")
L.append(f"{len(pairs)} module edges, {len(dom_pairs)} cross-domain pairs.")
L.append("")
if dom_pairs:
L.append("| from | to | module edges | pair also strict? |")
L.append("|---|---|---:|:---:|")
for (d1, d2) in sorted(dom_pairs):
also = "yes" if (d1, d2) in dom_edges else "**lazy-only**" if kind == LAZY else "no"
L.append(f"| {d1} | {d2} | {len(dom_pairs[(d1, d2)])} | {also} |")
L.append("")
lazy_only = [(k, v) for k, v in sorted(dom_pairs.items()) if k not in dom_edges]
if lazy_only:
L.append(f"Cross-domain pairs reachable ONLY through {kind} imports (hidden coupling):")
L.append("")
for (d1, d2), wit in lazy_only:
L.append(f"- **{d1} → {d2}**:")
for (src, dst) in wit[:20]:
L.append(f" - `{src}` → `{dst}`")
if len(wit) > 20:
L.append(f" - … and {len(wit) - 20} more")
L.append("")
if dynamic_unresolved:
L.append("Unresolved dynamic import call sites (non-literal target):")
L.append("")
for path, lineno in dynamic_unresolved:
L.append(f"- `{path}:{lineno}`")
L.append("")
# Sections append a trailing "" as their separator, so the last one leaves a
# blank element: joining it would end the file with a blank LINE, which the
# repository's whitespace gate (`git diff --check`) reports. Drop the
# trailing separators and write exactly one final newline.
while L and not L[-1]:
L.pop()
report = "\n".join(L) + "\n"
if args.output is None:
sys.stdout.write(report)
else:
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(report, encoding="utf-8")
print(f"wrote {args.output}", file=sys.stderr)
print(f"strict: {len(edges[STRICT])} module edges, {len(dom_edges)} domain edges, {len(cycles)} cycle groups",
file=sys.stderr)
for comp in cycles:
print(" cycle:", " ⇄ ".join(comp), file=sys.stderr)
return 0
if __name__ == "__main__":
sys.stdout.reconfigure(encoding="utf-8")
sys.stderr.reconfigure(encoding="utf-8")
sys.exit(main())