#!/usr/bin/env python3 """Validate ErsatzTV decision-record lifecycle invariants. Replaces the line-level append-only guard (ersatztv#303 H9) with lifecycle checks that preserve its spirit — rationale is never silently rewritten or deleted; every history touch is deliberate and reviewable. `[decisions-edit]` is kept ONLY for rationale-prose edits (see body-diff below); routine lifecycle metadata writes are token-free. Fail-open on tooling trouble (missing refs, git errors, parse issues), matching the old guard. """ from __future__ import annotations import argparse import subprocess import sys from pathlib import Path import scripts.decisions_lib as dl # noqa: E402 (run with PYTHONPATH=. or as module) SKIP_HEADINGS = {"Index", "Active catalog"} REQUIRED_META = ("key", "status") EDIT_TOKEN = "[decisions-edit]" # noqa: S105 (a commit-message marker, not a credential) def _key_of(ref: str | None) -> str | None: if not ref or ref == "none": return None return ref.split("@", 1)[0].strip() def validate(records, *, archive_keys, catalog_ok, budget_ok, removed, rewritten) -> list[str]: errs: list[str] = [] decision_recs = [r for r in records if r.heading not in SKIP_HEADINGS and r.status != "legacy-unmigrated"] active_by_key: dict[str, int] = {} known_keys = set(archive_keys) for r in decision_recs: if r.status not in dl.STATUSES: errs.append(f"{r.heading}: invalid status {r.status!r}") if not r.key or not dl.KEY_RE.match(r.key): errs.append(f"{r.heading}: bad key format {r.key!r}") continue known_keys.add(r.key) for f in REQUIRED_META: if getattr(r, f) in (None, ""): errs.append(f"{r.heading}: missing required metadata {f}") if r.status == "active": active_by_key[r.key] = active_by_key.get(r.key, 0) + 1 for key, n in active_by_key.items(): if n > 1: errs.append(f"key {key!r}: more than one active record ({n})") # reciprocal supersession (successor may live in active OR archive) for r in decision_recs: sk = _key_of(r.superseded_by) if sk and sk not in known_keys: errs.append(f"{r.heading}: superseded-by points to unknown key {sk!r}") pk = _key_of(r.supersedes) if pk and pk not in known_keys: errs.append(f"{r.heading}: supersedes points to unknown key {pk!r}") for h in removed: errs.append(f"record removed from the active set without an archive copy: {h!r}") for h in rewritten: errs.append(f"rationale prose of {h!r} changed without the {EDIT_TOKEN} token") if not catalog_ok: errs.append("docs/decisions/README.md active catalog is stale — run build_decisions_catalog.py") if not budget_ok: errs.append("aggregate active-corpus budget exceeded (see --budget)") return errs # ---- git helpers (all fail-open: return neutral values on any error) ---- def _run(args: list[str]) -> str | None: r = subprocess.run(args, capture_output=True, text=True) return r.stdout if r.returncode == 0 else None def _merge_base(base: str, head: str) -> str | None: out = _run(["git", "merge-base", base, head]) return out.strip() if out else None def _active_paths_at(ref: str) -> list[str]: """Active decision files at a ref: decisions.md + docs/decisions/*.md minus README/archive.""" out = _run(["git", "ls-tree", "-r", "--name-only", ref, "docs/decisions.md", "docs/decisions/"]) if out is None: return [] paths = [] for p in out.splitlines(): if not p.endswith(".md"): continue if p.startswith("docs/decisions/archive/") or p == "docs/decisions/README.md": continue paths.append(p) return paths def _records_at(ref: str, paths: list[str]): """{heading: Record} across the given paths at a ref.""" by_heading = {} for p in paths: blob = _run(["git", "show", f"{ref}:{p}"]) if blob is None: continue for rec in dl.parse_text(blob, Path(p)): if rec.heading not in SKIP_HEADINGS: by_heading[rec.heading] = rec return by_heading def _archive_records_at(ref: str): out = _run(["git", "ls-tree", "-r", "--name-only", ref, "docs/decisions/archive/"]) paths = [p for p in (out or "").splitlines() if p.endswith(".md")] by_heading = {} for p in paths: blob = _run(["git", "show", f"{ref}:{p}"]) if blob is None: continue for rec in dl.parse_text(blob, Path(p)): by_heading[rec.heading] = rec return by_heading def _rationale(rec) -> str: """Record body with the metadata block stripped, whitespace-normalized — the prose we protect.""" keep = [] seen_meta = False for line in rec.body.splitlines(): s = line.strip() if not seen_meta and ( s.startswith("`key:") or s.startswith("**Rule:**") or s.startswith("**Signals:**") or s.startswith("**Mechanics:**") ): seen_meta = True continue if seen_meta and ( s.startswith("**Rule:**") or s.startswith("**Signals:**") or s.startswith("**Mechanics:**") or s.startswith("`key:") ): continue keep.append(s) return "\n".join(x for x in keep if x) def _diff_findings(base: str, head: str) -> tuple[list[str], list[str]]: """(removed, rewritten) between merge-base(base,head) and head. Fail-open → ([], []).""" mb = _merge_base(base, head) if not mb: return [], [] if EDIT_TOKEN.lower() in (_run(["git", "log", "--format=%B", f"{mb}..{head}"]) or "").lower(): token = True else: token = False base_active = _records_at(mb, _active_paths_at(mb)) head_active = _records_at(head, _active_paths_at(head)) head_archive = _archive_records_at(head) gone = set(base_active) - set(head_active) removed = sorted(h for h in gone if h not in head_archive) rewritten: list[str] = [] if not token: # surviving records whose rationale prose changed for h in set(base_active) & set(head_active): if _rationale(base_active[h]) != _rationale(head_active[h]): rewritten.append(h) # archived records must body-match their base active copy (no laundering rewrites via archive) for h in gone & set(head_archive): if _rationale(base_active[h]) != _rationale(head_archive[h]): rewritten.append(h) return removed, sorted(rewritten) def _archive_keys() -> set[str]: keys: set[str] = set() if dl.ARCHIVE_DIR.exists(): for f in dl.ARCHIVE_DIR.glob("*.md"): for r in dl.parse_file(f): if r.key: keys.add(r.key) return keys def _budget_ok(limit: int) -> bool: total = 0 for f in dl.active_files(): if f.exists(): total += len(f.read_text(encoding="utf-8").splitlines()) cat = dl.TOPIC_DIR / "README.md" if cat.exists(): total += len(cat.read_text(encoding="utf-8").splitlines()) return total <= limit def _catalog_ok() -> bool: try: import scripts.build_decisions_catalog as bc # pyright: ignore[reportMissingImports] except Exception: return True # fail-open; the CI job also runs an independent --check step want = bc.render_catalog(dl.all_active_records()) cat = dl.TOPIC_DIR / "README.md" have = cat.read_text(encoding="utf-8") if cat.exists() else "" return want.strip() == have.strip() def main(argv=None) -> int: ap = argparse.ArgumentParser() ap.add_argument("--base") ap.add_argument("--head") ap.add_argument("--budget", type=int, default=4200) # aggregate; ratchet down as archive grows args = ap.parse_args(argv) records = dl.all_active_records() removed, rewritten = _diff_findings(args.base, args.head) if args.base and args.head else ([], []) errs = validate( records, archive_keys=_archive_keys(), catalog_ok=_catalog_ok(), budget_ok=_budget_ok(args.budget), removed=removed, rewritten=rewritten, ) unmigrated = [r for r in records if r.status == "legacy-unmigrated" and r.heading not in SKIP_HEADINGS] if unmigrated: print(f"::notice::{len(unmigrated)} legacy-unmigrated decision record(s) remain (must trend to 0).") if errs: for e in errs: print(f"decisions-validate: {e}", file=sys.stderr) return 1 print("decisions-validate: OK") return 0 if __name__ == "__main__": raise SystemExit(main())