#!/usr/bin/env python3 """What does each `kind` label in `authored/` actually REST on? Every authored entry carries a `kind` -- `measured`, `authored`, `name match, not measured` -- and a `why`. The label is the load-bearing part: `measured` means the port is repeating something observed off the running game, and a reader downstream will treat it as fact. Nothing has ever checked them. That is the point: **a discipline that has never visibly failed is the one nothing directs attention at.** The Decoder reached this from the input side -- Ⓐ and Ⓑ were delivery-confirmed because they had once broken, so the d-pad never was -- and on the same day a `measured` label of mine turned out to rest on a single entry that may have been measuring history. So this checks what is checkable about a label, and is explicit that the rest is not: citations resolvable references in the `why` -- a `docs/` path that exists on some ref, a commit sha that resolves, a capture filename BARE a label whose `why` cites nothing a reader could go and open DANGLING a citation that does not resolve anywhere in the repository 🔴 What it CANNOT do is read the cited page and confirm it says what the `why` claims. A label with three resolvable citations can still be wrong. This narrows "which labels rest on nothing" from unknown to a list; it does not audit meaning. """ import json, glob, os, re, subprocess, sys REFS = None def known_paths(): """Every path in the repo, across ALL refs -- docs/re/ lives on a branch. Checked against the working tree as well: a file added this iteration is not in any ref yet, and reporting a citation to it as unresolvable would make the audit fail every time it is itself referenced. """ global REFS if REFS is None: out = subprocess.run(["git", "rev-list", "--all", "--objects"], capture_output=True, text=True).stdout REFS = {l.split(" ", 1)[1] for l in out.splitlines() if " " in l} return REFS HANDOFF_TEXT = None def handoff(): """The live HANDOFF, so a cited Q number is checked against the real table.""" global HANDOFF_TEXT if HANDOFF_TEXT is None: sha = subprocess.run(["git", "log", "--all", "--format=%h", "--", "docs/port/HANDOFF.md"], capture_output=True, text=True).stdout.split()[0] HANDOFF_TEXT = subprocess.run(["git", "show", f"{sha}:docs/port/HANDOFF.md"], capture_output=True, text=True).stdout return HANDOFF_TEXT def sha_ok(s): r = subprocess.run(["git", "cat-file", "-e", s + "^{commit}"], capture_output=True) return r.returncode == 0 def text_of(why): if isinstance(why, str): return why if isinstance(why, list): return " ".join(str(x) for x in why) return "" def citations(t): """References a reader could actually follow.""" out = [] for p in re.findall(r"\b(?:docs|crates|port|tools|authored)/[\w./-]+\w", t): out.append(("path", p.rstrip(".,"))) for sha in re.findall(r"\b([0-9a-f]{7,40})\b", t): # 🔴 A PURE-DECIMAL RUN IS NOT A SHA. `1118268` and `1171516` are byte # counts in `voice/presentation_why`, and this reported them as # unresolvable commits -- a DANGLING verdict on a why that cites # nothing of the kind. A sha in this corpus always carries at least one # of a-f; requiring that removes the whole class without a length rule. if any(c in "abcdef" for c in sha): out.append(("sha", sha)) for p in re.findall(r"\b([\w-]+\.(?:png|txt|tsv|wav))\b", t): out.append(("file", p)) # The corpus cites two things that are not paths and are still followable: # a HANDOFF question number, and a MISSION section. Leaving these out made # the first run report four labels as resting on nothing when they rest on # the two documents the mission names -- an audit inventing defects is worse # than no audit, because its false positives are indistinguishable from its # true ones until each is opened. for q in re.findall(r"HANDOFF Q(\d+)", t): out.append(("handoff", "Q" + q)) for m in re.findall(r"(PORT-MISSION|MISSION)[ ]section[ ](\d+)", t): out.append(("mission", m[1])) for r in re.findall(r"MODDING rule (\d+)", t): out.append(("modding", r)) # 🔴 A CAPTURE FILENAME IS A CITATION and this could not see one. Five of the # sixteen `why` fields I reported as uncited name `live-extras.png` or an # equivalent -- openable, in `docs/re/captures/`, and exactly the evidence a # reader wants. My published "17 uncited" was inflated by a third by my own # extractor, which is the invents-defects failure aimed at my own backlog. for cap in re.findall(r"\b([\w-]+\.(?:png|txt|wav|tsv))\b", t): out.append(("capture", cap)) # ⚠️ A bare `HANDOFF` names the document and not the section. Counted, and # counted SEPARATELY, because "the contract says so" is a weaker pointer than # "Q5 says so" -- it sends a reader to 4 000 lines. if re.search(r"\bHANDOFF\b", t) and not re.search(r"HANDOFF Q\d+", t): out.append(("handoff-vague", "HANDOFF")) return out def walk(o, f, path, out): if isinstance(o, dict): for k, v in o.items(): if (k == "kind" or k.endswith("_kind")) and isinstance(v, str): stem = "" if k == "kind" else k[: -len("_kind")] own = o.get((stem + "_why") if stem else "why") # 🔴 An earlier version fell back to the parent's `why` when a # label had none of its own, and reported the result as `ok`. # That credits a label with evidence for a DIFFERENT claim: # every `goto_name_kind` scored on a sibling `why` about the # DESTINATION, while the label is about where the NAME came # from. Borrowed evidence is now its own outcome, because a # label resting on a neighbour's argument is exactly the case # this audit exists to surface. out.append((f, path + "/" + k, v, text_of(own), own is None and bool(text_of(o.get("why"))))) walk(v, f, path + "/" + k, out) elif isinstance(o, list): for x in o: walk(x, f, path, out) def selftest(): """Does this audit notice a label that rests on nothing? 🔴 THE GAP: `audit-kinds` has always reported what it found and never been asked whether it can find anything. A walk that matched no labels, a citation extractor that accepted everything, or a `main` that returned 0 regardless would all have produced the same clean run -- and clean runs from this tool are cited in `DECISIONS.md` as evidence that fifteen labels are grounded. Three synthetic rows are pushed through the REAL classifier, and its verdict is read rather than reasoned about: a `why` citing nothing -> must be BARE a `why` citing a path that exists -> must be ok a `why` citing a path that does not -> must be DANGLING Exit codes follow the convention the Decoder and I converged on: 0 all good, 1 a real audit failure, **2 the harness is broken** and no clean run from it means anything. """ paths = known_paths() # The liveness case belongs in the self-test too, driven as a subprocess so # its real exit code is read rather than reasoned about. empty = os.path.join(os.environ.get("TMPDIR", "/tmp"), "audit-kinds-liveness") os.makedirs(empty, exist_ok=True) got = subprocess.run([sys.executable, os.path.abspath(__file__)], cwd=empty, capture_output=True).returncode print(f" harness: an empty tree -> exit {got} (want 2) " f"{'✅' if got == 2 else '🔴 examined nothing and reported clean'}") live_ok = got == 2 cases = [ ("bare", "no citation of any kind here, just prose", "BARE"), ("ok", "see tools/port/audit-kinds for the method", "ok"), ("dangling", "see docs/port/NO-SUCH-FILE-XYZ.md", "DANGLING"), ] bad = 0 for name, why, want in cases: cites = citations(why) if not cites: got = "BARE" else: unresolved = [c for t, c in cites if t == "path" and c not in paths and not os.path.exists(c)] got = "DANGLING" if unresolved else "ok" mark = "✅" if got == want else "🔴" print(f" harness: a why that is {name:<9} -> {got:<8} (want {want:<8}) {mark}") if got != want: bad += 1 print() if not live_ok: bad += 1 if bad: print("🔴 the classifier cannot tell grounded labels from ungrounded ones,") print(" or it reports clean on an empty tree.") print(" Exit 2: nothing this tool has reported clean is trustworthy.") return 2 print("the classifier separates bare, dangling and grounded citations") return 0 def coverage(files): """How much of the authored corpus this audit can even see. 🔴 IT SEES 15 OF 70. Every `kind` label is checked for a citation, and a clean run has been quoted in `DECISIONS.md` as evidence that the authored data is grounded -- but a `why` with NO `kind` beside it is invisible to this walk entirely, and there are 55 of those against 15 labels. Found by reading the data rather than the tool: `audio.json`'s three SE cues carry measured provenance from HANDOFF Q8 and no `kind` field, so the audit that exists to check provenance never looked at them. ⚠️ NOT every `why` should have a `kind`. Section prose and `_` blocks explain a group rather than assert one value's provenance, and forcing a label there would invite mislabelling to satisfy a counter. So this REPORTS the ratio rather than demanding it be 1 -- a clean run must not read as full coverage. """ labelled = orphan = 0 for f in files: def walk(o): nonlocal labelled, orphan if isinstance(o, dict): for k, v in o.items(): if k.endswith("_why") or k == "why": stem = k[:-4] if k.endswith("_why") else "" kk = (stem + "_kind") if stem else "kind" if kk in o: labelled += 1 else: orphan += 1 walk(v) elif isinstance(o, list): for x in o: walk(x) walk(json.load(open(f, encoding="utf-8"))) return labelled, orphan def main(): if "--selftest" in sys.argv: return selftest() rows = [] for f in sorted(glob.glob("authored/*.json")): walk(json.load(open(f)), f, "", rows) # 🔴 LIVENESS. Run against a tree with no `authored/*.json` this printed # "0 kind label(s)" and exited 0 -- examined nothing, reported clean. The # Decoder's generalisation of my empty-band case, which is more general than # either instance: **a control that only compares two things cannot tell you # the comparison is happening.** An empty input makes a checker AGREEABLE # rather than wrong, and agreeable is indistinguishable from correct in a # log. if not rows: print("🔴 no `kind` labels found at all -- this audit examined NOTHING.") print(" Exit 2: the harness is broken (wrong directory, renamed files),") print(" not the corpus.") return 2 paths = known_paths() bare = dangling = 0 kinds = {} print(f" {len(rows)} kind label(s) in authored/\n") for f, where, kind, why, borrowed in rows: kinds.setdefault(kind, 0) kinds[kind] += 1 cites = citations(why) bad = [] for typ, c in cites: if typ == "capture": if c not in paths and not os.path.exists(c) \ and not any(p.endswith("/" + c) for p in paths): bad.append(c) elif typ == "handoff": if not re.search(rf"\|\s*{c}\s*\|", handoff()): bad.append(f"HANDOFF {c} (no such row)") elif typ == "path" and c not in paths and not os.path.exists(c): bad.append(c) elif typ == "sha" and not sha_ok(c): bad.append(c) mark = "ok " if not cites and borrowed: mark, bare = "🔴 BORROW", bare + 1 elif not cites: mark, bare = "🔴 BARE", bare + 1 elif bad: mark, dangling = "🔴 DANGL", dangling + 1 print(f" {mark} {kind:<24} {f.split('/')[-1]}{where}") if not cites and borrowed: print(" no `why` of its own; a sibling `why` argues a" " DIFFERENT claim") elif not cites: print(f" cites nothing openable -- {len(why)} chars of prose") elif bad: print(f" unresolvable: {', '.join(sorted(set(bad))[:4])}") else: print(f" {len(cites)} citation(s), all resolve") print() # Casing is checked because a consumer comparing == "measured" silently # misses "MEASURED", and a label that fails to match reads as absent. variants = [k for k in kinds if k.lower() == "measured"] if len(variants) > 1: print(f" ⚠️ {len(variants)} spellings of the same label: {variants}") print(" A consumer comparing == 'measured' misses the others, and a") print(" label that fails to match reads as ABSENT, not as wrong.\n") lab, orph = coverage(sorted(glob.glob("authored/*.json"))) print(f" COVERAGE: {lab} `why` field(s) carry a `kind` and were audited above;") print(f" {orph} carry NO `kind` and are INVISIBLE to this audit. A clean run") print(f" below is a statement about {lab} of {lab + orph} authored justifications.") print(" ⚠️ The denominator is not a target. Of the unlabelled ones, the great") print(" majority are SECTION PROSE -- `_` blocks and group explanations that") print(" assert no single value's provenance, where a label would be") print(" mislabelling to satisfy a counter. What was audited on 2026-09-01 is") print(" the other kind: a `why` sitting beside an actual VALUE. Thirteen of") print(" those existed unlabelled; all thirteen now carry a kind, and two of") print(" them failed the citation check the moment they became visible.") print() print(f" {bare} bare or borrowed, {dangling} dangling, {len(rows) - bare - dangling} with resolving citations") print(" 🔴 A resolving citation is not a verified label. Nothing here reads") print(" the cited page to confirm it says what the `why` claims.") return 1 if (bare or dangling) else 0 sys.exit(main())