"""Mechanically verify every claim row -> analysis/data/claims_verification.tsv. Closes the gap HANDOFF.md named as the analysis' largest hole: claims.tsv was extracted in full but verified selectively. This pass gives every one of its rows an explicit disposition instead of an implicit "not chased". What a script can and cannot do here is stated, not blurred: * checkable here -- backticked repo paths resolved against all six trees; issue references resolved against the group export, including expectation words ("geschlossen", "offen") checked against the actual state; legacy IDs against issue titles; version tokens against git tags. * not checkable -- host/runtime state (absolute paths, service versions, DNS), claims about the past, and claims whose truth lives outside the analysed repos. These are classed, counted and left honest, not silently passed. Rows that fail a mechanical check are FLAGged for human adjudication; the verdicts live in REPORT.md's appendix, not here -- this file is regenerated and must stay free of hand-written content. """ import csv import json import re from common import DATA_DIR, MGMT_REPO, cell, git, repos, write_tsv # --- reference data ------------------------------------------------------- PROJECT_ALIASES = { "gitops": "axion1337.chat-gitops", "axion1337.chat-gitops": "axion1337.chat-gitops", "threadnet-web": "ThreadNet-Web", "ThreadNet-Web": "ThreadNet-Web", "threadnet-call": "threadnet-call", "thread-net-git": "thread-net-git", "threadnet-operating": "threadnet-operating", "management": "management", } CLOSED_WORDS = re.compile(r"geschlossen|erledigt|abgeschlossen", re.I) OPEN_WORDS = re.compile(r"\boffen(?:e[rs]?)?\b", re.I) HISTORY_WORDS = re.compile( r"entfernt|gelöscht|removed|hieß|war\b|bis 2026|damals|Vorgänger|alte?[rn]?\b", re.I) PATH_EXT = (".md", ".yml", ".yaml", ".json", ".py", ".ts", ".toml", ".crt", ".sh", ".env", ".example", ".cjs", ".rst", ".txt", ".conf", ".rules", ".png", ".jpg", ".ico", ".icns", ".exe", ".production") DOMAIN_PREFIX = re.compile( r"^(https?://|git\.lab|registry\.|rohana|axionwiki|wiki\.lab|ghcr\.io|gcr\.io)") FORGE_REPO = re.compile(r"^(sorb|homelab|vendor|axion1337\.chat)/") NET_REF = re.compile(r"^\d{1,3}(\.\d{1,3}){3}(/\d+)?$") BARE_EXT = re.compile(r"^\.[a-z0-9]{1,6}$") VERSION_RE = re.compile(r"\bv?\d+\.\d+\.\d+(?:-[\w.]+)?\b") NAMED_REF_RE = re.compile(r"\b([A-Za-z][\w.-]*)#(\d{1,4})\b") URL_REF_RE = re.compile(r"/([\w.-]+)/-/issues/(\d{1,4})") BARE_REF_RE = re.compile(r"(? set of tracked paths, read from the committed tree files so a rerun without component clones still verifies identically.""" trees = {} for name, _ in repos(): tree_file = DATA_DIR / f"tree_{name}.txt" trees[name] = set(tree_file.read_text(encoding="utf-8").splitlines()) return trees def load_tags(): tags = {} for name, repo in repos(): try: tags[name] = {t for t in git(repo, "tag", "--list").splitlines() if t} except RuntimeError: tags[name] = set() return tags def load_branches(): """Remote branch names across all repos, with and without origin/.""" out = set() for _, repo in repos(): try: for b in git(repo, "branch", "-r", "--format=%(refname:short)").splitlines(): b = b.strip() if b and "HEAD" not in b: out.add(b) out.add(b.removeprefix("origin/")) except RuntimeError: pass return out # --- per-token checks ------------------------------------------------------ # Doc names that qualify a path with a repo alias ("gitops/CLAUDE.md"). TREE_ALIASES = {"gitops": "axion1337.chat-gitops", "management": "management"} def classify_path_token(tok, trees, branches, src_dir): """-> (check_note, flag_or_None)""" if tok.startswith(("/", "~", "$")) or DOMAIN_PREFIX.match(tok): return (f"runtime-path:{tok}", None) if NET_REF.match(tok): return (f"net-ref:{tok}", None) if FORGE_REPO.match(tok): return (f"forge-repo:{tok}", None) if tok.startswith("@"): return (f"package-ref:{tok}", None) if ":" in tok: return (f"image-ref:{tok}", None) if BARE_EXT.match(tok): return (None, None) if any(c in tok for c in " *<{=!") or tok.startswith("-"): return (None, None) looks_like_path = "/" in tok or tok.endswith(PATH_EXT) if not looks_like_path: return (None, None) clean = tok[2:] if tok.startswith("./") else tok stripped = clean.removeprefix("origin/") if stripped in branches or clean in branches: return (f"branch-ok:{clean}", None) first, _, rest = clean.partition("/") if rest and first in TREE_ALIASES and rest in trees[TREE_ALIASES[first]]: return (f"path-ok:{rest}@{TREE_ALIASES[first]}", None) candidates = [clean] if src_dir: # doc-relative resolution inside the management repo candidates.append(f"{src_dir}/{clean}") hits = [] for name, files in trees.items(): for cand in candidates: if cand in files or any(f.endswith("/" + cand) for f in files): hits.append(name) break if any(("/" + cand.rstrip("/") + "/") in ("/" + f) or f.startswith(cand.rstrip("/") + "/") for f in files): hits.append(name + "(dir)") break if hits: return (f"path-ok:{clean}@{','.join(sorted(hits)[:3])}", None) return (f"path-miss:{clean}", f"path-miss:{clean}") def issue_refs(line): """Extract (project, iid) refs; bare #N defaults to management.""" refs, spans = [], [] for m in list(NAMED_REF_RE.finditer(line)) + list(URL_REF_RE.finditer(line)): proj = PROJECT_ALIASES.get(m.group(1)) if proj: refs.append((proj, int(m.group(2)))) spans.append(m.span()) named_iids = {iid for _, iid in refs} for m in BARE_REF_RE.finditer(line): if any(a <= m.start() < b for a, b in spans): continue iid = int(m.group(1)) # "[#49](…/gitops/-/issues/49)" names the same issue twice; the # bare token is the link text, not a second (management) reference. if iid in named_iids: continue refs.append(("management", iid)) return refs def check_issue_ref(proj, iid, line, by_ref, single_ref): issue = by_ref.get((proj, iid)) if issue is None: return (f"issue-miss:{proj}#{iid}", f"issue-miss:{proj}#{iid}") expect = None # Expectation words are only attributable when the line references # exactly one issue ("#5 geschlossen, Folgethemen in #6" must not # expect #6 to be closed). if single_ref and CLOSED_WORDS.search(line) and not OPEN_WORDS.search(line): expect = "closed" elif single_ref and OPEN_WORDS.search(line) and not CLOSED_WORDS.search(line): expect = "opened" if expect and issue["state"] != expect: return (f"issue-state:{proj}#{iid}={issue['state']},text-says-{expect}", f"issue-state:{proj}#{iid}") return (f"issue-ok:{proj}#{iid}({issue['state']})", None) def main(): by_ref, ids_in_titles = load_issues() trees = load_trees() tags = load_tags() branches = load_branches() all_tags = {t for s in tags.values() for t in s} rows = [] counts = {} with (DATA_DIR / "claims.tsv").open(encoding="utf-8") as fh: for row in csv.DictReader(fh, delimiter="\t"): line = row["claim_text"] checks, flags = [], [] if row["in_code_block"] == "yes": status = "code-block" else: src_dir = row["path"].rsplit("/", 1)[0] if "/" in row["path"] else "" for tok in re.findall(r"`([^`]+)`", line): note, flag = classify_path_token(tok, trees, branches, src_dir) if note: checks.append(note) if flag: flags.append(flag) refs = issue_refs(line) for proj, iid in refs: note, flag = check_issue_ref(proj, iid, line, by_ref, single_ref=len(refs) == 1) checks.append(note) if flag: flags.append(flag) for lid in LEGACY_ID_RE.findall(line): checks.append(f"id-{'ok' if lid in ids_in_titles else 'no-issue'}:{lid}") for v in VERSION_RE.findall(line): if v in all_tags or "v" + v in all_tags: checks.append(f"tag-ok:{v}") if flags: status = "FLAG" elif any(c.startswith(("path-ok", "issue-ok")) for c in checks): status = "checked-ok" elif checks: status = "informational" else: status = "prose-or-runtime" hint = "historical-wording" if HISTORY_WORDS.search(line) else "" counts[status] = counts.get(status, 0) + 1 rows.append([row["path"], row["line"], status, ";".join(checks)[:400], ";".join(flags), hint, cell(line)[:200]]) rows.sort(key=lambda r: (r[0], int(r[1]))) write_tsv("claims_verification.tsv", ["path", "line", "status", "checks", "flags", "hint", "claim_text"], rows) for k in sorted(counts): print(f" {k:<18} {counts[k]}") print(" -- FLAG rows:") for r in rows: if r[2] == "FLAG": print(f" {r[0]}:{r[1]} [{r[4]}] {'(hist?)' if r[5] else ''} {r[6][:110]}") if __name__ == "__main__": main()