Files
management/analysis/scripts/inv_repo.py
T
Thore Cimbal b8b853163d analysis: add deterministic inventory scripts and raw data
run_all.sh reproduces every file under analysis/data/ from zero: it
clones the in-scope components if missing, exports group issue metadata
from git.lab, and regenerates the inventories. Reruns are diff-clean --
no wall-clock time enters an output; 'days since' is measured against
the management repo's HEAD date.

The management repo is inventoried at main, not at the analysis branch,
so this analysis does not observe its own commits. inv_repo.py aborts
the run if anything outside analysis/ was modified.
2026-08-09 12:00:00 +00:00

172 lines
6.4 KiB
Python

"""Per-repo inventory: file trees, doc inventory, markers, git activity.
Writes tree_<repo>.txt, docs_inventory.tsv, markers.tsv, git_activity.tsv.
"""
import re
from collections import defaultdict
from datetime import date
from common import (
DATA_DIR, DOC_SUFFIXES, assert_untouched, days_since, git, last_commit,
ref_of, repos, tracked_files, write_tsv,
)
MARKER_RE = re.compile(r"\b(TODO|FIXME|HACK|XXX|DEPRECATED)\b")
# git grep's ERE does not implement \b, which silently reports zero
# markers; the scan therefore runs in Python, where \b works.
MAX_SCAN_BYTES = 512 * 1024
def write_trees():
for name, repo in repos():
files = tracked_files(name, repo)
path = DATA_DIR / f"tree_{name}.txt"
path.write_text("\n".join(files) + "\n", encoding="utf-8")
print(f" tree_{name}.txt: {len(files)} files")
def write_docs_inventory():
rows = []
for name, repo in repos():
for f in tracked_files(name, repo):
if not any(f.endswith(s) for s in DOC_SUFFIXES):
continue
size = (repo / f).stat().st_size
iso, author, sha = last_commit(name, repo, f)
rows.append([name, f, size, iso, author, days_since(iso), sha])
rows.sort(key=lambda r: (r[0], r[1]))
write_tsv(
"docs_inventory.tsv",
["repo", "path", "bytes", "last_commit_date", "last_author",
"days_since_change", "last_commit"],
rows,
)
def _blame_dates(repo, path, lines_wanted):
"""Map line number -> (author date, short sha, author) via one blame pass."""
try:
out = git(repo, "blame", "--line-porcelain", "--", path)
except RuntimeError:
return {}
result, sha, author, when, lineno = {}, None, None, None, None
for line in out.splitlines():
if re.match(r"^[0-9a-f]{40} ", line):
parts = line.split()
sha, lineno = parts[0], int(parts[2])
elif line.startswith("author "):
author = line[len("author "):]
elif line.startswith("author-time "):
when = date.fromtimestamp(int(line[len("author-time "):])).isoformat()
elif line.startswith("\t"):
if lineno in lines_wanted:
result[lineno] = (when or "", (sha or "")[:9], author or "")
return result
def write_markers():
rows = []
for name, repo in repos():
for f in tracked_files(name, repo):
full = repo / f
try:
if full.stat().st_size > MAX_SCAN_BYTES:
continue
text = full.read_text(encoding="utf-8")
except (OSError, UnicodeDecodeError):
continue # binary or unreadable: not a marker carrier
hits = {}
for i, line in enumerate(text.splitlines(), start=1):
m = MARKER_RE.search(line)
if m:
hits[i] = (m.group(1), line.strip()[:200])
if not hits:
continue
blame = _blame_dates(repo, f, set(hits))
for lineno in sorted(hits):
marker, snippet = hits[lineno]
when, sha, author = blame.get(lineno, ("", "", ""))
rows.append([name, f, lineno, marker, snippet, when,
days_since(when), sha, author])
rows.sort(key=lambda r: (r[0], r[1], r[2]))
write_tsv(
"markers.tsv",
["repo", "path", "line", "marker", "text", "blame_date",
"days_since_blame", "blame_commit", "blame_author"],
rows,
)
def _is_doc(path):
return any(path.endswith(s) for s in DOC_SUFFIXES)
def write_git_activity():
rows = []
for name, repo in repos():
log = git(repo, "log", "--format=%x01%H\t%ad", "--date=short",
"--name-only", ref_of(name))
per_month = defaultdict(lambda: {"commits": 0, "docs": 0, "code": 0,
"docs_only": 0})
commit_days = []
for chunk in log.split("\x01"):
if not chunk.strip():
continue
head, _, body = chunk.partition("\n")
sha, iso = head.split("\t")
files = [ln for ln in body.splitlines() if ln.strip()]
touched_docs = any(_is_doc(f) for f in files)
touched_code = any(not _is_doc(f) for f in files)
bucket = per_month[iso[:7]]
bucket["commits"] += 1
bucket["docs"] += 1 if touched_docs else 0
bucket["code"] += 1 if touched_code else 0
bucket["docs_only"] += 1 if touched_docs and not touched_code else 0
commit_days.append(iso)
for month in sorted(per_month):
b = per_month[month]
for metric in ("commits", "docs", "code", "docs_only"):
key = {"docs": "commits_touching_docs",
"code": "commits_touching_code",
"docs_only": "commits_docs_only"}.get(metric, metric)
rows.append([name, month, key, b[metric]])
days = sorted({d for d in commit_days})
totals = {
"commits": len(commit_days),
"commits_touching_docs": sum(b["docs"] for b in per_month.values()),
"commits_touching_code": sum(b["code"] for b in per_month.values()),
"commits_docs_only": sum(b["docs_only"] for b in per_month.values()),
"active_days": len(days),
"first_commit": days[0] if days else "",
"last_commit": days[-1] if days else "",
"days_since_last_commit": days_since(days[-1]) if days else "",
}
gap, gap_from, gap_to = 0, "", ""
for a, b in zip(days, days[1:]):
ya, ma, da = (int(x) for x in a.split("-"))
yb, mb, db = (int(x) for x in b.split("-"))
delta = (date(yb, mb, db) - date(ya, ma, da)).days
if delta > gap:
gap, gap_from, gap_to = delta, a, b
totals["max_gap_days"] = gap
totals["max_gap_from"] = gap_from
totals["max_gap_to"] = gap_to
for metric in sorted(totals):
rows.append([name, "ALL", metric, totals[metric]])
rows.sort(key=lambda r: (r[0], r[1], r[2]))
write_tsv("git_activity.tsv", ["repo", "scope", "metric", "value"], rows)
if __name__ == "__main__":
assert_untouched()
DATA_DIR.mkdir(parents=True, exist_ok=True)
write_trees()
write_docs_inventory()
write_markers()
write_git_activity()