run_all.sh reproduces every file under analysis/data/ from zero: it clones the in-scope components if missing, exports group issue metadata from git.lab, and regenerates the inventories. Reruns are diff-clean -- no wall-clock time enters an output; 'days since' is measured against the management repo's HEAD date. The management repo is inventoried at main, not at the analysis branch, so this analysis does not observe its own commits. inv_repo.py aborts the run if anything outside analysis/ was modified.
147 lines
5.9 KiB
Python
147 lines
5.9 KiB
Python
"""Link inventory across all analysed repos -> analysis/data/links.tsv.
|
|
|
|
Statuses are the three the mandate fixes:
|
|
|
|
ok target verified to exist
|
|
broken target verified to be absent
|
|
points-outside-scope target lies outside the analysed set
|
|
|
|
A fourth column, `kind`, carries the nuance the status alone cannot:
|
|
whether the target was a relative file, a git.lab issue, a repo URL, an
|
|
anchor, or an external URL.
|
|
|
|
Issue URLs resolve offline against gitlab_issues.json when that file
|
|
exists, so a rerun needs no network. Without it they are reported as
|
|
kind=issue-url, status=points-outside-scope, note=unresolved-no-issue-data.
|
|
"""
|
|
|
|
import json
|
|
import re
|
|
from urllib.parse import unquote, urlsplit
|
|
|
|
from common import (
|
|
COMPONENTS, DATA_DIR, DOC_SUFFIXES, GROUP, OUT_OF_SCOPE_PROJECTS,
|
|
OUT_OF_SCOPE_SUBGROUPS, repos, tracked_files, write_tsv,
|
|
)
|
|
|
|
# Inline markdown links, minus image embeds; reference definitions too.
|
|
LINK_RE = re.compile(r"(?<!\!)\[[^\]]*\]\(\s*<?([^)\s>]+)[^)]*\)")
|
|
# A reference definition needs a target that actually looks like one.
|
|
# Without that guard, log lines such as "[ERROR]: Task failed: ..." in
|
|
# imported wiki dumps parse as links to a file named "Task".
|
|
REFDEF_RE = re.compile(r"^\s{0,3}\[[^\]]+\]:\s*<?([^\s>]*[/.:#][^\s>]*)")
|
|
|
|
IN_SCOPE_REPOS = {slug for slug, _ in COMPONENTS} | {"management"}
|
|
ISSUE_URL_RE = re.compile(r"^/([^/]+(?:/[^/]+)*?)/-/issues/(\d+)")
|
|
|
|
|
|
def load_issue_index():
|
|
"""(project, iid) pairs known to exist, from a previous API export."""
|
|
path = DATA_DIR / "gitlab_issues.json"
|
|
if not path.exists():
|
|
return None
|
|
data = json.loads(path.read_text(encoding="utf-8"))
|
|
return {(i["project"], i["iid"]) for i in data["issues"]}
|
|
|
|
|
|
def classify_url(target, issue_index):
|
|
"""-> (kind, status, note) for an absolute URL."""
|
|
parts = urlsplit(target)
|
|
host = parts.netloc.lower()
|
|
if host != "git.lab":
|
|
return ("external-url", "points-outside-scope", f"host={host}")
|
|
|
|
m = ISSUE_URL_RE.match(parts.path)
|
|
if m:
|
|
ns, iid = m.group(1), int(m.group(2))
|
|
group, _, project = ns.rpartition("/")
|
|
if group != GROUP or project not in IN_SCOPE_REPOS:
|
|
return ("issue-url", "points-outside-scope", f"project={ns}")
|
|
if issue_index is None:
|
|
return ("issue-url", "points-outside-scope", "unresolved-no-issue-data")
|
|
if (project, iid) in issue_index:
|
|
return ("issue-url", "ok", f"{project}#{iid}")
|
|
return ("issue-url", "broken", f"{project}#{iid} not in group export")
|
|
|
|
segs = [s for s in parts.path.split("/") if s]
|
|
if segs and segs[0] == GROUP and len(segs) >= 2:
|
|
project = segs[1]
|
|
if project in OUT_OF_SCOPE_PROJECTS or project in OUT_OF_SCOPE_SUBGROUPS:
|
|
return ("repo-url", "points-outside-scope", f"declared out of scope: {project}")
|
|
if project in IN_SCOPE_REPOS:
|
|
return ("repo-url", "ok", f"in-scope repo {project}")
|
|
return ("repo-url", "points-outside-scope", f"path={parts.path or '/'}")
|
|
|
|
|
|
def classify_relative(repo_name, src, target, tracked):
|
|
"""-> (kind, status, note) for a repo-relative link."""
|
|
path = unquote(target.split("#")[0].split("?")[0])
|
|
if not path:
|
|
return ("anchor", "ok", "same-document anchor")
|
|
|
|
base = src.rsplit("/", 1)[0] if "/" in src else ""
|
|
joined = f"{base}/{path}" if base and not path.startswith("/") else path.lstrip("/")
|
|
|
|
resolved, stack = [], joined.split("/")
|
|
for seg in stack:
|
|
if seg in ("", "."):
|
|
continue
|
|
if seg == "..":
|
|
if resolved:
|
|
resolved.pop()
|
|
else:
|
|
return ("relative-file", "points-outside-scope",
|
|
"traverses above the repo root")
|
|
else:
|
|
resolved.append(seg)
|
|
cand = "/".join(resolved)
|
|
|
|
if cand in tracked:
|
|
return ("relative-file", "ok", cand)
|
|
# Directory links are legitimate targets in this doc set.
|
|
if any(t.startswith(cand + "/") for t in tracked):
|
|
return ("relative-file", "ok", cand + "/ (directory)")
|
|
# Extensionless doc links ("[x](playwright#anchor)") are common in
|
|
# imported upstream docs and resolve in the renderers that serve them.
|
|
for suffix in sorted(DOC_SUFFIXES):
|
|
if cand + suffix in tracked:
|
|
return ("relative-file", "ok", f"{cand}{suffix} (extension implied)")
|
|
return ("relative-file", "broken", f"no tracked file at {cand}")
|
|
|
|
|
|
def main():
|
|
issue_index = load_issue_index()
|
|
rows = []
|
|
for name, repo in repos():
|
|
tracked = set(tracked_files(name, repo))
|
|
docs = [f for f in tracked if any(f.endswith(s) for s in DOC_SUFFIXES)]
|
|
for src in docs:
|
|
try:
|
|
text = (repo / src).read_text(encoding="utf-8")
|
|
except (OSError, UnicodeDecodeError):
|
|
continue
|
|
for lineno, line in enumerate(text.splitlines(), start=1):
|
|
targets = LINK_RE.findall(line)
|
|
ref = REFDEF_RE.match(line)
|
|
if ref:
|
|
targets = targets + [ref.group(1)]
|
|
for target in targets:
|
|
if target.startswith(("mailto:", "tel:")):
|
|
kind, status, note = ("mailto", "points-outside-scope", "")
|
|
elif target.startswith("#"):
|
|
kind, status, note = ("anchor", "ok", "same-document anchor")
|
|
elif re.match(r"^[a-zA-Z][a-zA-Z0-9+.-]*://", target):
|
|
kind, status, note = classify_url(target, issue_index)
|
|
else:
|
|
kind, status, note = classify_relative(name, src, target, tracked)
|
|
rows.append([name, src, lineno, kind, target, status, note])
|
|
|
|
rows.sort(key=lambda r: (r[0], r[1], r[2], r[4]))
|
|
write_tsv("links.tsv",
|
|
["repo", "src_path", "line", "kind", "target", "status", "note"],
|
|
rows)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|