Files
management/analysis/scripts/inv_links.py
T

147 lines
5.9 KiB
Python
Raw Normal View History

"""Link inventory across all analysed repos -> analysis/data/links.tsv.
Statuses are the three the mandate fixes:
ok target verified to exist
broken target verified to be absent
points-outside-scope target lies outside the analysed set
A fourth column, `kind`, carries the nuance the status alone cannot:
whether the target was a relative file, a git.lab issue, a repo URL, an
anchor, or an external URL.
Issue URLs resolve offline against gitlab_issues.json when that file
exists, so a rerun needs no network. Without it they are reported as
kind=issue-url, status=points-outside-scope, note=unresolved-no-issue-data.
"""
import json
import re
from urllib.parse import unquote, urlsplit
from common import (
COMPONENTS, DATA_DIR, DOC_SUFFIXES, GROUP, OUT_OF_SCOPE_PROJECTS,
OUT_OF_SCOPE_SUBGROUPS, repos, tracked_files, write_tsv,
)
# Inline markdown links, minus image embeds; reference definitions too.
LINK_RE = re.compile(r"(?<!\!)\[[^\]]*\]\(\s*<?([^)\s>]+)[^)]*\)")
# A reference definition needs a target that actually looks like one.
# Without that guard, log lines such as "[ERROR]: Task failed: ..." in
# imported wiki dumps parse as links to a file named "Task".
REFDEF_RE = re.compile(r"^\s{0,3}\[[^\]]+\]:\s*<?([^\s>]*[/.:#][^\s>]*)")
IN_SCOPE_REPOS = {slug for slug, _ in COMPONENTS} | {"management"}
ISSUE_URL_RE = re.compile(r"^/([^/]+(?:/[^/]+)*?)/-/issues/(\d+)")
def load_issue_index():
"""(project, iid) pairs known to exist, from a previous API export."""
path = DATA_DIR / "gitlab_issues.json"
if not path.exists():
return None
data = json.loads(path.read_text(encoding="utf-8"))
return {(i["project"], i["iid"]) for i in data["issues"]}
def classify_url(target, issue_index):
"""-> (kind, status, note) for an absolute URL."""
parts = urlsplit(target)
host = parts.netloc.lower()
if host != "git.lab":
return ("external-url", "points-outside-scope", f"host={host}")
m = ISSUE_URL_RE.match(parts.path)
if m:
ns, iid = m.group(1), int(m.group(2))
group, _, project = ns.rpartition("/")
if group != GROUP or project not in IN_SCOPE_REPOS:
return ("issue-url", "points-outside-scope", f"project={ns}")
if issue_index is None:
return ("issue-url", "points-outside-scope", "unresolved-no-issue-data")
if (project, iid) in issue_index:
return ("issue-url", "ok", f"{project}#{iid}")
return ("issue-url", "broken", f"{project}#{iid} not in group export")
segs = [s for s in parts.path.split("/") if s]
if segs and segs[0] == GROUP and len(segs) >= 2:
project = segs[1]
if project in OUT_OF_SCOPE_PROJECTS or project in OUT_OF_SCOPE_SUBGROUPS:
return ("repo-url", "points-outside-scope", f"declared out of scope: {project}")
if project in IN_SCOPE_REPOS:
return ("repo-url", "ok", f"in-scope repo {project}")
return ("repo-url", "points-outside-scope", f"path={parts.path or '/'}")
def classify_relative(repo_name, src, target, tracked):
"""-> (kind, status, note) for a repo-relative link."""
path = unquote(target.split("#")[0].split("?")[0])
if not path:
return ("anchor", "ok", "same-document anchor")
base = src.rsplit("/", 1)[0] if "/" in src else ""
joined = f"{base}/{path}" if base and not path.startswith("/") else path.lstrip("/")
resolved, stack = [], joined.split("/")
for seg in stack:
if seg in ("", "."):
continue
if seg == "..":
if resolved:
resolved.pop()
else:
return ("relative-file", "points-outside-scope",
"traverses above the repo root")
else:
resolved.append(seg)
cand = "/".join(resolved)
if cand in tracked:
return ("relative-file", "ok", cand)
# Directory links are legitimate targets in this doc set.
if any(t.startswith(cand + "/") for t in tracked):
return ("relative-file", "ok", cand + "/ (directory)")
# Extensionless doc links ("[x](playwright#anchor)") are common in
# imported upstream docs and resolve in the renderers that serve them.
for suffix in sorted(DOC_SUFFIXES):
if cand + suffix in tracked:
return ("relative-file", "ok", f"{cand}{suffix} (extension implied)")
return ("relative-file", "broken", f"no tracked file at {cand}")
def main():
issue_index = load_issue_index()
rows = []
for name, repo in repos():
tracked = set(tracked_files(name, repo))
docs = [f for f in tracked if any(f.endswith(s) for s in DOC_SUFFIXES)]
for src in docs:
try:
text = (repo / src).read_text(encoding="utf-8")
except (OSError, UnicodeDecodeError):
continue
for lineno, line in enumerate(text.splitlines(), start=1):
targets = LINK_RE.findall(line)
ref = REFDEF_RE.match(line)
if ref:
targets = targets + [ref.group(1)]
for target in targets:
if target.startswith(("mailto:", "tel:")):
kind, status, note = ("mailto", "points-outside-scope", "")
elif target.startswith("#"):
kind, status, note = ("anchor", "ok", "same-document anchor")
elif re.match(r"^[a-zA-Z][a-zA-Z0-9+.-]*://", target):
kind, status, note = classify_url(target, issue_index)
else:
kind, status, note = classify_relative(name, src, target, tracked)
rows.append([name, src, lineno, kind, target, status, note])
rows.sort(key=lambda r: (r[0], r[1], r[2], r[4]))
write_tsv("links.tsv",
["repo", "src_path", "line", "kind", "target", "status", "note"],
rows)
if __name__ == "__main__":
main()