validate-content.py (3907B)
1 #!/usr/bin/env python3 2 """Post-modernization sanity check on generated content.""" 3 import json, os, re, sys, glob 4 5 ROOT = os.path.normpath(os.path.join(os.path.dirname(os.path.abspath(__file__)), "..")) 6 SHEETS = os.path.join(ROOT, "src", "content", "sheets") 7 TAX = {"active-directory","enumeration","exploitation","privilege-escalation", 8 "password-attacks","web","tunneling-pivoting","cryptography","dfir", 9 "tools","linux-it","git-workflow","pentest-workflow","osint"} 10 11 manifest = json.load(open(os.path.join(ROOT, "content-manifest.json"))) 12 expected = {(s["category"], s["slug"]) for s in manifest["sheets"]} 13 14 files = glob.glob(os.path.join(SHEETS, "**", "*.md"), recursive=True) 15 found = set() 16 issues = [] 17 warns = [] 18 links = [] 19 20 for f in files: 21 rel = os.path.relpath(f, SHEETS) 22 txt = open(f, encoding="utf-8", errors="replace").read() 23 if not txt.startswith("---"): 24 issues.append(f"{rel}: no frontmatter"); continue 25 end = txt.find("\n---", 3) 26 fm = txt[3:end] 27 body = txt[end+4:] 28 def field(name): 29 m = re.search(rf"^{name}:\s*(.*)$", fm, re.M) 30 return m.group(1).strip() if m else None 31 cat = field("category") 32 title = field("title") 33 if cat not in TAX: issues.append(f"{rel}: bad/missing category '{cat}'") 34 if not title: issues.append(f"{rel}: missing title") 35 # slug from filename 36 slug = os.path.splitext(os.path.basename(f))[0] 37 found.add((cat, slug)) 38 # InternalAllTheThings is mirrored on-site under /internal, so a sheet 39 # linking the live upstream site walks the reader off the deployment. 40 for m in re.finditer(r"https?://swisskyrepo\.github\.io/InternalAllTheThings/(\S*?)[)\s]", body): 41 issues.append(f"{rel}: links live IATT site, use /internal/{m.group(1).strip('/')}") 42 # Internal /sheets/ cross-links are resolved against the files actually on 43 # disk. Slug renames and category moves silently rot these, so they are 44 # collected here and checked once the full slug set is known. 45 for m in re.finditer(r"\]\(/sheets/([a-z0-9][a-z0-9\-]*)/([a-z0-9][a-z0-9\-]*)/?(?:#[^)]*)?\)", body): 46 links.append((rel, m.group(1), m.group(2))) 47 48 # leftover Obsidian syntax 49 if re.search(r"\[\[[^\]]+\]\]", body): warns.append(f"{rel}: leftover [[wikilink]]") 50 if re.search(r"!\[\[", body): warns.append(f"{rel}: leftover ![[embed]]") 51 if "%%" in body: warns.append(f"{rel}: leftover %%comment%%") 52 # unlabeled code fences 53 fences = re.findall(r"^```(.*)$", body, re.M) 54 unl = sum(1 for i,x in enumerate(fences) if i % 2 == 0 and not x.strip()) 55 if unl: warns.append(f"{rel}: {unl} code fence(s) without a language") 56 57 for rel, lcat, lslug in links: 58 if (lcat, lslug) not in found: 59 issues.append(f"{rel}: broken internal link /sheets/{lcat}/{lslug}") 60 elif f"{lcat}/{lslug}" == os.path.splitext(rel)[0].replace(os.sep, "/"): 61 # A retarget that lands on the page doing the linking reads as a dead 62 # end. Past slug renames produced these, so they are called out too. 63 issues.append(f"{rel}: self-referential link /sheets/{lcat}/{lslug}") 64 65 missing = expected - found 66 extra = found - expected 67 68 print(f"files on disk : {len(files)}") 69 print(f"manifest wants: {len(expected)}") 70 print(f"matched : {len(expected & found)}") 71 print(f"sheet links : {len(links)}") 72 if missing: 73 print(f"\nMISSING ({len(missing)}):") 74 for c, s in sorted(missing): print(f" {c}/{s}") 75 if extra: 76 print(f"\nEXTRA/misplaced ({len(extra)}):") 77 for c, s in sorted(extra): print(f" {c}/{s}") 78 if issues: 79 print(f"\nISSUES ({len(issues)}):") 80 for i in issues: print(f" {i}") 81 if warns: 82 print(f"\nWARNINGS ({len(warns)}):") 83 for w in warns[:40]: print(f" {w}") 84 if len(warns) > 40: print(f" ... +{len(warns)-40} more") 85 86 ok = not missing and not issues 87 print("\n" + ("PASS" if ok else "NEEDS ATTENTION")) 88 sys.exit(0 if ok else 1)