build-candidates.py (3438B)
1 #!/usr/bin/env python3 2 """Build a metadata index of all candidate cheatsheet files for curation. 3 4 Scans the NetrunnerVault cheatsheets + the Ethical_Hacking-Cheatsheets repo, 5 emitting one JSON record per file with path, size, line count, title, 6 frontmatter fields, and a content preview. Output: candidates.json 7 """ 8 import json, os, re, sys 9 10 SOURCES = [ 11 ("vault", os.environ.get("VAULT_CHEATS", os.path.expanduser("~/git/NetrunnerVault/02Cybersecurity/Cheatsheets"))), 12 ("repo", "/Users/daemon1/git/Ethical_Hacking-Cheatsheets"), 13 ] 14 SKIP_DIRS = {".space", ".git", "attachments", "node_modules", ".obsidian"} 15 SKIP_NAMES = {".DS_Store"} 16 17 def parse_frontmatter(text): 18 fm = {} 19 if text.startswith("---"): 20 end = text.find("\n---", 3) 21 if end != -1: 22 block = text[3:end] 23 for line in block.splitlines(): 24 m = re.match(r"^([A-Za-z0-9_]+):\s*(.*)$", line) 25 if m: 26 fm[m.group(1).strip()] = m.group(2).strip() 27 return fm 28 29 def title_from(text, path): 30 for line in text.splitlines(): 31 s = line.strip() 32 if s.startswith("# "): 33 return s[2:].strip() 34 return os.path.splitext(os.path.basename(path))[0] 35 36 def preview(text, n=45): 37 # strip frontmatter for preview 38 if text.startswith("---"): 39 end = text.find("\n---", 3) 40 if end != -1: 41 text = text[end+4:] 42 lines = [l for l in text.splitlines()] 43 return "\n".join(lines[:n]) 44 45 records = [] 46 for src, root in SOURCES: 47 if not os.path.isdir(root): 48 print(f"WARN missing source {root}", file=sys.stderr); continue 49 for dp, dns, fns in os.walk(root): 50 dns[:] = [d for d in dns if d not in SKIP_DIRS] 51 for fn in fns: 52 if fn in SKIP_NAMES or fn.startswith(".fuse_hidden"): 53 continue 54 ext = os.path.splitext(fn)[1].lower() 55 if ext not in (".md", ".pdf", ".svg", ".html"): 56 continue 57 full = os.path.join(dp, fn) 58 rel = os.path.relpath(full, root) 59 try: 60 size = os.path.getsize(full) 61 except OSError: 62 continue 63 rec = {"source": src, "path": full, "rel": rel, "ext": ext.lstrip("."), 64 "bytes": size, "topdir": rel.split(os.sep)[0] if os.sep in rel else "(root)"} 65 if ext == ".md": 66 try: 67 with open(full, encoding="utf-8", errors="replace") as f: 68 text = f.read() 69 except OSError: 70 continue 71 rec["lines"] = text.count("\n") + 1 72 rec["title"] = title_from(text, full) 73 rec["frontmatter"] = parse_frontmatter(text) 74 rec["preview"] = preview(text) 75 # heuristics 76 rec["h2_count"] = len(re.findall(r"^##\s", text, re.M)) 77 rec["code_blocks"] = text.count("```") // 2 78 records.append(rec) 79 80 records.sort(key=lambda r: (r["source"], r["topdir"], r["rel"].lower())) 81 out = os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "candidates.json") 82 out = os.path.normpath(out) 83 with open(out, "w", encoding="utf-8") as f: 84 json.dump(records, f, indent=1, ensure_ascii=False) 85 86 md = [r for r in records if r["ext"] == "md"] 87 print(f"total files: {len(records)} md: {len(md)} pdf/svg/html: {len(records)-len(md)}") 88 print(f"written: {out}")