port-vault.py (16022B)
1 #!/usr/bin/env python3 2 """ 3 Port the NetrunnerVault Cheatsheets tree into the site's `sheets` collection. 4 5 Mechanical, no rewriting: content is preserved verbatim apart from four 6 deterministic clean-ups that only remove vault-local scaffolding — 7 1. Obsidian frontmatter (aliases/tags) is dropped and replaced with the 8 site's normalized frontmatter. 9 2. A chatbot preamble before the first H1 ("Right on cue, Netrunner…") 10 is trimmed — everything up to the first `# ` line goes. 11 3. `[[wikilinks]]` are flattened to their text (they point at vault pages 12 that don't exist on-site); `![[embeds]]` are dropped. 13 4. `[1][2]`-style citation markers are stripped OUTSIDE code fences. 14 15 Frontmatter (title/description/category/tags/tools/difficulty) is derived 16 from the filename, path and body. Nothing is paraphrased. 17 18 Dedup: a file whose canonical topic key already exists in the site (either 19 in the current sheets or earlier in this run) is SKIPPED, so the curated 60 20 are never clobbered and rustscan×3 collapses to the one already shipped. 21 22 Usage: 23 python3 scripts/port-vault.py --dry # decide only, write a manifest 24 python3 scripts/port-vault.py # apply (writes src/content/sheets) 25 """ 26 import json, os, re, sys, unicodedata 27 28 REPO = os.path.normpath(os.path.join(os.path.dirname(os.path.abspath(__file__)), "..")) 29 SRC = os.environ.get("VAULT_CHEATS", os.path.expanduser("~/git/NetrunnerVault/02Cybersecurity/Cheatsheets")) 30 SHEETS = os.path.join(REPO, "src", "content", "sheets") 31 MANIFEST = os.path.join(REPO, "port-vault-decisions.tsv") 32 UPDATED = "2026-08-10" 33 DRY = "--dry" in sys.argv 34 35 # ── Category: source top-level folder → site domain ───────────────────────── 36 TOP_MAP = { 37 "ActiveDirectory": "active-directory", 38 "Cryptography": "cryptography", 39 "HashingAndEncrypting": "cryptography", 40 "DFIR": "dfir", 41 "Enumeration": "enumeration", 42 "Exploitation": "exploitation", 43 "Git": "git-workflow", 44 "Linux": "linux-it", 45 "macOS": "linux-it", 46 "PasswordAttacks": "password-attacks", 47 "PrivEsc": "privilege-escalation", 48 "Tools": "tools", 49 "TunnelingAndPivoting": "tunneling-pivoting", 50 "Web": "web", 51 } 52 53 def category_for(rel): 54 parts = rel.split("/") 55 top = parts[0] 56 if top in TOP_MAP: 57 return TOP_MAP[top] 58 name = parts[-1].lower() 59 if top == "Misc": 60 if "tmux" in name: 61 return "linux-it" 62 if any(k in name for k in ("tunnel", "pivot", "portfw")): 63 return "tunneling-pivoting" 64 return "tools" 65 # Root-level strays (most are skipped as non-sheets before reaching here) 66 if "macos" in name: 67 return "linux-it" 68 if "git" in name: 69 return "git-workflow" 70 if "forensic" in name: 71 return "dfir" 72 return "tools" 73 74 # ── Skip: navigation / meta, not copy-ready cheatsheets ───────────────────── 75 SKIP_RE = re.compile(r"(roadmap|dashboard|attack-flow|most-used-commands|esc attack index|adcs dashboard)", re.I) 76 77 def is_non_sheet(rel): 78 base = rel.split("/")[-1] 79 # Strip a leading emoji/space so covers like "🔵 Attack.md" match — the 80 # emoji prefix is exactly what let one slip through the first run. 81 bare = strip_emoji(base).strip() 82 stem = bare[:-3] if bare.lower().endswith(".md") else bare 83 if not stem.strip(): 84 return True # Git/.md — empty stub 85 if stem.startswith("_"): 86 return True # _ADCS Dashboard / _index files 87 if bare.lower() in ("attack.md", "readme.md"): 88 return True # category cover / scripts readme 89 return bool(SKIP_RE.search(stem)) 90 91 # ── Slug / title / canonical key ──────────────────────────────────────────── 92 EMOJI_RE = re.compile( 93 "[\U0001F000-\U0001FAFF\U00002600-\U000027BF\U0001F1E6-\U0001F1FF" 94 "\U00002190-\U000021FF\U00002B00-\U00002BFF️]" 95 ) 96 97 def strip_emoji(s): 98 return EMOJI_RE.sub("", s) 99 100 def slugify(s): 101 s = strip_emoji(s) 102 s = unicodedata.normalize("NFKD", s).encode("ascii", "ignore").decode() 103 s = s.lower() 104 s = re.sub(r"[^a-z0-9]+", "-", s) 105 return re.sub(r"-+", "-", s).strip("-") or "x" 106 107 def title_from(rel): 108 base = rel.split("/")[-1][:-3] 109 t = strip_emoji(base).strip() 110 # Drop trailing "Cheatsheet" / "Cheat Sheet" / "markdown" noise words. 111 t = re.sub(r"\s*[-–—]?\s*(cheat\s*sheet|cheatsheet|markdown)\s*$", "", t, flags=re.I) 112 t = re.sub(r"\s{2,}", " ", t).strip(" -–—") 113 t = t or base 114 return TITLE_FIX.get(t, t) 115 116 # Suffix tokens that don't change the topic, plus explicit typo/alias fixes. 117 STRIP_TOKENS = {"cheatsheet", "cheat", "sheet", "guide", "usage", "full", 118 "quick", "markdown", "htb", "complete", "expanded", "fullguide"} 119 ALIAS = {"gog": "gpg", "volitility3": "volatility", "volitility": "volatility", 120 "redmd": "recmd", "emumeration": "enumeration", "nxc": "netexec", 121 "crackmapexec": "netexec", "meterpreter": "metasploit"} 122 123 def canonical(slug): 124 """Order-independent, de-duplicated topic key: 'privesc-windows' and 125 'windows-privesc' collapse to the same thing, and 'netexec-nxc' (nxc 126 aliases to netexec) to just 'netexec'. Sorting + set is what makes the 127 dedup catch reorderings the raw slug would miss.""" 128 toks = [t for t in slug.split("-") if t and not t.isdigit() and t not in STRIP_TOKENS] 129 toks = [ALIAS.get(t, t) for t in toks] 130 return "-".join(sorted(set(toks))) 131 132 # Semantic near-dups the canonical key can't catch — a second PowerShell 133 # sheet, a reset guide already covered by git-reset, an XSS page already 134 # covered by web/xss. Skip-if-present, by the user's call. 135 SKIP_SRC = { 136 "Tools/CMD-Powershell Cheat Sheet.md", # → windows-cmd-powershell 137 "Tools/Powershell.md", # → windows-cmd-powershell 138 "Tools/Certipy-ADCS-Cheatsheet.md", # → certipy + adcs-attack-methodology 139 "Git/Resetting.md", # → git-reset 140 "Git/Branches Expanded.md", # → git-branching 141 "Git/Branches.md", # → git-branching 142 "PasswordAttacks/john-cheatsheet.md", # → john-the-ripper 143 "PasswordAttacks/hashcat modes.md", # → hashcat 144 "Web/Cross-Site Scripting (XSS) - HTB Cheat Sheet.md", # → web/xss 145 } 146 147 # Source filename typos, fixed only in the on-site title (content untouched). 148 TITLE_FIX = {"Windows Emumeration": "Windows Enumeration"} 149 150 # ── Body clean-up (verbatim apart from vault-local scaffolding) ───────────── 151 def strip_frontmatter(txt): 152 if txt.startswith("---"): 153 end = txt.find("\n---", 3) 154 if end != -1: 155 nl = txt.find("\n", end + 1) 156 return txt[nl + 1:] if nl != -1 else "" 157 return txt 158 159 def trim_preamble(txt): 160 """Drop anything before the first H1 — that is where a chatbot intro, 161 Obsidian separators, or stray notes sit. If there is no H1, keep all.""" 162 m = re.search(r"^# .+$", txt, flags=re.M) 163 return txt[m.start():] if m else txt 164 165 def flatten_wikilinks(txt): 166 txt = re.sub(r"!\[\[[^\]]*\]\]", "", txt) # embeds → gone 167 txt = re.sub(r"\[\[([^\]|]+)\|([^\]]+)\]\]", r"\2", txt) # [[a|b]] → b 168 txt = re.sub(r"\[\[([^\]]+)\]\]", r"\1", txt) # [[a]] → a 169 return txt 170 171 def strip_citations(txt): 172 """Remove [1][2]-style markers, but never touch code fences (a shell 173 array index or regex must survive).""" 174 out, in_fence = [], False 175 for line in txt.split("\n"): 176 if line.lstrip().startswith("```"): 177 in_fence = not in_fence 178 out.append(line) 179 continue 180 if in_fence: 181 out.append(line) 182 continue 183 # Only runs of bracketed 1–3 digit numbers, i.e. citation clusters. 184 out.append(re.sub(r"(?:\[\d{1,3}\])+", "", line)) 185 return "\n".join(out) 186 187 def clean_body(txt): 188 txt = strip_frontmatter(txt) 189 txt = trim_preamble(txt) 190 txt = flatten_wikilinks(txt) 191 txt = strip_citations(txt) 192 return txt.strip() + "\n" 193 194 # ── Derived description ───────────────────────────────────────────────────── 195 def first_paragraph(body): 196 lines = body.split("\n") 197 skip_h1 = True 198 buf = [] 199 for ln in lines: 200 s = ln.strip() 201 if skip_h1 and s.startswith("# "): 202 skip_h1 = False 203 continue 204 if not s: 205 if buf: 206 break 207 continue 208 if s[0] in "#>|-*" or s.startswith("```") or s.startswith("**MITRE"): 209 if buf: 210 break 211 continue 212 buf.append(s) 213 para = " ".join(buf) 214 para = re.sub(r"`([^`]*)`", r"\1", para) 215 para = re.sub(r"\*\*([^*]*)\*\*", r"\1", para) 216 para = re.sub(r"\*([^*]*)\*", r"\1", para) 217 para = re.sub(r"\[([^\]]+)\]\([^)]+\)", r"\1", para) 218 para = re.sub(r"\s{2,}", " ", para).strip() 219 if len(para) > 155: 220 cut = para[:155].rsplit(" ", 1)[0] 221 para = cut.rstrip(",.;:") + "…" 222 return para 223 224 # ── tools / tags / difficulty ─────────────────────────────────────────────── 225 TOOL_DB = [ 226 ("nmap", "Nmap"), ("rustscan", "RustScan"), ("ffuf", "ffuf"), 227 ("gobuster", "Gobuster"), ("nuclei", "Nuclei"), ("nikto", "Nikto"), 228 ("wpscan", "WPScan"), ("smbmap", "smbmap"), ("netexec", "NetExec"), 229 ("nxc ", "NetExec"), ("crackmapexec", "NetExec"), ("impacket", "Impacket"), 230 ("secretsdump", "Impacket"), ("mimikatz", "Mimikatz"), ("rubeus", "Rubeus"), 231 ("certipy", "Certipy"), ("bloodhound", "BloodHound"), ("sharphound", "SharpHound"), 232 ("kerbrute", "Kerbrute"), ("ldapsearch", "ldapsearch"), ("hashcat", "Hashcat"), 233 ("john", "John"), ("sqlmap", "SQLMap"), ("metasploit", "Metasploit"), 234 ("meterpreter", "Meterpreter"), ("evil-winrm", "Evil-WinRM"), 235 ("chisel", "Chisel"), ("ligolo", "Ligolo-ng"), ("socat", "socat"), 236 ("proxychains", "proxychains"), ("responder", "Responder"), ("mitm6", "mitm6"), 237 ("snaffler", "Snaffler"), ("gitleaks", "Gitleaks"), ("trufflehog", "TruffleHog"), 238 ("tshark", "tshark"), ("volatility", "Volatility"), ("gpg", "GPG"), 239 ("openssl", "OpenSSL"), ("faketime", "faketime"), ("certify", "Certify"), 240 ("powershell", "PowerShell"), ("evil-winrm", "Evil-WinRM"), 241 ] 242 243 def tools_for(body): 244 low = body.lower() 245 seen, out = set(), [] 246 for needle, disp in TOOL_DB: 247 if disp in seen: 248 continue 249 if needle in low: 250 seen.add(disp) 251 out.append(disp) 252 if len(out) >= 5: 253 break 254 return out 255 256 TAG_MAP = [ 257 ("kerberos", "kerberos"), ("kerberoast", "kerberos"), ("adcs", "adcs"), 258 ("esc", "adcs"), ("certificate", "adcs"), ("dcsync", "credential-access"), 259 ("delegation", "delegation"), ("ntlm", "ntlm"), ("relay", "relay"), 260 ("ticket", "kerberos"), ("privilege", "privilege-escalation"), 261 ("persistence", "persistence"), ("lateral", "lateral-movement"), 262 ("xss", "xss"), ("sql", "sql-injection"), ("lfi", "file-inclusion"), 263 ("forensic", "forensics"), ("pivot", "pivoting"), ("tunnel", "tunneling"), 264 ("hash", "hashing"), ("spray", "password-attacks"), 265 ] 266 267 def tags_for(cat, slug, body): 268 hay = (slug + " " + body[:1500]).lower() 269 out = [cat] 270 for needle, tag in TAG_MAP: 271 if needle in hay and tag not in out: 272 out.append(tag) 273 if len(out) >= 5: 274 break 275 return out 276 277 ADV_HINT = re.compile(r"(esc\d|persist|theft|dcsync|delegation|golden|silver|" 278 r"diamond|sapphire|relay|adcs|zerologon|petitpotam|rbcd|" 279 r"skeleton|dsrm|sid-history|kerberoast|shadow-cred)", re.I) 280 281 def difficulty_for(rel, slug, cat): 282 p = rel.lower() 283 if "ad-attack" in p or "acl-esc" in p or "/kerberos/" in p: 284 return "advanced" 285 if cat == "privilege-escalation": 286 return "advanced" 287 if ADV_HINT.search(slug): 288 return "advanced" 289 return "intermediate" 290 291 # ── Existing sheets → canonical set (never clobber the curated 60) ────────── 292 def existing_canonicals(): 293 keys = {} 294 for root, _, files in os.walk(SHEETS): 295 for f in files: 296 if f.endswith(".md"): 297 slug = f[:-3] 298 keys[canonical(slug)] = os.path.relpath(os.path.join(root, f), SHEETS) 299 return keys 300 301 def yaml_scalar(s): 302 return json.dumps(s, ensure_ascii=False) 303 304 def yaml_list(xs): 305 return "[" + ", ".join(json.dumps(x, ensure_ascii=False) for x in xs) + "]" 306 307 def main(): 308 existing = existing_canonicals() 309 taken = dict(existing) # canonical → where (grows as we add) 310 rows = [] # (decision, cat, slug, title, rel, reason) 311 add_plan = [] # (dest_path, frontmatter+body) 312 313 all_md = [] 314 for root, _, files in os.walk(SRC): 315 for f in files: 316 if f.endswith(".md"): 317 all_md.append(os.path.relpath(os.path.join(root, f), SRC)) 318 all_md.sort() 319 320 for rel in all_md: 321 if is_non_sheet(rel): 322 rows.append(("SKIP", "", "", "", rel, "non-sheet (meta/index/roadmap)")) 323 continue 324 if rel in SKIP_SRC: 325 rows.append(("SKIP", "", "", "", rel, "semantic dup of existing sheet")) 326 continue 327 328 cat = category_for(rel) 329 title = title_from(rel) 330 slug = slugify(title) 331 key = canonical(slug) 332 333 if key in taken: 334 rows.append(("SKIP", cat, slug, title, rel, f"dup of {taken[key]}")) 335 continue 336 337 raw = open(os.path.join(SRC, rel), encoding="utf-8", errors="replace").read() 338 body = clean_body(raw) 339 desc = first_paragraph(body) or f"{title} — operator reference." 340 tools = tools_for(body) 341 tags = tags_for(cat, slug, body) 342 diff = difficulty_for(rel, slug, cat) 343 344 fm = [ 345 "---", 346 f"title: {yaml_scalar(title)}", 347 f"description: {yaml_scalar(desc)}", 348 f"category: {cat}", 349 f"tags: {yaml_list(tags)}", 350 f"tools: {yaml_list(tools)}", 351 f"difficulty: {diff}", 352 f'updated: "{UPDATED}"', 353 f"source: {yaml_scalar('vault:' + rel)}", 354 "---", 355 "", 356 ] 357 dest = os.path.join(SHEETS, cat, slug + ".md") 358 add_plan.append((dest, "\n".join(fm) + body)) 359 taken[key] = f"{cat}/{slug}.md (new)" 360 rows.append(("ADD", cat, slug, title, rel, f"{diff} · {len(tools)} tools")) 361 362 # Manifest 363 with open(MANIFEST, "w", encoding="utf-8") as fh: 364 fh.write("decision\tcategory\tslug\ttitle\tsource\treason\n") 365 for r in rows: 366 fh.write("\t".join(r) + "\n") 367 368 adds = [r for r in rows if r[0] == "ADD"] 369 skips = [r for r in rows if r[0] == "SKIP"] 370 percat = {} 371 for r in adds: 372 percat[r[1]] = percat.get(r[1], 0) + 1 373 374 print(f"scanned {len(all_md)} source .md") 375 print(f" ADD {len(adds)}") 376 print(f" SKIP {len(skips)} " 377 f"({sum(1 for r in skips if 'non-sheet' in r[5])} meta, " 378 f"{sum(1 for r in skips if r[5].startswith('dup'))} dup)") 379 print(" new per category:") 380 for c in sorted(percat): 381 print(f" {c:22} {percat[c]}") 382 print(f"manifest → {os.path.relpath(MANIFEST, REPO)}") 383 384 if DRY: 385 print("\nDRY RUN — no files written.") 386 return 387 388 for dest, content in add_plan: 389 os.makedirs(os.path.dirname(dest), exist_ok=True) 390 with open(dest, "w", encoding="utf-8") as fh: 391 fh.write(content) 392 print(f"\nwrote {len(add_plan)} sheets.") 393 394 if __name__ == "__main__": 395 main()