daemon-sec-cheatsheet

The cheatsheet vault for operators: AD, enumeration, exploitation, priv-esc, web, DFIR
git clone https://git.daemon-sec.xyz/daemon-sec-cheatsheet.git
Log | Files | Refs | README | LICENSE

build-candidates.py (3438B)


      1 #!/usr/bin/env python3
      2 """Build a metadata index of all candidate cheatsheet files for curation.
      3 
      4 Scans the NetrunnerVault cheatsheets + the Ethical_Hacking-Cheatsheets repo,
      5 emitting one JSON record per file with path, size, line count, title,
      6 frontmatter fields, and a content preview. Output: candidates.json
      7 """
      8 import json, os, re, sys
      9 
     10 SOURCES = [
     11     ("vault", os.environ.get("VAULT_CHEATS", os.path.expanduser("~/git/NetrunnerVault/02Cybersecurity/Cheatsheets"))),
     12     ("repo", "/Users/daemon1/git/Ethical_Hacking-Cheatsheets"),
     13 ]
     14 SKIP_DIRS = {".space", ".git", "attachments", "node_modules", ".obsidian"}
     15 SKIP_NAMES = {".DS_Store"}
     16 
     17 def parse_frontmatter(text):
     18     fm = {}
     19     if text.startswith("---"):
     20         end = text.find("\n---", 3)
     21         if end != -1:
     22             block = text[3:end]
     23             for line in block.splitlines():
     24                 m = re.match(r"^([A-Za-z0-9_]+):\s*(.*)$", line)
     25                 if m:
     26                     fm[m.group(1).strip()] = m.group(2).strip()
     27     return fm
     28 
     29 def title_from(text, path):
     30     for line in text.splitlines():
     31         s = line.strip()
     32         if s.startswith("# "):
     33             return s[2:].strip()
     34     return os.path.splitext(os.path.basename(path))[0]
     35 
     36 def preview(text, n=45):
     37     # strip frontmatter for preview
     38     if text.startswith("---"):
     39         end = text.find("\n---", 3)
     40         if end != -1:
     41             text = text[end+4:]
     42     lines = [l for l in text.splitlines()]
     43     return "\n".join(lines[:n])
     44 
     45 records = []
     46 for src, root in SOURCES:
     47     if not os.path.isdir(root):
     48         print(f"WARN missing source {root}", file=sys.stderr); continue
     49     for dp, dns, fns in os.walk(root):
     50         dns[:] = [d for d in dns if d not in SKIP_DIRS]
     51         for fn in fns:
     52             if fn in SKIP_NAMES or fn.startswith(".fuse_hidden"):
     53                 continue
     54             ext = os.path.splitext(fn)[1].lower()
     55             if ext not in (".md", ".pdf", ".svg", ".html"):
     56                 continue
     57             full = os.path.join(dp, fn)
     58             rel = os.path.relpath(full, root)
     59             try:
     60                 size = os.path.getsize(full)
     61             except OSError:
     62                 continue
     63             rec = {"source": src, "path": full, "rel": rel, "ext": ext.lstrip("."),
     64                    "bytes": size, "topdir": rel.split(os.sep)[0] if os.sep in rel else "(root)"}
     65             if ext == ".md":
     66                 try:
     67                     with open(full, encoding="utf-8", errors="replace") as f:
     68                         text = f.read()
     69                 except OSError:
     70                     continue
     71                 rec["lines"] = text.count("\n") + 1
     72                 rec["title"] = title_from(text, full)
     73                 rec["frontmatter"] = parse_frontmatter(text)
     74                 rec["preview"] = preview(text)
     75                 # heuristics
     76                 rec["h2_count"] = len(re.findall(r"^##\s", text, re.M))
     77                 rec["code_blocks"] = text.count("```") // 2
     78             records.append(rec)
     79 
     80 records.sort(key=lambda r: (r["source"], r["topdir"], r["rel"].lower()))
     81 out = os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "candidates.json")
     82 out = os.path.normpath(out)
     83 with open(out, "w", encoding="utf-8") as f:
     84     json.dump(records, f, indent=1, ensure_ascii=False)
     85 
     86 md = [r for r in records if r["ext"] == "md"]
     87 print(f"total files: {len(records)}  md: {len(md)}  pdf/svg/html: {len(records)-len(md)}")
     88 print(f"written: {out}")