daemon-sec-cheatsheet

The cheatsheet vault for operators: AD, enumeration, exploitation, priv-esc, web, DFIR
git clone https://git.daemon-sec.xyz/daemon-sec-cheatsheet.git
Log | Files | Refs | README | LICENSE

sync-payloads.py (12209B)


      1 #!/usr/bin/env python3
      2 """Mirror PayloadsAllTheThings (MIT, swisskyrepo) into the site as the `payloads`
      3 content collection. Faithful markdown mirror; images + raw payload files hotlink
      4 to upstream GitHub. Deterministic + re-runnable.
      5 
      6 Usage: python3 scripts/sync-payloads.py /path/to/patt-clone <sha>
      7 """
      8 import json, os, re, shutil, sys
      9 
     10 REPO = os.path.normpath(os.path.join(os.path.dirname(os.path.abspath(__file__)), ".."))
     11 OUT = os.path.join(REPO, "src", "content", "payloads")
     12 VENDOR = os.path.join(REPO, "vendor", "PayloadsAllTheThings")
     13 UPSTREAM = "https://github.com/swisskyrepo/PayloadsAllTheThings"
     14 
     15 SRC = sys.argv[1] if len(sys.argv) > 1 else None
     16 SHA = sys.argv[2] if len(sys.argv) > 2 else "master"
     17 if not SRC or not os.path.isdir(SRC):
     18     sys.exit("usage: sync-payloads.py <patt-clone-dir> <sha>")
     19 
     20 # Skip non-topic root entries.
     21 SKIP_TOP = {".git", ".github", "Images", "assets"}
     22 SKIP_FILES = {"readme.md", "contributing.md", "license.md", "code_of_conduct.md",
     23               "security.md", "changelog.md", "_template_vulnerability.md"}
     24 
     25 
     26 def slugify(s):
     27     s = s.strip().lower()
     28     s = re.sub(r"[^a-z0-9]+", "-", s)
     29     return re.sub(r"-+", "-", s).strip("-") or "x"
     30 
     31 
     32 def topic_slug(topic):
     33     return slugify(topic)
     34 
     35 
     36 def file_slug(name):  # name without .md
     37     return slugify(name)
     38 
     39 
     40 def blob_url(relpath):
     41     # GitHub blob URL for a repo-relative path (spaces -> %20)
     42     from urllib.parse import quote
     43     return f"{UPSTREAM}/blob/{SHA}/{quote(relpath)}"
     44 
     45 
     46 def raw_url(relpath):
     47     from urllib.parse import quote
     48     return f"https://raw.githubusercontent.com/swisskyrepo/PayloadsAllTheThings/{SHA}/{quote(relpath)}"
     49 
     50 
     51 # --- Discover all mirrored markdown, build a path->route map first --------------
     52 topics = sorted(d for d in os.listdir(SRC)
     53                 if os.path.isdir(os.path.join(SRC, d)) and d not in SKIP_TOP
     54                 and not d.startswith(".") and not d.startswith("_"))
     55 
     56 # map: repo-relative md path (posix) -> site route (/payloads/...)
     57 route_of = {}
     58 md_files = []  # (topic, relpath, abspath, is_readme)
     59 for topic in topics:
     60     tdir = os.path.join(SRC, topic)
     61     for root, _dirs, files in os.walk(tdir):
     62         for f in files:
     63             if not f.lower().endswith(".md"):
     64                 continue
     65             if f.lower() in SKIP_FILES and os.path.basename(root) != topic:
     66                 pass  # only skip template-ish names anywhere
     67             abspath = os.path.join(root, f)
     68             relpath = os.path.relpath(abspath, SRC).replace(os.sep, "/")
     69             is_readme = f.lower() == "readme.md"
     70             ts = topic_slug(topic)
     71             # sub-path inside the topic (for nested files), slugified per segment
     72             inner = os.path.relpath(abspath, tdir).replace(os.sep, "/")
     73             if is_readme and "/" not in inner:
     74                 route = f"/payloads/{ts}"
     75                 out_id = f"{ts}/index"
     76             else:
     77                 parts = inner[:-3].split("/")  # drop .md
     78                 seg = "/".join(slugify(p) for p in parts)
     79                 route = f"/payloads/{ts}/{seg}"
     80                 out_id = f"{ts}/{seg}"
     81             route_of[relpath] = route
     82             md_files.append((topic, relpath, abspath, is_readme, out_id, route))
     83 
     84 # --- Link rewriting -------------------------------------------------------------
     85 LINK = re.compile(r"(!?)\[([^\]]*)\]\(([^)]+)\)")
     86 
     87 
     88 # Absolute upstream URLs that actually point back into the repo. Two hosts:
     89 #   github.com/swisskyrepo/PayloadsAllTheThings/(blob|tree|raw)/<ref>/<path>
     90 #   raw.githubusercontent.com/swisskyrepo/PayloadsAllTheThings/<ref>/<path>
     91 UPSTREAM_URL = re.compile(
     92     r"^https?://(?:www\.)?github\.com/swisskyrepo/PayloadsAllTheThings/(?:blob|tree|raw)/[^/]+/(.*)$"
     93     r"|^https?://raw\.githubusercontent\.com/swisskyrepo/PayloadsAllTheThings/[^/]+/(.*)$",
     94     re.I,
     95 )
     96 _route_lc = None  # built lazily: lowercased path -> route
     97 
     98 # PayloadsAllTheThings moved its "Methodology and Resources" pages out to
     99 # InternalAllTheThings and left behind stubs whose whole body is links to the
    100 # live IATT site. We mirror IATT too (the `internal` collection), so those links
    101 # belong on-site rather than bouncing the reader to swisskyrepo.github.io.
    102 IATT_SITE = re.compile(r"^https?://swisskyrepo\.github\.io/InternalAllTheThings/?(.*)$", re.I)
    103 
    104 # The move was done with a find/replace that expanded every bare "#" on the
    105 # line into "<page-url>#", including the one in "Load C# assembly reflectively",
    106 # which left the site URL sitting in the link *text*. Undo that in labels: a
    107 # live IATT URL ending in "/#" is always the artifact, never a real link.
    108 IATT_IN_TEXT = re.compile(
    109     r"https?://swisskyrepo\.github\.io/InternalAllTheThings/[^\s\]]*?/#", re.I
    110 )
    111 
    112 # sync-internal.py owns the real route table; all we know here is the section
    113 # roots, so anything outside them is not ours to rewrite.
    114 IATT_SECTIONS = {"active-directory", "cheatsheets", "cloud", "command-control",
    115                  "containers", "databases", "devops", "methodology", "redteam"}
    116 
    117 # Pages upstream still links to but IATT has since deleted or renamed. Nothing
    118 # string-based can spot them, so they are listed and sent to their section index.
    119 IATT_STALE = {"active-directory/internal-mitm-relay", "cheatsheets/mssql-server-cheatsheet",
    120               "cheatsheets/source-code-management-ci", "cloud/aws/aws-pentest",
    121               "cloud/azure/azure-services"}
    122 
    123 
    124 def iatt_route(path):
    125     """Map an InternalAllTheThings live-site path to our /internal route, or None."""
    126     from urllib.parse import unquote
    127     path, _, anchor = path.partition("#")
    128     parts = [slugify(p) for p in unquote(path).strip("/").split("/") if p]
    129     if not parts:
    130         return "/internal"
    131     if parts[0] not in IATT_SECTIONS:
    132         return None
    133     seg = "/".join(parts)
    134     if seg in IATT_STALE:
    135         return f"/internal/{parts[0]}"  # anchor is meaningless on the index
    136     return f"/internal/{seg}" + ("#" + anchor if anchor else "")
    137 
    138 
    139 def _rlc():
    140     global _route_lc
    141     if _route_lc is None:
    142         _route_lc = {k.lower(): v for k, v in route_of.items()}
    143     return _route_lc
    144 
    145 
    146 def route_for_repo_path(resolved, anchor):
    147     """Map a repo-relative path to an internal route or a pinned upstream URL."""
    148     resolved = resolved.strip("/")
    149     if resolved == "":
    150         return "/payloads" + anchor
    151     low = resolved.lower()
    152     if low.endswith(".md"):
    153         r = _rlc().get(low)
    154         return (r + anchor) if r else blob_url(resolved) + anchor
    155     # a directory link (topic listing) -> that topic's index if mirrored
    156     idx = _rlc().get((resolved + "/readme.md").lower())
    157     if idx:
    158         return idx + anchor
    159     # non-md file (image, txt, py, ...) -> pinned raw upstream
    160     return raw_url(resolved) + anchor
    161 
    162 
    163 def resolve(cur_relpath, target):
    164     """Return rewritten URL for a markdown link target, or None to leave as-is."""
    165     from urllib.parse import unquote
    166     t = target.strip().strip("<>").strip()
    167     anchor = ""
    168 
    169     # 1) Absolute upstream URL pointing into the repo -> internalize.
    170     m = UPSTREAM_URL.match(t)
    171     if m:
    172         path = m.group(1) or m.group(2) or ""
    173         path = path.split("?", 1)[0]  # drop ?raw=true etc.
    174         if "#" in path:
    175             path, a = path.split("#", 1); anchor = "#" + a
    176         return route_for_repo_path(unquote(path), anchor)
    177 
    178     # 2) The live InternalAllTheThings site -> our mirror of it.
    179     m = IATT_SITE.match(t)
    180     if m:
    181         return iatt_route(m.group(1))
    182 
    183     # 3) Other absolute / protocol-relative / anchors / mailto -> leave.
    184     if re.match(r"^([a-z]+:|//|#|mailto:)", t, re.I):
    185         return None
    186 
    187     # 4) Relative in-repo link.
    188     if "#" in t:
    189         t, a = t.split("#", 1); anchor = "#" + a
    190     if not t:
    191         return None
    192     cur_dir = os.path.dirname(cur_relpath)
    193     resolved = os.path.normpath(os.path.join(cur_dir, unquote(t))).replace(os.sep, "/")
    194     return route_for_repo_path(resolved, anchor)
    195 
    196 
    197 def rewrite_body(cur_relpath, body):
    198     def repl(m):
    199         bang, text, target = m.group(1), m.group(2), m.group(3)
    200         text = IATT_IN_TEXT.sub("#", text)
    201         # ignore link targets wrapped in <...> or containing spaces we can't parse cleanly
    202         new = resolve(cur_relpath, target)
    203         if new is None:
    204             return m.group(0) if text == m.group(2) else f"{bang}[{text}]({target})"
    205         return f"{bang}[{text}]({new})"
    206     return LINK.sub(repl, body)
    207 
    208 
    209 def title_from(body, fallback):
    210     m = re.search(r"^\s*#\s+(.+?)\s*$", body, re.M)
    211     if m:
    212         return m.group(1).strip()
    213     return fallback
    214 
    215 
    216 def yaml_escape(s):
    217     return s.replace("\\", "\\\\").replace('"', '\\"')
    218 
    219 
    220 # --- Emit -----------------------------------------------------------------------
    221 if os.path.isdir(OUT):
    222     shutil.rmtree(OUT)
    223 os.makedirs(OUT, exist_ok=True)
    224 
    225 count = 0
    226 topics_seen = {}
    227 topics_with_readme = set()
    228 topic_pages = {}  # topic -> [(route, title)] for non-readme leaves, in title order
    229 for topic, relpath, abspath, is_readme, out_id, route in md_files:
    230     raw = open(abspath, encoding="utf-8", errors="replace").read()
    231     body = rewrite_body(relpath, raw)
    232     fallback = topic if is_readme else os.path.splitext(os.path.basename(abspath))[0]
    233     title = title_from(raw, fallback)
    234     if is_readme:
    235         topics_with_readme.add(topic)
    236     else:
    237         topic_pages.setdefault(topic, []).append((route, title))
    238     src_url = blob_url(relpath)
    239     fm = (
    240         "---\n"
    241         f'title: "{yaml_escape(title)}"\n'
    242         f'topic: "{yaml_escape(topic)}"\n'
    243         f'topicSlug: "{topic_slug(topic)}"\n'
    244         f'sourcePath: "{yaml_escape(relpath)}"\n'
    245         f'sourceUrl: "{src_url}"\n'
    246         f'sha: "{SHA}"\n'
    247         f"isReadme: {'true' if is_readme else 'false'}\n"
    248         "---\n\n"
    249     )
    250     dst = os.path.join(OUT, out_id + ".md")
    251     os.makedirs(os.path.dirname(dst), exist_ok=True)
    252     open(dst, "w", encoding="utf-8").write(fm + body.lstrip("\n"))
    253     count += 1
    254     topics_seen.setdefault(topic, 0)
    255     topics_seen[topic] += 1
    256 
    257 # --- Synthesize a topic index for README-less topics ----------------------------
    258 # Some upstream folders (e.g. "Methodology and Resources") are a flat set of
    259 # standalone pages with no README.md, so no `{slug}/index` root gets emitted and
    260 # the payloads index would link to a 404. Generate a listing page for each such
    261 # topic so its root route resolves and enumerates its pages.
    262 for topic, pages in topic_pages.items():
    263     if topic in topics_with_readme:
    264         continue
    265     ts = topic_slug(topic)
    266     listing = "\n".join(
    267         f"* [{title}]({route})" for route, title in sorted(pages, key=lambda p: p[1].lower())
    268     )
    269     from urllib.parse import quote
    270     tree_url = f"{UPSTREAM}/tree/{SHA}/" + quote(topic)
    271     fm = (
    272         "---\n"
    273         f'title: "{yaml_escape(topic)}"\n'
    274         f'topic: "{yaml_escape(topic)}"\n'
    275         f'topicSlug: "{ts}"\n'
    276         f'sourcePath: "{yaml_escape(topic)}"\n'
    277         f'sourceUrl: "{tree_url}"\n'
    278         f'sha: "{SHA}"\n'
    279         "isReadme: true\n"
    280         "---\n\n"
    281     )
    282     intro = (
    283         f"# {topic}\n\n"
    284         f"> {len(pages)} pages in this section, mirrored from PayloadsAllTheThings.\n\n"
    285         "## Pages\n\n"
    286     )
    287     dst = os.path.join(OUT, ts, "index.md")
    288     os.makedirs(os.path.dirname(dst), exist_ok=True)
    289     open(dst, "w", encoding="utf-8").write(fm + intro + listing + "\n")
    290     count += 1
    291     topics_seen[topic] = topics_seen.get(topic, 0) + 1
    292 
    293 # --- Vendor the license ---------------------------------------------------------
    294 os.makedirs(VENDOR, exist_ok=True)
    295 lic_src = None
    296 for cand in ("LICENSE.md", "LICENSE", "LICENSE.txt"):
    297     p = os.path.join(SRC, cand)
    298     if os.path.isfile(p):
    299         lic_src = p
    300         break
    301 if lic_src:
    302     shutil.copyfile(lic_src, os.path.join(VENDOR, os.path.basename(lic_src)))
    303 
    304 # --- Manifest -------------------------------------------------------------------
    305 manifest = {
    306     "upstream": UPSTREAM,
    307     "sha": SHA,
    308     "topics": len(topics_seen),
    309     "pages": count,
    310     "topicList": sorted(topics_seen.keys()),
    311 }
    312 open(os.path.join(REPO, "payloads-manifest.json"), "w").write(json.dumps(manifest, indent=2))
    313 
    314 print(f"mirrored {count} pages across {len(topics_seen)} topics @ {SHA}")
    315 print(f"license: {'copied' if lic_src else 'NOT FOUND'}")