sync-payloads.py (12209B)
1 #!/usr/bin/env python3 2 """Mirror PayloadsAllTheThings (MIT, swisskyrepo) into the site as the `payloads` 3 content collection. Faithful markdown mirror; images + raw payload files hotlink 4 to upstream GitHub. Deterministic + re-runnable. 5 6 Usage: python3 scripts/sync-payloads.py /path/to/patt-clone <sha> 7 """ 8 import json, os, re, shutil, sys 9 10 REPO = os.path.normpath(os.path.join(os.path.dirname(os.path.abspath(__file__)), "..")) 11 OUT = os.path.join(REPO, "src", "content", "payloads") 12 VENDOR = os.path.join(REPO, "vendor", "PayloadsAllTheThings") 13 UPSTREAM = "https://github.com/swisskyrepo/PayloadsAllTheThings" 14 15 SRC = sys.argv[1] if len(sys.argv) > 1 else None 16 SHA = sys.argv[2] if len(sys.argv) > 2 else "master" 17 if not SRC or not os.path.isdir(SRC): 18 sys.exit("usage: sync-payloads.py <patt-clone-dir> <sha>") 19 20 # Skip non-topic root entries. 21 SKIP_TOP = {".git", ".github", "Images", "assets"} 22 SKIP_FILES = {"readme.md", "contributing.md", "license.md", "code_of_conduct.md", 23 "security.md", "changelog.md", "_template_vulnerability.md"} 24 25 26 def slugify(s): 27 s = s.strip().lower() 28 s = re.sub(r"[^a-z0-9]+", "-", s) 29 return re.sub(r"-+", "-", s).strip("-") or "x" 30 31 32 def topic_slug(topic): 33 return slugify(topic) 34 35 36 def file_slug(name): # name without .md 37 return slugify(name) 38 39 40 def blob_url(relpath): 41 # GitHub blob URL for a repo-relative path (spaces -> %20) 42 from urllib.parse import quote 43 return f"{UPSTREAM}/blob/{SHA}/{quote(relpath)}" 44 45 46 def raw_url(relpath): 47 from urllib.parse import quote 48 return f"https://raw.githubusercontent.com/swisskyrepo/PayloadsAllTheThings/{SHA}/{quote(relpath)}" 49 50 51 # --- Discover all mirrored markdown, build a path->route map first -------------- 52 topics = sorted(d for d in os.listdir(SRC) 53 if os.path.isdir(os.path.join(SRC, d)) and d not in SKIP_TOP 54 and not d.startswith(".") and not d.startswith("_")) 55 56 # map: repo-relative md path (posix) -> site route (/payloads/...) 57 route_of = {} 58 md_files = [] # (topic, relpath, abspath, is_readme) 59 for topic in topics: 60 tdir = os.path.join(SRC, topic) 61 for root, _dirs, files in os.walk(tdir): 62 for f in files: 63 if not f.lower().endswith(".md"): 64 continue 65 if f.lower() in SKIP_FILES and os.path.basename(root) != topic: 66 pass # only skip template-ish names anywhere 67 abspath = os.path.join(root, f) 68 relpath = os.path.relpath(abspath, SRC).replace(os.sep, "/") 69 is_readme = f.lower() == "readme.md" 70 ts = topic_slug(topic) 71 # sub-path inside the topic (for nested files), slugified per segment 72 inner = os.path.relpath(abspath, tdir).replace(os.sep, "/") 73 if is_readme and "/" not in inner: 74 route = f"/payloads/{ts}" 75 out_id = f"{ts}/index" 76 else: 77 parts = inner[:-3].split("/") # drop .md 78 seg = "/".join(slugify(p) for p in parts) 79 route = f"/payloads/{ts}/{seg}" 80 out_id = f"{ts}/{seg}" 81 route_of[relpath] = route 82 md_files.append((topic, relpath, abspath, is_readme, out_id, route)) 83 84 # --- Link rewriting ------------------------------------------------------------- 85 LINK = re.compile(r"(!?)\[([^\]]*)\]\(([^)]+)\)") 86 87 88 # Absolute upstream URLs that actually point back into the repo. Two hosts: 89 # github.com/swisskyrepo/PayloadsAllTheThings/(blob|tree|raw)/<ref>/<path> 90 # raw.githubusercontent.com/swisskyrepo/PayloadsAllTheThings/<ref>/<path> 91 UPSTREAM_URL = re.compile( 92 r"^https?://(?:www\.)?github\.com/swisskyrepo/PayloadsAllTheThings/(?:blob|tree|raw)/[^/]+/(.*)$" 93 r"|^https?://raw\.githubusercontent\.com/swisskyrepo/PayloadsAllTheThings/[^/]+/(.*)$", 94 re.I, 95 ) 96 _route_lc = None # built lazily: lowercased path -> route 97 98 # PayloadsAllTheThings moved its "Methodology and Resources" pages out to 99 # InternalAllTheThings and left behind stubs whose whole body is links to the 100 # live IATT site. We mirror IATT too (the `internal` collection), so those links 101 # belong on-site rather than bouncing the reader to swisskyrepo.github.io. 102 IATT_SITE = re.compile(r"^https?://swisskyrepo\.github\.io/InternalAllTheThings/?(.*)$", re.I) 103 104 # The move was done with a find/replace that expanded every bare "#" on the 105 # line into "<page-url>#", including the one in "Load C# assembly reflectively", 106 # which left the site URL sitting in the link *text*. Undo that in labels: a 107 # live IATT URL ending in "/#" is always the artifact, never a real link. 108 IATT_IN_TEXT = re.compile( 109 r"https?://swisskyrepo\.github\.io/InternalAllTheThings/[^\s\]]*?/#", re.I 110 ) 111 112 # sync-internal.py owns the real route table; all we know here is the section 113 # roots, so anything outside them is not ours to rewrite. 114 IATT_SECTIONS = {"active-directory", "cheatsheets", "cloud", "command-control", 115 "containers", "databases", "devops", "methodology", "redteam"} 116 117 # Pages upstream still links to but IATT has since deleted or renamed. Nothing 118 # string-based can spot them, so they are listed and sent to their section index. 119 IATT_STALE = {"active-directory/internal-mitm-relay", "cheatsheets/mssql-server-cheatsheet", 120 "cheatsheets/source-code-management-ci", "cloud/aws/aws-pentest", 121 "cloud/azure/azure-services"} 122 123 124 def iatt_route(path): 125 """Map an InternalAllTheThings live-site path to our /internal route, or None.""" 126 from urllib.parse import unquote 127 path, _, anchor = path.partition("#") 128 parts = [slugify(p) for p in unquote(path).strip("/").split("/") if p] 129 if not parts: 130 return "/internal" 131 if parts[0] not in IATT_SECTIONS: 132 return None 133 seg = "/".join(parts) 134 if seg in IATT_STALE: 135 return f"/internal/{parts[0]}" # anchor is meaningless on the index 136 return f"/internal/{seg}" + ("#" + anchor if anchor else "") 137 138 139 def _rlc(): 140 global _route_lc 141 if _route_lc is None: 142 _route_lc = {k.lower(): v for k, v in route_of.items()} 143 return _route_lc 144 145 146 def route_for_repo_path(resolved, anchor): 147 """Map a repo-relative path to an internal route or a pinned upstream URL.""" 148 resolved = resolved.strip("/") 149 if resolved == "": 150 return "/payloads" + anchor 151 low = resolved.lower() 152 if low.endswith(".md"): 153 r = _rlc().get(low) 154 return (r + anchor) if r else blob_url(resolved) + anchor 155 # a directory link (topic listing) -> that topic's index if mirrored 156 idx = _rlc().get((resolved + "/readme.md").lower()) 157 if idx: 158 return idx + anchor 159 # non-md file (image, txt, py, ...) -> pinned raw upstream 160 return raw_url(resolved) + anchor 161 162 163 def resolve(cur_relpath, target): 164 """Return rewritten URL for a markdown link target, or None to leave as-is.""" 165 from urllib.parse import unquote 166 t = target.strip().strip("<>").strip() 167 anchor = "" 168 169 # 1) Absolute upstream URL pointing into the repo -> internalize. 170 m = UPSTREAM_URL.match(t) 171 if m: 172 path = m.group(1) or m.group(2) or "" 173 path = path.split("?", 1)[0] # drop ?raw=true etc. 174 if "#" in path: 175 path, a = path.split("#", 1); anchor = "#" + a 176 return route_for_repo_path(unquote(path), anchor) 177 178 # 2) The live InternalAllTheThings site -> our mirror of it. 179 m = IATT_SITE.match(t) 180 if m: 181 return iatt_route(m.group(1)) 182 183 # 3) Other absolute / protocol-relative / anchors / mailto -> leave. 184 if re.match(r"^([a-z]+:|//|#|mailto:)", t, re.I): 185 return None 186 187 # 4) Relative in-repo link. 188 if "#" in t: 189 t, a = t.split("#", 1); anchor = "#" + a 190 if not t: 191 return None 192 cur_dir = os.path.dirname(cur_relpath) 193 resolved = os.path.normpath(os.path.join(cur_dir, unquote(t))).replace(os.sep, "/") 194 return route_for_repo_path(resolved, anchor) 195 196 197 def rewrite_body(cur_relpath, body): 198 def repl(m): 199 bang, text, target = m.group(1), m.group(2), m.group(3) 200 text = IATT_IN_TEXT.sub("#", text) 201 # ignore link targets wrapped in <...> or containing spaces we can't parse cleanly 202 new = resolve(cur_relpath, target) 203 if new is None: 204 return m.group(0) if text == m.group(2) else f"{bang}[{text}]({target})" 205 return f"{bang}[{text}]({new})" 206 return LINK.sub(repl, body) 207 208 209 def title_from(body, fallback): 210 m = re.search(r"^\s*#\s+(.+?)\s*$", body, re.M) 211 if m: 212 return m.group(1).strip() 213 return fallback 214 215 216 def yaml_escape(s): 217 return s.replace("\\", "\\\\").replace('"', '\\"') 218 219 220 # --- Emit ----------------------------------------------------------------------- 221 if os.path.isdir(OUT): 222 shutil.rmtree(OUT) 223 os.makedirs(OUT, exist_ok=True) 224 225 count = 0 226 topics_seen = {} 227 topics_with_readme = set() 228 topic_pages = {} # topic -> [(route, title)] for non-readme leaves, in title order 229 for topic, relpath, abspath, is_readme, out_id, route in md_files: 230 raw = open(abspath, encoding="utf-8", errors="replace").read() 231 body = rewrite_body(relpath, raw) 232 fallback = topic if is_readme else os.path.splitext(os.path.basename(abspath))[0] 233 title = title_from(raw, fallback) 234 if is_readme: 235 topics_with_readme.add(topic) 236 else: 237 topic_pages.setdefault(topic, []).append((route, title)) 238 src_url = blob_url(relpath) 239 fm = ( 240 "---\n" 241 f'title: "{yaml_escape(title)}"\n' 242 f'topic: "{yaml_escape(topic)}"\n' 243 f'topicSlug: "{topic_slug(topic)}"\n' 244 f'sourcePath: "{yaml_escape(relpath)}"\n' 245 f'sourceUrl: "{src_url}"\n' 246 f'sha: "{SHA}"\n' 247 f"isReadme: {'true' if is_readme else 'false'}\n" 248 "---\n\n" 249 ) 250 dst = os.path.join(OUT, out_id + ".md") 251 os.makedirs(os.path.dirname(dst), exist_ok=True) 252 open(dst, "w", encoding="utf-8").write(fm + body.lstrip("\n")) 253 count += 1 254 topics_seen.setdefault(topic, 0) 255 topics_seen[topic] += 1 256 257 # --- Synthesize a topic index for README-less topics ---------------------------- 258 # Some upstream folders (e.g. "Methodology and Resources") are a flat set of 259 # standalone pages with no README.md, so no `{slug}/index` root gets emitted and 260 # the payloads index would link to a 404. Generate a listing page for each such 261 # topic so its root route resolves and enumerates its pages. 262 for topic, pages in topic_pages.items(): 263 if topic in topics_with_readme: 264 continue 265 ts = topic_slug(topic) 266 listing = "\n".join( 267 f"* [{title}]({route})" for route, title in sorted(pages, key=lambda p: p[1].lower()) 268 ) 269 from urllib.parse import quote 270 tree_url = f"{UPSTREAM}/tree/{SHA}/" + quote(topic) 271 fm = ( 272 "---\n" 273 f'title: "{yaml_escape(topic)}"\n' 274 f'topic: "{yaml_escape(topic)}"\n' 275 f'topicSlug: "{ts}"\n' 276 f'sourcePath: "{yaml_escape(topic)}"\n' 277 f'sourceUrl: "{tree_url}"\n' 278 f'sha: "{SHA}"\n' 279 "isReadme: true\n" 280 "---\n\n" 281 ) 282 intro = ( 283 f"# {topic}\n\n" 284 f"> {len(pages)} pages in this section, mirrored from PayloadsAllTheThings.\n\n" 285 "## Pages\n\n" 286 ) 287 dst = os.path.join(OUT, ts, "index.md") 288 os.makedirs(os.path.dirname(dst), exist_ok=True) 289 open(dst, "w", encoding="utf-8").write(fm + intro + listing + "\n") 290 count += 1 291 topics_seen[topic] = topics_seen.get(topic, 0) + 1 292 293 # --- Vendor the license --------------------------------------------------------- 294 os.makedirs(VENDOR, exist_ok=True) 295 lic_src = None 296 for cand in ("LICENSE.md", "LICENSE", "LICENSE.txt"): 297 p = os.path.join(SRC, cand) 298 if os.path.isfile(p): 299 lic_src = p 300 break 301 if lic_src: 302 shutil.copyfile(lic_src, os.path.join(VENDOR, os.path.basename(lic_src))) 303 304 # --- Manifest ------------------------------------------------------------------- 305 manifest = { 306 "upstream": UPSTREAM, 307 "sha": SHA, 308 "topics": len(topics_seen), 309 "pages": count, 310 "topicList": sorted(topics_seen.keys()), 311 } 312 open(os.path.join(REPO, "payloads-manifest.json"), "w").write(json.dumps(manifest, indent=2)) 313 314 print(f"mirrored {count} pages across {len(topics_seen)} topics @ {SHA}") 315 print(f"license: {'copied' if lic_src else 'NOT FOUND'}")