sync-internal.py (12357B)
1 #!/usr/bin/env python3 2 """Mirror InternalAllTheThings (swisskyrepo) into the site as the `internal` 3 content collection. Faithful markdown mirror; images + non-md assets hotlink to 4 upstream GitHub. Deterministic + re-runnable. IATT ships no LICENSE, so the 5 `sourceUrl` attribution on every page is the standing credit to the upstream. 6 7 Usage: python3 scripts/sync-internal.py /path/to/iatt-clone <sha> 8 """ 9 import json, os, re, shutil, sys 10 from urllib.parse import quote, unquote 11 12 REPO = os.path.normpath(os.path.join(os.path.dirname(os.path.abspath(__file__)), "..")) 13 OUT = os.path.join(REPO, "src", "content", "internal") 14 UPSTREAM = "https://github.com/swisskyrepo/InternalAllTheThings" 15 LIVE = "swisskyrepo.github.io/InternalAllTheThings" 16 17 SRC = sys.argv[1] if len(sys.argv) > 1 else None 18 SHA = sys.argv[2] if len(sys.argv) > 2 else "master" 19 if not SRC or not os.path.isdir(SRC): 20 sys.exit("usage: sync-internal.py <iatt-clone-dir> <sha>") 21 DOCS = os.path.join(SRC, "docs") 22 if not os.path.isdir(DOCS): 23 sys.exit(f"no docs/ under {SRC}") 24 25 # Human section titles keyed by the docs/<dir> name. Anything unmapped falls 26 # back to a title-cased dir name (hyphens -> spaces). 27 SECTION_TITLES = { 28 "active-directory": "Active Directory", 29 "cheatsheets": "Cheatsheets", 30 "cloud": "Cloud", 31 "command-control": "Command & Control", 32 "containers": "Containers", 33 "databases": "Databases", 34 "devops": "DevOps", 35 "methodology": "Methodology", 36 "redteam": "Red Team", 37 } 38 39 40 def slugify(s): 41 s = s.strip().lower() 42 s = re.sub(r"[^a-z0-9]+", "-", s) 43 return re.sub(r"-+", "-", s).strip("-") or "x" 44 45 46 def section_title(name): 47 return SECTION_TITLES.get(name, name.replace("-", " ").title()) 48 49 50 def blob_url(relpath): # repo-relative path -> pinned GitHub blob URL 51 return f"{UPSTREAM}/blob/{SHA}/{quote(relpath)}" 52 53 54 def raw_url(relpath): # repo-relative path -> pinned raw.githubusercontent URL 55 return f"https://raw.githubusercontent.com/swisskyrepo/InternalAllTheThings/{SHA}/{quote(relpath)}" 56 57 58 def tree_url(relpath): # repo-relative dir -> pinned GitHub tree URL 59 return f"{UPSTREAM}/tree/{SHA}/{quote(relpath)}" 60 61 62 # --- Discover mirrored markdown, build path->route maps before any rewrite ----- 63 # mkdocs.yml declares no `nav`, so mkdocs publishes every .md under docs/ — 64 # including DISCLAIMER.md, the liability notice for everything else here. Only 65 # docs/README.md is repo chrome rather than a page. 66 ROOT_SKIP = {"readme.md"} 67 68 sections = sorted( 69 d for d in os.listdir(DOCS) if os.path.isdir(os.path.join(DOCS, d)) 70 ) 71 72 route_of_md = {} # repo-relative md path (lowercased) -> internal route 73 route_of_dir = {} # repo-relative section dir (lowercased) -> section root route 74 routes = set() # every emitted route, for R1 resolution + verification 75 pages = [] # (section, rel, abspath, is_index, out_id, route) 76 root_pages = [] # (rel, abspath, slug, route) for docs/*.md outside a section 77 78 # Section roots resolve even for README-less sections (a synthesized index is 79 # emitted below), so register them up front — R1 links may target them. 80 for section in sections: 81 sec_route = f"/internal/{slugify(section)}" 82 route_of_dir[f"docs/{section}".lower()] = sec_route 83 routes.add(sec_route) 84 85 for section in sections: 86 sdir = os.path.join(DOCS, section) 87 for root, dirs, files in os.walk(sdir): 88 dirs.sort() 89 for f in sorted(files): 90 if not f.lower().endswith(".md"): 91 continue 92 abspath = os.path.join(root, f) 93 rel = os.path.relpath(abspath, SRC).replace(os.sep, "/") # docs/<...>.md 94 inner = os.path.relpath(abspath, sdir).replace(os.sep, "/") 95 sslug = slugify(section) 96 # A README.md directly in the section dir is that section's index. 97 is_index = inner.lower() == "readme.md" 98 if is_index: 99 out_id = f"{sslug}/index" 100 route = f"/internal/{sslug}" 101 else: 102 seg = "/".join(slugify(p) for p in inner[:-3].split("/")) 103 out_id = f"{sslug}/{seg}" 104 route = f"/internal/{sslug}/{seg}" 105 route_of_md[rel.lower()] = route 106 routes.add(route) 107 pages.append((section, rel, abspath, is_index, out_id, route)) 108 109 # A docs-root page belongs to no section, so it takes the top of the collection. 110 for f in sorted(os.listdir(DOCS)): 111 if not f.lower().endswith(".md") or f.lower() in ROOT_SKIP: 112 continue 113 abspath = os.path.join(DOCS, f) 114 rel = f"docs/{f}" 115 slug = slugify(os.path.splitext(f)[0]) 116 route = f"/internal/{slug}" 117 route_of_md[rel.lower()] = route 118 routes.add(route) 119 root_pages.append((rel, abspath, slug, route)) 120 121 # --- Link rewriting (rules R1-R5) ---------------------------------------------- 122 LINK = re.compile(r"(!?)\[([^\]]*)\]\(([^)]+)\)") 123 FENCE = re.compile(r"^\s*(```|~~~)") 124 125 # R1: the mkdocs live site — every link here must come home to /internal. 126 LIVE_URL = re.compile( 127 r"^https?://swisskyrepo\.github\.io/InternalAllTheThings/(.*)$", re.I 128 ) 129 # R2: GitHub source URLs that point back into the repo (two hosts). 130 UPSTREAM_URL = re.compile( 131 r"^https?://(?:www\.)?github\.com/swisskyrepo/InternalAllTheThings/(?:blob|tree|raw)/[^/]+/(.*)$" 132 r"|^https?://raw\.githubusercontent\.com/swisskyrepo/InternalAllTheThings/[^/]+/(.*)$", 133 re.I, 134 ) 135 136 137 def route_for_live_path(path, anchor): 138 """R1: map a live-site <path> to an internal route, else the section index, 139 else None (leave the original URL untouched).""" 140 path = unquote(path).split("?", 1)[0].strip("/") 141 if path == "": 142 return "/internal" + anchor # the collection root 143 cand = "/internal/" + "/".join(slugify(p) for p in path.split("/")) 144 if cand in routes: 145 return cand + anchor 146 sec = f"/internal/{slugify(path.split('/')[0])}" # fall back to the section 147 if sec in routes: 148 return sec + anchor 149 return None 150 151 152 def route_for_repo_path(resolved, anchor, pin): 153 """Map a repo-relative path to an internal route or a pinned upstream URL. 154 `pin` picks the fallback when nothing is mirrored: absolute GitHub links 155 (R2) always pin to raw upstream; relative links (R3/R4) only rewrite a path 156 that actually exists in the checkout, otherwise return None (R5).""" 157 resolved = resolved.strip("/") 158 low = resolved.lower() 159 if low.endswith(".md"): 160 r = route_of_md.get(low) 161 if r: 162 return r + anchor 163 if pin or os.path.isfile(os.path.join(SRC, resolved)): 164 return raw_url(resolved) + anchor 165 return None 166 # mkdocs omits the .md extension when linking a sibling page. 167 r = route_of_md.get(low + ".md") 168 if r: 169 return r + anchor 170 idx = route_of_dir.get(low) or route_of_md.get(low + "/readme.md") 171 if idx: 172 return idx + anchor 173 if pin or os.path.isfile(os.path.join(SRC, resolved)): 174 return raw_url(resolved) + anchor # image / script / other asset 175 return None 176 177 178 def resolve(cur_relpath, target): 179 """Return the rewritten URL for a markdown link target, or None to keep it.""" 180 t = target.strip().strip("<>").strip() 181 anchor = "" 182 183 m = LIVE_URL.match(t) # R1 184 if m: 185 path = m.group(1) 186 if "#" in path: 187 path, a = path.split("#", 1); anchor = "#" + a 188 return route_for_live_path(path, anchor) 189 190 m = UPSTREAM_URL.match(t) # R2 191 if m: 192 path = (m.group(1) or m.group(2) or "").split("?", 1)[0] 193 if "#" in path: 194 path, a = path.split("#", 1); anchor = "#" + a 195 return route_for_repo_path(unquote(path), anchor, pin=True) 196 197 # R5: other hosts, protocol-relative, bare anchors, mailto: -> leave. 198 if re.match(r"^([a-z][a-z0-9+.-]*:|//|#)", t, re.I): 199 return None 200 201 # R3/R4: relative in-repo link, resolved against the current file's dir. 202 if "#" in t: 203 t, a = t.split("#", 1); anchor = "#" + a 204 if not t: 205 return None 206 cur_dir = os.path.dirname(cur_relpath) 207 resolved = os.path.normpath(os.path.join(cur_dir, unquote(t))).replace(os.sep, "/") 208 return route_for_repo_path(resolved, anchor, pin=False) 209 210 211 def rewrite_body(cur_relpath, body): 212 """Rewrite link targets outside fenced code blocks; code stays verbatim.""" 213 def repl(m): 214 new = resolve(cur_relpath, m.group(3)) 215 return m.group(0) if new is None else f"{m.group(1)}[{m.group(2)}]({new})" 216 217 out, fenced = [], False 218 for line in body.split("\n"): 219 if FENCE.match(line): 220 fenced = not fenced 221 out.append(line if fenced else LINK.sub(repl, line)) 222 return "\n".join(out) 223 224 225 def title_from(body, fallback): 226 m = re.search(r"^\s*#\s+(.+?)\s*$", body, re.M) 227 return m.group(1).strip() if m else fallback 228 229 230 def yaml_escape(s): 231 return s.replace("\\", "\\\\").replace('"', '\\"') 232 233 234 def frontmatter(title, section, sslug, source_path, source_url, is_index): 235 return ( 236 "---\n" 237 f'title: "{yaml_escape(title)}"\n' 238 f'section: "{yaml_escape(section)}"\n' 239 f'sectionSlug: "{sslug}"\n' 240 f'sourcePath: "{yaml_escape(source_path)}"\n' 241 f'sourceUrl: "{source_url}"\n' 242 f'sha: "{SHA}"\n' 243 f"isIndex: {'true' if is_index else 'false'}\n" 244 "---\n\n" 245 ) 246 247 248 # --- Emit ----------------------------------------------------------------------- 249 if os.path.isdir(OUT): 250 shutil.rmtree(OUT) 251 os.makedirs(OUT, exist_ok=True) 252 253 count = 0 254 section_seen = {} 255 sections_with_index = set() 256 section_pages = {} # section -> [(route, title)] for non-index pages 257 for section, rel, abspath, is_index, out_id, route in pages: 258 raw = open(abspath, encoding="utf-8", errors="replace").read() 259 body = rewrite_body(rel, raw) 260 stitle = section_title(section) 261 fallback = stitle if is_index else os.path.splitext(os.path.basename(abspath))[0] 262 title = title_from(raw, fallback) 263 if is_index: 264 sections_with_index.add(section) 265 else: 266 section_pages.setdefault(section, []).append((route, title)) 267 fm = frontmatter(title, stitle, slugify(section), rel, blob_url(rel), is_index) 268 dst = os.path.join(OUT, out_id + ".md") 269 os.makedirs(os.path.dirname(dst), exist_ok=True) 270 open(dst, "w", encoding="utf-8").write(fm + body.lstrip("\n")) 271 count += 1 272 section_seen[section] = section_seen.get(section, 0) + 1 273 274 for rel, abspath, slug, route in root_pages: 275 raw = open(abspath, encoding="utf-8", errors="replace").read() 276 body = rewrite_body(rel, raw) 277 stitle = section_title(slug) 278 title = title_from(raw, stitle) 279 fm = frontmatter(title, stitle, slug, rel, blob_url(rel), True) 280 open(os.path.join(OUT, slug + ".md"), "w", encoding="utf-8").write( 281 fm + body.lstrip("\n") 282 ) 283 count += 1 284 285 # --- Synthesize a section index for any section lacking a README.md ------------- 286 # Most IATT sections are a flat/nested set of pages with no README, so no 287 # `{slug}/index` root gets emitted and the collection index would link to a 404. 288 # Generate a listing page for each so its root route resolves + enumerates pages. 289 for section in sections: 290 if section in sections_with_index: 291 continue 292 sslug = slugify(section) 293 stitle = section_title(section) 294 pgs = sorted(section_pages.get(section, []), key=lambda p: (p[1].lower(), p[0])) 295 listing = "\n".join(f"* [{title}]({route})" for route, title in pgs) 296 fm = frontmatter( 297 stitle, stitle, sslug, f"docs/{section}", tree_url(f"docs/{section}"), True 298 ) 299 intro = ( 300 f"# {stitle}\n\n" 301 f"> {len(pgs)} pages in this section, mirrored from InternalAllTheThings.\n\n" 302 "## Pages\n\n" 303 ) 304 dst = os.path.join(OUT, sslug, "index.md") 305 os.makedirs(os.path.dirname(dst), exist_ok=True) 306 open(dst, "w", encoding="utf-8").write(fm + intro + listing + "\n") 307 count += 1 308 section_seen[section] = section_seen.get(section, 0) + 1 309 310 # --- Manifest ------------------------------------------------------------------- 311 manifest = { 312 "upstream": UPSTREAM, 313 "sha": SHA, 314 "sections": len(section_seen), 315 "pages": count, 316 "sectionList": sorted(section_seen.keys()), 317 } 318 open(os.path.join(REPO, "internal-manifest.json"), "w").write( 319 json.dumps(manifest, indent=2) + "\n" 320 ) 321 322 print(f"mirrored {count} pages across {len(section_seen)} sections @ {SHA}")