sync-hacktricks.py (15868B)
1 #!/usr/bin/env python3 2 """Build the site's curated HackTricks collection from an upstream clone. 3 4 The selected trees are broad enough to be useful while excluding marketing, 5 privacy/attribution-evasion material, unfinished notes, and project metadata. 6 Markdown is adapted for Astro: GitBook-only wrappers and training banners are 7 removed, links are pinned or internalized, and images continue to load from the 8 upstream repository. The source is CC BY-NC 4.0; attribution is rendered by the 9 site on every generated page. 10 11 Usage: python3 scripts/sync-hacktricks.py /path/to/hacktricks-clone <sha> 12 """ 13 14 from __future__ import annotations 15 16 import json 17 import os 18 import posixpath 19 import re 20 import shutil 21 import sys 22 from pathlib import Path 23 from urllib.parse import quote, unquote 24 25 26 REPO = Path(__file__).resolve().parent.parent 27 OUT = REPO / "src" / "content" / "hacktricks" 28 UPSTREAM = "https://github.com/HackTricks-wiki/hacktricks" 29 30 SOURCE = Path(sys.argv[1]).resolve() if len(sys.argv) > 1 else None 31 SHA = sys.argv[2] if len(sys.argv) > 2 else "master" 32 if SOURCE is None or not (SOURCE / "src").is_dir(): 33 raise SystemExit("usage: sync-hacktricks.py <hacktricks-clone-dir> <sha>") 34 35 SOURCE_ROOT = SOURCE / "src" 36 37 # These five trees cover the material most useful in a field cheatsheet site. 38 # Keeping the selection explicit prevents a future upstream directory from 39 # silently becoming public here merely because it was added to the repository. 40 SECTIONS = { 41 "network-services-pentesting": "Network Services", 42 "pentesting-web": "Web Pentesting", 43 "windows-hardening": "Windows", 44 "linux-hardening": "Linux", 45 "mobile-pentesting": "Mobile", 46 } 47 48 SKIP_NAMES = {"license.md", "summary.md"} 49 50 51 def slugify(value: str) -> str: 52 value = value.strip().lower() 53 value = re.sub(r"[^a-z0-9]+", "-", value) 54 return re.sub(r"-+", "-", value).strip("-") or "x" 55 56 57 def quoted_blob(path: str) -> str: 58 return f"{UPSTREAM}/blob/{SHA}/{quote('src/' + path, safe='/')}" 59 60 61 def quoted_raw(path: str) -> str: 62 return ( 63 "https://raw.githubusercontent.com/HackTricks-wiki/hacktricks/" 64 f"{SHA}/{quote('src/' + path, safe='/')}" 65 ) 66 67 68 records: list[dict[str, object]] = [] 69 route_by_source: dict[str, str] = {} 70 71 for section_slug, section_title in SECTIONS.items(): 72 root = SOURCE_ROOT / section_slug 73 for source_file in sorted(root.rglob("*.md")): 74 if source_file.name.lower() in SKIP_NAMES: 75 continue 76 source_path = source_file.relative_to(SOURCE_ROOT).as_posix() 77 inner = source_file.relative_to(root).as_posix() 78 parts = inner[:-3].split("/") 79 is_index = parts[-1].lower() == "readme" 80 if is_index: 81 parts[-1] = "overview" 82 output_parts = [slugify(part) for part in parts] 83 output_id = "/".join([section_slug, *output_parts]) 84 route = f"/hacktricks/{output_id}" 85 route_by_source[source_path.lower()] = route 86 records.append( 87 { 88 "section_slug": section_slug, 89 "section_title": section_title, 90 "source_file": source_file, 91 "source_path": source_path, 92 "output_id": output_id, 93 "is_index": is_index, 94 } 95 ) 96 97 98 MARKDOWN_LINK = re.compile(r"(!?)\[([^\]]*)\]\(([^)]+)\)") 99 ANGLE_MARKDOWN_LINK = re.compile(r"(!?)\[([^\]]*)\]\(<([^>]+)>\)") 100 HTML_ASSET = re.compile(r'(?P<prefix>\b(?:src|href)=["\'])(?P<url>[^"\']+)(?P<suffix>["\'])', re.I) 101 UPSTREAM_GITHUB = re.compile( 102 r"^https?://(?:www\.)?github\.com/(?:HackTricks-wiki/hacktricks|carlospolop/hacktricks)/" 103 r"(?:blob|tree|raw)/[^/]+/(?:src/)?(.*)$", 104 re.I, 105 ) 106 107 108 def route_for_source(path: str, anchor: str = "") -> str | None: 109 normalized = posixpath.normpath(path).lstrip("./") 110 route = route_by_source.get(normalized.lower()) 111 return route + anchor if route else None 112 113 114 def resolve_target(current_source: str, target: str, image: bool) -> str | None: 115 target = target.strip().strip("<>") 116 if not target or target.startswith("#"): 117 return None 118 119 anchor = "" 120 if "#" in target: 121 target, fragment = target.split("#", 1) 122 anchor = f"#{fragment}" 123 124 upstream_match = UPSTREAM_GITHUB.match(target) 125 if upstream_match: 126 path = unquote(upstream_match.group(1)).split("?", 1)[0] 127 if path.lower().endswith(".md"): 128 return route_for_source(path, anchor) or quoted_blob(path) + anchor 129 return quoted_raw(path) + anchor 130 131 if re.match(r"^(?:[a-z][a-z0-9+.-]*:|//|mailto:)", target, re.I): 132 return None 133 134 current_dir = posixpath.dirname(current_source) 135 raw_target = unquote(target) 136 # A bare directory reference (upstream {{#ref}} blocks often point at a 137 # whole section, not a file) means "that section's index page". 138 if raw_target.endswith("/"): 139 raw_target += "README.md" 140 resolved = posixpath.normpath(posixpath.join(current_dir, raw_target)) 141 if resolved.startswith("../"): 142 return None 143 if resolved.lower().endswith(".md"): 144 return route_for_source(resolved, anchor) or quoted_blob(resolved) + anchor 145 if image or re.search(r"\.(?:png|jpe?g|gif|webp|svg|pdf|zip|txt|py|sh|ps1)$", resolved, re.I): 146 return quoted_raw(resolved) + anchor 147 return None 148 149 150 # Upstream has at least one typo'd close tag ({{/ref}} instead of 151 # {{#endref}} — see ssti-server-side-template-injection/README.md around its 152 # LESS section). Accepting that variant here matters: with only the correct 153 # spelling recognized, the non-greedy match instead runs on to the *next* 154 # real {{#endref}}, silently swallowing every heading and paragraph between 155 # the two into a single garbled link. MAX_BLOCK_LINES below is the backstop 156 # for whatever variant of this a future upstream edit introduces. 157 REF_BLOCK = re.compile(r"^[ \t]*\{\{#ref\}\}[ \t]*\n(.*?)\n[ \t]*\{\{(?:#end|/)ref\}\}[ \t]*$", re.M | re.S) 158 FILE_BLOCK = re.compile(r"^[ \t]*\{\{#file\}\}[ \t]*\n(.*?)\n[ \t]*\{\{(?:#end|/)file\}\}[ \t]*$", re.M | re.S | re.I) 159 MAX_BLOCK_LINES = 8 160 CONTENT_REF_BLOCK = re.compile( 161 r'^[ \t]*\{%\s*content-ref\s+url=["\']([^"\']+)["\'][^%]*%\}[ \t]*\n(.*?)\n[ \t]*\{%\s*endcontent-ref\s*%\}[ \t]*$', 162 re.M | re.S | re.I, 163 ) 164 WRAPPING_LINK = re.compile(r"^\[([^\]]*)\]\(([^)]+)\)$") 165 166 167 def humanize_ref_label(path: str) -> str: 168 """"protections/no-new-privileges.md" -> "No New Privileges" — same 169 dash-to-title-case fallback page_title() already uses for a page's own 170 title, so a ref link reads like a page name rather than a bare path.""" 171 stem = path.rstrip("/").rsplit("/", 1)[-1] 172 stem = re.sub(r"\.md$", "", stem, flags=re.I) 173 return stem.replace("-", " ").replace("_", " ").title() or path 174 175 176 def linkify_ref_blocks(body: str) -> str: 177 """ 178 {{#ref}}/{{#file}} wrap a bare relative path with no markdown link syntax 179 at all — upstream renders these as GitBook card links. Deleting the 180 wrapper (as this script used to) left the bare path behind as plain, 181 unclickable text throughout every synced page. Turning each path into a 182 real `[label](target)` link here lets rewrite_links() below resolve it 183 to the right on-site route exactly like an ordinary inline link. 184 """ 185 186 def ref_repl(match: "re.Match[str]") -> str: 187 lines = [line.strip() for line in match.group(1).splitlines() if line.strip()] 188 # A legitimate {{#ref}} block is a short list of bare paths. Anything 189 # longer, or containing a heading, is a sign the regex ran past its 190 # real closing tag onto unrelated content — leave it untouched rather 191 # than turn a chunk of the page into garbage nested links. 192 if not lines or len(lines) > MAX_BLOCK_LINES or any(line.startswith("#") for line in lines): 193 return match.group(0) 194 # A raw filename can contain spaces or literal parentheses (seen in 195 # an upstream PDF name); MARKDOWN_LINK's `([^)]+)` target group can't 196 # tell those apart from the link's own closing paren, so quote the 197 # target the same way quoted_blob()/quoted_raw() already do. 198 return "\n\n".join( 199 f"[{humanize_ref_label(line)}]({quote(line, safe='/')})" for line in lines 200 ) 201 202 body = REF_BLOCK.sub(ref_repl, body) 203 body = FILE_BLOCK.sub(ref_repl, body) 204 205 def content_ref_repl(match: "re.Match[str]") -> str: 206 target = match.group(1).strip() 207 label = match.group(2).strip() 208 # The label is sometimes already `[text](target)` — unwrap it rather 209 # than nesting a link inside a link. 210 wrapped = WRAPPING_LINK.match(label) 211 if wrapped: 212 label = wrapped.group(1) 213 label = label or humanize_ref_label(target) 214 return f"[{label}]({target})" 215 216 return CONTENT_REF_BLOCK.sub(content_ref_repl, body) 217 218 219 def clean_gitbook(body: str) -> str: 220 body = linkify_ref_blocks(body) 221 # The banner is repeated at the top and bottom of most upstream pages. 222 body = re.sub(r"^\s*\{\{#include\s+[^}]*banners/hacktricks-training\.md\}\}\s*$", "", body, flags=re.M | re.I) 223 # Other mdBook/GitBook includes cannot be expanded safely without copying 224 # arbitrary non-selected files. Leave a source-facing note instead. 225 body = re.sub( 226 r"^\s*\{\{#include\s+([^}]+)\}\}\s*$", 227 r"> Included upstream fragment: `\1` (see source link).", 228 body, 229 flags=re.M, 230 ) 231 # Newer mdBook-style tab blocks use double braces instead of Liquid. 232 body = re.sub( 233 r'^\s*\{\{#tab\s+name=["\']([^"\']+)["\'][^}]*\}\}\s*$', 234 r"### \1", 235 body, 236 flags=re.M | re.I, 237 ) 238 body = re.sub(r"^\s*\{\{#(?:end)?tabs?[^}]*\}\}\s*$", "", body, flags=re.M | re.I) 239 # Keep the useful content inside GitBook wrappers while dropping syntax 240 # Astro would otherwise display as literal template tags. 241 body = re.sub(r"^\s*\{%\s*tab\s+title=[\"']([^\"']+)[\"'][^%]*%\}\s*$", r"### \1", body, flags=re.M | re.I) 242 body = re.sub(r"^\s*\{%\s*(?:end)?(?:hint|tabs|tab|code|endcode|embed)[^%]*%\}\s*$", "", body, flags=re.M | re.I) 243 body = re.sub(r"\n{4,}", "\n\n\n", body) 244 return body.strip() + "\n" 245 246 247 def rewrite_links(source_path: str, body: str) -> str: 248 def markdown_replacement(match: re.Match[str]) -> str: 249 bang, label, target = match.groups() 250 replacement = resolve_target(source_path, target, image=bool(bang)) 251 if replacement is None: 252 return match.group(0) 253 return f"{bang}[{label}]({replacement})" 254 255 body = ANGLE_MARKDOWN_LINK.sub(markdown_replacement, body) 256 body = MARKDOWN_LINK.sub(markdown_replacement, body) 257 258 def html_replacement(match: re.Match[str]) -> str: 259 target = match.group("url") 260 replacement = resolve_target(source_path, target, image=True) 261 if replacement is None: 262 return match.group(0) 263 return f'{match.group("prefix")}{replacement}{match.group("suffix")}' 264 265 return HTML_ASSET.sub(html_replacement, body) 266 267 268 SUPPORTED_FENCES = { 269 "bash", "batch", "c", "cpp", "csharp", "css", "diff", "dockerfile", 270 "go", "graphql", "html", "http", "ini", "java", "javascript", "json", 271 "jsx", "kotlin", "lua", "makefile", "markdown", "nginx", "perl", "php", 272 "plaintext", "powershell", "python", "regex", "ruby", "rust", "sql", 273 "swift", "text", "toml", "tsx", "typescript", "xml", "yaml", 274 } 275 276 277 # A CommonMark fence is 3+ backticks. Longer runs exist upstream specifically 278 # so the block can contain a literal ``` in its own content (e.g. an example 279 # that itself shows markdown syntax) without that line closing the block 280 # early — see mssql-injection.md, which wraps a ```sql-quoting example in a 281 # 4-backtick fence. A closer must be a bare run of backticks at least as long 282 # as its opener; anything shorter, or followed by other text, is just content. 283 CANDIDATE_FENCE_LINE = re.compile(r"^(`{3,})([^`]*)$") 284 BARE_FENCE_LINE = re.compile(r"^(`{3,})\s*$") 285 286 287 def normalize_code_fences(body: str) -> str: 288 """ 289 Drop GitBook filename annotations and unsupported Shiki language ids — 290 on OPENING fences only. 291 292 A closing fence (bare ```) is syntactically identical to an opening 293 fence with an empty info string, and a naive regex substitution matches 294 both alike. An earlier version of this function treated every match the 295 same way: empty language -> not in SUPPORTED_FENCES -> default to 296 "text". That silently turned every *closing* ``` into ```text, which a 297 markdown parser reads as a new OPENING fence rather than a close. The 298 real closer never arrives, so the block swallows everything until the 299 next accidental ```text, and every fence after that is shifted the same 300 way for the rest of the file. Tracking open/close state (and the 301 opening fence's exact backtick count, for the 4-backtick case above) is 302 what CommonMark expects and what Astro's renderer needs. 303 """ 304 aliases = { 305 "bat": "batch", "cmd": "batch", "console": "text", "cs": "csharp", 306 "js": "javascript", "ps1": "powershell", "py": "python", "sh": "bash", 307 "shell": "bash", "yml": "yaml", 308 } 309 310 def opening_replacement(ticks: str, info: str) -> str: 311 raw = info.strip() 312 head = raw.split(":", 1)[0].strip() if raw else "" 313 tokens = head.split() 314 language = tokens[0].lower() if tokens else "" 315 language = aliases.get(language, language) 316 if language not in SUPPORTED_FENCES: 317 language = "text" 318 return f"{ticks}{language}" if language else ticks 319 320 lines = body.split("\n") 321 in_fence = False 322 fence_len = 0 323 for i, line in enumerate(lines): 324 if in_fence: 325 # Only a bare run of at least as many backticks as the opener 326 # closes a fence (CommonMark); anything shorter, or followed by 327 # other text, is fence *content* and must be left untouched. 328 bare = BARE_FENCE_LINE.match(line) 329 if bare and len(bare.group(1)) >= fence_len: 330 lines[i] = "`" * fence_len 331 in_fence = False 332 continue 333 match = CANDIDATE_FENCE_LINE.match(line) 334 if match: 335 fence_len = len(match.group(1)) 336 lines[i] = opening_replacement(match.group(1), match.group(2)) 337 in_fence = True 338 return "\n".join(lines) 339 340 341 def page_title(body: str, fallback: str) -> str: 342 match = re.search(r"^\s*#\s+(.+?)\s*$", body, re.M) 343 if not match: 344 return fallback 345 title = re.sub(r"[*_`]+", "", match.group(1)).strip() 346 return title or fallback 347 348 349 if OUT.exists(): 350 shutil.rmtree(OUT) 351 OUT.mkdir(parents=True) 352 353 for record in records: 354 source_file = record["source_file"] 355 assert isinstance(source_file, Path) 356 source_path = str(record["source_path"]) 357 raw = source_file.read_text(encoding="utf-8", errors="replace") 358 body = normalize_code_fences(rewrite_links(source_path, clean_gitbook(raw))) 359 fallback = source_file.parent.name if source_file.name.lower() == "readme.md" else source_file.stem 360 title = page_title(raw, fallback.replace("-", " ").title()) 361 362 frontmatter = { 363 "title": title, 364 "section": record["section_title"], 365 "sectionSlug": record["section_slug"], 366 "sourcePath": f"src/{source_path}", 367 "sourceUrl": quoted_blob(source_path), 368 "sha": SHA, 369 "isIndex": bool(record["is_index"]), 370 "modified": True, 371 "license": "CC-BY-NC-4.0", 372 } 373 output_path = OUT / f'{record["output_id"]}.md' 374 output_path.parent.mkdir(parents=True, exist_ok=True) 375 lines = ["---"] 376 for key, value in frontmatter.items(): 377 lines.append(f"{key}: {json.dumps(value, ensure_ascii=False)}") 378 lines.extend(["---", "", body]) 379 output_path.write_text("\n".join(lines), encoding="utf-8") 380 381 print(f"synced {len(records)} HackTricks pages from {SHA}")