daemon-sec-cheatsheet

The cheatsheet vault for operators: AD, enumeration, exploitation, priv-esc, web, DFIR
git clone https://git.daemon-sec.xyz/daemon-sec-cheatsheet.git
Log | Files | Refs | README | LICENSE

sync-hacktricks.py (15868B)


      1 #!/usr/bin/env python3
      2 """Build the site's curated HackTricks collection from an upstream clone.
      3 
      4 The selected trees are broad enough to be useful while excluding marketing,
      5 privacy/attribution-evasion material, unfinished notes, and project metadata.
      6 Markdown is adapted for Astro: GitBook-only wrappers and training banners are
      7 removed, links are pinned or internalized, and images continue to load from the
      8 upstream repository. The source is CC BY-NC 4.0; attribution is rendered by the
      9 site on every generated page.
     10 
     11 Usage: python3 scripts/sync-hacktricks.py /path/to/hacktricks-clone <sha>
     12 """
     13 
     14 from __future__ import annotations
     15 
     16 import json
     17 import os
     18 import posixpath
     19 import re
     20 import shutil
     21 import sys
     22 from pathlib import Path
     23 from urllib.parse import quote, unquote
     24 
     25 
     26 REPO = Path(__file__).resolve().parent.parent
     27 OUT = REPO / "src" / "content" / "hacktricks"
     28 UPSTREAM = "https://github.com/HackTricks-wiki/hacktricks"
     29 
     30 SOURCE = Path(sys.argv[1]).resolve() if len(sys.argv) > 1 else None
     31 SHA = sys.argv[2] if len(sys.argv) > 2 else "master"
     32 if SOURCE is None or not (SOURCE / "src").is_dir():
     33     raise SystemExit("usage: sync-hacktricks.py <hacktricks-clone-dir> <sha>")
     34 
     35 SOURCE_ROOT = SOURCE / "src"
     36 
     37 # These five trees cover the material most useful in a field cheatsheet site.
     38 # Keeping the selection explicit prevents a future upstream directory from
     39 # silently becoming public here merely because it was added to the repository.
     40 SECTIONS = {
     41     "network-services-pentesting": "Network Services",
     42     "pentesting-web": "Web Pentesting",
     43     "windows-hardening": "Windows",
     44     "linux-hardening": "Linux",
     45     "mobile-pentesting": "Mobile",
     46 }
     47 
     48 SKIP_NAMES = {"license.md", "summary.md"}
     49 
     50 
     51 def slugify(value: str) -> str:
     52     value = value.strip().lower()
     53     value = re.sub(r"[^a-z0-9]+", "-", value)
     54     return re.sub(r"-+", "-", value).strip("-") or "x"
     55 
     56 
     57 def quoted_blob(path: str) -> str:
     58     return f"{UPSTREAM}/blob/{SHA}/{quote('src/' + path, safe='/')}"
     59 
     60 
     61 def quoted_raw(path: str) -> str:
     62     return (
     63         "https://raw.githubusercontent.com/HackTricks-wiki/hacktricks/"
     64         f"{SHA}/{quote('src/' + path, safe='/')}"
     65     )
     66 
     67 
     68 records: list[dict[str, object]] = []
     69 route_by_source: dict[str, str] = {}
     70 
     71 for section_slug, section_title in SECTIONS.items():
     72     root = SOURCE_ROOT / section_slug
     73     for source_file in sorted(root.rglob("*.md")):
     74         if source_file.name.lower() in SKIP_NAMES:
     75             continue
     76         source_path = source_file.relative_to(SOURCE_ROOT).as_posix()
     77         inner = source_file.relative_to(root).as_posix()
     78         parts = inner[:-3].split("/")
     79         is_index = parts[-1].lower() == "readme"
     80         if is_index:
     81             parts[-1] = "overview"
     82         output_parts = [slugify(part) for part in parts]
     83         output_id = "/".join([section_slug, *output_parts])
     84         route = f"/hacktricks/{output_id}"
     85         route_by_source[source_path.lower()] = route
     86         records.append(
     87             {
     88                 "section_slug": section_slug,
     89                 "section_title": section_title,
     90                 "source_file": source_file,
     91                 "source_path": source_path,
     92                 "output_id": output_id,
     93                 "is_index": is_index,
     94             }
     95         )
     96 
     97 
     98 MARKDOWN_LINK = re.compile(r"(!?)\[([^\]]*)\]\(([^)]+)\)")
     99 ANGLE_MARKDOWN_LINK = re.compile(r"(!?)\[([^\]]*)\]\(<([^>]+)>\)")
    100 HTML_ASSET = re.compile(r'(?P<prefix>\b(?:src|href)=["\'])(?P<url>[^"\']+)(?P<suffix>["\'])', re.I)
    101 UPSTREAM_GITHUB = re.compile(
    102     r"^https?://(?:www\.)?github\.com/(?:HackTricks-wiki/hacktricks|carlospolop/hacktricks)/"
    103     r"(?:blob|tree|raw)/[^/]+/(?:src/)?(.*)$",
    104     re.I,
    105 )
    106 
    107 
    108 def route_for_source(path: str, anchor: str = "") -> str | None:
    109     normalized = posixpath.normpath(path).lstrip("./")
    110     route = route_by_source.get(normalized.lower())
    111     return route + anchor if route else None
    112 
    113 
    114 def resolve_target(current_source: str, target: str, image: bool) -> str | None:
    115     target = target.strip().strip("<>")
    116     if not target or target.startswith("#"):
    117         return None
    118 
    119     anchor = ""
    120     if "#" in target:
    121         target, fragment = target.split("#", 1)
    122         anchor = f"#{fragment}"
    123 
    124     upstream_match = UPSTREAM_GITHUB.match(target)
    125     if upstream_match:
    126         path = unquote(upstream_match.group(1)).split("?", 1)[0]
    127         if path.lower().endswith(".md"):
    128             return route_for_source(path, anchor) or quoted_blob(path) + anchor
    129         return quoted_raw(path) + anchor
    130 
    131     if re.match(r"^(?:[a-z][a-z0-9+.-]*:|//|mailto:)", target, re.I):
    132         return None
    133 
    134     current_dir = posixpath.dirname(current_source)
    135     raw_target = unquote(target)
    136     # A bare directory reference (upstream {{#ref}} blocks often point at a
    137     # whole section, not a file) means "that section's index page".
    138     if raw_target.endswith("/"):
    139         raw_target += "README.md"
    140     resolved = posixpath.normpath(posixpath.join(current_dir, raw_target))
    141     if resolved.startswith("../"):
    142         return None
    143     if resolved.lower().endswith(".md"):
    144         return route_for_source(resolved, anchor) or quoted_blob(resolved) + anchor
    145     if image or re.search(r"\.(?:png|jpe?g|gif|webp|svg|pdf|zip|txt|py|sh|ps1)$", resolved, re.I):
    146         return quoted_raw(resolved) + anchor
    147     return None
    148 
    149 
    150 # Upstream has at least one typo'd close tag ({{/ref}} instead of
    151 # {{#endref}} — see ssti-server-side-template-injection/README.md around its
    152 # LESS section). Accepting that variant here matters: with only the correct
    153 # spelling recognized, the non-greedy match instead runs on to the *next*
    154 # real {{#endref}}, silently swallowing every heading and paragraph between
    155 # the two into a single garbled link. MAX_BLOCK_LINES below is the backstop
    156 # for whatever variant of this a future upstream edit introduces.
    157 REF_BLOCK = re.compile(r"^[ \t]*\{\{#ref\}\}[ \t]*\n(.*?)\n[ \t]*\{\{(?:#end|/)ref\}\}[ \t]*$", re.M | re.S)
    158 FILE_BLOCK = re.compile(r"^[ \t]*\{\{#file\}\}[ \t]*\n(.*?)\n[ \t]*\{\{(?:#end|/)file\}\}[ \t]*$", re.M | re.S | re.I)
    159 MAX_BLOCK_LINES = 8
    160 CONTENT_REF_BLOCK = re.compile(
    161     r'^[ \t]*\{%\s*content-ref\s+url=["\']([^"\']+)["\'][^%]*%\}[ \t]*\n(.*?)\n[ \t]*\{%\s*endcontent-ref\s*%\}[ \t]*$',
    162     re.M | re.S | re.I,
    163 )
    164 WRAPPING_LINK = re.compile(r"^\[([^\]]*)\]\(([^)]+)\)$")
    165 
    166 
    167 def humanize_ref_label(path: str) -> str:
    168     """"protections/no-new-privileges.md" -> "No New Privileges" — same
    169     dash-to-title-case fallback page_title() already uses for a page's own
    170     title, so a ref link reads like a page name rather than a bare path."""
    171     stem = path.rstrip("/").rsplit("/", 1)[-1]
    172     stem = re.sub(r"\.md$", "", stem, flags=re.I)
    173     return stem.replace("-", " ").replace("_", " ").title() or path
    174 
    175 
    176 def linkify_ref_blocks(body: str) -> str:
    177     """
    178     {{#ref}}/{{#file}} wrap a bare relative path with no markdown link syntax
    179     at all — upstream renders these as GitBook card links. Deleting the
    180     wrapper (as this script used to) left the bare path behind as plain,
    181     unclickable text throughout every synced page. Turning each path into a
    182     real `[label](target)` link here lets rewrite_links() below resolve it
    183     to the right on-site route exactly like an ordinary inline link.
    184     """
    185 
    186     def ref_repl(match: "re.Match[str]") -> str:
    187         lines = [line.strip() for line in match.group(1).splitlines() if line.strip()]
    188         # A legitimate {{#ref}} block is a short list of bare paths. Anything
    189         # longer, or containing a heading, is a sign the regex ran past its
    190         # real closing tag onto unrelated content — leave it untouched rather
    191         # than turn a chunk of the page into garbage nested links.
    192         if not lines or len(lines) > MAX_BLOCK_LINES or any(line.startswith("#") for line in lines):
    193             return match.group(0)
    194         # A raw filename can contain spaces or literal parentheses (seen in
    195         # an upstream PDF name); MARKDOWN_LINK's `([^)]+)` target group can't
    196         # tell those apart from the link's own closing paren, so quote the
    197         # target the same way quoted_blob()/quoted_raw() already do.
    198         return "\n\n".join(
    199             f"[{humanize_ref_label(line)}]({quote(line, safe='/')})" for line in lines
    200         )
    201 
    202     body = REF_BLOCK.sub(ref_repl, body)
    203     body = FILE_BLOCK.sub(ref_repl, body)
    204 
    205     def content_ref_repl(match: "re.Match[str]") -> str:
    206         target = match.group(1).strip()
    207         label = match.group(2).strip()
    208         # The label is sometimes already `[text](target)` — unwrap it rather
    209         # than nesting a link inside a link.
    210         wrapped = WRAPPING_LINK.match(label)
    211         if wrapped:
    212             label = wrapped.group(1)
    213         label = label or humanize_ref_label(target)
    214         return f"[{label}]({target})"
    215 
    216     return CONTENT_REF_BLOCK.sub(content_ref_repl, body)
    217 
    218 
    219 def clean_gitbook(body: str) -> str:
    220     body = linkify_ref_blocks(body)
    221     # The banner is repeated at the top and bottom of most upstream pages.
    222     body = re.sub(r"^\s*\{\{#include\s+[^}]*banners/hacktricks-training\.md\}\}\s*$", "", body, flags=re.M | re.I)
    223     # Other mdBook/GitBook includes cannot be expanded safely without copying
    224     # arbitrary non-selected files. Leave a source-facing note instead.
    225     body = re.sub(
    226         r"^\s*\{\{#include\s+([^}]+)\}\}\s*$",
    227         r"> Included upstream fragment: `\1` (see source link).",
    228         body,
    229         flags=re.M,
    230     )
    231     # Newer mdBook-style tab blocks use double braces instead of Liquid.
    232     body = re.sub(
    233         r'^\s*\{\{#tab\s+name=["\']([^"\']+)["\'][^}]*\}\}\s*$',
    234         r"### \1",
    235         body,
    236         flags=re.M | re.I,
    237     )
    238     body = re.sub(r"^\s*\{\{#(?:end)?tabs?[^}]*\}\}\s*$", "", body, flags=re.M | re.I)
    239     # Keep the useful content inside GitBook wrappers while dropping syntax
    240     # Astro would otherwise display as literal template tags.
    241     body = re.sub(r"^\s*\{%\s*tab\s+title=[\"']([^\"']+)[\"'][^%]*%\}\s*$", r"### \1", body, flags=re.M | re.I)
    242     body = re.sub(r"^\s*\{%\s*(?:end)?(?:hint|tabs|tab|code|endcode|embed)[^%]*%\}\s*$", "", body, flags=re.M | re.I)
    243     body = re.sub(r"\n{4,}", "\n\n\n", body)
    244     return body.strip() + "\n"
    245 
    246 
    247 def rewrite_links(source_path: str, body: str) -> str:
    248     def markdown_replacement(match: re.Match[str]) -> str:
    249         bang, label, target = match.groups()
    250         replacement = resolve_target(source_path, target, image=bool(bang))
    251         if replacement is None:
    252             return match.group(0)
    253         return f"{bang}[{label}]({replacement})"
    254 
    255     body = ANGLE_MARKDOWN_LINK.sub(markdown_replacement, body)
    256     body = MARKDOWN_LINK.sub(markdown_replacement, body)
    257 
    258     def html_replacement(match: re.Match[str]) -> str:
    259         target = match.group("url")
    260         replacement = resolve_target(source_path, target, image=True)
    261         if replacement is None:
    262             return match.group(0)
    263         return f'{match.group("prefix")}{replacement}{match.group("suffix")}'
    264 
    265     return HTML_ASSET.sub(html_replacement, body)
    266 
    267 
    268 SUPPORTED_FENCES = {
    269     "bash", "batch", "c", "cpp", "csharp", "css", "diff", "dockerfile",
    270     "go", "graphql", "html", "http", "ini", "java", "javascript", "json",
    271     "jsx", "kotlin", "lua", "makefile", "markdown", "nginx", "perl", "php",
    272     "plaintext", "powershell", "python", "regex", "ruby", "rust", "sql",
    273     "swift", "text", "toml", "tsx", "typescript", "xml", "yaml",
    274 }
    275 
    276 
    277 # A CommonMark fence is 3+ backticks. Longer runs exist upstream specifically
    278 # so the block can contain a literal ``` in its own content (e.g. an example
    279 # that itself shows markdown syntax) without that line closing the block
    280 # early — see mssql-injection.md, which wraps a ```sql-quoting example in a
    281 # 4-backtick fence. A closer must be a bare run of backticks at least as long
    282 # as its opener; anything shorter, or followed by other text, is just content.
    283 CANDIDATE_FENCE_LINE = re.compile(r"^(`{3,})([^`]*)$")
    284 BARE_FENCE_LINE = re.compile(r"^(`{3,})\s*$")
    285 
    286 
    287 def normalize_code_fences(body: str) -> str:
    288     """
    289     Drop GitBook filename annotations and unsupported Shiki language ids —
    290     on OPENING fences only.
    291 
    292     A closing fence (bare ```) is syntactically identical to an opening
    293     fence with an empty info string, and a naive regex substitution matches
    294     both alike. An earlier version of this function treated every match the
    295     same way: empty language -> not in SUPPORTED_FENCES -> default to
    296     "text". That silently turned every *closing* ``` into ```text, which a
    297     markdown parser reads as a new OPENING fence rather than a close. The
    298     real closer never arrives, so the block swallows everything until the
    299     next accidental ```text, and every fence after that is shifted the same
    300     way for the rest of the file. Tracking open/close state (and the
    301     opening fence's exact backtick count, for the 4-backtick case above) is
    302     what CommonMark expects and what Astro's renderer needs.
    303     """
    304     aliases = {
    305         "bat": "batch", "cmd": "batch", "console": "text", "cs": "csharp",
    306         "js": "javascript", "ps1": "powershell", "py": "python", "sh": "bash",
    307         "shell": "bash", "yml": "yaml",
    308     }
    309 
    310     def opening_replacement(ticks: str, info: str) -> str:
    311         raw = info.strip()
    312         head = raw.split(":", 1)[0].strip() if raw else ""
    313         tokens = head.split()
    314         language = tokens[0].lower() if tokens else ""
    315         language = aliases.get(language, language)
    316         if language not in SUPPORTED_FENCES:
    317             language = "text"
    318         return f"{ticks}{language}" if language else ticks
    319 
    320     lines = body.split("\n")
    321     in_fence = False
    322     fence_len = 0
    323     for i, line in enumerate(lines):
    324         if in_fence:
    325             # Only a bare run of at least as many backticks as the opener
    326             # closes a fence (CommonMark); anything shorter, or followed by
    327             # other text, is fence *content* and must be left untouched.
    328             bare = BARE_FENCE_LINE.match(line)
    329             if bare and len(bare.group(1)) >= fence_len:
    330                 lines[i] = "`" * fence_len
    331                 in_fence = False
    332             continue
    333         match = CANDIDATE_FENCE_LINE.match(line)
    334         if match:
    335             fence_len = len(match.group(1))
    336             lines[i] = opening_replacement(match.group(1), match.group(2))
    337             in_fence = True
    338     return "\n".join(lines)
    339 
    340 
    341 def page_title(body: str, fallback: str) -> str:
    342     match = re.search(r"^\s*#\s+(.+?)\s*$", body, re.M)
    343     if not match:
    344         return fallback
    345     title = re.sub(r"[*_`]+", "", match.group(1)).strip()
    346     return title or fallback
    347 
    348 
    349 if OUT.exists():
    350     shutil.rmtree(OUT)
    351 OUT.mkdir(parents=True)
    352 
    353 for record in records:
    354     source_file = record["source_file"]
    355     assert isinstance(source_file, Path)
    356     source_path = str(record["source_path"])
    357     raw = source_file.read_text(encoding="utf-8", errors="replace")
    358     body = normalize_code_fences(rewrite_links(source_path, clean_gitbook(raw)))
    359     fallback = source_file.parent.name if source_file.name.lower() == "readme.md" else source_file.stem
    360     title = page_title(raw, fallback.replace("-", " ").title())
    361 
    362     frontmatter = {
    363         "title": title,
    364         "section": record["section_title"],
    365         "sectionSlug": record["section_slug"],
    366         "sourcePath": f"src/{source_path}",
    367         "sourceUrl": quoted_blob(source_path),
    368         "sha": SHA,
    369         "isIndex": bool(record["is_index"]),
    370         "modified": True,
    371         "license": "CC-BY-NC-4.0",
    372     }
    373     output_path = OUT / f'{record["output_id"]}.md'
    374     output_path.parent.mkdir(parents=True, exist_ok=True)
    375     lines = ["---"]
    376     for key, value in frontmatter.items():
    377         lines.append(f"{key}: {json.dumps(value, ensure_ascii=False)}")
    378     lines.extend(["---", "", body])
    379     output_path.write_text("\n".join(lines), encoding="utf-8")
    380 
    381 print(f"synced {len(records)} HackTricks pages from {SHA}")