fetch-ired.py (10650B)
1 #!/usr/bin/env python3 2 """Read-only research helper for ired.team (RedTeaming-Tactics-and-Techniques). 3 4 This is deliberately NOT a `sync-*` script. The sync scripts generate content 5 collections; this one writes nothing into `src/`. ired.team publishes no licence 6 (the GitHub API returns `"license": null`, there is no LICENSE file, `/license` 7 404s), so there is no grant of rights to copy its prose — the same situation 8 `docs/provenance-audit.md` adjudicated for awesome-nmap-grep, where it recorded 9 that credit alone does not cure a missing licence. 10 11 So this script only makes the upstream cheap to *read*, and makes image URLs 12 correct. Sheets are then written by hand in the house voice, and their 13 screenshots hotlink to the pinned commit rather than being redistributed. 14 15 ./scripts/fetch-ired.py --tree 16 ./scripts/fetch-ired.py --get offensive-security/persistence/t1015-sethc.md 17 ./scripts/fetch-ired.py --images offensive-security/persistence/t1015-sethc.md 18 ./scripts/fetch-ired.py --refresh-sha 19 20 Guide paths may be given with or without the `.md` suffix. 21 """ 22 from __future__ import annotations 23 24 import argparse 25 import json 26 import os 27 import re 28 import sys 29 import tempfile 30 import urllib.error 31 import urllib.request 32 from urllib.parse import quote, unquote 33 34 ROOT = os.path.normpath(os.path.join(os.path.dirname(os.path.abspath(__file__)), "..")) 35 SOURCE_JSON = os.path.join(ROOT, "src", "data", "ired-source.json") 36 # Scratch belongs outside the working copy. An earlier default of 37 # `<repo>/.ired-cache` meant every helper run dropped research markdown into 38 # `jj status`, which is noise at best and an accidental commit at worst. 39 DEFAULT_OUT = ( 40 os.environ.get("CLAUDE_SCRATCHPAD") 41 or os.path.join(tempfile.gettempdir(), "ired-cache") 42 ) 43 44 45 def load_source() -> dict: 46 with open(SOURCE_JSON, encoding="utf-8") as fh: 47 return json.load(fh) 48 49 50 def get(url: str) -> bytes: 51 req = urllib.request.Request(url, headers={"User-Agent": "daemon-sec-cheatsheet/fetch-ired"}) 52 with urllib.request.urlopen(req, timeout=30) as resp: 53 return resp.read() 54 55 56 def raw_url(src: dict, path: str) -> str: 57 """Pinned raw URL for an upstream path. 58 59 Every segment is percent-encoded. Upstream asset names are GitBook exports 60 like `Screenshot from 2019-04-28 16-28-59.png` and routinely carry spaces, 61 parentheses and the odd `+`, so hand-writing these URLs reliably produces 62 a silently broken image. 63 64 Encoding is idempotent: some guides carry an already-encoded target 65 (`image%20%28123%29.png`), and quoting that again yields `%2520%2528…`, 66 which 404s. Unquoting first means a raw name and an encoded name both 67 arrive at the same URL. 68 """ 69 encoded = "/".join(quote(unquote(seg), safe="") for seg in path.split("/")) 70 return f"https://raw.githubusercontent.com/{src['repo']}/{src['sha']}/{encoded}" 71 72 73 def site_url(src: dict, path: str) -> str: 74 """The human-facing ired.team URL for a guide, for the credit line.""" 75 slug = re.sub(r"\.md$", "", path) 76 slug = re.sub(r"/README$", "", slug) 77 return f"{src['site']}/{slug}" 78 79 80 def normalise(path: str) -> str: 81 return path if path.endswith(".md") else path + ".md" 82 83 84 # ---- tree ---------------------------------------------------------------- 85 86 def fetch_tree(src: dict) -> list[str]: 87 url = f"https://api.github.com/repos/{src['repo']}/git/trees/{src['sha']}?recursive=1" 88 data = json.loads(get(url)) 89 if data.get("truncated"): 90 print("warning: upstream tree response was truncated", file=sys.stderr) 91 return [t["path"] for t in data["tree"] if t["path"].endswith(".md")] 92 93 94 def cmd_tree(src: dict, grouped: bool) -> None: 95 paths = sorted(fetch_tree(src)) 96 if not grouped: 97 print("\n".join(paths)) 98 return 99 clusters: dict[str, list[str]] = {} 100 for p in paths: 101 parts = p.split("/") 102 key = "/".join(parts[:2]) if len(parts) > 2 else (parts[0] if len(parts) > 1 else "(root)") 103 clusters.setdefault(key, []).append(p) 104 for key in sorted(clusters): 105 print(f"\n## {key} ({len(clusters[key])})") 106 for p in clusters[key]: 107 print(f" {p}") 108 109 110 # ---- guide text ---------------------------------------------------------- 111 112 GITBOOK_HINT = re.compile(r"\{%\s*hint\s+style=\"?(\w+)\"?\s*%\}(.*?)\{%\s*endhint\s*%\}", re.S) 113 GITBOOK_CONTENT_REF = re.compile(r"\{%\s*content-ref\s+url=\"([^\"]+)\"\s*%\}(.*?)\{%\s*endcontent-ref\s*%\}", re.S) 114 GITBOOK_EMBED = re.compile(r"\{%\s*embed\s+url=\"([^\"]+)\"\s*%\}(?:\s*\{%\s*endembed\s*%\})?", re.S) 115 GITBOOK_CODE = re.compile(r"\{%\s*(?:end)?code[^%]*%\}") 116 GITBOOK_TABS = re.compile(r"\{%\s*(?:end)?tabs?[^%]*%\}") 117 GITBOOK_FILE = re.compile(r"\{%\s*file\s+src=\"([^\"]+)\"\s*%\}") 118 GITBOOK_LEFTOVER = re.compile(r"\{%.*?%\}", re.S) 119 120 121 def degitbook(text: str) -> str: 122 """Translate GitBook macros into plain markdown so the guide is readable. 123 124 Nothing here is about producing publishable text — it is about not having 125 to mentally parse `{% hint %}` blocks while reading 40 guides. 126 """ 127 text = GITBOOK_HINT.sub(lambda m: f"\n> **{m.group(1).upper()}:** {m.group(2).strip()}\n", text) 128 text = GITBOOK_CONTENT_REF.sub(lambda m: f"\n→ see `{m.group(1)}`\n", text) 129 text = GITBOOK_EMBED.sub(lambda m: f"\n→ {m.group(1)}\n", text) 130 text = GITBOOK_FILE.sub(lambda m: f"\n→ attached file: `{m.group(1)}`\n", text) 131 text = GITBOOK_CODE.sub("", text) 132 text = GITBOOK_TABS.sub("", text) 133 text = GITBOOK_LEFTOVER.sub("", text) 134 # GitBook escapes underscores in prose; unescape so identifiers read right. 135 text = text.replace("\\_", "_") 136 return re.sub(r"\n{3,}", "\n\n", text) 137 138 139 def cmd_get(src: dict, paths: list[str], out_dir: str) -> None: 140 os.makedirs(out_dir, exist_ok=True) 141 for path in paths: 142 path = normalise(path) 143 try: 144 body = get(raw_url(src, path)).decode("utf-8", "replace") 145 except urllib.error.HTTPError as exc: 146 print(f"!! {path}: HTTP {exc.code}", file=sys.stderr) 147 continue 148 dest = os.path.join(out_dir, path.replace("/", "__")) 149 header = ( 150 f"<!-- upstream: {path}\n" 151 f" site: {site_url(src, path)}\n" 152 f" commit: {src['sha']}\n" 153 f" licence: NONE PUBLISHED — do not copy this prose. Rewrite. -->\n\n" 154 ) 155 with open(dest, "w", encoding="utf-8") as fh: 156 fh.write(header + degitbook(body)) 157 print(dest) 158 159 160 # ---- images -------------------------------------------------------------- 161 162 # Markdown images, including GitBook's angle-bracket form for paths with spaces: 163 #  164 # 165 # The angle-bracket form needs its own branch rather than an optional `<`. A 166 # single branch that stops at `)` truncates `image (609).png` to `image (609` 167 # — the asset name's own closing paren is read as the markdown's — and that is 168 # the commonest name in the newer guides, so the emitted URL 404s silently. 169 # Inside `<…>` the delimiter is `>`, so parens in the name are just characters. 170 IMG = re.compile( 171 r"!\[([^\]]*)\]\(\s*(?:<([^>]+)>|([^)\s]+))\s*\)" 172 ) 173 174 175 def resolve(guide_path: str, target: str) -> str: 176 """Resolve a guide-relative image target to a repo-root-relative path.""" 177 if target.startswith(("http://", "https://")): 178 return target 179 base = os.path.dirname(guide_path) 180 return os.path.normpath(os.path.join(base, target)).replace(os.sep, "/") 181 182 183 def cmd_images(src: dict, paths: list[str]) -> None: 184 for path in paths: 185 path = normalise(path) 186 try: 187 body = get(raw_url(src, path)).decode("utf-8", "replace") 188 except urllib.error.HTTPError as exc: 189 print(f"!! {path}: HTTP {exc.code}", file=sys.stderr) 190 continue 191 # Two target branches in IMG, so collapse them: exactly one matches. 192 matches = [(alt, bracketed or bare) for alt, bracketed, bare in IMG.findall(body)] 193 print(f"\n## {path} ({len(matches)} images)") 194 print(f" credit: ired.team · {src['author']}") 195 print(f" guide: {site_url(src, path)}") 196 seen: set[str] = set() 197 for alt, target in matches: 198 resolved = resolve(path, target.strip()) 199 if resolved in seen: 200 continue 201 seen.add(resolved) 202 url = resolved if resolved.startswith("http") else raw_url(src, resolved) 203 print(f"\n alt-hint: {alt.strip() or '(none upstream — write one)'}") 204 print(f" {url}") 205 206 207 # ---- sha ----------------------------------------------------------------- 208 209 def cmd_refresh_sha(src: dict) -> None: 210 url = f"https://api.github.com/repos/{src['repo']}/commits/{src.get('default_branch', 'master')}" 211 head = json.loads(get(url)) 212 old, new = src["sha"], head["sha"] 213 if old == new: 214 print(f"already pinned at {new}") 215 return 216 print(f"{old}\n -> {new} ({head['commit']['committer']['date']})") 217 print( 218 "\nNOTE: every <figure class=\"shot\"> URL already in src/content/sheets still\n" 219 " points at the old commit. Re-pin those too, or the captions and the\n" 220 " images can drift apart.", 221 file=sys.stderr, 222 ) 223 src["sha"] = new 224 with open(SOURCE_JSON, "w", encoding="utf-8") as fh: 225 json.dump(src, fh, indent=2) 226 fh.write("\n") 227 228 229 def main() -> int: 230 ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) 231 ap.add_argument("--tree", action="store_true", help="list guide paths grouped by cluster") 232 ap.add_argument("--list", action="store_true", help="list guide paths, one per line") 233 ap.add_argument("--get", nargs="+", metavar="PATH", help="fetch guides as readable markdown") 234 ap.add_argument("--images", nargs="+", metavar="PATH", help="print pinned, encoded image URLs") 235 ap.add_argument("--out", default=DEFAULT_OUT, metavar="DIR", help=f"output dir for --get (default: {DEFAULT_OUT})") 236 ap.add_argument("--refresh-sha", action="store_true", help="re-pin to upstream HEAD") 237 args = ap.parse_args() 238 239 src = load_source() 240 if not any([args.tree, args.list, args.get, args.images, args.refresh_sha]): 241 ap.print_help() 242 return 1 243 if args.refresh_sha: 244 cmd_refresh_sha(src) 245 src = load_source() 246 if args.tree or args.list: 247 cmd_tree(src, grouped=bool(args.tree)) 248 if args.get: 249 cmd_get(src, args.get, args.out) 250 if args.images: 251 cmd_images(src, args.images) 252 return 0 253 254 255 if __name__ == "__main__": 256 sys.exit(main())