fetch-osint-catalogues.py (12158B)
1 #!/usr/bin/env python3 2 """Build a merged index of the two OSINT tool catalogues. 3 4 Sources: 5 - bellingcat/toolkit (~340 tools, gitbook/tools/<slug>/) 6 - The-OSINT-Newsletter/OSINT-Tools-Library (~305 tools, osint-tools/, branch `GitBook`) 7 8 **Neither publishes a licence.** No LICENSE file, `/license` returns 404, and 9 neither README states terms — the same position as ired.team, which 10 `docs/provenance-audit.md` and `src/content.config.ts` have already ruled on: 11 no grant of rights means nothing is mirrored. 12 13 So this script takes only the facts — that a tool exists, its name, its 14 homepage, and which category the catalogue filed it under. A tool name and a 15 URL are not anyone's creative expression. The catalogues' own descriptions are 16 deliberately **not** stored: `blurb` is left empty for a human to write from 17 the tool's own documentation. 18 19 What this produces is a worklist, not content: 20 21 src/data/osint-catalogue.json 22 23 Each entry carries a `cli` guess, from whether the tool looks like a package or 24 a repository rather than a website. The guess is a starting point for triage, 25 not a verdict — confirm before writing install instructions against it. 26 27 ./scripts/fetch-osint-catalogues.py # write the index 28 ./scripts/fetch-osint-catalogues.py --stats # summarise what is there 29 """ 30 from __future__ import annotations 31 32 import argparse 33 import json 34 import os 35 import re 36 import sys 37 import urllib.error 38 import urllib.request 39 40 ROOT = os.path.normpath(os.path.join(os.path.dirname(os.path.abspath(__file__)), "..")) 41 OUT = os.path.join(ROOT, "src", "data", "osint-catalogue.json") 42 43 BELLINGCAT = ("bellingcat/toolkit", "main") 44 NEWSLETTER = ("The-OSINT-Newsletter/OSINT-Tools-Library", "GitBook") 45 46 UA = {"User-Agent": "daemon-sec-cheatsheet/fetch-osint-catalogues"} 47 48 # A homepage that is really a package index or a repository is a hint that the 49 # tool is run from a terminal — but only a hint. Plenty of command-line tools 50 # link their docs site instead, so the page text is read for install markers 51 # too. Reading the catalogue's prose to decide a yes/no is not storing it. 52 CLI_HOST = re.compile( 53 r"(github\.com|gitlab\.com|codeberg\.org|pypi\.org|npmjs\.com|crates\.io|" 54 r"rubygems\.org|pkg\.go\.dev|sourceforge\.net|bitbucket\.org)", 55 re.I, 56 ) 57 58 INSTALL_MARK = re.compile( 59 r"(pipx?\s+install|npm\s+install|\bnpx\b|brew\s+install|go\s+install|" 60 r"cargo\s+install|gem\s+install|apt(?:-get)?\s+install|docker\s+(?:run|compose|pull)|" 61 r"git\s+clone|python3?\s+-m\s|requirements\.txt|\bmake\s+install\b)", 62 re.I, 63 ) 64 TERMINAL_MARK = re.compile( 65 r"(command[- ]line|\bCLI\b|\bterminal\b|\bshell\b|command prompt)", re.I 66 ) 67 68 69 def cli_signals(md: str, url: str) -> tuple[bool, list[str]]: 70 """Whether the tool looks like something you run in a terminal, and why. 71 72 The reasons travel with the guess so a human triaging the worklist can see 73 what it was based on instead of re-reading the page. 74 """ 75 why: list[str] = [] 76 if url and CLI_HOST.search(url): 77 why.append("code-host-url") 78 if INSTALL_MARK.search(md): 79 why.append("install-command") 80 if TERMINAL_MARK.search(md): 81 why.append("terminal-mention") 82 # A bare mention of "terminal" alone is weak — a web tool's page may say a 83 # competitor is command-line. Require a package/repo signal, or an install 84 # command, before calling it CLI. 85 strong = {"code-host-url", "install-command"} 86 return bool(strong & set(why)), why 87 88 89 def get(url: str) -> bytes: 90 req = urllib.request.Request(url, headers=UA) 91 with urllib.request.urlopen(req, timeout=30) as resp: 92 return resp.read() 93 94 95 def get_json(url: str): 96 return json.loads(get(url)) 97 98 99 def tree(repo: str, branch: str) -> list[dict]: 100 """Every blob in the repo at that branch.""" 101 data = get_json(f"https://api.github.com/repos/{repo}/git/trees/{branch}?recursive=1") 102 if data.get("truncated"): 103 print(f"!! {repo}: tree truncated, index will be incomplete", file=sys.stderr) 104 return [e for e in data.get("tree", []) if e.get("type") == "blob"] 105 106 107 # Bellingcat pages put the homepage under a `## URL` heading, and write it 108 # either as a markdown link or — in about a sixth of them — as a bare URL. A 109 # link-only regex silently drops that sixth, so the section is read first and 110 # both forms are accepted. 111 URL_SECTION = re.compile(r"^##\s+URL\s*$(.*?)(?=^##\s|\Z)", re.M | re.S) 112 ANY_URL = re.compile(r"(?:\]\(\s*<?)?(https?://[^\s)>\]]+)") 113 114 # An unfilled template page. The toolkit ships these as placeholders, and they 115 # carry `https://example.com` plus the template's own prompt text, so they are 116 # recorded and skipped rather than written up as if they described a tool. 117 STUB_MARKS = ( 118 "[[ A full description of the tool", 119 "A brief one line description of this tool", 120 ) 121 122 123 def extract_url(md: str) -> str: 124 """The tool's homepage: the `## URL` section if there is one, else the first 125 external link on the page.""" 126 sections = URL_SECTION.findall(md) 127 candidates = sections if sections else [md] 128 for chunk in candidates: 129 for m in ANY_URL.finditer(chunk): 130 url = m.group(1).rstrip(".,);\\") 131 low = url.lower() 132 if "gitbook.io" in low or "osintnewsletter.com" in low: 133 continue 134 if "example.com" in low: 135 continue 136 return url 137 return "" 138 139 140 def is_stub(md: str) -> bool: 141 return any(mark in md for mark in STUB_MARKS) or "https://example.com" in md 142 143 144 def title_of(md: str, fallback: str) -> str: 145 m = re.search(r"^#\s+(.+)$", md, re.M) 146 if m: 147 return m.group(1).strip().strip("#").strip() 148 m = re.search(r"^title:\s*(.+)$", md, re.M) 149 if m: 150 return m.group(1).strip().strip("\"'") 151 return fallback.replace("-", " ").title() 152 153 154 def harvest_bellingcat() -> list[dict]: 155 repo, branch = BELLINGCAT 156 out: list[dict] = [] 157 blobs = tree(repo, branch) 158 # gitbook/tools/<slug>/README.md is the tool page; tool.json beside it 159 # carries the category tags. 160 pages = {} 161 tags = {} 162 for e in blobs: 163 p = e["path"] 164 m = re.match(r"gitbook/tools/([^/]+)/(README\.md|tool\.json)$", p) 165 if not m: 166 continue 167 slug, kind = m.group(1), m.group(2) 168 (pages if kind == "README.md" else tags)[slug] = p 169 for slug in sorted(pages): 170 raw = f"https://raw.githubusercontent.com/{repo}/{branch}/{pages[slug]}" 171 try: 172 md = get(raw).decode("utf-8", "replace") 173 except urllib.error.HTTPError: 174 continue 175 cats: list[str] = [] 176 if slug in tags: 177 try: 178 meta = get_json( 179 f"https://raw.githubusercontent.com/{repo}/{branch}/{tags[slug]}" 180 ) 181 cats = [t for t in meta.get("tags", []) if isinstance(t, str)] 182 except Exception: 183 pass 184 url = extract_url(md) 185 cli, why = cli_signals(md, url) 186 out.append( 187 { 188 "slug": slug, 189 "name": title_of(md, slug), 190 "url": url, 191 "categories": cats, 192 "sources": ["bellingcat"], 193 "cli": cli, 194 "cliWhy": why, 195 "stub": is_stub(md), 196 "blurb": "", 197 } 198 ) 199 print(f" bellingcat/{slug}", file=sys.stderr) 200 return out 201 202 203 def harvest_newsletter() -> list[dict]: 204 repo, branch = NEWSLETTER 205 out: list[dict] = [] 206 for e in tree(repo, branch): 207 p = e["path"] 208 m = re.match(r"osint-tools/([^/]+)\.md$", p) 209 if not m or m.group(1) in {"README", "SUMMARY"}: 210 continue 211 slug = m.group(1) 212 try: 213 md = get( 214 f"https://raw.githubusercontent.com/{repo}/{branch}/{p}" 215 ).decode("utf-8", "replace") 216 except urllib.error.HTTPError: 217 continue 218 url = extract_url(md) 219 cli, why = cli_signals(md, url) 220 out.append( 221 { 222 "slug": slug, 223 "name": title_of(md, slug), 224 "url": url, 225 "categories": [], 226 "sources": ["osintnewsletter"], 227 "cli": cli, 228 "cliWhy": why, 229 "stub": is_stub(md), 230 "blurb": "", 231 } 232 ) 233 print(f" newsletter/{slug}", file=sys.stderr) 234 return out 235 236 237 def norm(name: str) -> str: 238 return re.sub(r"[^a-z0-9]", "", name.lower()) 239 240 241 def merge(groups: list[list[dict]]) -> list[dict]: 242 """One entry per tool. Two catalogue entries are the same tool when their 243 names normalise alike, or they point at the same homepage.""" 244 by_name: dict[str, dict] = {} 245 by_url: dict[str, dict] = {} 246 merged: list[dict] = [] 247 for group in groups: 248 for t in group: 249 key = norm(t["name"]) 250 urlkey = t["url"].rstrip("/").lower() 251 hit = by_name.get(key) or (by_url.get(urlkey) if urlkey else None) 252 if hit: 253 for s in t["sources"]: 254 if s not in hit["sources"]: 255 hit["sources"].append(s) 256 for c in t["categories"]: 257 if c not in hit["categories"]: 258 hit["categories"].append(c) 259 if not hit["url"]: 260 hit["url"] = t["url"] 261 hit["cli"] = hit["cli"] or t["cli"] 262 for w in t.get("cliWhy", []): 263 if w not in hit.setdefault("cliWhy", []): 264 hit["cliWhy"].append(w) 265 # A filled-in page on either side beats a stub on the other. 266 hit["stub"] = hit.get("stub", False) and t.get("stub", False) 267 continue 268 by_name[key] = t 269 if urlkey: 270 by_url[urlkey] = t 271 merged.append(t) 272 merged.sort(key=lambda t: norm(t["name"])) 273 return merged 274 275 276 def stats(tools: list[dict]) -> None: 277 import collections 278 279 print(f"tools : {len(tools)}") 280 print(f"likely CLI : {sum(1 for t in tools if t['cli'])}") 281 print(f"no homepage : {sum(1 for t in tools if not t['url'])}") 282 print(f"unfilled stubs : {sum(1 for t in tools if t.get('stub'))}") 283 both = sum(1 for t in tools if len(t["sources"]) > 1) 284 print(f"in both sources : {both}") 285 cats = collections.Counter(c for t in tools for c in t["categories"]) 286 print("categories :") 287 for c, n in cats.most_common(): 288 print(f" {c:<28} {n}") 289 290 291 def main() -> int: 292 ap = argparse.ArgumentParser() 293 ap.add_argument("--stats", action="store_true", help="summarise the written index") 294 args = ap.parse_args() 295 296 if args.stats: 297 with open(OUT, encoding="utf-8") as fh: 298 stats(json.load(fh)["tools"]) 299 return 0 300 301 tools = merge([harvest_bellingcat(), harvest_newsletter()]) 302 payload = { 303 "note": ( 304 "Discovery index only. Neither catalogue publishes a licence, so no " 305 "description or prose from either is stored here — only the fact that " 306 "a tool exists, its name, its homepage and the category it was filed " 307 "under. Write every word of a sheet from the tool's own documentation." 308 ), 309 "sources": [ 310 { 311 "name": "Bellingcat's Online Investigation Toolkit", 312 "url": "https://bellingcat.gitbook.io/toolkit", 313 "repo": BELLINGCAT[0], 314 "license": None, 315 }, 316 { 317 "name": "The OSINT Newsletter — OSINT Tools Library", 318 "url": "https://tools.osintnewsletter.com/", 319 "repo": NEWSLETTER[0], 320 "license": None, 321 }, 322 ], 323 "tools": tools, 324 } 325 os.makedirs(os.path.dirname(OUT), exist_ok=True) 326 with open(OUT, "w", encoding="utf-8") as fh: 327 json.dump(payload, fh, indent=2, ensure_ascii=False) 328 fh.write("\n") 329 print(f"\nwrote {OUT}") 330 stats(tools) 331 return 0 332 333 334 if __name__ == "__main__": 335 sys.exit(main())