daemon-sec-cheatsheet

The cheatsheet vault for operators: AD, enumeration, exploitation, priv-esc, web, DFIR
git clone https://git.daemon-sec.xyz/daemon-sec-cheatsheet.git
Log | Files | Refs | README | LICENSE

fetch-osint-catalogues.py (12158B)


      1 #!/usr/bin/env python3
      2 """Build a merged index of the two OSINT tool catalogues.
      3 
      4 Sources:
      5   - bellingcat/toolkit                      (~340 tools, gitbook/tools/<slug>/)
      6   - The-OSINT-Newsletter/OSINT-Tools-Library (~305 tools, osint-tools/, branch `GitBook`)
      7 
      8 **Neither publishes a licence.** No LICENSE file, `/license` returns 404, and
      9 neither README states terms — the same position as ired.team, which
     10 `docs/provenance-audit.md` and `src/content.config.ts` have already ruled on:
     11 no grant of rights means nothing is mirrored.
     12 
     13 So this script takes only the facts — that a tool exists, its name, its
     14 homepage, and which category the catalogue filed it under. A tool name and a
     15 URL are not anyone's creative expression. The catalogues' own descriptions are
     16 deliberately **not** stored: `blurb` is left empty for a human to write from
     17 the tool's own documentation.
     18 
     19 What this produces is a worklist, not content:
     20 
     21     src/data/osint-catalogue.json
     22 
     23 Each entry carries a `cli` guess, from whether the tool looks like a package or
     24 a repository rather than a website. The guess is a starting point for triage,
     25 not a verdict — confirm before writing install instructions against it.
     26 
     27     ./scripts/fetch-osint-catalogues.py            # write the index
     28     ./scripts/fetch-osint-catalogues.py --stats    # summarise what is there
     29 """
     30 from __future__ import annotations
     31 
     32 import argparse
     33 import json
     34 import os
     35 import re
     36 import sys
     37 import urllib.error
     38 import urllib.request
     39 
     40 ROOT = os.path.normpath(os.path.join(os.path.dirname(os.path.abspath(__file__)), ".."))
     41 OUT = os.path.join(ROOT, "src", "data", "osint-catalogue.json")
     42 
     43 BELLINGCAT = ("bellingcat/toolkit", "main")
     44 NEWSLETTER = ("The-OSINT-Newsletter/OSINT-Tools-Library", "GitBook")
     45 
     46 UA = {"User-Agent": "daemon-sec-cheatsheet/fetch-osint-catalogues"}
     47 
     48 # A homepage that is really a package index or a repository is a hint that the
     49 # tool is run from a terminal — but only a hint. Plenty of command-line tools
     50 # link their docs site instead, so the page text is read for install markers
     51 # too. Reading the catalogue's prose to decide a yes/no is not storing it.
     52 CLI_HOST = re.compile(
     53     r"(github\.com|gitlab\.com|codeberg\.org|pypi\.org|npmjs\.com|crates\.io|"
     54     r"rubygems\.org|pkg\.go\.dev|sourceforge\.net|bitbucket\.org)",
     55     re.I,
     56 )
     57 
     58 INSTALL_MARK = re.compile(
     59     r"(pipx?\s+install|npm\s+install|\bnpx\b|brew\s+install|go\s+install|"
     60     r"cargo\s+install|gem\s+install|apt(?:-get)?\s+install|docker\s+(?:run|compose|pull)|"
     61     r"git\s+clone|python3?\s+-m\s|requirements\.txt|\bmake\s+install\b)",
     62     re.I,
     63 )
     64 TERMINAL_MARK = re.compile(
     65     r"(command[- ]line|\bCLI\b|\bterminal\b|\bshell\b|command prompt)", re.I
     66 )
     67 
     68 
     69 def cli_signals(md: str, url: str) -> tuple[bool, list[str]]:
     70     """Whether the tool looks like something you run in a terminal, and why.
     71 
     72     The reasons travel with the guess so a human triaging the worklist can see
     73     what it was based on instead of re-reading the page.
     74     """
     75     why: list[str] = []
     76     if url and CLI_HOST.search(url):
     77         why.append("code-host-url")
     78     if INSTALL_MARK.search(md):
     79         why.append("install-command")
     80     if TERMINAL_MARK.search(md):
     81         why.append("terminal-mention")
     82     # A bare mention of "terminal" alone is weak — a web tool's page may say a
     83     # competitor is command-line. Require a package/repo signal, or an install
     84     # command, before calling it CLI.
     85     strong = {"code-host-url", "install-command"}
     86     return bool(strong & set(why)), why
     87 
     88 
     89 def get(url: str) -> bytes:
     90     req = urllib.request.Request(url, headers=UA)
     91     with urllib.request.urlopen(req, timeout=30) as resp:
     92         return resp.read()
     93 
     94 
     95 def get_json(url: str):
     96     return json.loads(get(url))
     97 
     98 
     99 def tree(repo: str, branch: str) -> list[dict]:
    100     """Every blob in the repo at that branch."""
    101     data = get_json(f"https://api.github.com/repos/{repo}/git/trees/{branch}?recursive=1")
    102     if data.get("truncated"):
    103         print(f"!! {repo}: tree truncated, index will be incomplete", file=sys.stderr)
    104     return [e for e in data.get("tree", []) if e.get("type") == "blob"]
    105 
    106 
    107 # Bellingcat pages put the homepage under a `## URL` heading, and write it
    108 # either as a markdown link or — in about a sixth of them — as a bare URL. A
    109 # link-only regex silently drops that sixth, so the section is read first and
    110 # both forms are accepted.
    111 URL_SECTION = re.compile(r"^##\s+URL\s*$(.*?)(?=^##\s|\Z)", re.M | re.S)
    112 ANY_URL = re.compile(r"(?:\]\(\s*<?)?(https?://[^\s)>\]]+)")
    113 
    114 # An unfilled template page. The toolkit ships these as placeholders, and they
    115 # carry `https://example.com` plus the template's own prompt text, so they are
    116 # recorded and skipped rather than written up as if they described a tool.
    117 STUB_MARKS = (
    118     "[[ A full description of the tool",
    119     "A brief one line description of this tool",
    120 )
    121 
    122 
    123 def extract_url(md: str) -> str:
    124     """The tool's homepage: the `## URL` section if there is one, else the first
    125     external link on the page."""
    126     sections = URL_SECTION.findall(md)
    127     candidates = sections if sections else [md]
    128     for chunk in candidates:
    129         for m in ANY_URL.finditer(chunk):
    130             url = m.group(1).rstrip(".,);\\")
    131             low = url.lower()
    132             if "gitbook.io" in low or "osintnewsletter.com" in low:
    133                 continue
    134             if "example.com" in low:
    135                 continue
    136             return url
    137     return ""
    138 
    139 
    140 def is_stub(md: str) -> bool:
    141     return any(mark in md for mark in STUB_MARKS) or "https://example.com" in md
    142 
    143 
    144 def title_of(md: str, fallback: str) -> str:
    145     m = re.search(r"^#\s+(.+)$", md, re.M)
    146     if m:
    147         return m.group(1).strip().strip("#").strip()
    148     m = re.search(r"^title:\s*(.+)$", md, re.M)
    149     if m:
    150         return m.group(1).strip().strip("\"'")
    151     return fallback.replace("-", " ").title()
    152 
    153 
    154 def harvest_bellingcat() -> list[dict]:
    155     repo, branch = BELLINGCAT
    156     out: list[dict] = []
    157     blobs = tree(repo, branch)
    158     # gitbook/tools/<slug>/README.md is the tool page; tool.json beside it
    159     # carries the category tags.
    160     pages = {}
    161     tags = {}
    162     for e in blobs:
    163         p = e["path"]
    164         m = re.match(r"gitbook/tools/([^/]+)/(README\.md|tool\.json)$", p)
    165         if not m:
    166             continue
    167         slug, kind = m.group(1), m.group(2)
    168         (pages if kind == "README.md" else tags)[slug] = p
    169     for slug in sorted(pages):
    170         raw = f"https://raw.githubusercontent.com/{repo}/{branch}/{pages[slug]}"
    171         try:
    172             md = get(raw).decode("utf-8", "replace")
    173         except urllib.error.HTTPError:
    174             continue
    175         cats: list[str] = []
    176         if slug in tags:
    177             try:
    178                 meta = get_json(
    179                     f"https://raw.githubusercontent.com/{repo}/{branch}/{tags[slug]}"
    180                 )
    181                 cats = [t for t in meta.get("tags", []) if isinstance(t, str)]
    182             except Exception:
    183                 pass
    184         url = extract_url(md)
    185         cli, why = cli_signals(md, url)
    186         out.append(
    187             {
    188                 "slug": slug,
    189                 "name": title_of(md, slug),
    190                 "url": url,
    191                 "categories": cats,
    192                 "sources": ["bellingcat"],
    193                 "cli": cli,
    194                 "cliWhy": why,
    195                 "stub": is_stub(md),
    196                 "blurb": "",
    197             }
    198         )
    199         print(f"  bellingcat/{slug}", file=sys.stderr)
    200     return out
    201 
    202 
    203 def harvest_newsletter() -> list[dict]:
    204     repo, branch = NEWSLETTER
    205     out: list[dict] = []
    206     for e in tree(repo, branch):
    207         p = e["path"]
    208         m = re.match(r"osint-tools/([^/]+)\.md$", p)
    209         if not m or m.group(1) in {"README", "SUMMARY"}:
    210             continue
    211         slug = m.group(1)
    212         try:
    213             md = get(
    214                 f"https://raw.githubusercontent.com/{repo}/{branch}/{p}"
    215             ).decode("utf-8", "replace")
    216         except urllib.error.HTTPError:
    217             continue
    218         url = extract_url(md)
    219         cli, why = cli_signals(md, url)
    220         out.append(
    221             {
    222                 "slug": slug,
    223                 "name": title_of(md, slug),
    224                 "url": url,
    225                 "categories": [],
    226                 "sources": ["osintnewsletter"],
    227                 "cli": cli,
    228                 "cliWhy": why,
    229                 "stub": is_stub(md),
    230                 "blurb": "",
    231             }
    232         )
    233         print(f"  newsletter/{slug}", file=sys.stderr)
    234     return out
    235 
    236 
    237 def norm(name: str) -> str:
    238     return re.sub(r"[^a-z0-9]", "", name.lower())
    239 
    240 
    241 def merge(groups: list[list[dict]]) -> list[dict]:
    242     """One entry per tool. Two catalogue entries are the same tool when their
    243     names normalise alike, or they point at the same homepage."""
    244     by_name: dict[str, dict] = {}
    245     by_url: dict[str, dict] = {}
    246     merged: list[dict] = []
    247     for group in groups:
    248         for t in group:
    249             key = norm(t["name"])
    250             urlkey = t["url"].rstrip("/").lower()
    251             hit = by_name.get(key) or (by_url.get(urlkey) if urlkey else None)
    252             if hit:
    253                 for s in t["sources"]:
    254                     if s not in hit["sources"]:
    255                         hit["sources"].append(s)
    256                 for c in t["categories"]:
    257                     if c not in hit["categories"]:
    258                         hit["categories"].append(c)
    259                 if not hit["url"]:
    260                     hit["url"] = t["url"]
    261                 hit["cli"] = hit["cli"] or t["cli"]
    262                 for w in t.get("cliWhy", []):
    263                     if w not in hit.setdefault("cliWhy", []):
    264                         hit["cliWhy"].append(w)
    265                 # A filled-in page on either side beats a stub on the other.
    266                 hit["stub"] = hit.get("stub", False) and t.get("stub", False)
    267                 continue
    268             by_name[key] = t
    269             if urlkey:
    270                 by_url[urlkey] = t
    271             merged.append(t)
    272     merged.sort(key=lambda t: norm(t["name"]))
    273     return merged
    274 
    275 
    276 def stats(tools: list[dict]) -> None:
    277     import collections
    278 
    279     print(f"tools           : {len(tools)}")
    280     print(f"likely CLI      : {sum(1 for t in tools if t['cli'])}")
    281     print(f"no homepage     : {sum(1 for t in tools if not t['url'])}")
    282     print(f"unfilled stubs  : {sum(1 for t in tools if t.get('stub'))}")
    283     both = sum(1 for t in tools if len(t["sources"]) > 1)
    284     print(f"in both sources : {both}")
    285     cats = collections.Counter(c for t in tools for c in t["categories"])
    286     print("categories      :")
    287     for c, n in cats.most_common():
    288         print(f"  {c:<28} {n}")
    289 
    290 
    291 def main() -> int:
    292     ap = argparse.ArgumentParser()
    293     ap.add_argument("--stats", action="store_true", help="summarise the written index")
    294     args = ap.parse_args()
    295 
    296     if args.stats:
    297         with open(OUT, encoding="utf-8") as fh:
    298             stats(json.load(fh)["tools"])
    299         return 0
    300 
    301     tools = merge([harvest_bellingcat(), harvest_newsletter()])
    302     payload = {
    303         "note": (
    304             "Discovery index only. Neither catalogue publishes a licence, so no "
    305             "description or prose from either is stored here — only the fact that "
    306             "a tool exists, its name, its homepage and the category it was filed "
    307             "under. Write every word of a sheet from the tool's own documentation."
    308         ),
    309         "sources": [
    310             {
    311                 "name": "Bellingcat's Online Investigation Toolkit",
    312                 "url": "https://bellingcat.gitbook.io/toolkit",
    313                 "repo": BELLINGCAT[0],
    314                 "license": None,
    315             },
    316             {
    317                 "name": "The OSINT Newsletter — OSINT Tools Library",
    318                 "url": "https://tools.osintnewsletter.com/",
    319                 "repo": NEWSLETTER[0],
    320                 "license": None,
    321             },
    322         ],
    323         "tools": tools,
    324     }
    325     os.makedirs(os.path.dirname(OUT), exist_ok=True)
    326     with open(OUT, "w", encoding="utf-8") as fh:
    327         json.dump(payload, fh, indent=2, ensure_ascii=False)
    328         fh.write("\n")
    329     print(f"\nwrote {OUT}")
    330     stats(tools)
    331     return 0
    332 
    333 
    334 if __name__ == "__main__":
    335     sys.exit(main())