mirror of
https://github.com/iimp0ster/detection-chokepoints
synced 2026-08-09 12:41:00 +00:00
Re-source ClickGrab ingest off the dead Git-LFS path onto raw GitHub blobs and re-architect the ClickFix trends page around three feeds, each used only for what it is reliable for (DECISIONS #010-012). Ingest (#010-011): fetch MHaggis ClickGrab as raw blobs (upstream LFS quota exhausted); append-only idempotent volume generator + daily GHA for volume and Carson gist landscape count. Behaviour (#012): rebuild the per-domain command classification from Carson's ClickFix Hunter export (build_domain_monthly.py) and re-plumb charts/cards to it, separating hex-XOR from base64 (the prior site-crawl source conflated them and measured ~93-99% noise). Trends prose corrected to the honest figures: May base64 69% (316/458), inline 95.2%, Nov msiexec 87% (669/767). Workstream B: rank MHaggis lure-page HTML keywords (build_lure_keywords.py) into data-driven URLScan OSINT pivots on the clickfix chokepoint and enrich the multilingual IOK matcher. Validated: scripts/validate_schema.py passes (13 chokepoints, 3 trends files). Deferred: two classifier regex bugs distort Dec-Apr months only; headline figures robust (DECISIONS #013). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
148 lines
5.5 KiB
Python
148 lines
5.5 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
build_lure_keywords.py
|
|
Aggregate the lure-page HTML keywords MHaggis ClickGrab extracts from each crawled
|
|
ClickFix site into ranked, data-driven OSINT signal.
|
|
|
|
WHY MHaggis here (see docs/DECISIONS.md #012): MHaggis CRAWLS the live lure pages,
|
|
so its SuspiciousKeywords list is the right material for lure-page fingerprinting —
|
|
URLScan page.body pivots and IOK html|contains matchers — even though its site-level
|
|
CRADLE classification is too noisy for the behavioural trend (that's Carson's job).
|
|
|
|
The SuspiciousKeywords list mixes genuine lure phrases ("i am not a robot",
|
|
"checking if you are human", the Dutch "captcha-verificatie-id") with command
|
|
fragments ("cmd /c curl ...", "powershell -w h ..."). We keep only natural-language
|
|
LURE phrases — those are what a URLScan page.body query or an IOK html matcher hunts.
|
|
|
|
INPUT: cache/clickgrab/days/<date>.json (MHaggis per-site records)
|
|
OUTPUT: _data/clickfix_lure_keywords.yml (ranked keywords + lure families + pivots)
|
|
|
|
Usage:
|
|
python scripts/build_lure_keywords.py
|
|
"""
|
|
|
|
import glob
|
|
import json
|
|
import re
|
|
import sys
|
|
from collections import Counter
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
from urllib.parse import quote
|
|
|
|
import yaml
|
|
|
|
REPO_ROOT = Path(__file__).parent.parent
|
|
DAYS_DIR = REPO_ROOT / "cache" / "clickgrab" / "days"
|
|
OUT_PATH = REPO_ROOT / "_data" / "clickfix_lure_keywords.yml"
|
|
|
|
LURE_FLAGS = ["CaptchaElements", "FakeCloudflare", "FakeGlitchLures", "BotDetection",
|
|
"ClickFixInstructions", "FakeWindowsUpdate", "FakeBrowserUpdate",
|
|
"FakeVideoConferencing", "FakeSoftwareDownloads"]
|
|
|
|
# Drop anything that is a command fragment, code identifier, or shell token rather
|
|
# than a human-readable lure phrase a defender would pivot on.
|
|
CMD_TOKENS = re.compile(
|
|
r"cmd|powershell|\bcurl\b|\biwr\b|\biex\b|http|//|\\|\.ps1|\.exe|\.vbs|\.bat|"
|
|
r"bypass|-enc|frombase64|command\s*=|responsetext|wscript|hidden|%temp%|"
|
|
r"-w\b|const\b|webclient|mshta|conhost|invoke-", re.I)
|
|
# A lure phrase: 3-60 chars, letters/digits/space/hyphen/apostrophe only.
|
|
LURE_SHAPE = re.compile(r"^[a-z0-9 '\-]{3,60}$")
|
|
|
|
|
|
def is_lure_phrase(kw: str) -> bool:
|
|
k = kw.strip().lower()
|
|
if not LURE_SHAPE.match(k):
|
|
return False
|
|
if CMD_TOKENS.search(k):
|
|
return False
|
|
# Drop bare single-char-ish noise and pure numbers.
|
|
return not k.isdigit()
|
|
|
|
|
|
def _truthy(v) -> bool:
|
|
return (v is True) or (isinstance(v, str) and v.lower() == "true") \
|
|
or (isinstance(v, list) and len(v) > 0) or (isinstance(v, (int, float)) and v > 0)
|
|
|
|
|
|
def main() -> None:
|
|
files = sorted(glob.glob(str(DAYS_DIR / "*.json")))
|
|
if not files:
|
|
print(f"ERROR: no cached MHaggis days in {DAYS_DIR}", file=sys.stderr)
|
|
sys.exit(1)
|
|
|
|
kw_sites = Counter() # lure phrase -> # sites it appeared on
|
|
fam_sites = Counter() # lure-family flag -> # sites
|
|
n_sites = 0
|
|
|
|
for f in files:
|
|
try:
|
|
recs = json.loads(Path(f).read_text(encoding="utf-8"))
|
|
except Exception:
|
|
continue
|
|
for r in recs:
|
|
if not isinstance(r, dict):
|
|
continue
|
|
n_sites += 1
|
|
sk = r.get("SuspiciousKeywords") or []
|
|
if isinstance(sk, str):
|
|
sk = [sk]
|
|
# dedup per site so a phrase repeated on one page counts once
|
|
phrases = {k.strip().lower() for k in sk if isinstance(k, str) and is_lure_phrase(k)}
|
|
for p in phrases:
|
|
kw_sites[p] += 1
|
|
for flag in LURE_FLAGS:
|
|
if _truthy(r.get(flag)):
|
|
fam_sites[flag] += 1
|
|
|
|
top = kw_sites.most_common(40)
|
|
families = fam_sites.most_common()
|
|
|
|
# Build URLScan pivots for the highest-signal phrases (skip the very generic
|
|
# single words; pair each with the clipboard-write API for precision).
|
|
GENERIC = {"robot", "captcha", "verification", "verify", "human", "checking"}
|
|
pivots = []
|
|
for phrase, sites in top:
|
|
if phrase in GENERIC or len(phrase) < 8:
|
|
continue
|
|
q = f'page.body:navigator.clipboard AND page.body:"{phrase}"'
|
|
pivots.append({
|
|
"phrase": phrase,
|
|
"sites": sites,
|
|
"query": q,
|
|
"url": "https://urlscan.io/search/#" + quote(q),
|
|
})
|
|
if len(pivots) >= 8:
|
|
break
|
|
|
|
data = {
|
|
"meta": {
|
|
"generated": datetime.now(timezone.utc).strftime("%Y-%m-%d"),
|
|
"source": "MHaggis ClickGrab nightly crawls (SuspiciousKeywords)",
|
|
"sites_analyzed": n_sites,
|
|
},
|
|
"lure_keywords": [{"phrase": p, "sites": c} for p, c in top],
|
|
"lure_families": [{"name": n, "sites": c} for n, c in families],
|
|
"urlscan_pivots": pivots,
|
|
}
|
|
header = ("# Generated by scripts/build_lure_keywords.py — ranked ClickFix lure-page HTML\n"
|
|
"# keywords from MHaggis crawls, for chokepoint OSINT pivots + IOK matchers (#012).\n\n")
|
|
OUT_PATH.write_text(header + yaml.safe_dump(data, sort_keys=False, allow_unicode=True, width=4096),
|
|
encoding="utf-8")
|
|
|
|
print(f"sites: {n_sites} | distinct lure phrases: {len(kw_sites)}")
|
|
print("--- top lure phrases (phrase : #sites) ---")
|
|
for p, c in top[:20]:
|
|
print(f" {c:4} {p}")
|
|
print("--- lure families ---")
|
|
for n, c in families:
|
|
print(f" {c:4} {n}")
|
|
print("--- suggested URLScan pivots ---")
|
|
for pv in pivots:
|
|
print(f" [{pv['sites']:4}] {pv['query']}")
|
|
print(f"\nWritten: {OUT_PATH}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|