Files
iimp0ster-detection-chokepo…/scripts/build_lure_keywords.py
T
imposterandClaude Opus 4.8 66034135c3 feat(clickgrab): re-source ingest + clean Carson three-feed trend model
Re-source ClickGrab ingest off the dead Git-LFS path onto raw GitHub blobs and re-architect the ClickFix trends page around three feeds, each used only for what it is reliable for (DECISIONS #010-012).

Ingest (#010-011): fetch MHaggis ClickGrab as raw blobs (upstream LFS quota exhausted); append-only idempotent volume generator + daily GHA for volume and Carson gist landscape count.

Behaviour (#012): rebuild the per-domain command classification from Carson's ClickFix Hunter export (build_domain_monthly.py) and re-plumb charts/cards to it, separating hex-XOR from base64 (the prior site-crawl source conflated them and measured ~93-99% noise). Trends prose corrected to the honest figures: May base64 69% (316/458), inline 95.2%, Nov msiexec 87% (669/767).

Workstream B: rank MHaggis lure-page HTML keywords (build_lure_keywords.py) into data-driven URLScan OSINT pivots on the clickfix chokepoint and enrich the multilingual IOK matcher.

Validated: scripts/validate_schema.py passes (13 chokepoints, 3 trends files). Deferred: two classifier regex bugs distort Dec-Apr months only; headline figures robust (DECISIONS #013).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-15 17:59:48 -06:00

148 lines
5.5 KiB
Python

#!/usr/bin/env python3
"""
build_lure_keywords.py
Aggregate the lure-page HTML keywords MHaggis ClickGrab extracts from each crawled
ClickFix site into ranked, data-driven OSINT signal.
WHY MHaggis here (see docs/DECISIONS.md #012): MHaggis CRAWLS the live lure pages,
so its SuspiciousKeywords list is the right material for lure-page fingerprinting —
URLScan page.body pivots and IOK html|contains matchers — even though its site-level
CRADLE classification is too noisy for the behavioural trend (that's Carson's job).
The SuspiciousKeywords list mixes genuine lure phrases ("i am not a robot",
"checking if you are human", the Dutch "captcha-verificatie-id") with command
fragments ("cmd /c curl ...", "powershell -w h ..."). We keep only natural-language
LURE phrases — those are what a URLScan page.body query or an IOK html matcher hunts.
INPUT: cache/clickgrab/days/<date>.json (MHaggis per-site records)
OUTPUT: _data/clickfix_lure_keywords.yml (ranked keywords + lure families + pivots)
Usage:
python scripts/build_lure_keywords.py
"""
import glob
import json
import re
import sys
from collections import Counter
from datetime import datetime, timezone
from pathlib import Path
from urllib.parse import quote
import yaml
REPO_ROOT = Path(__file__).parent.parent
DAYS_DIR = REPO_ROOT / "cache" / "clickgrab" / "days"
OUT_PATH = REPO_ROOT / "_data" / "clickfix_lure_keywords.yml"
LURE_FLAGS = ["CaptchaElements", "FakeCloudflare", "FakeGlitchLures", "BotDetection",
"ClickFixInstructions", "FakeWindowsUpdate", "FakeBrowserUpdate",
"FakeVideoConferencing", "FakeSoftwareDownloads"]
# Drop anything that is a command fragment, code identifier, or shell token rather
# than a human-readable lure phrase a defender would pivot on.
CMD_TOKENS = re.compile(
r"cmd|powershell|\bcurl\b|\biwr\b|\biex\b|http|//|\\|\.ps1|\.exe|\.vbs|\.bat|"
r"bypass|-enc|frombase64|command\s*=|responsetext|wscript|hidden|%temp%|"
r"-w\b|const\b|webclient|mshta|conhost|invoke-", re.I)
# A lure phrase: 3-60 chars, letters/digits/space/hyphen/apostrophe only.
LURE_SHAPE = re.compile(r"^[a-z0-9 '\-]{3,60}$")
def is_lure_phrase(kw: str) -> bool:
k = kw.strip().lower()
if not LURE_SHAPE.match(k):
return False
if CMD_TOKENS.search(k):
return False
# Drop bare single-char-ish noise and pure numbers.
return not k.isdigit()
def _truthy(v) -> bool:
return (v is True) or (isinstance(v, str) and v.lower() == "true") \
or (isinstance(v, list) and len(v) > 0) or (isinstance(v, (int, float)) and v > 0)
def main() -> None:
files = sorted(glob.glob(str(DAYS_DIR / "*.json")))
if not files:
print(f"ERROR: no cached MHaggis days in {DAYS_DIR}", file=sys.stderr)
sys.exit(1)
kw_sites = Counter() # lure phrase -> # sites it appeared on
fam_sites = Counter() # lure-family flag -> # sites
n_sites = 0
for f in files:
try:
recs = json.loads(Path(f).read_text(encoding="utf-8"))
except Exception:
continue
for r in recs:
if not isinstance(r, dict):
continue
n_sites += 1
sk = r.get("SuspiciousKeywords") or []
if isinstance(sk, str):
sk = [sk]
# dedup per site so a phrase repeated on one page counts once
phrases = {k.strip().lower() for k in sk if isinstance(k, str) and is_lure_phrase(k)}
for p in phrases:
kw_sites[p] += 1
for flag in LURE_FLAGS:
if _truthy(r.get(flag)):
fam_sites[flag] += 1
top = kw_sites.most_common(40)
families = fam_sites.most_common()
# Build URLScan pivots for the highest-signal phrases (skip the very generic
# single words; pair each with the clipboard-write API for precision).
GENERIC = {"robot", "captcha", "verification", "verify", "human", "checking"}
pivots = []
for phrase, sites in top:
if phrase in GENERIC or len(phrase) < 8:
continue
q = f'page.body:navigator.clipboard AND page.body:"{phrase}"'
pivots.append({
"phrase": phrase,
"sites": sites,
"query": q,
"url": "https://urlscan.io/search/#" + quote(q),
})
if len(pivots) >= 8:
break
data = {
"meta": {
"generated": datetime.now(timezone.utc).strftime("%Y-%m-%d"),
"source": "MHaggis ClickGrab nightly crawls (SuspiciousKeywords)",
"sites_analyzed": n_sites,
},
"lure_keywords": [{"phrase": p, "sites": c} for p, c in top],
"lure_families": [{"name": n, "sites": c} for n, c in families],
"urlscan_pivots": pivots,
}
header = ("# Generated by scripts/build_lure_keywords.py — ranked ClickFix lure-page HTML\n"
"# keywords from MHaggis crawls, for chokepoint OSINT pivots + IOK matchers (#012).\n\n")
OUT_PATH.write_text(header + yaml.safe_dump(data, sort_keys=False, allow_unicode=True, width=4096),
encoding="utf-8")
print(f"sites: {n_sites} | distinct lure phrases: {len(kw_sites)}")
print("--- top lure phrases (phrase : #sites) ---")
for p, c in top[:20]:
print(f" {c:4} {p}")
print("--- lure families ---")
for n, c in families:
print(f" {c:4} {n}")
print("--- suggested URLScan pivots ---")
for pv in pivots:
print(f" [{pv['sites']:4}] {pv['query']}")
print(f"\nWritten: {OUT_PATH}")
if __name__ == "__main__":
main()