mirror of
https://github.com/iimp0ster/detection-chokepoints
synced 2026-08-09 12:41:00 +00:00
feat: add Streamlit Cloud data persistence via GitHub API
The Streamlit app is hosted on Streamlit Community Cloud which has an ephemeral filesystem — collected data was lost on every restart with no way to push it back to the repo for the GitHub Pages site. Adds GitHub API-based commit and PR functions so collected data can be published from the cloud app. Also adds missing VT_API_KEY and GH_TOKEN to the secrets pipeline, and creates a secrets.toml.example for cloud configuration. Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,30 @@
|
||||
# Streamlit Cloud Secrets Configuration
|
||||
#
|
||||
# On Streamlit Community Cloud, paste these key-value pairs into the app's
|
||||
# "Secrets" panel (Settings > Secrets). The app reads them via st.secrets.
|
||||
#
|
||||
# For local development, copy this file to .streamlit/secrets.toml and fill
|
||||
# in your keys. secrets.toml is gitignored and never committed.
|
||||
|
||||
# ── abuse.ch feeds ───────────────────────────────────────────────────────────
|
||||
MB_API_KEY = "" # MalwareBazaar
|
||||
THREATFOX_API_KEY = "" # ThreatFox
|
||||
URLHAUS_API_KEY = "" # URLhaus
|
||||
|
||||
# ── Infrastructure hunting ───────────────────────────────────────────────────
|
||||
SHODAN_API_KEY = "" # Shodan (favicon hunts, host lookups)
|
||||
URLSCAN_API_KEY = "" # URLScan (filename pivots, chain reconstruction)
|
||||
|
||||
# ── Enrichment ───────────────────────────────────────────────────────────────
|
||||
VALIDIN_API_KEY = "" # Validin (DNS pivots, CF origin unmasking)
|
||||
VT_API_KEY = "" # VirusTotal (domain cross-reference)
|
||||
|
||||
# ── AI triage ────────────────────────────────────────────────────────────────
|
||||
ANTHROPIC_API_KEY = "" # Claude API (payload classification)
|
||||
|
||||
# ── Data persistence (REQUIRED for publishing from Streamlit Cloud) ──────────
|
||||
# A GitHub personal access token (classic or fine-grained) with:
|
||||
# - contents: write (push data commits)
|
||||
# - pull-requests: write (create review PRs)
|
||||
# Scoped to this repo only.
|
||||
GH_TOKEN = ""
|
||||
@@ -1,15 +1,25 @@
|
||||
"""Collection tab — runs IOC feeds, infrastructure hunts, chain reconstruction,
|
||||
and AI triage scripts, then shows cache file status."""
|
||||
and AI triage scripts, then shows cache file status.
|
||||
|
||||
On Streamlit Cloud the filesystem is ephemeral — collected data must be pushed
|
||||
back to the GitHub repo so the Jekyll/GitHub Pages site can serve it."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from datetime import datetime
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
import streamlit as st
|
||||
|
||||
from app.utils.helpers import REPO_ROOT, load_records, run_script
|
||||
from app.utils.helpers import (
|
||||
REPO_ROOT,
|
||||
github_commit_data,
|
||||
github_create_pr,
|
||||
load_records,
|
||||
run_script,
|
||||
)
|
||||
|
||||
_SCRIPTS = Path(REPO_ROOT / "scripts")
|
||||
_CACHE = Path(REPO_ROOT / "cache")
|
||||
@@ -35,6 +45,7 @@ def _build_env(secrets: dict) -> dict:
|
||||
"SHODAN_API_KEY": secrets.get("shodan", ""),
|
||||
"ANTHROPIC_API_KEY": secrets.get("anthropic", ""),
|
||||
"VALIDIN_API_KEY": secrets.get("validin", ""),
|
||||
"VT_API_KEY": secrets.get("vt", ""),
|
||||
}
|
||||
|
||||
|
||||
@@ -183,3 +194,139 @@ def render(secrets: dict) -> None:
|
||||
cols = st.columns(len(cache_files))
|
||||
for col, (rel_path, label) in zip(cols, cache_files.items()):
|
||||
col.metric(label, _file_status(rel_path))
|
||||
|
||||
st.divider()
|
||||
|
||||
# ── Publish to GitHub ──────────────────────────────────────────────────
|
||||
st.subheader("Publish to GitHub")
|
||||
st.info(
|
||||
"Streamlit Cloud has an **ephemeral filesystem** — collected data is lost "
|
||||
"when the app restarts. Use this section to push `_data/masq_infra.json` "
|
||||
"back to the GitHub repo so the GitHub Pages site displays the latest data."
|
||||
)
|
||||
|
||||
gh_token = secrets.get("github", "")
|
||||
if not gh_token:
|
||||
st.warning(
|
||||
"GH_TOKEN not configured in Streamlit secrets. "
|
||||
"Add a GitHub personal access token with `contents: write` and "
|
||||
"`pull-requests: write` permissions to enable publishing."
|
||||
)
|
||||
else:
|
||||
masq_path = REPO_ROOT / "_data" / "masq_infra.json"
|
||||
history_path = REPO_ROOT / "_data" / "masq_infra_history.json"
|
||||
|
||||
has_masq = masq_path.exists()
|
||||
has_history = history_path.exists()
|
||||
|
||||
if not has_masq:
|
||||
st.caption(
|
||||
"_data/masq_infra.json not found. Run the full pipeline "
|
||||
"(IOC Collection → Build Data export) first."
|
||||
)
|
||||
|
||||
pub_col1, pub_col2 = st.columns(2)
|
||||
|
||||
with pub_col1:
|
||||
if st.button(
|
||||
"Commit directly to main",
|
||||
key="btn_publish_direct",
|
||||
disabled=not has_masq,
|
||||
):
|
||||
files: dict[str, str] = {}
|
||||
files["_data/masq_infra.json"] = masq_path.read_text(encoding="utf-8")
|
||||
if has_history:
|
||||
files["_data/masq_infra_history.json"] = history_path.read_text(
|
||||
encoding="utf-8"
|
||||
)
|
||||
|
||||
ts = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M")
|
||||
msg = f"chore: update masq-infra data [{ts}]"
|
||||
|
||||
with st.spinner("Committing to GitHub …"):
|
||||
result = github_commit_data(gh_token, files, msg)
|
||||
|
||||
if result["ok"]:
|
||||
st.success(
|
||||
f"Committed to main ({result['sha'][:8]}). "
|
||||
"GitHub Pages will rebuild automatically."
|
||||
)
|
||||
else:
|
||||
st.error(f"Commit failed: {result['error']}")
|
||||
|
||||
with pub_col2:
|
||||
if st.button(
|
||||
"Open a review PR",
|
||||
key="btn_publish_pr",
|
||||
disabled=not has_masq,
|
||||
):
|
||||
date_str = datetime.now(timezone.utc).strftime("%Y-%m-%d")
|
||||
pr_branch = f"data/masq-infra-{date_str}"
|
||||
ts = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M")
|
||||
msg = f"chore: update masq-infra data [{ts}]"
|
||||
|
||||
files = {}
|
||||
files["_data/masq_infra.json"] = masq_path.read_text(encoding="utf-8")
|
||||
if has_history:
|
||||
files["_data/masq_infra_history.json"] = history_path.read_text(
|
||||
encoding="utf-8"
|
||||
)
|
||||
|
||||
with st.spinner("Creating branch and PR …"):
|
||||
# Create the branch via the API
|
||||
import requests as req # noqa: PLC0415
|
||||
from app.utils.helpers import _get_repo_slug, _GITHUB_API # noqa: PLC0415
|
||||
|
||||
slug = _get_repo_slug()
|
||||
headers = {
|
||||
"Authorization": f"token {gh_token}",
|
||||
"Accept": "application/vnd.github+json",
|
||||
}
|
||||
|
||||
# Get main branch SHA
|
||||
r = req.get(
|
||||
f"{_GITHUB_API}/repos/{slug}/git/ref/heads/main",
|
||||
headers=headers, timeout=15,
|
||||
)
|
||||
if r.status_code != 200:
|
||||
st.error(f"Cannot resolve main branch: {r.status_code}")
|
||||
else:
|
||||
main_sha = r.json()["object"]["sha"]
|
||||
|
||||
# Create or update the PR branch
|
||||
r = req.post(
|
||||
f"{_GITHUB_API}/repos/{slug}/git/refs",
|
||||
headers=headers, timeout=15,
|
||||
json={"ref": f"refs/heads/{pr_branch}", "sha": main_sha},
|
||||
)
|
||||
# 422 = branch already exists; update it instead
|
||||
if r.status_code == 422:
|
||||
req.patch(
|
||||
f"{_GITHUB_API}/repos/{slug}/git/refs/heads/{pr_branch}",
|
||||
headers=headers, timeout=15,
|
||||
json={"sha": main_sha, "force": True},
|
||||
)
|
||||
|
||||
# Commit data to the PR branch
|
||||
result = github_commit_data(
|
||||
gh_token, files, msg, branch=pr_branch,
|
||||
)
|
||||
if not result["ok"]:
|
||||
st.error(f"Commit failed: {result['error']}")
|
||||
else:
|
||||
# Create the PR
|
||||
pr_result = github_create_pr(
|
||||
gh_token,
|
||||
head_branch=pr_branch,
|
||||
base_branch="main",
|
||||
title=f"chore: update masq-infra data [{date_str}]",
|
||||
body=(
|
||||
"Pipeline data update from Streamlit Cloud.\n\n"
|
||||
"Review the changes, then merge to publish "
|
||||
"to the live site."
|
||||
),
|
||||
)
|
||||
if pr_result["ok"]:
|
||||
st.success(f"PR created: {pr_result['url']}")
|
||||
else:
|
||||
st.error(f"PR failed: {pr_result['error']}")
|
||||
|
||||
@@ -128,6 +128,11 @@ if not secrets.get("urlscan") or not secrets.get("mb"):
|
||||
st.warning(
|
||||
"⚠️ URLSCAN_API_KEY or MB_API_KEY not configured. Collection will be limited."
|
||||
)
|
||||
if not secrets.get("github"):
|
||||
st.warning(
|
||||
"⚠️ GH_TOKEN not configured. Collected data cannot be published back to "
|
||||
"the GitHub repo and will be lost when the app restarts."
|
||||
)
|
||||
|
||||
# ── Route to selected tab ─────────────────────────────────────────────────────
|
||||
if selected == "📡 Collection":
|
||||
|
||||
@@ -20,6 +20,8 @@ _KEY_MAP: dict[str, str] = {
|
||||
"urlhaus": "URLHAUS_API_KEY",
|
||||
"validin": "VALIDIN_API_KEY",
|
||||
"anthropic": "ANTHROPIC_API_KEY",
|
||||
"vt": "VT_API_KEY",
|
||||
"github": "GH_TOKEN",
|
||||
}
|
||||
|
||||
|
||||
@@ -134,3 +136,188 @@ def run_script(script_path: str, env: dict) -> tuple[int, str, str]:
|
||||
cwd=str(REPO_ROOT),
|
||||
)
|
||||
return result.returncode, result.stdout, result.stderr
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# GitHub data persistence — commit collected data back to the repo via API
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_GITHUB_API = "https://api.github.com"
|
||||
|
||||
|
||||
def _get_repo_slug() -> str | None:
|
||||
"""Derive 'owner/repo' from the git remote origin URL.
|
||||
|
||||
Falls back to the GITHUB_REPOSITORY env var (set on Streamlit Cloud
|
||||
when the app is linked to a repo).
|
||||
"""
|
||||
slug = os.environ.get("GITHUB_REPOSITORY", "")
|
||||
if slug:
|
||||
return slug
|
||||
# Try parsing from git remote
|
||||
try:
|
||||
result = subprocess.run(
|
||||
["git", "remote", "get-url", "origin"],
|
||||
capture_output=True, text=True, cwd=str(REPO_ROOT),
|
||||
)
|
||||
url = result.stdout.strip()
|
||||
# Handle both HTTPS and SSH URLs
|
||||
if "github.com" in url:
|
||||
# https://github.com/owner/repo.git or git@github.com:owner/repo.git
|
||||
import re # noqa: PLC0415
|
||||
m = re.search(r"github\.com[:/](.+?)(?:\.git)?$", url)
|
||||
if m:
|
||||
return m.group(1)
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def github_commit_data(
|
||||
token: str,
|
||||
files: dict[str, str],
|
||||
message: str,
|
||||
branch: str | None = None,
|
||||
) -> dict:
|
||||
"""Commit one or more files to the repo via the GitHub API.
|
||||
|
||||
This is the persistence mechanism for Streamlit Cloud: since the filesystem
|
||||
is ephemeral, collected data must be pushed back to the repo so GitHub Pages
|
||||
can serve it.
|
||||
|
||||
Args:
|
||||
token: GitHub personal access token (GH_TOKEN from st.secrets).
|
||||
files: Mapping of repo-relative paths to file contents.
|
||||
message: Commit message.
|
||||
branch: Target branch (defaults to repo default branch).
|
||||
|
||||
Returns:
|
||||
Dict with 'ok' bool and either 'sha' or 'error' string.
|
||||
"""
|
||||
import base64 as b64 # noqa: PLC0415
|
||||
|
||||
slug = _get_repo_slug()
|
||||
if not slug:
|
||||
return {"ok": False, "error": "Could not determine GitHub repo slug"}
|
||||
|
||||
headers = {
|
||||
"Authorization": f"token {token}",
|
||||
"Accept": "application/vnd.github+json",
|
||||
}
|
||||
|
||||
import requests as req # noqa: PLC0415
|
||||
|
||||
# Resolve default branch if none specified
|
||||
if not branch:
|
||||
r = req.get(f"{_GITHUB_API}/repos/{slug}", headers=headers, timeout=15)
|
||||
if r.status_code != 200:
|
||||
return {"ok": False, "error": f"Cannot fetch repo info: {r.status_code}"}
|
||||
branch = r.json().get("default_branch", "main")
|
||||
|
||||
# Get the latest commit SHA on the target branch
|
||||
r = req.get(
|
||||
f"{_GITHUB_API}/repos/{slug}/git/ref/heads/{branch}",
|
||||
headers=headers, timeout=15,
|
||||
)
|
||||
if r.status_code != 200:
|
||||
return {"ok": False, "error": f"Cannot resolve branch '{branch}': {r.status_code}"}
|
||||
base_sha = r.json()["object"]["sha"]
|
||||
|
||||
# Get the tree SHA of that commit
|
||||
r = req.get(
|
||||
f"{_GITHUB_API}/repos/{slug}/git/commits/{base_sha}",
|
||||
headers=headers, timeout=15,
|
||||
)
|
||||
if r.status_code != 200:
|
||||
return {"ok": False, "error": f"Cannot fetch commit: {r.status_code}"}
|
||||
base_tree_sha = r.json()["tree"]["sha"]
|
||||
|
||||
# Create blobs for each file
|
||||
tree_entries = []
|
||||
for path, content in files.items():
|
||||
r = req.post(
|
||||
f"{_GITHUB_API}/repos/{slug}/git/blobs",
|
||||
headers=headers, timeout=30,
|
||||
json={"content": b64.b64encode(content.encode()).decode(), "encoding": "base64"},
|
||||
)
|
||||
if r.status_code != 201:
|
||||
return {"ok": False, "error": f"Cannot create blob for {path}: {r.status_code}"}
|
||||
blob_sha = r.json()["sha"]
|
||||
tree_entries.append({
|
||||
"path": path, "mode": "100644", "type": "blob", "sha": blob_sha,
|
||||
})
|
||||
|
||||
# Create a new tree
|
||||
r = req.post(
|
||||
f"{_GITHUB_API}/repos/{slug}/git/trees",
|
||||
headers=headers, timeout=30,
|
||||
json={"base_tree": base_tree_sha, "tree": tree_entries},
|
||||
)
|
||||
if r.status_code != 201:
|
||||
return {"ok": False, "error": f"Cannot create tree: {r.status_code}"}
|
||||
new_tree_sha = r.json()["sha"]
|
||||
|
||||
# Create commit
|
||||
r = req.post(
|
||||
f"{_GITHUB_API}/repos/{slug}/git/commits",
|
||||
headers=headers, timeout=30,
|
||||
json={
|
||||
"message": message,
|
||||
"tree": new_tree_sha,
|
||||
"parents": [base_sha],
|
||||
},
|
||||
)
|
||||
if r.status_code != 201:
|
||||
return {"ok": False, "error": f"Cannot create commit: {r.status_code}"}
|
||||
new_commit_sha = r.json()["sha"]
|
||||
|
||||
# Update branch reference
|
||||
r = req.patch(
|
||||
f"{_GITHUB_API}/repos/{slug}/git/refs/heads/{branch}",
|
||||
headers=headers, timeout=15,
|
||||
json={"sha": new_commit_sha},
|
||||
)
|
||||
if r.status_code != 200:
|
||||
return {"ok": False, "error": f"Cannot update ref: {r.status_code}"}
|
||||
|
||||
return {"ok": True, "sha": new_commit_sha}
|
||||
|
||||
|
||||
def github_create_pr(
|
||||
token: str,
|
||||
head_branch: str,
|
||||
base_branch: str = "main",
|
||||
title: str = "",
|
||||
body: str = "",
|
||||
) -> dict:
|
||||
"""Create a pull request via the GitHub API.
|
||||
|
||||
Returns dict with 'ok' bool and either 'url' or 'error'.
|
||||
"""
|
||||
import requests as req # noqa: PLC0415
|
||||
|
||||
slug = _get_repo_slug()
|
||||
if not slug:
|
||||
return {"ok": False, "error": "Could not determine GitHub repo slug"}
|
||||
|
||||
headers = {
|
||||
"Authorization": f"token {token}",
|
||||
"Accept": "application/vnd.github+json",
|
||||
}
|
||||
|
||||
r = req.post(
|
||||
f"{_GITHUB_API}/repos/{slug}/pulls",
|
||||
headers=headers, timeout=30,
|
||||
json={
|
||||
"title": title,
|
||||
"body": body,
|
||||
"head": head_branch,
|
||||
"base": base_branch,
|
||||
},
|
||||
)
|
||||
if r.status_code == 201:
|
||||
return {"ok": True, "url": r.json().get("html_url", "")}
|
||||
if r.status_code == 422:
|
||||
# PR already exists
|
||||
return {"ok": True, "url": "(PR already exists for this branch)"}
|
||||
return {"ok": False, "error": f"PR creation failed: {r.status_code} {r.text[:500]}"}
|
||||
|
||||
Reference in New Issue
Block a user