feat: add Streamlit Cloud data persistence via GitHub API

The Streamlit app is hosted on Streamlit Community Cloud which has an
ephemeral filesystem — collected data was lost on every restart with no
way to push it back to the repo for the GitHub Pages site.

Adds GitHub API-based commit and PR functions so collected data can be
published from the cloud app. Also adds missing VT_API_KEY and GH_TOKEN
to the secrets pipeline, and creates a secrets.toml.example for cloud
configuration.

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
Claude
2026-03-28 05:13:48 +00:00
parent 259af6159a
commit fcb02022b5
4 changed files with 372 additions and 3 deletions
+30
View File
@@ -0,0 +1,30 @@
# Streamlit Cloud Secrets Configuration
#
# On Streamlit Community Cloud, paste these key-value pairs into the app's
# "Secrets" panel (Settings > Secrets). The app reads them via st.secrets.
#
# For local development, copy this file to .streamlit/secrets.toml and fill
# in your keys. secrets.toml is gitignored and never committed.
# ── abuse.ch feeds ───────────────────────────────────────────────────────────
MB_API_KEY = "" # MalwareBazaar
THREATFOX_API_KEY = "" # ThreatFox
URLHAUS_API_KEY = "" # URLhaus
# ── Infrastructure hunting ───────────────────────────────────────────────────
SHODAN_API_KEY = "" # Shodan (favicon hunts, host lookups)
URLSCAN_API_KEY = "" # URLScan (filename pivots, chain reconstruction)
# ── Enrichment ───────────────────────────────────────────────────────────────
VALIDIN_API_KEY = "" # Validin (DNS pivots, CF origin unmasking)
VT_API_KEY = "" # VirusTotal (domain cross-reference)
# ── AI triage ────────────────────────────────────────────────────────────────
ANTHROPIC_API_KEY = "" # Claude API (payload classification)
# ── Data persistence (REQUIRED for publishing from Streamlit Cloud) ──────────
# A GitHub personal access token (classic or fine-grained) with:
# - contents: write (push data commits)
# - pull-requests: write (create review PRs)
# Scoped to this repo only.
GH_TOKEN = ""
+150 -3
View File
@@ -1,15 +1,25 @@
"""Collection tab — runs IOC feeds, infrastructure hunts, chain reconstruction,
and AI triage scripts, then shows cache file status."""
and AI triage scripts, then shows cache file status.
On Streamlit Cloud the filesystem is ephemeral — collected data must be pushed
back to the GitHub repo so the Jekyll/GitHub Pages site can serve it."""
from __future__ import annotations
import json
import re
from datetime import datetime
from datetime import datetime, timezone
from pathlib import Path
import streamlit as st
from app.utils.helpers import REPO_ROOT, load_records, run_script
from app.utils.helpers import (
REPO_ROOT,
github_commit_data,
github_create_pr,
load_records,
run_script,
)
_SCRIPTS = Path(REPO_ROOT / "scripts")
_CACHE = Path(REPO_ROOT / "cache")
@@ -35,6 +45,7 @@ def _build_env(secrets: dict) -> dict:
"SHODAN_API_KEY": secrets.get("shodan", ""),
"ANTHROPIC_API_KEY": secrets.get("anthropic", ""),
"VALIDIN_API_KEY": secrets.get("validin", ""),
"VT_API_KEY": secrets.get("vt", ""),
}
@@ -183,3 +194,139 @@ def render(secrets: dict) -> None:
cols = st.columns(len(cache_files))
for col, (rel_path, label) in zip(cols, cache_files.items()):
col.metric(label, _file_status(rel_path))
st.divider()
# ── Publish to GitHub ──────────────────────────────────────────────────
st.subheader("Publish to GitHub")
st.info(
"Streamlit Cloud has an **ephemeral filesystem** — collected data is lost "
"when the app restarts. Use this section to push `_data/masq_infra.json` "
"back to the GitHub repo so the GitHub Pages site displays the latest data."
)
gh_token = secrets.get("github", "")
if not gh_token:
st.warning(
"GH_TOKEN not configured in Streamlit secrets. "
"Add a GitHub personal access token with `contents: write` and "
"`pull-requests: write` permissions to enable publishing."
)
else:
masq_path = REPO_ROOT / "_data" / "masq_infra.json"
history_path = REPO_ROOT / "_data" / "masq_infra_history.json"
has_masq = masq_path.exists()
has_history = history_path.exists()
if not has_masq:
st.caption(
"_data/masq_infra.json not found. Run the full pipeline "
"(IOC Collection → Build Data export) first."
)
pub_col1, pub_col2 = st.columns(2)
with pub_col1:
if st.button(
"Commit directly to main",
key="btn_publish_direct",
disabled=not has_masq,
):
files: dict[str, str] = {}
files["_data/masq_infra.json"] = masq_path.read_text(encoding="utf-8")
if has_history:
files["_data/masq_infra_history.json"] = history_path.read_text(
encoding="utf-8"
)
ts = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M")
msg = f"chore: update masq-infra data [{ts}]"
with st.spinner("Committing to GitHub …"):
result = github_commit_data(gh_token, files, msg)
if result["ok"]:
st.success(
f"Committed to main ({result['sha'][:8]}). "
"GitHub Pages will rebuild automatically."
)
else:
st.error(f"Commit failed: {result['error']}")
with pub_col2:
if st.button(
"Open a review PR",
key="btn_publish_pr",
disabled=not has_masq,
):
date_str = datetime.now(timezone.utc).strftime("%Y-%m-%d")
pr_branch = f"data/masq-infra-{date_str}"
ts = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M")
msg = f"chore: update masq-infra data [{ts}]"
files = {}
files["_data/masq_infra.json"] = masq_path.read_text(encoding="utf-8")
if has_history:
files["_data/masq_infra_history.json"] = history_path.read_text(
encoding="utf-8"
)
with st.spinner("Creating branch and PR …"):
# Create the branch via the API
import requests as req # noqa: PLC0415
from app.utils.helpers import _get_repo_slug, _GITHUB_API # noqa: PLC0415
slug = _get_repo_slug()
headers = {
"Authorization": f"token {gh_token}",
"Accept": "application/vnd.github+json",
}
# Get main branch SHA
r = req.get(
f"{_GITHUB_API}/repos/{slug}/git/ref/heads/main",
headers=headers, timeout=15,
)
if r.status_code != 200:
st.error(f"Cannot resolve main branch: {r.status_code}")
else:
main_sha = r.json()["object"]["sha"]
# Create or update the PR branch
r = req.post(
f"{_GITHUB_API}/repos/{slug}/git/refs",
headers=headers, timeout=15,
json={"ref": f"refs/heads/{pr_branch}", "sha": main_sha},
)
# 422 = branch already exists; update it instead
if r.status_code == 422:
req.patch(
f"{_GITHUB_API}/repos/{slug}/git/refs/heads/{pr_branch}",
headers=headers, timeout=15,
json={"sha": main_sha, "force": True},
)
# Commit data to the PR branch
result = github_commit_data(
gh_token, files, msg, branch=pr_branch,
)
if not result["ok"]:
st.error(f"Commit failed: {result['error']}")
else:
# Create the PR
pr_result = github_create_pr(
gh_token,
head_branch=pr_branch,
base_branch="main",
title=f"chore: update masq-infra data [{date_str}]",
body=(
"Pipeline data update from Streamlit Cloud.\n\n"
"Review the changes, then merge to publish "
"to the live site."
),
)
if pr_result["ok"]:
st.success(f"PR created: {pr_result['url']}")
else:
st.error(f"PR failed: {pr_result['error']}")
+5
View File
@@ -128,6 +128,11 @@ if not secrets.get("urlscan") or not secrets.get("mb"):
st.warning(
"⚠️ URLSCAN_API_KEY or MB_API_KEY not configured. Collection will be limited."
)
if not secrets.get("github"):
st.warning(
"⚠️ GH_TOKEN not configured. Collected data cannot be published back to "
"the GitHub repo and will be lost when the app restarts."
)
# ── Route to selected tab ─────────────────────────────────────────────────────
if selected == "📡 Collection":
+187
View File
@@ -20,6 +20,8 @@ _KEY_MAP: dict[str, str] = {
"urlhaus": "URLHAUS_API_KEY",
"validin": "VALIDIN_API_KEY",
"anthropic": "ANTHROPIC_API_KEY",
"vt": "VT_API_KEY",
"github": "GH_TOKEN",
}
@@ -134,3 +136,188 @@ def run_script(script_path: str, env: dict) -> tuple[int, str, str]:
cwd=str(REPO_ROOT),
)
return result.returncode, result.stdout, result.stderr
# ---------------------------------------------------------------------------
# GitHub data persistence — commit collected data back to the repo via API
# ---------------------------------------------------------------------------
_GITHUB_API = "https://api.github.com"
def _get_repo_slug() -> str | None:
"""Derive 'owner/repo' from the git remote origin URL.
Falls back to the GITHUB_REPOSITORY env var (set on Streamlit Cloud
when the app is linked to a repo).
"""
slug = os.environ.get("GITHUB_REPOSITORY", "")
if slug:
return slug
# Try parsing from git remote
try:
result = subprocess.run(
["git", "remote", "get-url", "origin"],
capture_output=True, text=True, cwd=str(REPO_ROOT),
)
url = result.stdout.strip()
# Handle both HTTPS and SSH URLs
if "github.com" in url:
# https://github.com/owner/repo.git or git@github.com:owner/repo.git
import re # noqa: PLC0415
m = re.search(r"github\.com[:/](.+?)(?:\.git)?$", url)
if m:
return m.group(1)
except Exception:
pass
return None
def github_commit_data(
token: str,
files: dict[str, str],
message: str,
branch: str | None = None,
) -> dict:
"""Commit one or more files to the repo via the GitHub API.
This is the persistence mechanism for Streamlit Cloud: since the filesystem
is ephemeral, collected data must be pushed back to the repo so GitHub Pages
can serve it.
Args:
token: GitHub personal access token (GH_TOKEN from st.secrets).
files: Mapping of repo-relative paths to file contents.
message: Commit message.
branch: Target branch (defaults to repo default branch).
Returns:
Dict with 'ok' bool and either 'sha' or 'error' string.
"""
import base64 as b64 # noqa: PLC0415
slug = _get_repo_slug()
if not slug:
return {"ok": False, "error": "Could not determine GitHub repo slug"}
headers = {
"Authorization": f"token {token}",
"Accept": "application/vnd.github+json",
}
import requests as req # noqa: PLC0415
# Resolve default branch if none specified
if not branch:
r = req.get(f"{_GITHUB_API}/repos/{slug}", headers=headers, timeout=15)
if r.status_code != 200:
return {"ok": False, "error": f"Cannot fetch repo info: {r.status_code}"}
branch = r.json().get("default_branch", "main")
# Get the latest commit SHA on the target branch
r = req.get(
f"{_GITHUB_API}/repos/{slug}/git/ref/heads/{branch}",
headers=headers, timeout=15,
)
if r.status_code != 200:
return {"ok": False, "error": f"Cannot resolve branch '{branch}': {r.status_code}"}
base_sha = r.json()["object"]["sha"]
# Get the tree SHA of that commit
r = req.get(
f"{_GITHUB_API}/repos/{slug}/git/commits/{base_sha}",
headers=headers, timeout=15,
)
if r.status_code != 200:
return {"ok": False, "error": f"Cannot fetch commit: {r.status_code}"}
base_tree_sha = r.json()["tree"]["sha"]
# Create blobs for each file
tree_entries = []
for path, content in files.items():
r = req.post(
f"{_GITHUB_API}/repos/{slug}/git/blobs",
headers=headers, timeout=30,
json={"content": b64.b64encode(content.encode()).decode(), "encoding": "base64"},
)
if r.status_code != 201:
return {"ok": False, "error": f"Cannot create blob for {path}: {r.status_code}"}
blob_sha = r.json()["sha"]
tree_entries.append({
"path": path, "mode": "100644", "type": "blob", "sha": blob_sha,
})
# Create a new tree
r = req.post(
f"{_GITHUB_API}/repos/{slug}/git/trees",
headers=headers, timeout=30,
json={"base_tree": base_tree_sha, "tree": tree_entries},
)
if r.status_code != 201:
return {"ok": False, "error": f"Cannot create tree: {r.status_code}"}
new_tree_sha = r.json()["sha"]
# Create commit
r = req.post(
f"{_GITHUB_API}/repos/{slug}/git/commits",
headers=headers, timeout=30,
json={
"message": message,
"tree": new_tree_sha,
"parents": [base_sha],
},
)
if r.status_code != 201:
return {"ok": False, "error": f"Cannot create commit: {r.status_code}"}
new_commit_sha = r.json()["sha"]
# Update branch reference
r = req.patch(
f"{_GITHUB_API}/repos/{slug}/git/refs/heads/{branch}",
headers=headers, timeout=15,
json={"sha": new_commit_sha},
)
if r.status_code != 200:
return {"ok": False, "error": f"Cannot update ref: {r.status_code}"}
return {"ok": True, "sha": new_commit_sha}
def github_create_pr(
token: str,
head_branch: str,
base_branch: str = "main",
title: str = "",
body: str = "",
) -> dict:
"""Create a pull request via the GitHub API.
Returns dict with 'ok' bool and either 'url' or 'error'.
"""
import requests as req # noqa: PLC0415
slug = _get_repo_slug()
if not slug:
return {"ok": False, "error": "Could not determine GitHub repo slug"}
headers = {
"Authorization": f"token {token}",
"Accept": "application/vnd.github+json",
}
r = req.post(
f"{_GITHUB_API}/repos/{slug}/pulls",
headers=headers, timeout=30,
json={
"title": title,
"body": body,
"head": head_branch,
"base": base_branch,
},
)
if r.status_code == 201:
return {"ok": True, "url": r.json().get("html_url", "")}
if r.status_code == 422:
# PR already exists
return {"ok": True, "url": "(PR already exists for this branch)"}
return {"ok": False, "error": f"PR creation failed: {r.status_code} {r.text[:500]}"}