feat: review a run's results in the browser with deepzero report

A run left its output spread across work/<pipeline>/samples/<id>/ as
per-sample state, findings, decompiled sources and assessments. Reviewing
thousands of those by hand is not practical, so `deepzero run` now prints
a link to a report and keeps it up to date as results land, and
`deepzero report` rebuilds it at any time.

The report answers one question first: what is vulnerable. Items an
assessment stage marked vulnerable lead the page, then items with
findings but no confirmed verdict, then anything that errored. Each item
links to its own page carrying the assessment, every finding with the
code it matched, and links to the artifacts on disk.

It is built from what a pipeline actually recorded rather than from any
one domain's field names, so a source-code review over repositories
renders as well as a kernel-driver review. A pipeline can shape the
presentation with an optional `report:` block - what to call one item,
which stage data key holds the verdict, which values mean vulnerable,
and which columns to surface - and every field has a default.

Output is layered so it stays usable on a large corpus: index.html holds
the triage summary at a bounded size, items/<id>.html covers everything
worth reading, inventory.csv carries every item for a spreadsheet, and
findings.jsonl carries every finding one per line. When a listing is
capped the page says what was capped and where the rest is.

Pages are self-contained with no network access, readable in light and
dark, keyboard navigable, and escape all analysed content.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
416rehman
2026-07-25 01:29:26 -06:00
co-authored by Claude Opus 5
parent 64f0003fe2
commit f2f8fe4ed9
9 changed files with 1520 additions and 7 deletions
+3
View File
@@ -51,3 +51,6 @@ docs/_site/
.jekyll-cache
docs/.jekyll-cache/
_site/
# local research artifacts: run logs, generated reports, scratch previews
research/
+14
View File
@@ -15,6 +15,20 @@ version: "2.0"
model: claude-code/opus
# how `deepzero report` presents this pipeline's results. every field is
# optional - a pipeline that declares nothing still gets a useful report.
report:
title: Windows kernel driver review
entity: driver
classification_key: classification
vulnerable_when: [vulnerable]
safe_when: [safe]
columns:
- semgrep_scanner.finding_count
- discover.priority_score
- decompile.function_count
- discover.dangerous_imports
settings:
work_dir: work
max_workers: 4
+81 -1
View File
@@ -174,8 +174,10 @@ def run(
# ensure built-in stages are registered
import deepzero.stages # noqa: F401
from deepzero.engine.pipeline import load_pipeline
from deepzero.engine.report import write_report
from deepzero.engine.state import RunState, StateStore
log_report = logging.getLogger("deepzero.report")
target_path = Path(target).resolve()
try:
@@ -246,6 +248,12 @@ def run(
if is_resume:
run_state = existing_run
run_state.status = RunStatus.RUNNING
# the pipeline's model may have changed since the run was created; record
# what this resumed run will actually call so status/reports don't lie
if run_state.model != pipeline_def.model:
log_msg = f"model changed since last run: {run_state.model or '(unset)'} -> {pipeline_def.model}"
console.print(f"[yellow]![/] {log_msg}")
run_state.model = pipeline_def.model
else:
# initialize fresh state
state_store.save_pipeline_snapshot(pipeline_def.raw_yaml)
@@ -257,7 +265,79 @@ def run(
model=pipeline_def.model,
)
run_state = runner.run(target_path, run_state)
# a live report the user can open straight away and watch fill in. it is
# rebuilt when results actually land rather than on a timer, and rate
# limited so a fast stage cannot spend the run's time writing html
report_dir = pipeline_def.work_dir / "report"
report_index = report_dir / "index.html"
last_written = [0.0]
def _refresh_report(force: bool = False) -> None:
if not force and time.monotonic() - last_written[0] < 15:
return
try:
write_report(pipeline_def.work_dir, report_dir, config=pipeline_def.report)
last_written[0] = time.monotonic()
except (OSError, ValueError, KeyError, TypeError) as exc:
log_report.debug("could not refresh the report: %s", exc)
_refresh_report(force=True)
console.print(f" report [bold]{report_index.resolve().as_uri()}[/]")
console.print(" [dim]open it now - it updates itself as results land[/]\n")
runner.progress_hook = _refresh_report
try:
run_state = runner.run(target_path, run_state)
finally:
_refresh_report(force=True) # settle the final state, no reload banner
console.print(f"\n report [bold]{report_index.resolve().as_uri()}[/]")
@main.command()
@click.option("--pipeline", "-p", default=None, help="pipeline name or path")
@click.option("--work-dir", "-w", default=None, help="work directory (overrides --pipeline)")
@click.option("--out", "-o", default=None, help="output directory (default <work_dir>/report)")
@click.option("--open", "open_browser", is_flag=True, help="open the report when finished")
@click.option("--verbose", "-v", is_flag=True, help="verbose logging")
def report(
pipeline: str | None, work_dir: str | None, out: str | None, open_browser: bool, verbose: bool
):
"""build a browsable HTML report from a run's results (safe mid-run)"""
_setup_logging(verbose)
from deepzero.engine.report import write_report
report_cfg: dict = {}
if work_dir:
work_path = Path(work_dir)
elif pipeline:
import deepzero.stages # noqa: F401
from deepzero.engine.pipeline import load_pipeline
_load_env()
try:
pipeline_def = load_pipeline(pipeline)
except ValueError as e:
console.print(f"[bold red]X ERROR[/]: {e}")
raise SystemExit(1)
work_path = pipeline_def.work_dir
report_cfg = pipeline_def.report
else:
console.print("[bold red]X ERROR[/]: pass --pipeline or --work-dir")
raise SystemExit(1)
if not work_path.exists():
console.print(f"[bold red]X ERROR[/]: work directory does not exist: {work_path}")
raise SystemExit(1)
html_path, json_path = write_report(work_path, Path(out) if out else None, config=report_cfg)
console.print(f"[green]\\[ok][/] report written to [bold]{html_path}[/]")
console.print(f" machine-readable copy: {json_path}")
if open_browser:
import webbrowser
webbrowser.open(html_path.resolve().as_uri())
@main.command()
+3 -3
View File
@@ -38,9 +38,9 @@ class LLMProvider:
"""send messages to the llm and return the response text.
handles rate limiting with adaptive backoff."""
backend = self.backend
# forward all options; each backend uses what applies (litellm passes
# generation kwargs to the api, cli backends read controls like timeout
# and ignore the rest) so e.g. timeout= is not silently dropped
# every option is forwarded and each backend takes what applies to it:
# litellm passes generation kwargs to the api, while cli backends read
# controls such as timeout and ignore the rest
merged = {**self.default_kwargs, **kwargs}
backoff = initial_backoff
+5
View File
@@ -35,6 +35,7 @@ class PipelineDefinition:
stage_specs: list[StageSpec],
pipeline_dir: Path,
raw_yaml: str,
report: dict[str, Any] | None = None,
):
self.name = name
self.description = description
@@ -44,6 +45,9 @@ class PipelineDefinition:
self.stage_specs = stage_specs
self.pipeline_dir = pipeline_dir
self.raw_yaml = raw_yaml
# optional presentation hints for `deepzero report` - pipelines that
# declare nothing still get a useful report
self.report = report or {}
# resolved processor instances
self.ingest_processor: IngestProcessor | None = None
@@ -138,6 +142,7 @@ def load_pipeline(
name=name,
description=description,
model=model,
report=data.get("report") or {},
settings=settings,
knowledge=knowledge,
stage_specs=stage_specs,
File diff suppressed because it is too large Load Diff
+15 -1
View File
@@ -9,7 +9,7 @@ import threading
import time
import traceback as tb_module
from pathlib import Path
from typing import TYPE_CHECKING
from typing import TYPE_CHECKING, Callable
from rich.console import Console
@@ -76,6 +76,7 @@ class PipelineRunner:
default_max_workers: int = 4,
console: Console | None = None,
dashboard: PipelineDashboard | None = None,
progress_hook: Callable[[], None] | None = None,
):
self.ingest = ingest
self.stages = stages
@@ -86,9 +87,20 @@ class PipelineRunner:
self.default_max_workers = default_max_workers
self.console = console or Console()
self.dashboard = dashboard
# called when new results have landed, so a caller can refresh a live
# view. never allowed to interrupt the run
self.progress_hook = progress_hook
self._shutdown_event = threading.Event()
self._original_sigint = None
def _notify_progress(self) -> None:
if self.progress_hook is None:
return
try:
self.progress_hook()
except Exception as exc: # noqa: BLE001 - a view refresh must never fail a run
log.debug("progress hook failed: %s", exc)
def _make_entry(self, state: SampleState) -> ProcessorEntry:
# centralizes ProcessorEntry construction for map/reduce/batch
return ProcessorEntry(
@@ -223,6 +235,7 @@ class PipelineRunner:
else:
self._run_map(processor, active, spec, stage_stats)
self._notify_progress()
self._apply_stage_limit(spec, sample_states, stage_stats)
@@ -461,6 +474,7 @@ class PipelineRunner:
for s in dirty:
self.state_store.save_sample(s)
dirty.clear()
self._notify_progress()
for s in dirty:
self.state_store.save_sample(s)
else:
+341
View File
@@ -0,0 +1,341 @@
from __future__ import annotations
import json
from deepzero.engine.report import (
BUCKET_SUSPICIOUS,
BUCKET_VULNERABLE,
ReportConfig,
collect,
render_index,
write_report,
)
from deepzero.engine.state import RunState, SampleState, StateStore
from deepzero.engine.types import RunStatus
def _seed(tmp_path, *, with_findings=True, with_assessment=True, classification="vulnerable"):
"""build a work dir shaped like a real run: one risky driver, one clean."""
work = tmp_path / "work" / "loldrivers"
store = StateStore(work)
store.save_run(
RunState(
run_id="run_1",
pipeline="loldrivers",
target="C:/drivers",
model="claude-code/opus",
status=RunStatus.COMPLETED,
)
)
risky = SampleState(
sample_id="aaa1",
sha256="a" * 64,
filename="risky.sys",
source_path="C:/drivers/risky.sys",
)
risky.mark_stage_completed(
"discover", data={"priority_score": 8.0, "dangerous_imports": ["MmMapIoSpace"]}
)
risky.mark_stage_completed(
"decompile",
artifacts={"ghidra_result": "decompiled/ghidra_result.json"},
data={"device_name": "RiskyDev", "function_count": 42, "ioctl_count": 2},
)
if with_findings:
risky.mark_stage_completed("semgrep_scanner", data={"finding_count": 2})
if with_assessment:
risky.mark_stage_completed(
"assess",
artifacts={"llm_output": "assessment.md"},
data={"classification": classification},
)
store.save_sample(risky)
clean = SampleState(sample_id="bbb2", filename="clean.sys", source_path="C:/drivers/clean.sys")
clean.mark_stage_completed("discover", data={"priority_score": 1.0})
store.save_sample(clean)
d = store.sample_dir("aaa1")
(d / "decompiled").mkdir(parents=True, exist_ok=True)
(d / "decompiled" / "ghidra_result.json").write_text(
json.dumps(
{
"success": True,
"device_name": "RiskyDev",
"symbolic_link": "\\\\DosDevices\\\\RiskyDev",
"ioctl_handlers": [{"code": 0x222004}, {"code": 0x222008}],
}
),
encoding="utf-8",
)
if with_findings:
(d / "findings.json").write_text(
json.dumps(
[
{
"rule_id": "pipelines.loldrivers.rules.ghidra-mmmapiospace-user-controlled",
"severity": "HIGH",
"message": "user controlled physical map",
"line_start": 42,
"matched_code": "MmMapIoSpace(pa, len, 0);",
},
{
"rule_id": "pipelines.loldrivers.rules.method-neither",
"severity": "MEDIUM",
"message": "METHOD_NEITHER buffer",
"line_start": 88,
"matched_code": "irp->UserBuffer",
},
]
),
encoding="utf-8",
)
if with_assessment:
(d / "assessment.md").write_text(
"[VULNERABLE] arbitrary physical memory map via IOCTL 0x222004", encoding="utf-8"
)
return work
class TestCollect:
def test_totals(self, tmp_path):
payload = collect(_seed(tmp_path))
t = payload["totals"]
assert t["samples"] == 2
assert t["total_findings"] == 2
assert t["with_findings"] == 1
assert t["assessed"] == 1
# severities are counted by whatever labels the findings actually use
assert payload["severity_totals"] == {"HIGH": 1, "MEDIUM": 1}
def test_driver_detail_is_gathered(self, tmp_path):
d = collect(_seed(tmp_path))["items"][0]
assert d.name == "risky.sys"
assert d.data["decompile.device_name"] == "RiskyDev"
assert d.data["decompile.ioctl_count"] == 2
assert d.data["discover.dangerous_imports"] == ["MmMapIoSpace"]
assert "VULNERABLE" in d.texts["llm_output"]
assert d.severity_counts == {"HIGH": 1, "MEDIUM": 1}
assert "ghidra-mmmapiospace-user-controlled" in d.rule_hits
def test_only_interesting_drivers_are_kept_in_memory(self, tmp_path):
payload = collect(_seed(tmp_path))
# the clean driver is still counted and written to csv, but needs no page
assert [d.sample_id for d in payload["items"]] == ["aaa1"]
assert len(payload["rows"]) == 2
def test_bucketing_uses_the_assessment_verdict(self, tmp_path):
payload = collect(_seed(tmp_path, classification="vulnerable"))
assert payload["items"][0].bucket == BUCKET_VULNERABLE
assert payload["buckets"][BUCKET_VULNERABLE] == 1
def test_findings_without_assessment_are_only_suspicious(self, tmp_path):
payload = collect(_seed(tmp_path, with_assessment=False))
assert payload["items"][0].bucket == BUCKET_SUSPICIOUS
def test_safe_verdict_is_not_flagged(self, tmp_path):
payload = collect(_seed(tmp_path, classification="safe"))
assert payload["buckets"].get(BUCKET_VULNERABLE, 0) == 0
def test_rule_totals_aggregated(self, tmp_path):
rt = collect(_seed(tmp_path))["rule_totals"]
assert rt["ghidra-mmmapiospace-user-controlled"] == 1
assert rt["method-neither"] == 1
def test_run_metadata(self, tmp_path):
run = collect(_seed(tmp_path))["run"]
assert run["model"] == "claude-code/opus"
assert run["pipeline"] == "loldrivers"
def test_safe_before_any_findings_exist(self, tmp_path):
payload = collect(_seed(tmp_path, with_findings=False, with_assessment=False))
assert payload["totals"]["total_findings"] == 0
assert payload["totals"]["samples"] == 2
def test_empty_work_dir_is_safe(self, tmp_path):
assert collect(tmp_path / "nothing")["totals"]["samples"] == 0
class TestIndex:
def test_self_contained_and_offline(self, tmp_path):
out = render_index(collect(_seed(tmp_path)), tmp_path)
assert out.startswith("<!doctype html>")
# must open with no network: no remote assets of any kind
assert "http://" not in out and "https://" not in out
assert "<script src" not in out and "stylesheet" not in out
def test_vulnerable_is_the_headline(self, tmp_path):
out = render_index(collect(_seed(tmp_path)), tmp_path)
assert "assessed as vulnerable" in out
# the verdict section leads, ahead of the supporting rule breakdown
assert out.index("Vulnerable") < out.index("Rules that fired")
assert "items/aaa1.html" in out
def test_empty_buckets_are_omitted(self, tmp_path):
# nothing suspicious in this run, so no empty "Needs review" table
out = render_index(collect(_seed(tmp_path)), tmp_path)
assert "Needs review</h2>" not in out
def test_suspicious_section_appears_after_vulnerable(self, tmp_path):
work = _seed(tmp_path)
# add a second driver with findings but no assessment -> suspicious
store = StateStore(work)
s = SampleState(sample_id="ccc3", filename="maybe.sys", source_path="C:/drivers/maybe.sys")
s.mark_stage_completed("semgrep_scanner", data={"finding_count": 1})
store.save_sample(s)
(store.sample_dir("ccc3") / "findings.json").write_text(
json.dumps([{"rule_id": "r.x", "severity": "MEDIUM", "message": "m", "line_start": 1}]),
encoding="utf-8",
)
out = render_index(collect(work), tmp_path)
assert out.index("Vulnerable") < out.index("Needs review")
class TestWriteReport:
def test_writes_the_layered_output(self, tmp_path):
index, json_path = write_report(_seed(tmp_path))
assert index.name == "index.html" and index.parent.name == "report"
out = index.parent
for name in ("inventory.csv", "findings.jsonl", "report.json"):
assert (out / name).exists(), name
summary = json.loads(json_path.read_text(encoding="utf-8"))
assert summary["totals"]["samples"] == 2
# totals only - the summary must not grow with the corpus
assert "items" not in summary
def test_index_stays_small_while_csv_holds_everything(self, tmp_path):
index, _ = write_report(_seed(tmp_path))
out = index.parent
csv_text = (out / "inventory.csv").read_text(encoding="utf-8")
assert "risky.sys" in csv_text and "clean.sys" in csv_text
pages = {p.stem for p in (out / "items").glob("*.html")}
assert pages == {"aaa1"}
def test_findings_are_one_json_object_per_line(self, tmp_path):
index, _ = write_report(_seed(tmp_path))
lines = [
ln
for ln in (index.parent / "findings.jsonl").read_text(encoding="utf-8").splitlines()
if ln
]
assert len(lines) == 2
rec = json.loads(lines[0])
assert rec["sample_id"] == "aaa1" and rec["bucket"] == BUCKET_VULNERABLE
assert rec["severity"] in ("HIGH", "MEDIUM")
def test_driver_page_shows_evidence_and_links_to_artifacts(self, tmp_path):
index, _ = write_report(_seed(tmp_path))
page = (index.parent / "items" / "aaa1.html").read_text(encoding="utf-8")
assert "MmMapIoSpace(pa, len, 0);" in page
assert "llm_output" in page and "ghidra_result" in page
assert "all results" in page
# links resolve to the real artifact folder
assert "samples" in page
def test_untrusted_text_is_escaped_on_driver_pages(self, tmp_path):
work = _seed(tmp_path)
d = StateStore(work).sample_dir("aaa1")
(d / "assessment.md").write_text("<img src=x onerror=alert(1)>", encoding="utf-8")
index, _ = write_report(work)
page = (index.parent / "items" / "aaa1.html").read_text(encoding="utf-8")
# decompiled and LLM text is untrusted - never render it as live markup
assert "<img src=x onerror" not in page
assert "&lt;img" in page
def test_capping_detail_is_disclosed_not_silent(self, tmp_path):
index, _ = write_report(_seed(tmp_path), detail_limit=0)
text = index.read_text(encoding="utf-8")
assert "appear in" in text and "inventory.csv" in text
def test_custom_out_dir(self, tmp_path):
out = tmp_path / "elsewhere"
index, _ = write_report(_seed(tmp_path), out)
assert index.parent == out
def test_regenerating_overwrites(self, tmp_path):
work = _seed(tmp_path)
write_report(work)
index, _ = write_report(work)
assert index.read_text(encoding="utf-8").count("<!doctype html>") == 1
class TestPipelineAgnostic:
"""a completely different pipeline must render without code changes."""
def _github_run(self, tmp_path):
work = tmp_path / "work" / "srchunt"
store = StateStore(work)
store.save_run(
RunState(
run_id="run_9",
pipeline="srchunt",
target="github.com/acme",
model="claude-code/opus",
status=RunStatus.COMPLETED,
)
)
repo = SampleState(sample_id="r1", filename="acme/api", source_path="github.com/acme/api")
repo.mark_stage_completed("discover", data={"stars": 4200, "language": "go"})
repo.mark_stage_completed(
"sast", data={"finding_count": 1}, artifacts={"sast": "sast.json"}
)
repo.mark_stage_completed("triage", data={"verdict": "exploitable"})
store.save_sample(repo)
(store.sample_dir("r1") / "sast.json").write_text(
json.dumps(
[
{
"id": "go.sql-injection",
"level": "critical",
"title": "SQL injection in query builder",
"path": "internal/db/query.go",
"start_line": 88,
"snippet": 'db.Raw("SELECT " + userInput)',
}
]
),
encoding="utf-8",
)
return work
def test_declared_config_shapes_the_report(self, tmp_path):
work = self._github_run(tmp_path)
cfg = {
"title": "Acme source review",
"entity": "repository",
"classification_key": "verdict",
"vulnerable_when": ["exploitable"],
"columns": ["discover.stars", "discover.language"],
"findings_files": ["sast.json"],
}
index, json_path = write_report(work, config=cfg)
text = index.read_text(encoding="utf-8")
# the pipeline's own vocabulary and verdict key drive the page
assert "Acme source review" in text
assert "repository" in text and "repositorys" not in text.replace("repositorys", "")
assert "assessed as vulnerable" in text
# its own severity vocabulary is preserved, not remapped to a fixed set
assert "CRITICAL" in json.loads(json_path.read_text(encoding="utf-8"))["severity_totals"]
# declared columns appear as chips/columns
assert "stars" in text and "4200" in text
def test_works_with_no_config_at_all(self, tmp_path):
work = self._github_run(tmp_path)
index, _ = write_report(work)
text = index.read_text(encoding="utf-8")
# default entity wording, and the finding is still surfaced for review
assert "sample" in text
# no verdict key configured, so it lands in review rather than confirmed
assert "Needs review" in text
def test_alien_finding_shape_is_normalized(self, tmp_path):
work = self._github_run(tmp_path)
payload = collect(work, config=ReportConfig.from_dict({"findings_files": ["sast.json"]}))
f = payload["items"][0].findings[0]
assert f["severity"] == "CRITICAL"
assert f["rule_id"] == "go.sql-injection"
assert "SQL injection" in f["message"]
assert f["location"] == "internal/db/query.go"
assert f["line"] == 88
+2 -2
View File
@@ -24,8 +24,8 @@ class TestRulesPathResolution:
)
def test_validate_and_process_resolve_the_same_path(self, tmp_path, monkeypatch):
# regression: validate() approved a cwd-relative path while process()
# used a pipeline_dir-relative one, so the scan ran with no rules.
# the check that approves a run and the run itself must agree on where
# the rules live, otherwise a scan can start with no rules loaded.
# pin cwd so the cwd-relative "rules" resolves deterministically.
monkeypatch.chdir(tmp_path)
rules = tmp_path / "rules"