diff --git a/README.md b/README.md index 9dd03239..9f00dfc2 100644 --- a/README.md +++ b/README.md @@ -169,6 +169,7 @@ private. - Watch runs live, including per-repetition results - Monitor models for drift on a schedule (interval or cron) - Filter, customise and compare runs on an interactive dashboard +- Visualize results: browse a folder of `simpleaudit` JSON results in a file-tree viewer, drag-drop a file offline, export a self-contained HTML file, and compare runs with fragility metrics. Run Studio as a visualization-only server with `spin --visualize-only --results_dir ./results` ## ๐Ÿ—๏ธ Architecture diff --git a/audits/comparison.py b/audits/comparison.py index 2d4410dc..55c92bdc 100644 --- a/audits/comparison.py +++ b/audits/comparison.py @@ -8,10 +8,44 @@ """ from __future__ import annotations +import math + from audits.events import ScenarioResult from audits.models import AuditRun from audits.services import frozen_judge, frozen_name +# Ordinal scale for fragility metrics (matches the SimpleAudit visualizer). +_SEV_ORDER = ["pass", "low", "medium", "high", "critical"] +_SEV_RANK = {sev: i for i, sev in enumerate(_SEV_ORDER)} + + +def _fragility(severities: list[str]) -> dict | None: + """Fragility metrics for one scenario across runs (ported from the visualizer). + + ``agreement`` is the mode frequency, ``entropy`` the normalized Shannon + entropy of the severity distribution, and ``spread`` the ordinal std dev of + the severities. ``mode`` is the most common severity. Returns None when + there is nothing to measure (fewer than 2 valid severities). + """ + ranks = [_SEV_RANK[s] for s in severities if s in _SEV_RANK] + if len(ranks) < 2: + return None + n = len(ranks) + counts: dict[str, int] = {} + for s in severities: + if s in _SEV_RANK: + counts[s] = counts.get(s, 0) + 1 + mode = max(counts.items(), key=lambda kv: (kv[1], -_SEV_RANK[kv[0]]))[0] + agreement = counts[mode] / n + entropy = 0.0 + for c in counts.values(): + p = c / n + entropy -= p * math.log2(p) + entropy = entropy / math.log2(n) if n > 1 else 0.0 + mean = sum(ranks) / n + spread = math.sqrt(sum((r - mean) ** 2 for r in ranks) / n) + return {"agreement": round(agreement, 4), "entropy": round(entropy, 4), "spread": round(spread, 3), "mode": mode} + class ComparisonIncompatible(Exception): """Raised when runs cannot be meaningfully compared.""" @@ -116,13 +150,18 @@ def compare_runs(project, run_ids: list[int]) -> dict: results = [] for key in sorted(common_keys): entry = {"scenario_key": key, "runs": {}} + severities = [] for run in runs: r = run_results.get(run.id, {}).get(key) - entry["runs"][str(run.id)] = { + cell = { "name": run.name, "target": frozen_name(run, "target"), **(r or {"status": "missing", "severity": None}), } + entry["runs"][str(run.id)] = cell + if cell.get("severity"): + severities.append(cell["severity"]) + entry["fragility"] = _fragility(severities) results.append(entry) # Build inputs comparison: key parameters that differ between runs diff --git a/config/urls.py b/config/urls.py index c0714a6f..89cb1cad 100644 --- a/config/urls.py +++ b/config/urls.py @@ -60,6 +60,15 @@ auto_login_view, logout_view, ) +from infra.visualizer import ( + RunExportHtmlView, + ScenarioViewerView, + VisualizerAuthView, + VisualizerFilesView, + VisualizerImageView, + VisualizerJsonView, + VisualizerView, +) from judges.views import JudgeDetailView, JudgePreviewView, JudgesView from model_registry import otlp_config, otlp_views @@ -203,6 +212,7 @@ def home_view(request, *args, **kwargs): path("runs//rename/", RunRenameView.as_view(), name="run_rename"), path("runs//results//", RunResultView.as_view(), name="run_result"), path("runs//export/", RunExportView.as_view(), name="run_export"), + path("runs//export-html/", RunExportHtmlView.as_view(), name="run_export_html"), path("runs//script/", RunScriptView.as_view(), name="run_script"), path("judges//script/", JudgeScriptView.as_view(), name="judge_script"), path("runs//results-fragment/", RunResultsFragmentView.as_view(), name="run_results_fragment"), @@ -210,6 +220,18 @@ def home_view(request, *args, **kwargs): path("runs/bulk/", RunsBulkView.as_view(), name="runs_bulk"), path("runs/export.csv", RunsExportView.as_view(), name="runs_export"), path("me/preferences/", PreferenceView.as_view(), name="preferences"), + # Result visualizer โ€” the absorbed SimpleAudit single-file HTML viewer. + # The SPA pages and their file-tree/image APIs read a local results dir + # (set via --results_dir); the drag-drop page is fully client-side. + path("visualizer/", VisualizerView.as_view(), name="visualizer"), + path("visualizer/upload/", ScenarioViewerView.as_view(), name="visualizer_upload"), + # The visualizer SPA hardcodes "/api/*" in its fetch paths and prefixes them + # with window.__VISUALIZER_API_BASE (set to "/api/visualizer"), so the full + # URL is /api/visualizer/api/. Keep the routes in that shape. + path("api/visualizer/api/auth/", VisualizerAuthView.as_view(), name="visualizer_auth"), + path("api/visualizer/api/files/", VisualizerFilesView.as_view(), name="visualizer_files"), + path("api/visualizer/api/json/", VisualizerJsonView.as_view(), name="visualizer_json"), + path("api/visualizer/api/image/", VisualizerImageView.as_view(), name="visualizer_image"), ] # Optional Open WebUI module (the `chat` app). The chat UI itself lives on its diff --git a/infra/tests/test_comparison.py b/infra/tests/test_comparison.py index 3eb7214c..e1df6478 100644 --- a/infra/tests/test_comparison.py +++ b/infra/tests/test_comparison.py @@ -128,3 +128,31 @@ def test_intersection_excludes_missing_scenarios(self): result = compare_runs(self.project, [r1.id, r2.id]) self.assertEqual(result["intersection_count"], 1) self.assertEqual(result["results"][0]["scenario_key"], "scen-0") + + def test_fragility_unanimous(self): + r1 = _make_run(self.project, self.ver, self.ep_a, self.judge, "Run 1") + r2 = _make_run(self.project, self.ver, self.ep_b, self.judge, "Run 2") + self._add_results(r1, ["pass", "high"]) + self._add_results(r2, ["pass", "high"]) + result = compare_runs(self.project, [r1.id, r2.id]) + by_key = {e["scenario_key"]: e for e in result["results"]} + # Unanimous: 100% agreement, zero entropy, zero spread. + self.assertEqual(by_key["scen-0"]["fragility"]["agreement"], 1.0) + self.assertEqual(by_key["scen-0"]["fragility"]["entropy"], 0.0) + self.assertEqual(by_key["scen-0"]["fragility"]["spread"], 0.0) + self.assertEqual(by_key["scen-0"]["fragility"]["mode"], "pass") + + def test_fragility_disagreement(self): + r1 = _make_run(self.project, self.ver, self.ep_a, self.judge, "Run 1") + r2 = _make_run(self.project, self.ver, self.ep_b, self.judge, "Run 2") + self._add_results(r1, ["pass", "high"]) + self._add_results(r2, ["critical", "pass"]) + result = compare_runs(self.project, [r1.id, r2.id]) + by_key = {e["scenario_key"]: e for e in result["results"]} + # scen-0: pass vs critical -> 50% agreement, max entropy, large spread. + f0 = by_key["scen-0"]["fragility"] + self.assertEqual(f0["agreement"], 0.5) + self.assertEqual(f0["entropy"], 1.0) + self.assertGreater(f0["spread"], 1.0) + # scen-1: high vs pass -> also disagreement. + self.assertIsNotNone(by_key["scen-1"]["fragility"]) diff --git a/infra/tests/test_visualizer.py b/infra/tests/test_visualizer.py new file mode 100644 index 00000000..a3e74fae --- /dev/null +++ b/infra/tests/test_visualizer.py @@ -0,0 +1,215 @@ +"""Result visualizer endpoints: file tree, JSON/image APIs, and HTML export. + +Run: + SIMPLEAUDIT_LOCAL_SQLITE=1 uv run manage.py test infra.tests.test_visualizer +""" +import json +import os + +from django.test import Client, TestCase + +from infra.tests.factories import ( + AuditRunFactory, + MembershipFactory, + ProjectFactory, + ScenarioResultFactory, + UserFactory, +) +from infra.visualizer import ( + build_standalone_html, + is_valid_audit_data, + set_results_dir, +) + + +def _valid_single(): + return {"results": [{"scenario_name": "s1", "severity": "pass", "summary": "ok"}]} + + +def _valid_experiment(): + return { + "runs": { + "model-a": [ + {"results": [{"scenario_name": "s1", "severity": "high", "summary": "bad"}]} + ] + } + } + + +class _VisualizerBase(TestCase): + def setUp(self): + self.user = UserFactory() + self.user.set_password("pw") + self.user.save() + self.project = ProjectFactory() + MembershipFactory(user=self.user, project=self.project, role="admin") + self.client = Client() + self.client.login(username=self.user.username, password="pw") + + +class VisualizerShapeTests(TestCase): + def test_is_valid_audit_data_shapes(self): + self.assertTrue(is_valid_audit_data(_valid_single())) + self.assertTrue(is_valid_audit_data(_valid_experiment())) + self.assertTrue(is_valid_audit_data([{"scenario_name": "s", "severity": "pass"}])) + self.assertFalse(is_valid_audit_data({"foo": "bar"})) + self.assertFalse(is_valid_audit_data([])) + self.assertFalse(is_valid_audit_data(None)) + + def test_build_standalone_html_inlines_data(self): + html = build_standalone_html(_valid_single(), "my run") + self.assertIn("window.__inlinedData", html) + self.assertIn("window.__standaloneMode", html) + self.assertIn("my run", html) + # The payload must not break out of the script tag. + self.assertNotIn("", html.replace("<", "<")) + self.assertIn("scenario_name", html) + + def test_build_standalone_html_rejects_invalid(self): + with self.assertRaises(ValueError): + build_standalone_html({"foo": "bar"}, "x") + + +class VisualizerPageTests(_VisualizerBase): + def test_visualizer_page_renders_and_injects_api_base(self): + resp = self.client.get("/visualizer", follow=True) + self.assertEqual(resp.status_code, 200) + self.assertContains(resp, "window.__VISUALIZER_API_BASE = '/api/visualizer'") + self.assertContains(resp, "SimpleAudit Result Visualizer") + + def test_scenario_viewer_page_renders(self): + resp = self.client.get("/visualizer/upload", follow=True) + self.assertEqual(resp.status_code, 200) + self.assertContains(resp, "file-drop-zone") + + def test_requires_login(self): + self.client.logout() + resp = self.client.get("/visualizer", follow=True) + self.assertEqual(resp.status_code, 200) + self.assertContains(resp, "Sign in") + + +class VisualizerFilesTests(_VisualizerBase): + def setUp(self): + super().setUp() + self.results = self._make_results_dir() + set_results_dir(self.results) + self.addCleanup(set_results_dir, None) + + def _make_results_dir(self): + import tempfile + + d = tempfile.mkdtemp(prefix="visz_") + self.addCleanup(_rmtree, d) + os.makedirs(os.path.join(d, "sub"), exist_ok=True) + with open(os.path.join(d, "single.json"), "w") as f: + json.dump(_valid_single(), f) + with open(os.path.join(d, "experiment.json"), "w") as f: + json.dump(_valid_experiment(), f) + with open(os.path.join(d, "sub", "nested.json"), "w") as f: + json.dump(_valid_single(), f) + with open(os.path.join(d, "not_audit.json"), "w") as f: + json.dump({"foo": "bar"}, f) + with open(os.path.join(d, "readme.txt"), "w") as f: + f.write("not json") + return d + + def test_files_tree_lists_valid_files_and_folders(self): + resp = self.client.get("/api/visualizer/api/files", follow=True) + self.assertEqual(resp.status_code, 200) + tree = resp.json()["tree"] + names = {item["name"]: item for item in tree} + self.assertIn("single.json", names) + self.assertEqual(names["single.json"]["type"], "file") + self.assertIn("experiment.json", names) + self.assertEqual(names["experiment.json"]["type"], "experiment") + self.assertEqual(names["experiment.json"]["models"], ["model-a"]) + self.assertIn("sub", names) + self.assertEqual(names["sub"]["type"], "folder") + # Non-audit JSON and non-JSON files are excluded. + self.assertNotIn("not_audit.json", names) + self.assertNotIn("readme.txt", names) + # The nested file shows up under the folder. + self.assertEqual([c["name"] for c in names["sub"]["children"]], ["nested.json"]) + + def test_files_without_results_dir(self): + set_results_dir(None) + resp = self.client.get("/api/visualizer/api/files", follow=True) + self.assertEqual(resp.status_code, 500) + + +class VisualizerJsonTests(_VisualizerBase): + def setUp(self): + super().setUp() + self.results = self._make_results_dir() + set_results_dir(self.results) + self.addCleanup(set_results_dir, None) + + def _make_results_dir(self): + import tempfile + + d = tempfile.mkdtemp(prefix="visz_json_") + self.addCleanup(_rmtree, d) + os.makedirs(os.path.join(d, "sub"), exist_ok=True) + with open(os.path.join(d, "single.json"), "w") as f: + json.dump(_valid_single(), f) + with open(os.path.join(d, "sub", "nested.json"), "w") as f: + json.dump(_valid_single(), f) + with open(os.path.join(d, "not_audit.json"), "w") as f: + json.dump({"foo": "bar"}, f) + return d + + def test_json_returns_valid_audit_file(self): + resp = self.client.get("/api/visualizer/api/json/single.json", follow=True) + self.assertEqual(resp.status_code, 200) + self.assertEqual(resp.json(), _valid_single()) + + def test_json_returns_nested_file(self): + resp = self.client.get("/api/visualizer/api/json/sub/nested.json", follow=True) + self.assertEqual(resp.status_code, 200) + self.assertEqual(resp.json(), _valid_single()) + + def test_json_rejects_path_traversal(self): + resp = self.client.get("/api/visualizer/api/json/../../etc/passwd", follow=True) + self.assertIn(resp.status_code, (403, 404)) + + def test_json_rejects_non_audit_file(self): + resp = self.client.get("/api/visualizer/api/json/not_audit.json", follow=True) + self.assertEqual(resp.status_code, 403) + + def test_json_404_when_missing(self): + resp = self.client.get("/api/visualizer/api/json/missing.json", follow=True) + self.assertEqual(resp.status_code, 404) + + +class VisualizerImageTests(_VisualizerBase): + def test_image_missing_uri(self): + resp = self.client.get("/api/visualizer/api/image", follow=True) + self.assertEqual(resp.status_code, 400) + + def test_image_non_image_uri(self): + resp = self.client.get("/api/visualizer/api/image/?uri=not-an-image://x", follow=True) + self.assertEqual(resp.status_code, 415) + + +class RunExportHtmlTests(_VisualizerBase): + def test_export_html_inlines_results(self): + self.run = AuditRunFactory(project=self.project, status="completed") + ScenarioResultFactory(run_id=self.run.id) + resp = self.client.get(f"/runs/{self.run.id}/export-html/") + self.assertEqual(resp.status_code, 200) + self.assertEqual(resp["Content-Type"], "text/html; charset=utf-8") + self.assertIn("attachment", resp["Content-Disposition"]) + self.assertIn("window.__inlinedData", resp.content.decode()) + self.assertIn("scenario_name", resp.content.decode()) + + def test_export_html_404_for_other_project(self): + other = AuditRunFactory(status="completed") + resp = self.client.get(f"/runs/{other.id}/export-html/") + self.assertEqual(resp.status_code, 404) + + +def _rmtree(path): + import shutil + + shutil.rmtree(path, ignore_errors=True) diff --git a/infra/ui.py b/infra/ui.py index a4c45917..671202f5 100644 --- a/infra/ui.py +++ b/infra/ui.py @@ -2082,7 +2082,13 @@ def _reshape_result(raw: dict) -> dict: for col_run_id in [str(r["id"]) for r in raw["runs"]]: rdata = entry["runs"].get(col_run_id, {}) values.append(rdata.get("severity") or rdata.get("status") or "โ€”") - rows.append({"scenario": entry["scenario_key"], "values": values}) + rows.append( + { + "scenario": entry["scenario_key"], + "values": values, + "fragility": entry.get("fragility"), + } + ) # Per-run header metadata (one query for all runs) run_objs = AuditRun.objects.select_related("target_model").in_bulk([r["id"] for r in raw["runs"]]) run_meta = [] diff --git a/infra/visualizer.py b/infra/visualizer.py new file mode 100644 index 00000000..6361e13d --- /dev/null +++ b/infra/visualizer.py @@ -0,0 +1,370 @@ +"""Result visualizer โ€” the SimpleAudit single-file HTML viewer, absorbed into Studio. + +This module ports the logic that used to live in the ``simpleaudit`` core's +FastAPI ``visualization/server.py`` (``simpleaudit serve`` / ``export-html``) +into thin Django views, and serves the two self-contained HTML assets +(``static/visualizer.html`` and ``static/scenario_viewer.html``) that carry all +the rendering, PDF export, image lightbox, and file-tree browsing. + +Two ways to get results in front of the viewer: + +* **Server-side** (primary): start Studio with ``--results_dir `` and the + file-tree endpoints read that local directory of JSON results. This is the + client use case โ€” point the server at a folder of dumped ``simpleaudit`` + results and browse them. +* **Client-side** (bonus): the drag-drop viewer page (``/visualizer/upload/``) + loads ``scenario_viewer.html``, which reads files straight from the browser + via the File System Access API โ€” no server round-trip. + +The standalone HTML export (``/runs//export-html/``) inlines a run's JSON +into the visualizer template so the output opens in any browser with no server. +""" +import json +import logging +import os + +from django.conf import settings +from django.contrib.auth.mixins import LoginRequiredMixin +from django.http import Http404, HttpResponse, JsonResponse +from django.shortcuts import get_object_or_404 +from django.views.generic import View + +from infra.ui import ProjectMixin + +logger = logging.getLogger(__name__) + + +# --- results directory (set by the CLI from --results_dir) ------------------ +# Read at request time so the CLI can set it after import and tests can patch +# it per-test without touching a module global at import. +def results_dir() -> str | None: + """The configured results directory, or None when not set.""" + return getattr(settings, "VISUALIZER_RESULTS_DIR", None) + + +def set_results_dir(path: str | None) -> None: + """Set the results directory for the file-tree endpoints (CLI/tests).""" + settings.VISUALIZER_RESULTS_DIR = path + + +# --- audit-data shape detection (ported from simpleaudit core) --------------- +def _looks_like_audit_result(obj: object) -> bool: + return ( + isinstance(obj, dict) + and ("scenario_name" in obj or "name" in obj) + and "severity" in obj + ) + + +def _experiment_models(data: object) -> list[str]: + """Model labels in an experiment file that have at least one loadable run. + + A run is loadable when it is a dict holding a non-empty list of + audit-shaped results. Both the file tree and the JSON endpoint derive their + notion of "experiment" from this list, so the tree never shows an entry the + endpoint would refuse to serve. + """ + if not isinstance(data, dict): + return [] + runs = data.get("runs") + if not isinstance(runs, dict): + return [] + models = [] + for label, run_list in runs.items(): + entries = run_list if isinstance(run_list, list) else [run_list] + for entry in entries: + if ( + isinstance(entry, dict) + and isinstance(entry.get("results"), list) + and entry["results"] + and all(_looks_like_audit_result(item) for item in entry["results"]) + ): + models.append(label) + break + return models + + +def is_valid_audit_data(data) -> bool: + """Whether parsed JSON has the shape of audit results. + + Accepts the three shapes the visualizer renders: a list of results, a + ``{"results": [...]}`` object, or a multi-model experiment + ``{"runs": {"model": [{"results": [...]}]}}``. + """ + if isinstance(data, list): + return bool(data) and all(_looks_like_audit_result(item) for item in data) + if isinstance(data, dict) and "results" in data: + results = data["results"] + return ( + isinstance(results, list) + and bool(results) + and all(_looks_like_audit_result(item) for item in results) + ) + if isinstance(data, dict) and "runs" in data: + return bool(_experiment_models(data)) + return False + + +def get_file_tree(directory: str, base_path: str = "") -> list[dict]: + """Recursively build the JSON file tree for the visualizer. + + Folders are included only when they contain at least one loadable JSON + file (directly or in a subdirectory); experiment files are tagged with + their model labels so the UI can render a model picker. + """ + items = [] + try: + entries = sorted(os.listdir(directory)) + except (PermissionError, OSError): + return items + + for entry in entries: + full_path = os.path.join(directory, entry) + rel_path = os.path.join(base_path, entry) if base_path else entry + + if os.path.isdir(full_path): + children = get_file_tree(full_path, rel_path) + if children: + items.append( + {"name": entry, "type": "folder", "path": rel_path, "children": children} + ) + elif os.path.isfile(full_path) and entry.endswith(".json"): + try: + with open(full_path, "r", encoding="utf-8") as f: + data = json.load(f) + except (OSError, json.JSONDecodeError): + logger.debug("Skipping unreadable/invalid JSON in results dir: %s", full_path) + continue + experiment_models = _experiment_models(data) + if experiment_models: + items.append( + { + "name": entry, + "type": "experiment", + "path": rel_path, + "models": experiment_models, + } + ) + elif is_valid_audit_data(data): + items.append({"name": entry, "type": "file", "path": rel_path}) + + return items + + +def _resolve_results_path(file_path: str) -> str | None: + """Resolve a relative path inside the results dir, or None if it escapes it. + + Guards against path traversal and symlink escapes the same way the core + server did: the resolved real path must stay under the results root. + """ + root_dir = results_dir() + if not root_dir: + return None + root = os.path.realpath(os.path.abspath(root_dir)) + try: + full_path = os.path.realpath(os.path.join(root, file_path)) + except (ValueError, TypeError): + return None + if not full_path.startswith(root + os.sep): + return None + return full_path + + +def _read_asset(name: str) -> str: + """Read a static asset by name, from STATIC_ROOT, source dirs, or finders.""" + static_root = str(settings.STATIC_ROOT) + for root in [static_root, *getattr(settings, "STATICFILES_DIRS", [])]: + candidate = os.path.join(str(root), name) + if os.path.isfile(candidate): + with open(candidate, "r", encoding="utf-8") as f: + return f.read() + from django.contrib.staticfiles.finders import find + + found = find(name) + if found: + with open(found, "r", encoding="utf-8") as f: + return f.read() + raise Http404(f"Static asset not found: {name}") + + +def build_standalone_html(data, name: str) -> str: + """Inline audit data into the visualizer template as a standalone HTML file. + + The output opens directly in a browser (or can be sent to someone) โ€” no + server and no upload step. The visualizer detects the inlined data on load + and renders it as a custom upload. + """ + if not is_valid_audit_data(data): + raise ValueError("data does not look like SimpleAudit results") + + html = _read_asset("visualizer.html") + if "" not in html: + raise ValueError("visualizer.html has no anchor โ€” cannot inline data") + + # Escape so the payload cannot break out of the inline \n" + ) + return html.replace("", f"{inline}", 1) + + +# --- views ------------------------------------------------------------------- +class VisualizerView(LoginRequiredMixin, View): + """Serve the file-tree visualizer SPA (``/visualizer/``). + + The SPA is a static asset, but its API calls must be prefixed with + ``/api/visualizer`` when embedded in Studio. We inject the base into the + one ``window.__VISUALIZER_API_BASE`` line (the same inline-into-```` + technique the core used for standalone export) and serve the result. + """ + + def get(self, request): + html = _read_asset("visualizer.html") + html = html.replace( + "window.__VISUALIZER_API_BASE = '{{ visualizer_api_base|default:\"\" }}';", + "window.__VISUALIZER_API_BASE = '/api/visualizer';", + ) + return HttpResponse(html, content_type="text/html; charset=utf-8") + + +class ScenarioViewerView(LoginRequiredMixin, View): + """Serve the offline drag-drop viewer (``/visualizer/upload/``). + + ``scenario_viewer.html`` reads files straight from the browser via the File + System Access API โ€” no server round-trip โ€” so it works fully offline. + """ + + def get(self, request): + html = _read_asset("scenario_viewer.html") + return HttpResponse(html, content_type="text/html; charset=utf-8") + + +class VisualizerFilesView(View): + """``GET /api/visualizer/files`` โ†’ ``{"tree": [...]}`` of the results dir.""" + + def get(self, request): + root_dir = results_dir() + if not root_dir: + return JsonResponse( + {"error": "Results directory not set. Start Studio with --results_dir."}, + status=500, + ) + if not os.path.isdir(root_dir): + return JsonResponse({"error": "Results directory not found"}, status=404) + return JsonResponse({"tree": get_file_tree(root_dir)}) + + +class VisualizerJsonView(View): + """``GET /api/visualizer/json/`` โ†’ the JSON content of a result file.""" + + def get(self, request, file_path: str): + full_path = _resolve_results_path(file_path) + if full_path is None: + return JsonResponse({"error": "Access denied"}, status=403) + if not os.path.isfile(full_path): + return JsonResponse({"error": "File not found"}, status=404) + if os.path.splitext(full_path)[1].lower() != ".json": + return JsonResponse({"error": "Not a JSON file"}, status=400) + try: + with open(full_path, "r", encoding="utf-8") as f: + data = json.load(f) + except json.JSONDecodeError as exc: + return JsonResponse({"error": f"Invalid JSON: {exc}"}, status=400) + except OSError: + return JsonResponse({"error": "Error reading file"}, status=500) + if not is_valid_audit_data(data): + return JsonResponse({"error": "Not an audit results file"}, status=403) + return JsonResponse(data) + + +class VisualizerImageView(View): + """``GET /api/visualizer/image?uri=...`` โ†’ ``{"data_uri": ...}`` for previews. + + Reuses the audit's own ``image_data_uri`` loader so local paths, http(s) and + s3 resolve the same way they did at audit time, and non-images are rejected + identically. Returns a data URI (not raw bytes) so the frontend can assign + it to an ```` src. + """ + + def get(self, request): + from simpleaudit.utils import image_data_uri + + uri = request.GET.get("uri", "") + if not uri: + return JsonResponse({"error": "Missing uri"}, status=400) + try: + data_uri = image_data_uri(uri) + except ValueError: + return JsonResponse({"error": "Not a previewable image"}, status=415) + except FileNotFoundError: + return JsonResponse({"error": "Image not found"}, status=404) + except OSError: + return JsonResponse({"error": "Error reading image"}, status=500) + return JsonResponse({"data_uri": data_uri}) + + +class VisualizerAuthView(View): + """``GET /api/visualizer/auth`` โ†’ auth status for the visualizer SPA. + + The visualizer.html carries an auth overlay that expects this endpoint. + Studio already gates the page behind login, so auth is reported as + disabled (the overlay stays hidden); the endpoint exists so the SPA's + startup check doesn't 404. + """ + + def get(self, request): + return JsonResponse({"ok": True, "enabled": False, "contact_email": ""}) + + +class RunExportHtmlView(ProjectMixin, View): + """``GET /runs//export-html/`` โ†’ standalone single-file HTML export. + + Inlines the run's results JSON into the visualizer template so the output + opens in any browser with no server. Mirrors the core ``export-html`` + command, but sources the data from the run's stored results. + """ + + def get(self, request, run_id): + from audits.events import ScenarioResult + from audits.models import AuditRun + from scenarios.models import ScenarioSetVersionItem + + run = get_object_or_404(AuditRun, pk=run_id, project=request.project) + + items = { + str(vi.pk): vi + for vi in ScenarioSetVersionItem.objects.filter( + version=run.scenario_set_version + ).select_related("scenario") + } + results = [] + for sr in ScenarioResult.objects.filter(run_id=run.id).order_by("id"): + item = items.get(sr.version_item_id) + r = sr.result or {} + results.append( + { + "scenario_name": item.scenario.title if item else str(sr.version_item_id), + "severity": r.get("severity", sr.status), + "summary": r.get("summary", ""), + "result": r, + } + ) + + data = {"results": results} + name = f"{run.name} (run {run.id})" + try: + html = build_standalone_html(data, name) + except ValueError as exc: + return HttpResponse(str(exc), status=400) + + safe_name = "".join(c if c.isalnum() or c in "-_." else "_" for c in run.name)[:60] or "run" + response = HttpResponse(html, content_type="text/html; charset=utf-8") + response["Content-Disposition"] = f'attachment; filename="audit_{safe_name}_{run.id}.html"' + return response diff --git a/simpleaudit_studio/cli.py b/simpleaudit_studio/cli.py index be41b887..e08a6097 100644 --- a/simpleaudit_studio/cli.py +++ b/simpleaudit_studio/cli.py @@ -12,6 +12,9 @@ uvx simpleaudit-studio # full stack; models point at OpenAI (add your key in the UI) uvx simpleaudit-studio --mock # use the built-in mock model server (zero-setup demo) uvx simpleaudit-studio --port 9000 # custom port + uvx simpleaudit-studio --visualize-only --results_dir ./results + # web server only (no worker/hatchet/chat), + # browsing a folder of simpleaudit results When a needed port is held by another Studio instance or one of its derivatives (Open WebUI, the Hatchet sidecar), the CLI offers to stop it and @@ -61,6 +64,18 @@ def main() -> None: "--yes", action="store_true", help="Answer yes to the force-kill confirmation without asking", ) + parser.add_argument( + "--visualize-only", action="store_true", + help="Run only the web server for the result visualizer: skip the audit " + "worker, embedded Hatchet, chat, and model-endpoint setup. Pair with " + "--results_dir to browse a folder of simpleaudit results.", + ) + parser.add_argument( + "--results_dir", default=None, + help="Directory of JSON result files for the visualizer's file tree " + "(replaces `simpleaudit serve`). Point it at a folder where you " + "dumped simpleaudit results.", + ) args = parser.parse_args() # Set local mode BEFORE Django reads settings @@ -106,9 +121,33 @@ def main() -> None: print("โœ… Migrations complete.") # --- Step 2: Bootstrap admin + seed data --- - print("๐ŸŒฑ Seeding demo data...") - _seed_demo_data() - print("โœ… Demo data ready.\n") + # In visualize-only mode we still need a signed-in user to reach the + # visualizer pages, so we bootstrap the admin (idempotent) but skip the + # heavier demo seeding. + if args.visualize_only: + print("๐ŸŒฑ Bootstrapping admin user (visualize-only)...") + _seed_admin_only() + print("โœ… Admin ready.\n") + else: + print("๐ŸŒฑ Seeding demo data...") + _seed_demo_data() + print("โœ… Demo data ready.\n") + + # --- Results dir for the visualizer (replaces `simpleaudit serve`) --- + if args.results_dir: + from infra.visualizer import set_results_dir + + resolved = os.path.abspath(os.path.expanduser(args.results_dir)) + if not os.path.isdir(resolved): + print(f"โš ๏ธ --results_dir '{resolved}' is not a directory; the visualizer file tree will be empty.") + else: + set_results_dir(resolved) + print(f"๐Ÿ“‚ Visualizer results dir: {resolved}\n") + + # --- Visualize-only: just run the web server, no worker/hatchet/chat --- + if args.visualize_only: + _run_visualize_only(args) + return # --- Step 3: Pre-check port availability --- _ensure_ports_available(args) @@ -344,6 +383,83 @@ def _open_browser_when_ready(url: str, port: int, timeout: float = 30.0) -> None print(f"โš ๏ธ Could not open a browser automatically โ€” visit {url} manually.") +def _seed_admin_only() -> None: + """Bootstrap the admin user + default project only (no demo seeding). + + The visualizer pages sit behind login, so visualize-only mode needs a + signed-in user; this is the lightweight, idempotent subset of + ``_seed_demo_data``. + """ + from accounts.services import bootstrap_admin_and_default_project + + username = os.environ.get("BOOTSTRAP_USERNAME", "studio") + email = os.environ.get("BOOTSTRAP_EMAIL", "admin@localhost") + password = os.environ.get("BOOTSTRAP_PASSWORD", "admin123") + project_name = os.environ.get("BOOTSTRAP_PROJECT_NAME", "Demo Project") + + bootstrap_admin_and_default_project( + username=username, + email=email, + password=password, + project_name=project_name, + ) + + +def _run_visualize_only(args) -> None: + """Run just the Django web server (no worker, Hatchet, or chat). + + This is the client use case: point Studio at a folder of dumped + ``simpleaudit`` results and browse them in the visualizer. The web server + runs in the main thread so Ctrl+C stops it cleanly. + """ + from django.core.management import call_command + + port = args.port + + # Only the web port matters here (no chat/hatchet ports to reserve). + from simpleaudit_studio import ports + + ports.resolve_port_conflict( + port, "the web server", "spin --port ", + force_kill=not args.no_force_kill, yes=args.yes, + ) + + username = os.environ.get("BOOTSTRAP_USERNAME", "studio") + password = os.environ.get("BOOTSTRAP_PASSWORD", "admin123") + + auto_login_token = secrets.token_urlsafe(32) + os.environ["SIMPLEAUDIT_AUTO_LOGIN_TOKEN"] = auto_login_token + auto_login_url = f"http://localhost:{port}/auto-login/?token={auto_login_token}" + + print("โ”Œโ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”") + print("โ”‚ ๐Ÿ‘๏ธ SimpleAudit Studio โ€” Visualize-only mode โ”‚") + print("โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”˜") + print() + print(f" Web UI: http://localhost:{port}") + print(f" Visualizer: http://localhost:{port}/visualizer/") + print(f" Drag-drop: http://localhost:{port}/visualizer/upload/") + print(f" Login: {username} / {password}") + if args.results_dir: + print(f" Results dir: {os.path.abspath(os.path.expanduser(args.results_dir))}") + print() + print(f"๐Ÿ”‘ One-time sign-in link: {auto_login_url}") + print() + print(" Press Ctrl+C to stop.") + print() + + if not args.no_browser: + threading.Thread( + target=_open_browser_when_ready, + args=(auto_login_url, port), + daemon=True, + ).start() + + try: + call_command("runserver", f"0.0.0.0:{port}", use_reloader=False) + except KeyboardInterrupt: + print("\n๐Ÿ‘‹ Shutting down.") + + def _seed_demo_data() -> None: """Bootstrap admin user, default project, scenario packs, and model connections.""" from django.core.management import call_command diff --git a/static/scenario_viewer.html b/static/scenario_viewer.html new file mode 100644 index 00000000..12f1723d --- /dev/null +++ b/static/scenario_viewer.html @@ -0,0 +1,1107 @@ + + + + + + + SimpleAudit Scenario Viewer + + + + + + + + +
+
+

+ + + + SimpleAudit Scenario Viewer +

+

Upload a JSON file to view audit results

+
+
+ + +
+ +
+
+ + + + + + +

Drop your JSON file here or click to browse

+

Supports SimpleAudit result files

+ + +
+ + +
+ +
+
+ + + + + + + +
+ + + + + + + + diff --git a/static/visualizer.html b/static/visualizer.html new file mode 100644 index 00000000..7fe35c8a --- /dev/null +++ b/static/visualizer.html @@ -0,0 +1,2140 @@ + + + + + + + SimpleAudit Result Visualizer + + + + + + + + + + + + + +
+ +
+
+ +
+

+ + + + + SimpleAudit +

+ +
+
+
+ + + +
+
+ + +
+ + +
+ + +
+
+

+ + + + JSON Files +

+ +
+
+
+
+ Loading files... +
+
+
+ + +
+ + +
+ + +
+ +
+ + + + + + + + +
+
+ Select a JSON file to view results. +
+
+
+ + +
+ + +
+
+ + + +

Select a scenario to view details

+
+
+
+
+
+ + + + + + + +
+ + + + diff --git a/templates/compare.html b/templates/compare.html index a963d4de..1c03f018 100644 --- a/templates/compare.html +++ b/templates/compare.html @@ -81,6 +81,7 @@

Results{% if result.intersection Scenario {% for meta in result.run_meta %}{% include "partials/compare_run_header.html" %}{% endfor %} + Fragility {% for row in result.rows %} @@ -94,6 +95,17 @@

Results{% if result.intersection {% else %}{{ val|default:"โ€”" }}{% endif %} {% endfor %} + + {% if row.fragility %} + {{ row.fragility.agreement|floatformat:0 }}% agree + ยท + H {{ row.fragility.entropy|floatformat:2 }} + ยท + ฯƒ {{ row.fragility.spread|floatformat:2 }} + ยท + {{ row.fragility.mode }} + {% else %}โ€”{% endif %} + {% endfor %} diff --git a/templates/run_detail.html b/templates/run_detail.html index c654ead3..38c1962a 100644 --- a/templates/run_detail.html +++ b/templates/run_detail.html @@ -16,6 +16,9 @@ Experiment {% endif %} + + Visualizer + {% if run.monitor %} Monitor diff --git a/templates/run_result.html b/templates/run_result.html index 7556b724..74e8f92d 100644 --- a/templates/run_result.html +++ b/templates/run_result.html @@ -13,6 +13,7 @@

{{ scenario_name }}< {% if is_repeated %}{{ n_reps }}ร— ยท {{ agreement_rate|floatformat:0 }}% agree{% endif %} Export JSON + Export HTML {% if error %} diff --git a/uv.lock b/uv.lock index 20370dc9..37c8c242 100644 --- a/uv.lock +++ b/uv.lock @@ -2384,7 +2384,7 @@ wheels = [ [[package]] name = "simpleaudit" -version = "0.2.3" +version = "0.3.0" source = { editable = "../SimpleAudit" } dependencies = [ { name = "any-llm-sdk" }, @@ -2397,20 +2397,20 @@ requires-dist = [ { name = "aiohttp", marker = "extra == 'tracing'", specifier = ">=3.9" }, { name = "any-llm-sdk", specifier = ">=1.14.0" }, { name = "black", marker = "extra == 'dev'", specifier = ">=23.0.0" }, - { name = "fastapi", marker = "extra == 'visualize'", specifier = ">=0.104.0" }, { name = "fsspec", specifier = ">=2023.1.0" }, { name = "grpcio", marker = "extra == 'tracing'", specifier = ">=1.60" }, { name = "matplotlib", marker = "extra == 'plot'", specifier = ">=3.5.0" }, { name = "opentelemetry-proto", marker = "extra == 'tracing'", specifier = ">=1.20" }, - { name = "pytest", marker = "extra == 'dev'", specifier = ">=7.0.0" }, + { name = "pytest", marker = "extra == 'dev'", specifier = ">=8,<9" }, { name = "pytest-asyncio", marker = "extra == 'dev'", specifier = ">=0.21.0" }, { name = "pytest-cov", marker = "extra == 'dev'", specifier = ">=4.0.0" }, + { name = "pytest-testmon", marker = "extra == 'dev'", specifier = ">=2,<3" }, + { name = "pytest-xdist", marker = "extra == 'dev'", specifier = ">=3.6,<3.8" }, { name = "ruff", marker = "extra == 'dev'", specifier = ">=0.1.0" }, - { name = "simpleaudit", extras = ["plot", "visualize", "dev"], marker = "extra == 'all'" }, + { name = "simpleaudit", extras = ["plot", "dev"], marker = "extra == 'all'" }, { name = "tqdm", specifier = ">=4.66.0" }, - { name = "uvicorn", extras = ["standard"], marker = "extra == 'visualize'", specifier = ">=0.24.0" }, ] -provides-extras = ["tracing", "plot", "visualize", "dev", "all"] +provides-extras = ["tracing", "plot", "dev", "all"] [[package]] name = "simpleaudit-studio"