Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions .env.dev.example
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,11 @@ SIMPLEAUDIT_SQLITE_PATH=dev.sqlite3
# JSON result files shown by /visualizer/. The folder is created on demand;
# set an absolute path when results live elsewhere.
VISUALIZER_RESULTS_DIR=./results
# Bounded-scan limits for the visualizer file tree (defaults shown):
# VISUALIZER_MAX_INSPECTED_FILES=5000
# VISUALIZER_SCAN_TIME_BUDGET_S=5
# VISUALIZER_MAX_TREE_DEPTH=8
# VISUALIZER_MAX_FILE_SIZE_MB=100

# --- Chat (Open WebUI) ----------------------------------------------------------
# On by default in dev; no line needed. To force it off:
Expand Down
4 changes: 4 additions & 0 deletions docs/deployment.md
Original file line number Diff line number Diff line change
Expand Up @@ -169,6 +169,10 @@ override them when you deliberately need to.
| `SIMPLEAUDIT_LOCAL_SQLITE` **internal** | `1` = local SQLite database at the repository's `dev.sqlite3`. Set automatically by the `dev` entry point and by the test suite. |
| `SIMPLEAUDIT_SQLITE_PATH` | Override where the local dev SQLite file lives. Default: `<repo>/dev.sqlite3`. Relative paths resolve from the repo root; absolute paths are used as-is. Ignored unless `SIMPLEAUDIT_LOCAL_SQLITE=1`. |
| `VISUALIZER_RESULTS_DIR` | Folder of JSON audit-result files shown by `/visualizer/`. Dev defaults to `./results`; the folder is created on demand. |
| `VISUALIZER_MAX_INSPECTED_FILES` | Max JSON files to inspect per scan (default `5000`). When hit, the tree is returned with `truncated: true, reason: "file_limit"`. |
| `VISUALIZER_SCAN_TIME_BUDGET_S` | Wall-clock time budget for the scan in seconds (default `5`). When hit, the tree is returned with `truncated: true, reason: "time_budget"`. |
| `VISUALIZER_MAX_TREE_DEPTH` | Max directory depth to walk (default `8`). When hit, the tree is returned with `truncated: true, reason: "depth_limit"`. |
| `VISUALIZER_MAX_FILE_SIZE_MB` | Max JSON file size in MB to parse (default `100`). Larger files are skipped. |
| `SIMPLEAUDIT_MINIMAL` **internal** | `1` = single-process demo mode. Set by the CLI. |
| `SIMPLEAUDIT_CHAT` | Chat mode: `embedded` / `docker`, or `off` (and `disabled`, `false`, `no`, `0`, unset). On by default in every mode (`embedded` in the single-process modes, `docker` in compose — set in `.env`). Turn off with this var or the `--disable-chat` flag; the flag beats the env value (flag > env > mode default). |
| `SIMPLEAUDIT_CHAT_OTLP` | Export Open WebUI structural spans to Studio. Embedded `uvx simpleaudit-studio` defaults to `true`; other modes remain off unless enabled. Set `false` to opt out of embedded tracing. Content capture remains independently opt-in. |
Expand Down
8 changes: 8 additions & 0 deletions infra/tests/test_api_e2e.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,8 @@
This is the "another developer can run this" integration test that validates
the entire user journey without requiring a live worker or Docker stack.
"""
from unittest import mock

from django.test import tag
from rest_framework.test import APITestCase

Expand All @@ -27,6 +29,12 @@ def setUp(self):
self.project = Project.objects.create(name="E2E", slug="e2e")
ProjectMembership.objects.create(project=self.project, user=self.user, role=ProjectMembership.Role.AUDITOR)
self.pid = self.project.id
# No live Hatchet server in tests: stub the enqueue step so run
# creation doesn't attempt a real gRPC call (which hangs when the
# configured host is unreachable, e.g. the compose default in CI).
self._submit_patcher = mock.patch("audits.views.submit_audit_run")
self._submit_patcher.start()
self.addCleanup(self._submit_patcher.stop)

def _auth(self):
"""Get a token via the auth endpoint."""
Expand Down
215 changes: 215 additions & 0 deletions infra/tests/test_visualizer.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@
"""
import json
import os
import tempfile

from django.test import Client, TestCase

Expand All @@ -16,6 +17,8 @@
UserFactory,
)
from infra.visualizer import (
_metadata_index,
_scan_results_dir,
build_standalone_html,
is_valid_audit_data,
set_results_dir,
Expand Down Expand Up @@ -214,3 +217,215 @@ def _rmtree(path):
import shutil

shutil.rmtree(path, ignore_errors=True)


def _make_valid_json(path: str) -> None:
"""Write a minimal valid audit-results JSON file."""
with open(path, "w") as f:
json.dump({"results": [{"scenario_name": "s", "severity": "pass", "summary": "ok"}]}, f)


class BoundedScanTests(TestCase):
"""Tests for the bounded, incremental file-tree scan."""

def setUp(self):
self.tmp = tempfile.mkdtemp(prefix="visz_bounded_")
self.addCleanup(_rmtree, self.tmp)
# Clear the metadata index between tests.
_metadata_index.clear()

def _scan(self, **kwargs):
defaults = {
"max_depth": 8,
"max_files": 5000,
"time_budget": 5.0,
}
defaults.update(kwargs)
return _scan_results_dir(self.tmp, **defaults)

def test_small_tree_not_truncated(self):
_make_valid_json(os.path.join(self.tmp, "a.json"))
tree, meta = self._scan()
self.assertFalse(meta["truncated"])
self.assertIsNone(meta["reason"])
self.assertEqual(meta["inspected"], 1)
self.assertGreaterEqual(meta["elapsed_seconds"], 0)
self.assertEqual(len(tree), 1)
self.assertEqual(tree[0]["name"], "a.json")

def test_file_limit_truncates(self):
# Create 10 valid JSON files, limit to 3.
for i in range(10):
_make_valid_json(os.path.join(self.tmp, f"f{i:02d}.json"))
tree, meta = self._scan(max_files=3)
self.assertTrue(meta["truncated"])
self.assertEqual(meta["reason"], "file_limit")
self.assertEqual(meta["inspected"], 3)
self.assertEqual(len(tree), 3)

def test_depth_limit_truncates(self):
# Create a file at depth 3, limit depth to 2.
deep = os.path.join(self.tmp, "a", "b", "c")
os.makedirs(deep, exist_ok=True)
_make_valid_json(os.path.join(deep, "deep.json"))
tree, meta = self._scan(max_depth=2)
self.assertTrue(meta["truncated"])
self.assertEqual(meta["reason"], "depth_limit")
# The deep file should not be in the tree.
self.assertEqual(tree, [])

def test_time_budget_truncates(self):
# Create many files and use a tight time budget (5ms).
# The scan should inspect some files but not all 500.
for i in range(500):
_make_valid_json(os.path.join(self.tmp, f"f{i:03d}.json"))
_, meta = self._scan(time_budget=0.005) # 5ms
self.assertTrue(meta["truncated"])
self.assertEqual(meta["reason"], "time_budget")
self.assertGreater(meta["inspected"], 0)
self.assertLess(meta["inspected"], 500)

def test_symlinks_are_skipped(self):
_make_valid_json(os.path.join(self.tmp, "real.json"))
os.symlink(os.path.join(self.tmp, "real.json"), os.path.join(self.tmp, "link.json"))
tree, _ = self._scan()
names = [item["name"] for item in tree]
self.assertIn("real.json", names)
self.assertNotIn("link.json", names)

def test_symlinked_directory_not_followed(self):
# Create a real dir with a file, and a symlink to it.
real_dir = os.path.join(self.tmp, "realdir")
os.makedirs(real_dir)
_make_valid_json(os.path.join(real_dir, "inner.json"))
os.symlink(real_dir, os.path.join(self.tmp, "linkdir"))
tree, _ = self._scan()
# The real dir's file should appear; the symlinked dir should not.
folder_names = [item["name"] for item in tree if item["type"] == "folder"]
self.assertIn("realdir", folder_names)
self.assertNotIn("linkdir", folder_names)

def test_oversized_file_skipped(self):
# Create a file larger than the 100MB limit (use a sparse file).
big_path = os.path.join(self.tmp, "big.json")
with open(big_path, "w") as f:
f.seek(101 * 1024 * 1024 - 1)
f.write("\0")
_make_valid_json(os.path.join(self.tmp, "small.json"))
tree, _ = self._scan()
names = [item["name"] for item in tree]
self.assertIn("small.json", names)
self.assertNotIn("big.json", names)

def test_valid_results_beyond_cap_still_truncated(self):
# 10 valid files, cap at 5 → truncated, but the 5 shown are valid.
for i in range(10):
_make_valid_json(os.path.join(self.tmp, f"v{i}.json"))
tree, meta = self._scan(max_files=5)
self.assertTrue(meta["truncated"])
self.assertEqual(meta["reason"], "file_limit")
self.assertEqual(len(tree), 5)
for item in tree:
self.assertEqual(item["type"], "file")

def test_metadata_index_reuse(self):
# Scan once, then scan again — the second scan should be faster
# (metadata index hit) and produce the same tree.
_make_valid_json(os.path.join(self.tmp, "a.json"))
tree1, meta1 = self._scan()
self.assertEqual(meta1["inspected"], 1)
# Clear the in-memory tree cache (not the metadata index).
from infra.visualizer import _file_tree_cache

_file_tree_cache["data"] = None
_file_tree_cache["key"] = None
tree2, meta2 = self._scan()
self.assertEqual(tree1, tree2)
self.assertEqual(meta2["inspected"], 1)
# The metadata index should have an entry for the file.
self.assertGreater(len(_metadata_index), 0)

def test_pruned_dirs_skipped(self):
# Files inside pruned dirs should not appear.
for d in ["node_modules", ".git", "__pycache__"]:
p = os.path.join(self.tmp, d)
os.makedirs(p, exist_ok=True)
_make_valid_json(os.path.join(p, "hidden.json"))
_make_valid_json(os.path.join(self.tmp, "visible.json"))
tree, _ = self._scan()
names = [item["name"] for item in tree]
self.assertIn("visible.json", names)
self.assertNotIn("node_modules", names)
self.assertNotIn(".git", names)
self.assertNotIn("__pycache__", names)

def test_experiment_file_in_tree(self):
exp = {
"runs": {
"model-a": [
{"results": [{"scenario_name": "s", "severity": "high", "summary": "x"}]}
]
}
}
with open(os.path.join(self.tmp, "exp.json"), "w") as f:
json.dump(exp, f)
tree, _ = self._scan()
self.assertEqual(len(tree), 1)
self.assertEqual(tree[0]["type"], "experiment")
self.assertEqual(tree[0]["models"], ["model-a"])

def test_non_audit_json_excluded(self):
with open(os.path.join(self.tmp, "notaudit.json"), "w") as f:
json.dump({"foo": "bar"}, f)
tree, meta = self._scan()
self.assertEqual(tree, [])
self.assertEqual(meta["inspected"], 1)


class VisualizerFilesMetaTests(_VisualizerBase):
"""Tests for the /api/files response metadata."""

def setUp(self):
super().setUp()
self.results = tempfile.mkdtemp(prefix="visz_meta_")
self.addCleanup(_rmtree, self.results)
_make_valid_json(os.path.join(self.results, "a.json"))
set_results_dir(self.results)
self.addCleanup(set_results_dir, None)

def test_response_includes_metadata(self):
resp = self.client.get("/api/visualizer/api/files/", follow=True)
self.assertEqual(resp.status_code, 200)
data = resp.json()
self.assertIn("tree", data)
self.assertIn("truncated", data)
self.assertIn("reason", data)
self.assertIn("inspected", data)
self.assertIn("elapsed_seconds", data)
self.assertIn("configured", data)
self.assertTrue(data["configured"])
self.assertFalse(data["truncated"])
self.assertIsNone(data["reason"])
self.assertEqual(data["inspected"], 1)
self.assertGreaterEqual(data["elapsed_seconds"], 0)

def test_response_truncated_when_limit_hit(self):
# Add more files than the default cap (5000) — use a low cap via env.
import os as _os

old = _os.environ.get("VISUALIZER_MAX_INSPECTED_FILES")
_os.environ["VISUALIZER_MAX_INSPECTED_FILES"] = "2"
self.addCleanup(lambda: _os.environ.pop("VISUALIZER_MAX_INSPECTED_FILES", None) if old is None else _os.environ.__setitem__("VISUALIZER_MAX_INSPECTED_FILES", old))
# Clear the tree cache so the new limit takes effect.
from infra.visualizer import _file_tree_cache

_file_tree_cache["data"] = None
_file_tree_cache["key"] = None
# Add a third file.
_make_valid_json(os.path.join(self.results, "b.json"))
_make_valid_json(os.path.join(self.results, "c.json"))
resp = self.client.get("/api/visualizer/api/files/", follow=True)
data = resp.json()
self.assertTrue(data["truncated"])
self.assertEqual(data["reason"], "file_limit")
self.assertEqual(data["inspected"], 2)
Loading
Loading