From d6d266ef3e2f834754f74b22bd147d94e2bd3f07 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mats=20B=C3=B8e=20Bergmann?= Date: Tue, 6 Oct 2026 10:53:24 +0200 Subject: [PATCH 01/10] chore: pin stock OpenCode 2.0.23 --- Containerfile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Containerfile b/Containerfile index 97fb1f9..3ec8003 100644 --- a/Containerfile +++ b/Containerfile @@ -1,5 +1,5 @@ FROM node:24-bookworm-slim@sha256:0e0ff40c39bc087845bfb27465a0df4ea419520094bc35842ff83dd8cbe6f9b6 AS opencode-builder -ARG OPENCODE_VERSION=2.0.18 +ARG OPENCODE_VERSION=2.0.23 RUN npm install --global "@opencode/cli@${OPENCODE_VERSION}" \ && resolved="$(readlink -f "$(command -v opencode)")" \ && test -x "$resolved" \ From f9dc9bf494f4ec90367f8c88d4cdc2840d9ca87e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mats=20B=C3=B8e=20Bergmann?= Date: Tue, 6 Oct 2026 10:53:26 +0200 Subject: [PATCH 02/10] docs: define trusted-checkout evaluation model --- README.md | 18 +++++++++++++++--- 1 file changed, 15 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index 0f5c5a7..faa29c5 100644 --- a/README.md +++ b/README.md @@ -72,7 +72,7 @@ Known API-key environment variables are passed when present: Additional variables require explicit `--env NAME`. -Reasoning can be pinned explicitly with `--reasoning LEVEL`. The pinned OpenCode 2.0.18 CLI represents a model variant in the model reference, so the runner maps `--model provider/model --reasoning LEVEL` to `opencode run --model provider/model#LEVEL`. Supplying both a `#variant` in `--model` and `--reasoning` is rejected as ambiguous. If the model reference already contains a variant and `--reasoning` is omitted, the result records that variant with `"reasoning_source": "model-variant"`. If neither form supplies a level, the runner leaves OpenCode's provider/model default untouched and records `"reasoning": "provider-default"`. +Reasoning can be pinned explicitly with `--reasoning LEVEL`. The pinned stock OpenCode 2.0.23 CLI represents a model variant in the model reference, so the runner maps `--model provider/model --reasoning LEVEL` to `opencode run --model provider/model#LEVEL`. Supplying both a `#variant` in `--model` and `--reasoning` is rejected as ambiguous. If the model reference already contains a variant and `--reasoning` is omitted, the result records that variant with `"reasoning_source": "model-variant"`. If neither form supplies a level, the runner leaves OpenCode's provider/model default untouched and records `"reasoning": "provider-default"`. ### `github-copilot-cli` @@ -173,7 +173,7 @@ opencode-eval-runner invoke \ ... ``` -OpenCode 2.0.18 does not expose the old singular `debug agent ` command that returned a resolved tool map. The runner therefore performs the strongest supported zero-inference preflight: it requires the expected plugin entrypoint to be materialized in the isolated OpenCode config, runs `opencode debug agents` to prove the configured location starts successfully with plugins active, and requires the selected agent to resolve. Missing plugin materialization, plugin/startup failure, or missing agent is infrastructure/non-evidence, never a behavioral FAIL. Actual tool use remains a repository-owned behavioral assertion in the eval corpus. +Stock OpenCode 2.0.23 does not expose the old singular `debug agent ` command that returned a resolved tool map. The runner therefore performs the strongest supported zero-inference preflight: it requires the expected plugin entrypoint to be materialized in the isolated OpenCode config, runs `opencode debug agents` to prove the configured location starts successfully with plugins active, and requires the selected agent to resolve. Missing plugin materialization, plugin/startup failure, or missing agent is infrastructure/non-evidence, never a behavioral FAIL. Actual tool use remains a repository-owned behavioral assertion in the eval corpus. ### Evaluating a skill @@ -315,7 +315,7 @@ The eval repository decides whether that observed behavior is PASS, FAIL, or non The transport images currently pin: -- OpenCode CLI `2.0.18` +- OpenCode CLI `2.0.23` - GitHub Copilot CLI `1.0.83` The two CLIs are not bundled together. OpenCode's npm package is used only as a build-time native-binary selector; GitHub Copilot CLI is installed from its native release installer. Node/npm are absent from the final runtime images. @@ -340,6 +340,18 @@ OPENCODE_EVAL_RUNNER_COPILOT_IMAGE=... Tags matching `v*` are published with `opencode-` and `copilot-` prefixes. +## Evaluation trust model + +The normal evaluation profile is a **trusted-checkout** profile. It assumes the runner, pinned stock OpenCode runtime, reviewed instrumentation, and explicitly selected evaluated checkout/dependencies are trusted components of the evaluation environment. + +They are not trusted merely because they produce data that looks like evidence. Model prose, tool-returned collector-shaped JSON, target-writable files, requested actions, inferred identities, and reconstructed results do not establish that an event occurred. + +Authoritative runtime observations must come from reviewed instrumentation observing actual execution. Missing, partial, ambiguous, or unsupported required observations are non-evidence and must fail closed for the affected assertion. + +This profile does **not** claim resistance to an evaluated plugin that deliberately compromises the trusted runtime or instrumentation. Hostile-plugin isolation is a separate optional profile, not a prerequisite for normal Loom evaluation. + +See [Trusted-checkout runtime evidence](docs/trusted-checkout-evidence.md). + ## Security boundary The runner: From 97589bb7c492d81949c84d4dfbfd8f1b00d7232c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mats=20B=C3=B8e=20Bergmann?= Date: Tue, 6 Oct 2026 10:53:28 +0200 Subject: [PATCH 03/10] docs: replace TRUST-001 with trusted-checkout evidence contract --- docs/trusted-checkout-evidence.md | 125 ++++++++++++++++++++++++++++++ 1 file changed, 125 insertions(+) create mode 100644 docs/trusted-checkout-evidence.md diff --git a/docs/trusted-checkout-evidence.md b/docs/trusted-checkout-evidence.md new file mode 100644 index 0000000..350c793 --- /dev/null +++ b/docs/trusted-checkout-evidence.md @@ -0,0 +1,125 @@ +# Trusted-checkout runtime evidence + +Status: **replacement direction for TRUST-001**. + +This document supersedes the hostile-runtime direction explored in PR #41 and PR #43. Those PRs remain useful research/reference material, but normal Loom evaluation does not require the runner to defend itself from a deliberately malicious Loom checkout that shares its runtime authority. + +## Contract + +### TRUST-001 — authoritative runtime observation + +For evaluation of an explicitly trusted checkout, evidence used for scoring MUST originate from reviewed runtime instrumentation observing actual execution. + +The following MUST NOT independently establish that an event occurred: + +- model assertions or generated prose; +- tool payloads shaped like collector/evidence records; +- requested or intended actions; +- inferred actor, parent, or execution identity; +- reconstructed results; +- target-writable evidence files. + +Required observations MUST preserve enough runtime identity and ordering to evaluate the consumer contract, including actor/session/call identity, input, result or error, parent binding where applicable, and execution order. + +Missing, partial, ambiguous, lost, or unsupported required observations MUST make the affected assertion ineligible for PASS. The runner MUST NOT fill gaps from model text, stdout, workspace files, or guessed correlations. + +The trusted-checkout profile does not claim protection against malicious modification of the runner, stock OpenCode process, reviewed instrumentation, evaluated checkout, or their dependencies. + +## Trust model + +Trusted components: + +- the selected `opencode-eval-runner` revision; +- pinned **stock OpenCode 2.0.23**; +- reviewed runtime instrumentation; +- the explicitly selected Loom checkout and its reviewed dependencies; +- host-side evidence projection/persistence code. + +Not trusted as evidence authority: + +- model output; +- agent claims; +- tool-returned collector-shaped data; +- normal product/session/workspace files; +- caller-supplied identity or completeness claims. + +This is an evaluation-correctness boundary, not a hostile-code security boundary. + +## Required evidence behavior + +The target behavior remains strict even though the security scope is smaller: + +- **Native calls:** observe the actual runtime call, actor/session/call identity, accepted/executable input, and terminal result/error. +- **Code Mode inner calls:** assign a unique runtime observation identity per actual inner invocation, bind it to the real outer `execute` call, and observe the final value/error that Code Mode exposes to the script. +- **Delegation:** derive child Session identity and ancestry from runtime facts, not a parent result payload. +- **Ordering:** preserve runtime observation order; do not correlate concurrent calls by FIFO or input equality. +- **Completeness:** explicitly report missing starts/terminals, capture loss, unsupported boundaries, and incomplete scope. +- **Confidentiality:** redact or omit credentials before the runner first persists, clips, logs, or exports evidence. +- **Noninterference:** observation must not add product retries or change normal Loom execution semantics. + +If stock OpenCode's supported interfaces cannot expose an exact required boundary, the result is `unsupported`/ineligible for that assertion. The response is not to invent evidence and not to turn the normal profile into a hostile-code isolation project. + +## Implementation direction + +Keep the normal path: + +```text +Loom eval harness + -> opencode-eval-runner invoke + -> stock OpenCode 2.0.23 + + reviewed runner-owned observation instrumentation + + trusted Loom checkout + -> safe host projection + -> Loom judging +``` + +The preferred implementation is same-process reviewed instrumentation using supported stock OpenCode plugin/runtime surfaces. It may use a runner-owned observer plugin, tool/session hooks, live runtime events, and reviewed wrappers where those surfaces preserve the required boundary. + +OpenCode source patches, forks, remote PluginHost isolation, evidence signing, and a capability broker are not requirements of this profile. + +Provider-free integration tests must prove the exact observation/correlation behavior before a field becomes eligible evidence. + +## Reuse from PR #41 + +| Work | Disposition | +| --- | --- | +| Evidence-safety projection/redaction and fail-closed field handling | **Reuse/adapt**; keep the behavior, decouple it from hostile-runtime image/signing assumptions | +| Credential protection before host/file/print sinks | **Reuse** | +| Disposable OpenCode state/profile work | **Reuse where useful** for deterministic eval isolation | +| Normal `invoke` compatibility and provider-free integration probes | **Reuse/adapt** to stock 2.0.23 | +| Native/Code Mode observation schemas and concurrency tests | **Reuse as behavioral requirements/tests** | +| Delegated-session identity/ancestry probes | **Reuse** | +| Patched OpenCode runtime | **Drop** | +| HMAC observer/import trust boundary | **Drop** for the normal profile | +| protected-channel / remote tool service | **Drop** | +| plugin isolation / remote PluginHost work | **Drop** | +| Cosign evidence-authenticity machinery | **Drop** as a TRUST-001 prerequisite | +| adversarial same-authority attack tests | **Move to optional future untrusted profile** | + +## Reuse from PR #43 + +| Work | Disposition | +| --- | --- | +| Stock OpenCode 2.0.23 source/capability assessment | **Reuse** | +| Identification of public Session/event/tool surfaces | **Reuse** | +| Scope/completeness rules that prevent false absence/PASS | **Reuse and simplify** | +| First-sink confidentiality inventory | **Reuse and simplify** | +| Loom callback/capability inventory | **Reference when needed for compatibility** | +| hostile-runtime TCB/authority model | **Drop** from the normal profile | +| isolated Loom execution domain | **Drop** | +| capability/evidence-channel peer-authentication requirements | **Drop** | +| OCI adversarial boundary experiment/gates | **Drop** | + +## Implementation sequence + +This is normal engineering work, not a multi-authorization security experiment: + +1. Pin and verify stock OpenCode 2.0.23. +2. Add the smallest reviewed observation instrumentation that can capture native and Code Mode execution without changing product semantics. +3. Port the useful PR #41 evidence-safety projection so captured values are protected before persistence/export. +4. Add explicit completeness/loss fields and fail closed when required data is missing. +5. Exercise direct, Code Mode, delegation, error, timeout, and concurrent reverse-completion cases provider-free. +6. Compose through Loom's existing `eval:live -> run-evals.py -> invoke` path. +7. Only after those checks pass should Loom consume the new evidence schema for PASS/FAIL decisions. + +A future **untrusted-plugin execution profile** may add isolation if there is a real need to evaluate hostile plugin code. It must remain optional and separate from the normal trusted-checkout path. From b6e6f8548c2885c4ab040d0c0c26bb3d8dbe1b24 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mats=20B=C3=B8e=20Bergmann?= Date: Tue, 6 Oct 2026 10:54:49 +0200 Subject: [PATCH 04/10] test: expect stock OpenCode 2.0.23 --- tests/test_invoke.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/test_invoke.py b/tests/test_invoke.py index b1495e7..5622b59 100644 --- a/tests/test_invoke.py +++ b/tests/test_invoke.py @@ -706,11 +706,11 @@ def test_workflows_pin_external_actions_by_commit(self): ): self.assertNotIn(mutable, ci + publish) - def test_container_pins_opencode_2_0_18(self): + def test_container_pins_stock_opencode_2_0_23(self): containerfile = (Path(__file__).resolve().parents[1] / "Containerfile").read_text( encoding="utf-8" ) - self.assertIn("ARG OPENCODE_VERSION=2.0.18", containerfile) + self.assertIn("ARG OPENCODE_VERSION=2.0.23", containerfile) self.assertNotIn("ARG OPENCODE_VERSION=2.0.15", containerfile) def test_container_pins_base_images_and_copilot_release_asset(self): From bda7264e7c2e4ba9cece315f9f3a86a5651b3ed6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mats=20B=C3=B8e=20Bergmann?= Date: Tue, 6 Oct 2026 10:56:09 +0200 Subject: [PATCH 05/10] docs: retain stock 2.0.23 observation findings --- docs/stock-opencode-2.0.23-observation.md | 59 +++++++++++++++++++++++ 1 file changed, 59 insertions(+) create mode 100644 docs/stock-opencode-2.0.23-observation.md diff --git a/docs/stock-opencode-2.0.23-observation.md b/docs/stock-opencode-2.0.23-observation.md new file mode 100644 index 0000000..e1653be --- /dev/null +++ b/docs/stock-opencode-2.0.23-observation.md @@ -0,0 +1,59 @@ +# Stock OpenCode 2.0.23 observation surface + +Purpose: implementation reference for the trusted-checkout evidence profile. + +Source checkpoint: stock OpenCode **v2.0.23** (`0fd7e2829449b052abf0078666669302923d77af`). This is distilled from the source assessment performed in superseded PR #43. + +OpenCode remains stock and immutable. A missing observation boundary is reported as unsupported; it is not a reason to patch OpenCode or add a hostile-runtime broker. + +## Useful stock surfaces + +| Observation need | Stock surface | Status | +| --- | --- | --- | +| Live runtime events | `ctx.event.subscribe()` | supported source; ordering/drain must be proven by integration test | +| Session creation / ancestry | `session.created` + Session API | supported | +| Agent for a step | Session step/message events | supported | +| Native tool call identity/input | Session tool input/called events | supported source | +| Native terminal success/failure | Session tool success/failed events | supported source | +| Tool pre-execution hook | `ctx.tool.hook("execute.before")` | supported; occurs before tool decode/execution | +| Tool post-handler hook | `ctx.tool.hook("execute.after")` | supported; occurs after handler result but before later core normalization | +| Tool registration wrapping | `ctx.tool.transform(...)` | supported candidate for reviewed same-process instrumentation | +| Code Mode inner name/input/status | Code Mode metadata + tool hooks | supported source | +| Code Mode unique inner invocation + exact final caller value/error | no single public final boundary demonstrated | **must be proven or marked unsupported** | + +## Native calls + +Stock Session events are the preferred source for native terminal facts because they represent the runtime's own Session lifecycle rather than model or tool payload claims. + +The observer must bind call identity, Session, agent/message context, input and terminal result/error without reconstructing them from prose or matching by value. + +## Code Mode + +Code Mode executes inner tools through the normal tool registry, so same-process reviewed instrumentation can observe real inner execution without isolating Loom. + +The difficult part is not security; it is exact correlation and finality: + +- inner calls share the outer `execute` context in stock OpenCode; +- public Code Mode metadata records name/input/status but not each inner returned value/error; +- `execute.after` is before later core normalization; +- concurrent identical inner calls must not be paired by FIFO, input equality, or completion order. + +The first implementation should test a runner-owned observer plugin using supported tool transforms/hooks and runtime events. It must allocate a unique observation identity at an actual execution boundary and prove how that identity reaches the final inner value/error. + +If that exact binding cannot be demonstrated for a case, the affected result field remains unavailable and the assertion cannot PASS. + +## Ordering and completeness + +A monotonic observer sequence is useful, but sequence alone is not completeness. The integration must also account for starts, terminals, observer loss, process interruption, and required descendant Sessions. + +Absence assertions are eligible only when the relevant scope is complete. Missing capture is never interpreted as "did not happen". + +## What is intentionally not required + +- plugin/process isolation from the trusted Loom checkout; +- remote PluginHost or capability broker; +- evidence-channel peer authentication against same-authority attackers; +- patched/forked OpenCode; +- cryptographic evidence authenticity after runtime compromise. + +Those belong only to a future optional untrusted-plugin profile. From 618b2e9056f993ea504c08316f3e31499413d726 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mats=20B=C3=B8e=20Bergmann?= Date: Tue, 6 Oct 2026 10:56:12 +0200 Subject: [PATCH 06/10] docs: link retained stock observation assessment --- docs/trusted-checkout-evidence.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/docs/trusted-checkout-evidence.md b/docs/trusted-checkout-evidence.md index 350c793..e0a503c 100644 --- a/docs/trusted-checkout-evidence.md +++ b/docs/trusted-checkout-evidence.md @@ -79,6 +79,8 @@ OpenCode source patches, forks, remote PluginHost isolation, evidence signing, a Provider-free integration tests must prove the exact observation/correlation behavior before a field becomes eligible evidence. +See [Stock OpenCode 2.0.23 observation surface](stock-opencode-2.0.23-observation.md) for the retained source/capability findings from PR #43. + ## Reuse from PR #41 | Work | Disposition | From a8c44dcb120da7987aba6d38eacb9415d2b222d9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mats=20B=C3=B8e=20Bergmann?= Date: Tue, 6 Oct 2026 17:14:39 +0200 Subject: [PATCH 07/10] feat: add pre-sink evidence sanitizer --- container/evidence_safety.py | 403 +++++++++++++++++++++++++++++++++++ 1 file changed, 403 insertions(+) create mode 100644 container/evidence_safety.py diff --git a/container/evidence_safety.py b/container/evidence_safety.py new file mode 100644 index 0000000..5cf7d43 --- /dev/null +++ b/container/evidence_safety.py @@ -0,0 +1,403 @@ +"""Small pre-sink sanitization layer for runtime evidence. + +This module is deliberately not an authenticity boundary. It protects known +credentials before evidence or convenience output is clipped, serialized, or +persisted by the runner. +""" +from __future__ import annotations + +import json +import math +import re +import sqlite3 +from pathlib import Path +from typing import Any, Iterable, Mapping + +SCHEMA = "opencode-eval-runner/evidence-safety/v1" +REDACTED = "***REDACTED***" +OMITTED = object() +SUPPORTED_JSON_ESCAPE_LAYERS = 3 +JSON_SOURCE_LIMIT = 4_000_000 +MAX_DEPTH = 32 +MAX_NODES = 20_000 + +REASONS = frozenset({ + "credential_match", + "sensitive_key", + "size_limit", + "missing", + "unsupported_representation", + "credential_inventory_unavailable", + "event_limit", +}) + + +def sensitive_key(key: str) -> bool: + key = re.sub(r"([A-Z]+)([A-Z][a-z])", r"\1_\2", key) + key = re.sub(r"([a-z0-9])([A-Z])", r"\1_\2", key) + normalized = "_".join(re.findall(r"[a-z0-9]+", key.lower())) + return ( + normalized in {"key", "apikey", "api_key", "access", "refresh", "token"} + or normalized.endswith(("_key", "_token")) + or any( + normalized == value or normalized.endswith("_" + value) + for value in ("secret", "password", "credential", "authorization", "cookie") + ) + ) + + +class UnsafeEvidence(ValueError): + def __init__(self, reason: str): + if reason not in REASONS: + reason = "unsupported_representation" + super().__init__(reason) + self.reason = reason + + +def _owned(value: Any, depth: int = 0, budget: list[int] | None = None) -> Any: + budget = [MAX_NODES] if budget is None else budget + budget[0] -= 1 + if depth > MAX_DEPTH or budget[0] < 0: + raise UnsafeEvidence("unsupported_representation") + if value is None or type(value) in (bool, int): + return value + if type(value) is float: + if not math.isfinite(value): + raise UnsafeEvidence("unsupported_representation") + return value + if type(value) is str: + try: + value.encode("utf-8", errors="strict") + except UnicodeError as exc: + raise UnsafeEvidence("unsupported_representation") from exc + return value + if type(value) is list: + return [_owned(item, depth + 1, budget) for item in value] + if type(value) is dict: + if not all(type(key) is str for key in value): + raise UnsafeEvidence("unsupported_representation") + return {key: _owned(item, depth + 1, budget) for key, item in value.items()} + raise UnsafeEvidence("unsupported_representation") + + +def _encoded_size(value: Any) -> int: + try: + return len( + json.dumps( + value, + ensure_ascii=False, + allow_nan=False, + separators=(",", ":"), + ).encode("utf-8") + ) + except (TypeError, ValueError, UnicodeError, RecursionError, OverflowError) as exc: + raise UnsafeEvidence("unsupported_representation") from exc + + +def _collect_sensitive_values(value: Any, out: set[str], depth: int = 0) -> None: + if depth > MAX_DEPTH: + raise UnsafeEvidence("credential_inventory_unavailable") + if isinstance(value, dict): + for key, item in value.items(): + if not isinstance(key, str): + raise UnsafeEvidence("credential_inventory_unavailable") + if sensitive_key(key): + _collect_scalar_credentials(item, out) + _collect_sensitive_values(item, out, depth + 1) + elif isinstance(value, list): + for item in value: + _collect_sensitive_values(item, out, depth + 1) + + +def _collect_scalar_credentials(value: Any, out: set[str]) -> None: + if isinstance(value, str): + if value: + out.add(value) + try: + nested = json.loads(value) + except (json.JSONDecodeError, TypeError): + return + if isinstance(nested, (dict, list)): + _collect_sensitive_values(nested, out) + return + if type(value) in (int, float) and not isinstance(value, bool): + if type(value) is float and not math.isfinite(value): + return + out.add(json.dumps(value, allow_nan=False)) + elif isinstance(value, (dict, list)): + _collect_sensitive_values(value, out) + + +def _json_source_credentials(path: Path, out: set[str]) -> bool: + if not path.is_file(): + return True + try: + if path.stat().st_size > JSON_SOURCE_LIMIT: + return False + value = json.loads(path.read_text(encoding="utf-8")) + _collect_sensitive_values(value, out) + return True + except (OSError, UnicodeError, json.JSONDecodeError, UnsafeEvidence, RecursionError): + return False + + +def _database_credentials(path: Path, out: set[str]) -> bool: + if not path.is_file(): + return True + try: + with sqlite3.connect(f"file:{path}?mode=ro", uri=True) as db: + columns = [row[1] for row in db.execute('PRAGMA table_info("credential")')] + if not columns: + return True + quoted = ", ".join('"' + col.replace('"', '""') + '"' for col in columns) + for row in db.execute(f'SELECT {quoted} FROM "credential"'): + for column, value in zip(columns, row): + if value is None or isinstance(value, bytes): + continue + if sensitive_key(column) or column.lower() in {"data", "value", "payload"}: + _collect_scalar_credentials(value, out) + elif isinstance(value, str) and value.lstrip().startswith(("{", "[")): + try: + nested = json.loads(value) + except json.JSONDecodeError: + continue + _collect_sensitive_values(nested, out) + return True + except (sqlite3.Error, OSError, UnsafeEvidence, RecursionError): + return False + + +class Sanitizer: + def __init__(self, credentials: Iterable[str] = (), *, inventory_complete: bool = True): + values = {value for value in credentials if isinstance(value, str) and value} + self.credentials = tuple(sorted(values, key=lambda value: (-len(value), value))) + self.inventory_complete = inventory_complete + + variants = set(self.credentials) + frontier = set(self.credentials) + for _ in range(SUPPORTED_JSON_ESCAPE_LAYERS): + generated = { + json.dumps(value, ensure_ascii=ascii_only)[1:-1] + for value in frontier + for ascii_only in (False, True) + } - variants + variants.update(generated) + frontier = generated + + self._matcher = ( + re.compile( + "|".join( + re.escape(value) + for value in sorted(variants, key=lambda value: (-len(value), value)) + ) + ) + if variants + else None + ) + + @classmethod + def from_runtime( + cls, + env: Mapping[str, str], + *, + json_sources: Iterable[Path] = (), + database_sources: Iterable[Path] = (), + ) -> "Sanitizer": + values: set[str] = set() + for name, value in env.items(): + if isinstance(value, str) and value and sensitive_key(name): + values.add(value) + + complete = True + for path in json_sources: + complete = _json_source_credentials(path, values) and complete + for path in database_sources: + complete = _database_credentials(path, values) and complete + return cls(values, inventory_complete=complete) + + def matches(self, value: str) -> bool: + return self._matcher is not None and self._matcher.search(value) is not None + + def redact_text(self, value: str) -> tuple[str, bool]: + if not self.inventory_complete: + return REDACTED, True + if self._matcher is None: + return value, False + safe = self._matcher.sub(REDACTED, value) + return safe, safe != value + + def evidence_value(self, value: Any) -> tuple[Any, bool]: + if not self.inventory_complete: + raise UnsafeEvidence("credential_inventory_unavailable") + value = _owned(value) + return self._evidence_value(value) + + def _evidence_value(self, value: Any) -> tuple[Any, bool]: + if type(value) is str: + return self.redact_text(value) + if type(value) is list: + safe_items = [] + changed = False + for item in value: + safe, item_changed = self._evidence_value(item) + safe_items.append(safe) + changed = changed or item_changed + return safe_items, changed + if type(value) is dict: + if any(sensitive_key(key) or self.matches(key) for key in value): + raise UnsafeEvidence("sensitive_key") + safe: dict[str, Any] = {} + changed = False + for key, item in value.items(): + safe_item, item_changed = self._evidence_value(item) + safe[key] = safe_item + changed = changed or item_changed + return safe, changed + if value is None or type(value) is bool: + return value, False + + encoded = json.dumps( + value, + ensure_ascii=False, + allow_nan=False, + separators=(",", ":"), + ) + if encoded in self.credentials: + raise UnsafeEvidence("credential_match") + return value, False + + def output_value(self, value: Any) -> Any: + if not self.inventory_complete: + return REDACTED + try: + value = _owned(value) + except UnsafeEvidence: + return REDACTED + return self._output_value(value) + + def _output_value(self, value: Any) -> Any: + if type(value) is dict: + safe: dict[str, Any] = {} + for child_key, item in value.items(): + if self.matches(child_key): + return {"__omitted__": "sensitive_key"} + if sensitive_key(child_key): + safe[child_key] = REDACTED + else: + safe[child_key] = self._output_value(item) + return safe + if type(value) is list: + return [self._output_value(item) for item in value] + if type(value) is str: + return self.redact_text(value)[0] + if value is None or type(value) is bool: + return value + + encoded = json.dumps( + value, + ensure_ascii=False, + allow_nan=False, + separators=(",", ":"), + ) + return REDACTED if encoded in self.credentials else value + + def json_lines(self, text: str) -> tuple[str, bool]: + if not text: + return text, False + + safe_lines: list[str] = [] + changed = False + for line in text.splitlines(): + try: + parsed = json.loads(line) + except json.JSONDecodeError: + safe, line_changed = self.redact_text(line) + safe_lines.append(safe) + changed = changed or line_changed + continue + + safe = self.output_value(parsed) + encoded = json.dumps( + safe, + ensure_ascii=False, + allow_nan=False, + separators=(",", ":"), + ) + safe_lines.append(encoded) + changed = changed or encoded != line + + suffix = "\n" if text.endswith("\n") else "" + return "\n".join(safe_lines) + suffix, changed + + +class Projection: + def __init__(self, sanitizer: Sanitizer, *, stage: str = "before_sink"): + self.sanitizer = sanitizer + self.stage = stage + self.fields: list[dict[str, Any]] = [] + self.losses = {reason: 0 for reason in REASONS} + + def _record( + self, + field: str, + state: str, + *, + event: int | None, + reason: str | None = None, + ) -> None: + item: dict[str, Any] = {"event": event, "field": field, "state": state} + if reason is not None: + item.update({"reason": reason, "stage": self.stage}) + self.losses[reason] += 1 + self.fields.append(item) + + def omit(self, field: str, reason: str, *, event: int | None = None) -> object: + if reason not in REASONS: + reason = "unsupported_representation" + self._record(field, "omitted", event=event, reason=reason) + return OMITTED + + def field( + self, + field: str, + value: Any = OMITTED, + *, + event: int | None = None, + limit: int = 6000, + protocol: bool = False, + ) -> Any: + if value is OMITTED: + return self.omit(field, "missing", event=event) + try: + if protocol: + safe = _owned(value) + changed = False + else: + safe, changed = self.sanitizer.evidence_value(value) + if _encoded_size(safe) > limit: + return self.omit(field, "size_limit", event=event) + except UnsafeEvidence as exc: + return self.omit(field, exc.reason, event=event) + except (TypeError, ValueError, UnicodeError, RecursionError, OverflowError): + return self.omit(field, "unsupported_representation", event=event) + + if changed: + self._record(field, "redacted", event=event, reason="credential_match") + else: + self._record(field, "exact", event=event) + return safe + + def loss(self, reason: str, *, count: int = 1) -> None: + if reason not in REASONS: + reason = "unsupported_representation" + self.losses[reason] += max(count, 0) + + def summary(self) -> dict[str, Any]: + losses = {key: value for key, value in self.losses.items() if value} + return { + "schema": SCHEMA, + "inventory_complete": self.sanitizer.inventory_complete, + "evidence_eligible": self.sanitizer.inventory_complete and not losses, + "fields": self.fields, + "losses": losses, + } From 9b37bfeefdeaee2bf99c4dcabf26ceb283159d86 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mats=20B=C3=B8e=20Bergmann?= Date: Tue, 6 Oct 2026 17:18:39 +0200 Subject: [PATCH 08/10] feat: sanitize evidence before clipping and output --- container/invoke.py | 298 ++++++++++++++++++++++++++++++++++---------- 1 file changed, 234 insertions(+), 64 deletions(-) diff --git a/container/invoke.py b/container/invoke.py index 1dc9111..86bc187 100644 --- a/container/invoke.py +++ b/container/invoke.py @@ -16,6 +16,11 @@ from pathlib import Path from typing import Any +try: + from .evidence_safety import OMITTED, Projection, Sanitizer +except ImportError: + from evidence_safety import OMITTED, Projection, Sanitizer + RESULT_SCHEMA = "opencode-eval-runner/v1" OPENCODE_EVAL_TITLE = "opencode-eval-runner" COPILOT_AGENT_NAME = "eval-runner" @@ -160,8 +165,13 @@ def _tool_result_text(value: Any, limit: int) -> tuple[str, bool]: return text[:head] + marker + text[-(retained - head):], True -def extract_tool_result_evidence(events: list[dict[str, Any]]) -> dict[str, Any]: - """Bound tool results from the full structured event stream before stdout clipping.""" +def extract_tool_result_evidence( + events: list[dict[str, Any]], + sanitizer: Sanitizer | None = None, +) -> dict[str, Any]: + """Project tool evidence safely before any field or total-size decision.""" + sanitizer = sanitizer or Sanitizer() + projection = Projection(sanitizer, stage="before_evidence_size") evidence: dict[str, Any] = { "schema": "opencode-eval-runner/tool-results/v1", "source": "opencode.event-stream.full", @@ -170,44 +180,144 @@ def extract_tool_result_evidence(events: list[dict[str, Any]]) -> dict[str, Any] "events": [], } recent: list[dict[str, Any]] = [] + + def project( + item: dict[str, Any], + event_index: int, + field: str, + value: Any = OMITTED, + *, + limit: int, + protocol: bool = False, + ) -> None: + safe = projection.field( + field, + value, + event=event_index, + limit=limit, + protocol=protocol, + ) + if safe is not OMITTED: + item[field] = safe + + def drop_event_fields(sequence: int) -> None: + event_index = sequence - 1 + projection.fields = [ + field + for field in projection.fields + if field.get("event") != event_index + ] + for event in events: if event.get("type") != "tool_use": continue - part = event.get("part") - if not isinstance(part, dict) or part.get("type") != "tool" or not isinstance(part.get("tool"), str): - continue - state = part.get("state") - if not isinstance(state, dict): - continue + evidence["observed_events"] += 1 - item: dict[str, Any] = { - "sequence": evidence["observed_events"], - "truncated_fields": [], - } - fields: dict[str, tuple[Any, int]] = { - "tool": (part["tool"], 256), - "status": (state.get("status", "unknown"), 256), - "input": (state.get("input", {}), 2000), - } - for key, value in (("call_id", part.get("callID")), ("session_id", event.get("sessionID"))): - if value is not None: - fields[key] = (value, 256) - for key in ("output", "error"): - if key in state: - fields[key] = (state[key], TOOL_RESULT_FIELD_LIMIT) - for key, (value, limit) in fields.items(): - item[key], clipped = _tool_result_text(value, limit) - if clipped: - item["truncated_fields"].append(key) + sequence = evidence["observed_events"] + event_index = sequence - 1 + item: dict[str, Any] = {"sequence": sequence} + part = event.get("part") + + if not isinstance(part, dict): + reason = "missing" if part is None else "unsupported_representation" + for field in ("tool", "status", "input", "call_id", "session_id"): + projection.omit(field, reason, event=event_index) + else: + project(item, event_index, "tool", part.get("tool", OMITTED), limit=256) + project( + item, + event_index, + "call_id", + part.get("callID", part.get("id", OMITTED)), + limit=256, + ) + project( + item, + event_index, + "session_id", + event.get("sessionID", event.get("sessionId", OMITTED)), + limit=256, + ) + + state = part.get("state", OMITTED) + if not isinstance(state, dict): + reason = "missing" if state is OMITTED else "unsupported_representation" + projection.omit("status", reason, event=event_index) + projection.omit("input", reason, event=event_index) + else: + status = state.get("status", OMITTED) + project( + item, + event_index, + "status", + status, + limit=256, + protocol=True, + ) + project( + item, + event_index, + "input", + state.get("input", OMITTED), + limit=2000, + ) + if status == "completed": + project( + item, + event_index, + "output", + state.get("output", OMITTED), + limit=TOOL_RESULT_FIELD_LIMIT, + ) + elif status in {"error", "failed"}: + project( + item, + event_index, + "error", + state.get("error", OMITTED), + limit=TOOL_RESULT_FIELD_LIMIT, + ) + else: + if "output" in state: + project( + item, + event_index, + "output", + state["output"], + limit=TOOL_RESULT_FIELD_LIMIT, + ) + if "error" in state: + project( + item, + event_index, + "error", + state["error"], + limit=TOOL_RESULT_FIELD_LIMIT, + ) + recent.append(item) if len(recent) > TOOL_RESULT_EVENT_LIMIT: - recent.pop(0) + dropped = recent.pop(0) + drop_event_fields(dropped["sequence"]) + evidence["omitted_events"] += 1 + projection.loss("event_limit") evidence["events"] = recent - evidence["omitted_events"] = evidence["observed_events"] - len(recent) - while len(json.dumps(evidence, ensure_ascii=False)) > TOOL_RESULT_TOTAL_LIMIT and evidence["events"]: - evidence["events"].pop(0) + evidence["safety"] = projection.summary() + evidence["evidence_eligible"] = evidence["safety"]["evidence_eligible"] + + while ( + len(json.dumps(evidence, ensure_ascii=False, separators=(",", ":"))) + > TOOL_RESULT_TOTAL_LIMIT + and evidence["events"] + ): + dropped = evidence["events"].pop(0) + drop_event_fields(dropped["sequence"]) evidence["omitted_events"] += 1 + projection.loss("size_limit") + evidence["safety"] = projection.summary() + evidence["evidence_eligible"] = False + return evidence @@ -674,6 +784,50 @@ def resolve_opencode_reasoning(model: str, reasoning: str) -> tuple[str, str, st return model, "provider-default", "provider-default" +RUNTIME_JSON_CREDENTIAL_SOURCES = ( + Path("/seed/auth.json"), + Path("/seed/opencode.json"), + Path("/seed/models.json"), +) +RUNTIME_DATABASE_CREDENTIAL_SOURCES = (Path("/seed/opencode.db"),) +SAFE_RESULT_PROTOCOL_FIELDS = frozenset({ + "schema", + "transport", + "reasoning_source", + "exit_code", + "timed_out", + "infrastructure_error", + "timing", + "stdout_truncated", + "stderr_truncated", + "stdout_total_chars", + "stderr_total_chars", + "tool_result_evidence", +}) + + +def runtime_sanitizer(env: dict[str, str]) -> Sanitizer: + return Sanitizer.from_runtime( + env, + json_sources=RUNTIME_JSON_CREDENTIAL_SOURCES, + database_sources=RUNTIME_DATABASE_CREDENTIAL_SOURCES, + ) + + +def sanitize_result_for_output( + result: dict[str, Any], + sanitizer: Sanitizer, +) -> dict[str, Any]: + """Final sink guard; product status stays separate from evidence eligibility.""" + safe: dict[str, Any] = {} + for key, value in result.items(): + if key in SAFE_RESULT_PROTOCOL_FIELDS: + safe[key] = value + else: + safe[key] = sanitizer.output_value(value) + return safe + + def invoke_opencode( model: str, agent: str, @@ -683,6 +837,7 @@ def invoke_opencode( reasoning: str = "", ) -> dict[str, Any]: env = prepare_opencode_env() + sanitizer = runtime_sanitizer(env) plugins = plugin_diagnostic(env) expected_plugin = os.environ.get("EVAL_EXPECT_PLUGIN", "").strip() plugin_preflight = verify_expected_plugin(env, agent, model, expected_plugin, timeout) @@ -715,9 +870,10 @@ def invoke_opencode( proc = run(command, Path("/workspace"), env, timeout) except subprocess.TimeoutExpired as exc: run_seconds = time.perf_counter() - run_started - stdout = timeout_output(exc.stdout) - stderr = timeout_output(exc.stderr) - events = parse_events(stdout) + raw_stdout = timeout_output(exc.stdout) + raw_stderr = timeout_output(exc.stderr) + events = parse_events(raw_stdout) + safe_stdout, _ = sanitizer.json_lines(raw_stdout) sid = session_id(events) summary = last_event_summary(events) detail = ( @@ -725,9 +881,10 @@ def invoke_opencode( f"partial_events={len(events)}" + (f"; last_event={json.dumps(summary, sort_keys=True)}" if summary else "") ) - if stderr.strip(): - detail += "\n" + stderr.strip() - return { + if raw_stderr.strip(): + detail += "\n" + raw_stderr.strip() + safe_detail, _ = sanitizer.json_lines(detail) + result = { "schema": RESULT_SCHEMA, "transport": "opencode", "model": model, @@ -748,19 +905,22 @@ def invoke_opencode( "export_exit_code": None, "total_seconds": round(run_seconds, 3), }, - "stderr": detail[:STDERR_CAPTURE_LIMIT], - "stderr_truncated": len(detail) > STDERR_CAPTURE_LIMIT, - "stderr_total_chars": len(detail), - "stdout": stdout[:STDOUT_CAPTURE_LIMIT], - "stdout_truncated": len(stdout) > STDOUT_CAPTURE_LIMIT, - "stdout_total_chars": len(stdout), - "tool_result_evidence": extract_tool_result_evidence(events), + "stderr": safe_detail[:STDERR_CAPTURE_LIMIT], + "stderr_truncated": len(safe_detail) > STDERR_CAPTURE_LIMIT, + "stderr_total_chars": len(safe_detail), + "stdout": safe_stdout[:STDOUT_CAPTURE_LIMIT], + "stdout_truncated": len(safe_stdout) > STDOUT_CAPTURE_LIMIT, + "stdout_total_chars": len(safe_stdout), + "tool_result_evidence": extract_tool_result_evidence(events, sanitizer), "plugin_diagnostic": plugins, "plugin_preflight": plugin_preflight, } + return sanitize_result_for_output(result, sanitizer) run_seconds = time.perf_counter() - run_started events = parse_events(proc.stdout) + safe_stdout, _ = sanitizer.json_lines(proc.stdout) + safe_stderr, _ = sanitizer.json_lines(proc.stderr) sid = session_id(events) # The structured `opencode run --format json` event stream is the @@ -775,7 +935,7 @@ def invoke_opencode( export_seconds = 0.0 export_exit_code: int | None = None - return { + result = { "schema": RESULT_SCHEMA, "transport": "opencode", "model": model, @@ -795,16 +955,17 @@ def invoke_opencode( "export_exit_code": export_exit_code, "total_seconds": round(run_seconds + export_seconds, 3), }, - "stderr": proc.stderr[:STDERR_CAPTURE_LIMIT], - "stderr_truncated": len(proc.stderr) > STDERR_CAPTURE_LIMIT, - "stderr_total_chars": len(proc.stderr), - "stdout": proc.stdout[:STDOUT_CAPTURE_LIMIT], - "stdout_truncated": len(proc.stdout) > STDOUT_CAPTURE_LIMIT, - "stdout_total_chars": len(proc.stdout), - "tool_result_evidence": extract_tool_result_evidence(events), + "stderr": safe_stderr[:STDERR_CAPTURE_LIMIT], + "stderr_truncated": len(safe_stderr) > STDERR_CAPTURE_LIMIT, + "stderr_total_chars": len(safe_stderr), + "stdout": safe_stdout[:STDOUT_CAPTURE_LIMIT], + "stdout_truncated": len(safe_stdout) > STDOUT_CAPTURE_LIMIT, + "stdout_total_chars": len(safe_stdout), + "tool_result_evidence": extract_tool_result_evidence(events, sanitizer), "plugin_diagnostic": plugins, "plugin_preflight": plugin_preflight, } + return sanitize_result_for_output(result, sanitizer) def copilot_auth_source(env: dict[str, str]) -> str | None: @@ -834,9 +995,10 @@ def invoke_copilot( reasoning: str = "", ) -> dict[str, Any]: env = dict(os.environ) + sanitizer = runtime_sanitizer(env) auth_source = copilot_auth_source(env) if not auth_source: - return { + return sanitize_result_for_output({ "schema": RESULT_SCHEMA, "transport": "github-copilot-cli", "model": model, @@ -852,7 +1014,7 @@ def invoke_copilot( "skills_loaded": [], "stderr": "github-copilot-cli requires COPILOT_GITHUB_TOKEN, GH_TOKEN, or GITHUB_TOKEN", "stdout": "", - } + }, sanitizer) root = Path("/tmp/copilot") work = root / "work" @@ -862,7 +1024,10 @@ def invoke_copilot( for path in (agent_dir, home, cache): path.mkdir(parents=True, exist_ok=True) - (agent_dir / f"{COPILOT_AGENT_NAME}.agent.md").write_text(copilot_profile(system), encoding="utf-8") + (agent_dir / f"{COPILOT_AGENT_NAME}.agent.md").write_text( + copilot_profile(system), + encoding="utf-8", + ) env["COPILOT_HOME"] = str(home) env["COPILOT_CACHE_HOME"] = str(cache) env["COPILOT_AUTO_UPDATE"] = "false" @@ -885,7 +1050,9 @@ def invoke_copilot( "--deny-tool", ",".join(COPILOT_DENIED_PERMISSIONS), ] proc = run(command, work, env, timeout) - return { + safe_stdout, _ = sanitizer.json_lines(proc.stdout) + safe_stderr, _ = sanitizer.json_lines(proc.stderr) + result = { "schema": RESULT_SCHEMA, "transport": "github-copilot-cli", "model": model, @@ -896,21 +1063,24 @@ def invoke_copilot( "credential_source": auth_source, "exit_code": proc.returncode, "session_id": None, - "text": proc.stdout.strip() if proc.returncode == 0 else "", + "text": safe_stdout.strip() if proc.returncode == 0 else "", "tools": [], "actions": [], "skills_loaded": [], - "stderr": proc.stderr[:STDERR_CAPTURE_LIMIT], - "stderr_truncated": len(proc.stderr) > STDERR_CAPTURE_LIMIT, - "stderr_total_chars": len(proc.stderr), - "stdout": proc.stdout[:STDOUT_CAPTURE_LIMIT], - "stdout_truncated": len(proc.stdout) > STDOUT_CAPTURE_LIMIT, - "stdout_total_chars": len(proc.stdout), + "stderr": safe_stderr[:STDERR_CAPTURE_LIMIT], + "stderr_truncated": len(safe_stderr) > STDERR_CAPTURE_LIMIT, + "stderr_total_chars": len(safe_stderr), + "stdout": safe_stdout[:STDOUT_CAPTURE_LIMIT], + "stdout_truncated": len(safe_stdout) > STDOUT_CAPTURE_LIMIT, + "stdout_total_chars": len(safe_stdout), } + return sanitize_result_for_output(result, sanitizer) def emit_result(result: dict[str, Any]) -> None: - sys.stdout.write(json.dumps(result, separators=(",", ":")) + "\n") + sanitizer = runtime_sanitizer(dict(os.environ)) + safe = sanitize_result_for_output(result, sanitizer) + sys.stdout.write(json.dumps(safe, separators=(",", ":")) + "\n") sys.stdout.flush() From a83a89986bd3476f016332ad7d111f20bd17c549 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mats=20B=C3=B8e=20Bergmann?= Date: Tue, 6 Oct 2026 17:20:50 +0200 Subject: [PATCH 09/10] test: cover trusted-checkout evidence safety --- tests/test_evidence_safety.py | 356 ++++++++++++++++++++++++++++++++++ 1 file changed, 356 insertions(+) create mode 100644 tests/test_evidence_safety.py diff --git a/tests/test_evidence_safety.py b/tests/test_evidence_safety.py new file mode 100644 index 0000000..c386af6 --- /dev/null +++ b/tests/test_evidence_safety.py @@ -0,0 +1,356 @@ +from __future__ import annotations + +import io +import json +import os +from pathlib import Path +import sqlite3 +import subprocess +import tempfile +import unittest +from unittest.mock import patch + +from container.evidence_safety import OMITTED, REDACTED, Projection, Sanitizer +from container.invoke import ( + emit_result, + extract_tool_result_evidence, + invoke_opencode, +) + + +def disposition(summary: dict, field: str, event: int | None = None) -> dict: + return next( + item + for item in summary["fields"] + if item["field"] == field and item["event"] == event + ) + + +class EvidenceSafetyTests(unittest.TestCase): + def test_short_credentials_preserve_json_structure_and_protocol_fields(self): + sanitizer = Sanitizer(["0", "1", "text", "low"]) + projection = Projection(sanitizer) + + status = projection.field("status", "completed", event=0, protocol=True) + payload = projection.field( + "output", + {"choice": "text", "priority": "low", "ok": True}, + event=0, + ) + numeric = projection.field("count", 0, event=0) + + self.assertEqual(status, "completed") + self.assertEqual( + payload, + {"choice": REDACTED, "priority": REDACTED, "ok": True}, + ) + self.assertIs(numeric, OMITTED) + self.assertEqual(disposition(projection.summary(), "output", 0)["state"], "redacted") + self.assertEqual( + disposition(projection.summary(), "count", 0)["reason"], + "credential_match", + ) + json.dumps(payload) + + def test_nested_sensitive_key_omits_enclosing_evidence_field(self): + projection = Projection(Sanitizer()) + raw = {"nested": {"apiKey": "not-in-inventory", "public": "ok"}} + + self.assertIs(projection.field("input", raw, event=0), OMITTED) + item = disposition(projection.summary(), "input", 0) + self.assertEqual(item["state"], "omitted") + self.assertEqual(item["reason"], "sensitive_key") + self.assertNotIn("not-in-inventory", json.dumps(projection.summary())) + + def test_escaped_credential_representations_are_protected(self): + secret = 'tok-"line\n\\ending' + sanitizer = Sanitizer([secret]) + value = secret + + for _ in range(4): + safe, changed = sanitizer.redact_text(value) + self.assertTrue(changed) + self.assertNotIn(secret, safe) + value = json.dumps(value)[1:-1] + + def test_redaction_happens_before_size_decision(self): + secret = "z" * 20_000 + projection = Projection(Sanitizer([secret])) + + safe = projection.field("output", secret, event=0, limit=64) + + self.assertEqual(safe, REDACTED) + self.assertEqual(disposition(projection.summary(), "output", 0)["state"], "redacted") + self.assertNotIn(secret, json.dumps(projection.summary())) + + def test_oversize_public_value_is_omitted_not_partially_kept(self): + projection = Projection(Sanitizer()) + raw = "safe-prefix-" + ("x" * 500) + + self.assertIs(projection.field("output", raw, event=0, limit=64), OMITTED) + item = disposition(projection.summary(), "output", 0) + self.assertEqual(item["reason"], "size_limit") + self.assertNotIn(raw[:32], json.dumps(projection.summary())) + + def test_missing_and_malformed_values_use_fixed_safe_reasons(self): + projection = Projection(Sanitizer()) + cycle: dict[str, object] = {} + cycle["self"] = cycle + + self.assertIs(projection.field("missing", event=0), OMITTED) + self.assertIs(projection.field("cycle", cycle, event=0), OMITTED) + self.assertIs(projection.field("nan", float("nan"), event=0), OMITTED) + + summary = projection.summary() + self.assertEqual(disposition(summary, "missing", 0)["reason"], "missing") + self.assertEqual( + disposition(summary, "cycle", 0)["reason"], + "unsupported_representation", + ) + self.assertEqual( + disposition(summary, "nan", 0)["reason"], + "unsupported_representation", + ) + self.assertNotIn("self", json.dumps(summary)) + + def test_runtime_inventory_reads_env_json_and_credential_database(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + auth = root / "auth.json" + auth.write_text( + json.dumps({"provider": {"key": "json-secret"}}), + encoding="utf-8", + ) + database = root / "opencode.db" + with sqlite3.connect(database) as db: + db.execute("CREATE TABLE credential (provider TEXT, data TEXT)") + db.execute( + "INSERT INTO credential(provider, data) VALUES (?, ?)", + ("fixture", json.dumps({"token": "db-secret"})), + ) + db.commit() + + sanitizer = Sanitizer.from_runtime( + {"OPENAI_API_KEY": "env-secret"}, + json_sources=(auth,), + database_sources=(database,), + ) + + self.assertTrue(sanitizer.inventory_complete) + self.assertEqual( + set(sanitizer.credentials), + {"env-secret", "json-secret", "db-secret"}, + ) + + def test_malformed_selected_inventory_fails_closed(self): + with tempfile.TemporaryDirectory() as tmp: + auth = Path(tmp) / "auth.json" + auth.write_text("{bad", encoding="utf-8") + sanitizer = Sanitizer.from_runtime({}, json_sources=(auth,)) + + self.assertFalse(sanitizer.inventory_complete) + projection = Projection(sanitizer) + self.assertIs(projection.field("output", "possibly private", event=0), OMITTED) + self.assertEqual( + disposition(projection.summary(), "output", 0)["reason"], + "credential_inventory_unavailable", + ) + self.assertFalse(projection.summary()["evidence_eligible"]) + + def test_success_failure_and_running_events_are_projected_without_invention(self): + secret = "fixture-secret" + events = [ + { + "type": "tool_use", + "sessionID": "ses-1", + "part": { + "type": "tool", + "tool": "demo", + "callID": "call-1", + "state": { + "status": "completed", + "input": {"value": "public"}, + "output": {"value": secret}, + }, + }, + }, + { + "type": "tool_use", + "sessionID": "ses-1", + "part": { + "type": "tool", + "tool": "demo", + "callID": "call-2", + "state": { + "status": "error", + "input": {"value": "public"}, + "error": f"failed: {secret}", + }, + }, + }, + { + "type": "tool_use", + "sessionID": "ses-1", + "part": { + "type": "tool", + "tool": "demo", + "callID": "call-3", + "state": { + "status": "running", + "input": {"value": "public"}, + }, + }, + }, + ] + + evidence = extract_tool_result_evidence(events, Sanitizer([secret])) + + self.assertEqual(evidence["observed_events"], 3) + self.assertEqual(evidence["events"][0]["status"], "completed") + self.assertEqual(evidence["events"][0]["output"], {"value": REDACTED}) + self.assertEqual(evidence["events"][1]["status"], "error") + self.assertEqual(evidence["events"][1]["error"], f"failed: {REDACTED}") + self.assertEqual(evidence["events"][2]["status"], "running") + self.assertNotIn("output", evidence["events"][2]) + self.assertNotIn("error", evidence["events"][2]) + self.assertFalse(evidence["evidence_eligible"]) + self.assertNotIn(secret, json.dumps(evidence)) + + def test_oversize_runtime_evidence_field_is_explicitly_omitted(self): + raw = "safe-prefix-" + ("x" * 7000) + events = [{ + "type": "tool_use", + "sessionID": "ses-1", + "part": { + "type": "tool", + "tool": "demo", + "callID": "call-1", + "state": { + "status": "completed", + "input": {"value": "public"}, + "output": raw, + }, + }, + }] + + evidence = extract_tool_result_evidence(events, Sanitizer()) + + self.assertNotIn("output", evidence["events"][0]) + self.assertEqual( + disposition(evidence["safety"], "output", 0)["reason"], + "size_limit", + ) + self.assertFalse(evidence["evidence_eligible"]) + self.assertNotIn(raw[:64], json.dumps(evidence)) + + def test_actual_invoke_path_never_exports_secret_and_keeps_product_failure(self): + secret = "ACTUAL-INVOKE-SECRET" + event = { + "type": "tool_use", + "sessionID": "ses-safe", + "part": { + "type": "tool", + "tool": "demo", + "callID": "call-safe", + "state": { + "status": "completed", + "input": {"query": "public"}, + "output": {"answer": secret}, + }, + }, + } + text_event = { + "type": "text", + "sessionID": "ses-safe", + "part": {"type": "text", "text": f"model echo {secret}"}, + } + + class Result: + returncode = 1 + stdout = "\n".join((json.dumps(event), json.dumps(text_event))) + stderr = f"provider failure {secret}" + + env = { + "OPENAI_API_KEY": secret, + "OPENCODE_CONFIG_DIR": "/tmp/nonexistent-opencode-config", + } + with patch("container.invoke.prepare_opencode_env", return_value=env), patch( + "container.invoke.run", + return_value=Result(), + ), patch.dict( + os.environ, + {"OPENAI_API_KEY": secret, "EVAL_EXPECT_PLUGIN": ""}, + clear=True, + ): + result = invoke_opencode( + "openai/fixture", + "general", + "prompt", + 30, + ) + output = io.StringIO() + with patch("container.invoke.sys.stdout", output): + emit_result(result) + + wire = output.getvalue() + parsed = json.loads(wire) + self.assertNotIn(secret, wire) + self.assertEqual(parsed["exit_code"], 1) + self.assertEqual( + parsed["tool_result_evidence"]["events"][0]["output"], + {"answer": REDACTED}, + ) + self.assertFalse(parsed["tool_result_evidence"]["evidence_eligible"]) + + def test_timeout_path_sanitizes_before_clipping_and_keeps_timeout_status(self): + secret = "TIMEOUT-SECRET" + event = { + "type": "tool_use", + "sessionID": "ses-timeout", + "part": { + "type": "tool", + "tool": "demo", + "callID": "call-timeout", + "state": { + "status": "running", + "input": {"query": secret}, + }, + }, + } + + def fail(command, cwd, env, timeout): + raise subprocess.TimeoutExpired( + command, + timeout, + output=json.dumps(event), + stderr=f"still running {secret}", + ) + + env = { + "OPENAI_API_KEY": secret, + "OPENCODE_CONFIG_DIR": "/tmp/nonexistent-opencode-config", + } + with patch("container.invoke.prepare_opencode_env", return_value=env), patch( + "container.invoke.run", + side_effect=fail, + ), patch.dict( + os.environ, + {"OPENAI_API_KEY": secret, "EVAL_EXPECT_PLUGIN": ""}, + clear=True, + ): + result = invoke_opencode( + "openai/fixture", + "general", + "prompt", + 1, + ) + + wire = json.dumps(result) + self.assertNotIn(secret, wire) + self.assertEqual(result["exit_code"], 124) + self.assertTrue(result["timed_out"]) + self.assertFalse(result["tool_result_evidence"]["evidence_eligible"]) + + +if __name__ == "__main__": + unittest.main() From 65ba6757b51a619d892fcf93689f31f539715714 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Mats=20B=C3=B8e=20Bergmann?= Date: Tue, 6 Oct 2026 17:22:05 +0200 Subject: [PATCH 10/10] fix: avoid treating credential JSON wrapper as a secret --- container/evidence_safety.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/container/evidence_safety.py b/container/evidence_safety.py index 5cf7d43..6ceb981 100644 --- a/container/evidence_safety.py +++ b/container/evidence_safety.py @@ -111,14 +111,17 @@ def _collect_sensitive_values(value: Any, out: set[str], depth: int = 0) -> None def _collect_scalar_credentials(value: Any, out: set[str]) -> None: if isinstance(value, str): - if value: - out.add(value) try: nested = json.loads(value) except (json.JSONDecodeError, TypeError): + if value: + out.add(value) return if isinstance(nested, (dict, list)): _collect_sensitive_values(nested, out) + return + if value: + out.add(value) return if type(value) in (int, float) and not isinstance(value, bool): if type(value) is float and not math.isfinite(value):