diff --git a/Containerfile b/Containerfile index 97fb1f9..3ec8003 100644 --- a/Containerfile +++ b/Containerfile @@ -1,5 +1,5 @@ FROM node:24-bookworm-slim@sha256:0e0ff40c39bc087845bfb27465a0df4ea419520094bc35842ff83dd8cbe6f9b6 AS opencode-builder -ARG OPENCODE_VERSION=2.0.18 +ARG OPENCODE_VERSION=2.0.23 RUN npm install --global "@opencode/cli@${OPENCODE_VERSION}" \ && resolved="$(readlink -f "$(command -v opencode)")" \ && test -x "$resolved" \ diff --git a/README.md b/README.md index 0f5c5a7..faa29c5 100644 --- a/README.md +++ b/README.md @@ -72,7 +72,7 @@ Known API-key environment variables are passed when present: Additional variables require explicit `--env NAME`. -Reasoning can be pinned explicitly with `--reasoning LEVEL`. The pinned OpenCode 2.0.18 CLI represents a model variant in the model reference, so the runner maps `--model provider/model --reasoning LEVEL` to `opencode run --model provider/model#LEVEL`. Supplying both a `#variant` in `--model` and `--reasoning` is rejected as ambiguous. If the model reference already contains a variant and `--reasoning` is omitted, the result records that variant with `"reasoning_source": "model-variant"`. If neither form supplies a level, the runner leaves OpenCode's provider/model default untouched and records `"reasoning": "provider-default"`. +Reasoning can be pinned explicitly with `--reasoning LEVEL`. The pinned stock OpenCode 2.0.23 CLI represents a model variant in the model reference, so the runner maps `--model provider/model --reasoning LEVEL` to `opencode run --model provider/model#LEVEL`. Supplying both a `#variant` in `--model` and `--reasoning` is rejected as ambiguous. If the model reference already contains a variant and `--reasoning` is omitted, the result records that variant with `"reasoning_source": "model-variant"`. If neither form supplies a level, the runner leaves OpenCode's provider/model default untouched and records `"reasoning": "provider-default"`. ### `github-copilot-cli` @@ -173,7 +173,7 @@ opencode-eval-runner invoke \ ... ``` -OpenCode 2.0.18 does not expose the old singular `debug agent ` command that returned a resolved tool map. The runner therefore performs the strongest supported zero-inference preflight: it requires the expected plugin entrypoint to be materialized in the isolated OpenCode config, runs `opencode debug agents` to prove the configured location starts successfully with plugins active, and requires the selected agent to resolve. Missing plugin materialization, plugin/startup failure, or missing agent is infrastructure/non-evidence, never a behavioral FAIL. Actual tool use remains a repository-owned behavioral assertion in the eval corpus. +Stock OpenCode 2.0.23 does not expose the old singular `debug agent ` command that returned a resolved tool map. The runner therefore performs the strongest supported zero-inference preflight: it requires the expected plugin entrypoint to be materialized in the isolated OpenCode config, runs `opencode debug agents` to prove the configured location starts successfully with plugins active, and requires the selected agent to resolve. Missing plugin materialization, plugin/startup failure, or missing agent is infrastructure/non-evidence, never a behavioral FAIL. Actual tool use remains a repository-owned behavioral assertion in the eval corpus. ### Evaluating a skill @@ -315,7 +315,7 @@ The eval repository decides whether that observed behavior is PASS, FAIL, or non The transport images currently pin: -- OpenCode CLI `2.0.18` +- OpenCode CLI `2.0.23` - GitHub Copilot CLI `1.0.83` The two CLIs are not bundled together. OpenCode's npm package is used only as a build-time native-binary selector; GitHub Copilot CLI is installed from its native release installer. Node/npm are absent from the final runtime images. @@ -340,6 +340,18 @@ OPENCODE_EVAL_RUNNER_COPILOT_IMAGE=... Tags matching `v*` are published with `opencode-` and `copilot-` prefixes. +## Evaluation trust model + +The normal evaluation profile is a **trusted-checkout** profile. It assumes the runner, pinned stock OpenCode runtime, reviewed instrumentation, and explicitly selected evaluated checkout/dependencies are trusted components of the evaluation environment. + +They are not trusted merely because they produce data that looks like evidence. Model prose, tool-returned collector-shaped JSON, target-writable files, requested actions, inferred identities, and reconstructed results do not establish that an event occurred. + +Authoritative runtime observations must come from reviewed instrumentation observing actual execution. Missing, partial, ambiguous, or unsupported required observations are non-evidence and must fail closed for the affected assertion. + +This profile does **not** claim resistance to an evaluated plugin that deliberately compromises the trusted runtime or instrumentation. Hostile-plugin isolation is a separate optional profile, not a prerequisite for normal Loom evaluation. + +See [Trusted-checkout runtime evidence](docs/trusted-checkout-evidence.md). + ## Security boundary The runner: diff --git a/docs/stock-opencode-2.0.23-observation.md b/docs/stock-opencode-2.0.23-observation.md new file mode 100644 index 0000000..e1653be --- /dev/null +++ b/docs/stock-opencode-2.0.23-observation.md @@ -0,0 +1,59 @@ +# Stock OpenCode 2.0.23 observation surface + +Purpose: implementation reference for the trusted-checkout evidence profile. + +Source checkpoint: stock OpenCode **v2.0.23** (`0fd7e2829449b052abf0078666669302923d77af`). This is distilled from the source assessment performed in superseded PR #43. + +OpenCode remains stock and immutable. A missing observation boundary is reported as unsupported; it is not a reason to patch OpenCode or add a hostile-runtime broker. + +## Useful stock surfaces + +| Observation need | Stock surface | Status | +| --- | --- | --- | +| Live runtime events | `ctx.event.subscribe()` | supported source; ordering/drain must be proven by integration test | +| Session creation / ancestry | `session.created` + Session API | supported | +| Agent for a step | Session step/message events | supported | +| Native tool call identity/input | Session tool input/called events | supported source | +| Native terminal success/failure | Session tool success/failed events | supported source | +| Tool pre-execution hook | `ctx.tool.hook("execute.before")` | supported; occurs before tool decode/execution | +| Tool post-handler hook | `ctx.tool.hook("execute.after")` | supported; occurs after handler result but before later core normalization | +| Tool registration wrapping | `ctx.tool.transform(...)` | supported candidate for reviewed same-process instrumentation | +| Code Mode inner name/input/status | Code Mode metadata + tool hooks | supported source | +| Code Mode unique inner invocation + exact final caller value/error | no single public final boundary demonstrated | **must be proven or marked unsupported** | + +## Native calls + +Stock Session events are the preferred source for native terminal facts because they represent the runtime's own Session lifecycle rather than model or tool payload claims. + +The observer must bind call identity, Session, agent/message context, input and terminal result/error without reconstructing them from prose or matching by value. + +## Code Mode + +Code Mode executes inner tools through the normal tool registry, so same-process reviewed instrumentation can observe real inner execution without isolating Loom. + +The difficult part is not security; it is exact correlation and finality: + +- inner calls share the outer `execute` context in stock OpenCode; +- public Code Mode metadata records name/input/status but not each inner returned value/error; +- `execute.after` is before later core normalization; +- concurrent identical inner calls must not be paired by FIFO, input equality, or completion order. + +The first implementation should test a runner-owned observer plugin using supported tool transforms/hooks and runtime events. It must allocate a unique observation identity at an actual execution boundary and prove how that identity reaches the final inner value/error. + +If that exact binding cannot be demonstrated for a case, the affected result field remains unavailable and the assertion cannot PASS. + +## Ordering and completeness + +A monotonic observer sequence is useful, but sequence alone is not completeness. The integration must also account for starts, terminals, observer loss, process interruption, and required descendant Sessions. + +Absence assertions are eligible only when the relevant scope is complete. Missing capture is never interpreted as "did not happen". + +## What is intentionally not required + +- plugin/process isolation from the trusted Loom checkout; +- remote PluginHost or capability broker; +- evidence-channel peer authentication against same-authority attackers; +- patched/forked OpenCode; +- cryptographic evidence authenticity after runtime compromise. + +Those belong only to a future optional untrusted-plugin profile. diff --git a/docs/trusted-checkout-evidence.md b/docs/trusted-checkout-evidence.md new file mode 100644 index 0000000..e0a503c --- /dev/null +++ b/docs/trusted-checkout-evidence.md @@ -0,0 +1,127 @@ +# Trusted-checkout runtime evidence + +Status: **replacement direction for TRUST-001**. + +This document supersedes the hostile-runtime direction explored in PR #41 and PR #43. Those PRs remain useful research/reference material, but normal Loom evaluation does not require the runner to defend itself from a deliberately malicious Loom checkout that shares its runtime authority. + +## Contract + +### TRUST-001 — authoritative runtime observation + +For evaluation of an explicitly trusted checkout, evidence used for scoring MUST originate from reviewed runtime instrumentation observing actual execution. + +The following MUST NOT independently establish that an event occurred: + +- model assertions or generated prose; +- tool payloads shaped like collector/evidence records; +- requested or intended actions; +- inferred actor, parent, or execution identity; +- reconstructed results; +- target-writable evidence files. + +Required observations MUST preserve enough runtime identity and ordering to evaluate the consumer contract, including actor/session/call identity, input, result or error, parent binding where applicable, and execution order. + +Missing, partial, ambiguous, lost, or unsupported required observations MUST make the affected assertion ineligible for PASS. The runner MUST NOT fill gaps from model text, stdout, workspace files, or guessed correlations. + +The trusted-checkout profile does not claim protection against malicious modification of the runner, stock OpenCode process, reviewed instrumentation, evaluated checkout, or their dependencies. + +## Trust model + +Trusted components: + +- the selected `opencode-eval-runner` revision; +- pinned **stock OpenCode 2.0.23**; +- reviewed runtime instrumentation; +- the explicitly selected Loom checkout and its reviewed dependencies; +- host-side evidence projection/persistence code. + +Not trusted as evidence authority: + +- model output; +- agent claims; +- tool-returned collector-shaped data; +- normal product/session/workspace files; +- caller-supplied identity or completeness claims. + +This is an evaluation-correctness boundary, not a hostile-code security boundary. + +## Required evidence behavior + +The target behavior remains strict even though the security scope is smaller: + +- **Native calls:** observe the actual runtime call, actor/session/call identity, accepted/executable input, and terminal result/error. +- **Code Mode inner calls:** assign a unique runtime observation identity per actual inner invocation, bind it to the real outer `execute` call, and observe the final value/error that Code Mode exposes to the script. +- **Delegation:** derive child Session identity and ancestry from runtime facts, not a parent result payload. +- **Ordering:** preserve runtime observation order; do not correlate concurrent calls by FIFO or input equality. +- **Completeness:** explicitly report missing starts/terminals, capture loss, unsupported boundaries, and incomplete scope. +- **Confidentiality:** redact or omit credentials before the runner first persists, clips, logs, or exports evidence. +- **Noninterference:** observation must not add product retries or change normal Loom execution semantics. + +If stock OpenCode's supported interfaces cannot expose an exact required boundary, the result is `unsupported`/ineligible for that assertion. The response is not to invent evidence and not to turn the normal profile into a hostile-code isolation project. + +## Implementation direction + +Keep the normal path: + +```text +Loom eval harness + -> opencode-eval-runner invoke + -> stock OpenCode 2.0.23 + + reviewed runner-owned observation instrumentation + + trusted Loom checkout + -> safe host projection + -> Loom judging +``` + +The preferred implementation is same-process reviewed instrumentation using supported stock OpenCode plugin/runtime surfaces. It may use a runner-owned observer plugin, tool/session hooks, live runtime events, and reviewed wrappers where those surfaces preserve the required boundary. + +OpenCode source patches, forks, remote PluginHost isolation, evidence signing, and a capability broker are not requirements of this profile. + +Provider-free integration tests must prove the exact observation/correlation behavior before a field becomes eligible evidence. + +See [Stock OpenCode 2.0.23 observation surface](stock-opencode-2.0.23-observation.md) for the retained source/capability findings from PR #43. + +## Reuse from PR #41 + +| Work | Disposition | +| --- | --- | +| Evidence-safety projection/redaction and fail-closed field handling | **Reuse/adapt**; keep the behavior, decouple it from hostile-runtime image/signing assumptions | +| Credential protection before host/file/print sinks | **Reuse** | +| Disposable OpenCode state/profile work | **Reuse where useful** for deterministic eval isolation | +| Normal `invoke` compatibility and provider-free integration probes | **Reuse/adapt** to stock 2.0.23 | +| Native/Code Mode observation schemas and concurrency tests | **Reuse as behavioral requirements/tests** | +| Delegated-session identity/ancestry probes | **Reuse** | +| Patched OpenCode runtime | **Drop** | +| HMAC observer/import trust boundary | **Drop** for the normal profile | +| protected-channel / remote tool service | **Drop** | +| plugin isolation / remote PluginHost work | **Drop** | +| Cosign evidence-authenticity machinery | **Drop** as a TRUST-001 prerequisite | +| adversarial same-authority attack tests | **Move to optional future untrusted profile** | + +## Reuse from PR #43 + +| Work | Disposition | +| --- | --- | +| Stock OpenCode 2.0.23 source/capability assessment | **Reuse** | +| Identification of public Session/event/tool surfaces | **Reuse** | +| Scope/completeness rules that prevent false absence/PASS | **Reuse and simplify** | +| First-sink confidentiality inventory | **Reuse and simplify** | +| Loom callback/capability inventory | **Reference when needed for compatibility** | +| hostile-runtime TCB/authority model | **Drop** from the normal profile | +| isolated Loom execution domain | **Drop** | +| capability/evidence-channel peer-authentication requirements | **Drop** | +| OCI adversarial boundary experiment/gates | **Drop** | + +## Implementation sequence + +This is normal engineering work, not a multi-authorization security experiment: + +1. Pin and verify stock OpenCode 2.0.23. +2. Add the smallest reviewed observation instrumentation that can capture native and Code Mode execution without changing product semantics. +3. Port the useful PR #41 evidence-safety projection so captured values are protected before persistence/export. +4. Add explicit completeness/loss fields and fail closed when required data is missing. +5. Exercise direct, Code Mode, delegation, error, timeout, and concurrent reverse-completion cases provider-free. +6. Compose through Loom's existing `eval:live -> run-evals.py -> invoke` path. +7. Only after those checks pass should Loom consume the new evidence schema for PASS/FAIL decisions. + +A future **untrusted-plugin execution profile** may add isolation if there is a real need to evaluate hostile plugin code. It must remain optional and separate from the normal trusted-checkout path. diff --git a/runner/evidence_accounting.py b/runner/evidence_accounting.py new file mode 100644 index 0000000..955ff25 --- /dev/null +++ b/runner/evidence_accounting.py @@ -0,0 +1,374 @@ +"""Fail-closed completeness accounting for normalized runtime observations. + +The capture adapters for native and Code Mode calls may differ. This module only +accounts for a small normalized stream and explicit capture-health signals. It +does not infer missing work from product output, stdout, or absent JSON fields. +""" +from __future__ import annotations + +from collections.abc import Iterable, Mapping +from typing import Any + +STATUSES = ("complete", "incomplete", "unsupported", "invalid") +FIELD_STATES = frozenset({"available", "redacted", "omitted", "truncated", "unsupported"}) +PROCESS_STATES = frozenset({"completed", "timeout", "interrupted"}) +EVENT_KINDS = frozenset({"start", "terminal"}) + +_GLOBAL_INVALID = frozenset({ + "invalid_accounting_input", + "malformed_observation", + "duplicate_sequence", + "duplicate_invocation", + "ambiguous_invocation", +}) +_GLOBAL_INCOMPLETE = frozenset({ + "observation_not_closed", + "observer_failure", + "callback_failure", + "observation_loss", + "runtime_timeout", + "process_interrupted", +}) + + +def _counter(value: Any) -> int | None: + return value if type(value) is int and value >= 0 else None + + +def _name(value: Any) -> str | None: + if isinstance(value, str) and 0 < len(value) <= 256: + return value + return None + + +def _new_boundary(*, declared_supported: bool = False, declared_unsupported: bool = False) -> dict[str, Any]: + return { + "status": "unsupported" if declared_unsupported else "complete", + "evidence_eligible": not declared_unsupported, + "declared_supported": declared_supported, + "declared_unsupported": declared_unsupported, + "starts": 0, + "terminals": 0, + "missing_terminals": 0, + "required_fields_omitted": 0, + "required_fields_truncated": 0, + "required_fields_unsupported": 0, + "issues": ["unsupported_boundary"] if declared_unsupported else [], + } + + +def _add_issue(target: list[str], code: str) -> None: + if code not in target: + target.append(code) + + +def _set_boundary_status(boundary: dict[str, Any], status: str, issue: str) -> None: + priority = {"complete": 0, "unsupported": 1, "incomplete": 2, "invalid": 3} + if priority[status] > priority[boundary["status"]]: + boundary["status"] = status + boundary["evidence_eligible"] = boundary["status"] == "complete" + _add_issue(boundary["issues"], issue) + + +def _invalid_result(issues: list[str], coverage: dict[str, Any]) -> dict[str, Any]: + return { + "status": "invalid", + "evidence_eligible": False, + "issues": issues or ["invalid_accounting_input"], + "coverage": coverage, + } + + +def account_runtime_evidence( + observations: Iterable[Mapping[str, Any]], + *, + observation_closed: bool, + supported_boundaries: Iterable[str] = (), + unsupported_boundaries: Iterable[str] = (), + observer_failures: int = 0, + callback_failures: int = 0, + losses: int = 0, + process_state: str = "completed", +) -> dict[str, Any]: + """Account for normalized runtime observation starts and terminals. + + Each observation must contain kind, sequence, invocation_id, + boundary, and required_fields. required_fields maps semantic + field names to available, redacted, omitted, truncated, or + unsupported. Adapters decide which fields are required; this layer only + accounts for their explicit states. + + observation_closed is an ordinary correctness signal from the capture + adapter that no more in-scope observations are expected. It is not a + cryptographic seal. Without it, absence is never complete evidence. + """ + coverage: dict[str, Any] = { + "observation_closed": observation_closed if type(observation_closed) is bool else False, + "starts": 0, + "terminals": 0, + "missing_terminals": 0, + "observer_failures": 0, + "callback_failures": 0, + "losses": 0, + "malformed_observations": 0, + "duplicate_invocations": 0, + "ambiguous_invocations": 0, + "duplicate_sequences": 0, + "required_fields_omitted": 0, + "required_fields_truncated": 0, + "required_fields_unsupported": 0, + "process_state": process_state if process_state in PROCESS_STATES else "invalid", + "supported_boundaries": [], + "unsupported_boundaries": [], + "by_boundary": {}, + } + issues: list[str] = [] + + failures = _counter(observer_failures) + callback_failure_count = _counter(callback_failures) + loss_count = _counter(losses) + if ( + type(observation_closed) is not bool + or failures is None + or callback_failure_count is None + or loss_count is None + or process_state not in PROCESS_STATES + ): + _add_issue(issues, "invalid_accounting_input") + return _invalid_result(issues, coverage) + coverage["observer_failures"] = failures + coverage["callback_failures"] = callback_failure_count + coverage["losses"] = loss_count + + supported: set[str] = set() + unsupported: set[str] = set() + for raw, target in ((supported_boundaries, supported), (unsupported_boundaries, unsupported)): + try: + values = list(raw) + except TypeError: + _add_issue(issues, "invalid_accounting_input") + return _invalid_result(issues, coverage) + for value in values: + name = _name(value) + if name is None: + _add_issue(issues, "invalid_accounting_input") + return _invalid_result(issues, coverage) + target.add(name) + if supported & unsupported: + _add_issue(issues, "invalid_accounting_input") + return _invalid_result(issues, coverage) + + coverage["supported_boundaries"] = sorted(supported) + coverage["unsupported_boundaries"] = sorted(unsupported) + boundaries: dict[str, dict[str, Any]] = { + name: _new_boundary(declared_supported=True) for name in sorted(supported) + } + for name in sorted(unsupported): + boundaries[name] = _new_boundary(declared_unsupported=True) + + calls: dict[str, dict[str, Any]] = {} + seen_sequences: set[int] = set() + + try: + stream = list(observations) + except TypeError: + _add_issue(issues, "invalid_accounting_input") + return _invalid_result(issues, coverage) + + for item in stream: + if not isinstance(item, Mapping): + coverage["malformed_observations"] += 1 + _add_issue(issues, "malformed_observation") + continue + + kind = item.get("kind") + sequence = item.get("sequence") + invocation_id = _name(item.get("invocation_id")) + boundary_name = _name(item.get("boundary")) + fields = item.get("required_fields") + valid_sequence = type(sequence) is int and sequence >= 0 + valid_fields = isinstance(fields, Mapping) + if ( + kind not in EVENT_KINDS + or not valid_sequence + or invocation_id is None + or boundary_name is None + or not valid_fields + ): + coverage["malformed_observations"] += 1 + _add_issue(issues, "malformed_observation") + continue + + field_states: dict[str, str] = {} + malformed_field = False + for field_name, state in fields.items(): + safe_name = _name(field_name) + if safe_name is None or state not in FIELD_STATES: + malformed_field = True + break + field_states[safe_name] = state + if malformed_field: + coverage["malformed_observations"] += 1 + _add_issue(issues, "malformed_observation") + continue + + boundary = boundaries.setdefault(boundary_name, _new_boundary()) + if boundary_name not in supported and boundary_name not in unsupported: + _set_boundary_status(boundary, "unsupported", "undeclared_boundary") + + if sequence in seen_sequences: + coverage["duplicate_sequences"] += 1 + _add_issue(issues, "duplicate_sequence") + _set_boundary_status(boundary, "invalid", "duplicate_sequence") + continue + seen_sequences.add(sequence) + + for state in field_states.values(): + if state == "omitted": + coverage["required_fields_omitted"] += 1 + boundary["required_fields_omitted"] += 1 + _set_boundary_status(boundary, "incomplete", "required_field_omitted") + elif state == "truncated": + coverage["required_fields_truncated"] += 1 + boundary["required_fields_truncated"] += 1 + _set_boundary_status(boundary, "incomplete", "required_field_truncated") + elif state == "unsupported": + coverage["required_fields_unsupported"] += 1 + boundary["required_fields_unsupported"] += 1 + _set_boundary_status(boundary, "unsupported", "required_field_unsupported") + + if kind == "start": + if invocation_id in calls: + coverage["duplicate_invocations"] += 1 + _add_issue(issues, "duplicate_invocation") + _set_boundary_status(boundary, "invalid", "duplicate_invocation") + continue + calls[invocation_id] = { + "boundary": boundary_name, + "start_sequence": sequence, + "terminal_sequence": None, + } + coverage["starts"] += 1 + boundary["starts"] += 1 + continue + + call = calls.get(invocation_id) + if call is None: + coverage["ambiguous_invocations"] += 1 + _add_issue(issues, "ambiguous_invocation") + _set_boundary_status(boundary, "invalid", "terminal_without_start") + continue + start_boundary = boundaries[call["boundary"]] + if call["boundary"] != boundary_name or call["terminal_sequence"] is not None or sequence <= call["start_sequence"]: + coverage["ambiguous_invocations"] += 1 + _add_issue(issues, "ambiguous_invocation") + _set_boundary_status(boundary, "invalid", "ambiguous_terminal") + _set_boundary_status(start_boundary, "invalid", "ambiguous_terminal") + continue + call["terminal_sequence"] = sequence + coverage["terminals"] += 1 + boundary["terminals"] += 1 + + for call in calls.values(): + if call["terminal_sequence"] is None: + coverage["missing_terminals"] += 1 + boundary = boundaries[call["boundary"]] + boundary["missing_terminals"] += 1 + _set_boundary_status(boundary, "incomplete", "missing_terminal") + + if not observation_closed: + _add_issue(issues, "observation_not_closed") + if failures: + _add_issue(issues, "observer_failure") + if callback_failure_count: + _add_issue(issues, "callback_failure") + if loss_count: + _add_issue(issues, "observation_loss") + if process_state == "timeout": + _add_issue(issues, "runtime_timeout") + elif process_state == "interrupted": + _add_issue(issues, "process_interrupted") + + if coverage["missing_terminals"]: + _add_issue(issues, "missing_terminal") + if coverage["required_fields_omitted"]: + _add_issue(issues, "required_field_omitted") + if coverage["required_fields_truncated"]: + _add_issue(issues, "required_field_truncated") + if coverage["required_fields_unsupported"]: + _add_issue(issues, "required_field_unsupported") + if unsupported: + _add_issue(issues, "unsupported_boundary") + undeclared = sorted(name for name in boundaries if name not in supported and name not in unsupported) + if undeclared: + _add_issue(issues, "undeclared_boundary") + + global_invalid = any(code in _GLOBAL_INVALID for code in issues) + global_incomplete = any(code in _GLOBAL_INCOMPLETE for code in issues) + for boundary in boundaries.values(): + if global_invalid: + _set_boundary_status(boundary, "invalid", "global_invalid_capture") + elif global_incomplete: + _set_boundary_status(boundary, "incomplete", "global_incomplete_capture") + + coverage["by_boundary"] = {name: boundaries[name] for name in sorted(boundaries)} + + if global_invalid or any(boundary["status"] == "invalid" for boundary in boundaries.values()): + status = "invalid" + elif global_incomplete or any(boundary["status"] == "incomplete" for boundary in boundaries.values()): + status = "incomplete" + elif any(boundary["status"] == "unsupported" for boundary in boundaries.values()): + status = "unsupported" + else: + status = "complete" + + return { + "status": status, + "evidence_eligible": status == "complete", + "issues": issues, + "coverage": coverage, + } + + +def assertion_status(accounting: Mapping[str, Any], required_boundaries: Iterable[str]) -> str: + """Return the fail-closed status for one assertion's required boundaries. + + This lets an unsupported Code Mode boundary avoid poisoning an unrelated + native-only assertion while global loss/invalidity still fails every scope. + """ + try: + coverage = accounting["coverage"] + boundaries = coverage["by_boundary"] + issues = accounting["issues"] + except (KeyError, TypeError): + return "invalid" + if not isinstance(coverage, Mapping) or not isinstance(boundaries, Mapping) or not isinstance(issues, list): + return "invalid" + if any(code in _GLOBAL_INVALID for code in issues): + return "invalid" + if any(code in _GLOBAL_INCOMPLETE for code in issues): + return "incomplete" + + try: + names = list(required_boundaries) + except TypeError: + return "invalid" + statuses: list[str] = [] + for raw_name in names: + name = _name(raw_name) + if name is None: + return "invalid" + boundary = boundaries.get(name) + if not isinstance(boundary, Mapping): + return "unsupported" + status = boundary.get("status") + if status not in STATUSES: + return "invalid" + statuses.append(status) + if "invalid" in statuses: + return "invalid" + if "incomplete" in statuses: + return "incomplete" + if "unsupported" in statuses: + return "unsupported" + return "complete" diff --git a/tests/test_evidence_accounting.py b/tests/test_evidence_accounting.py new file mode 100644 index 0000000..2e84949 --- /dev/null +++ b/tests/test_evidence_accounting.py @@ -0,0 +1,175 @@ +import unittest + +from runner.evidence_accounting import account_runtime_evidence, assertion_status + + +def start(invocation_id="a", sequence=1, boundary="native", **fields): + return { + "kind": "start", + "sequence": sequence, + "invocation_id": invocation_id, + "boundary": boundary, + "required_fields": fields or {"input": "available", "identity": "available"}, + } + + +def terminal(invocation_id="a", sequence=2, boundary="native", **fields): + return { + "kind": "terminal", + "sequence": sequence, + "invocation_id": invocation_id, + "boundary": boundary, + "required_fields": fields or {"outcome": "available", "result_or_error": "available"}, + } + + +class RuntimeEvidenceAccountingTests(unittest.TestCase): + def account(self, observations=(), **kwargs): + kwargs.setdefault("observation_closed", True) + kwargs.setdefault("supported_boundaries", ("native",)) + return account_runtime_evidence(observations, **kwargs) + + def test_complete_balanced_capture(self): + result = self.account([start(), terminal()]) + self.assertEqual(result["status"], "complete") + self.assertTrue(result["evidence_eligible"]) + self.assertEqual(result["coverage"]["starts"], 1) + self.assertEqual(result["coverage"]["terminals"], 1) + self.assertEqual(result["coverage"]["missing_terminals"], 0) + + def test_closed_supported_boundary_can_prove_zero_observed_calls(self): + result = self.account([]) + self.assertEqual(assertion_status(result, ["native"]), "complete") + self.assertEqual(result["coverage"]["by_boundary"]["native"]["starts"], 0) + + def test_unclosed_scope_cannot_prove_absence(self): + result = self.account([], observation_closed=False) + self.assertEqual(result["status"], "incomplete") + self.assertEqual(assertion_status(result, ["native"]), "incomplete") + self.assertIn("observation_not_closed", result["issues"]) + + def test_start_without_terminal_is_incomplete(self): + result = self.account([start()]) + self.assertEqual(result["status"], "incomplete") + self.assertEqual(result["coverage"]["missing_terminals"], 1) + self.assertIn("missing_terminal", result["issues"]) + + def test_observer_failure_and_loss_are_incomplete(self): + for kwargs, issue in ( + ({"observer_failures": 1}, "observer_failure"), + ({"callback_failures": 1}, "callback_failure"), + ({"losses": 2}, "observation_loss"), + ): + with self.subTest(issue=issue): + result = self.account([start(), terminal()], **kwargs) + self.assertEqual(result["status"], "incomplete") + self.assertIn(issue, result["issues"]) + + def test_malformed_observation_is_invalid(self): + result = self.account([{"kind": "start"}]) + self.assertEqual(result["status"], "invalid") + self.assertEqual(result["coverage"]["malformed_observations"], 1) + + def test_timeout_and_interruption_never_become_complete(self): + for state, issue in (("timeout", "runtime_timeout"), ("interrupted", "process_interrupted")): + with self.subTest(state=state): + result = self.account([start(), terminal()], process_state=state) + self.assertEqual(result["status"], "incomplete") + self.assertIn(issue, result["issues"]) + + def test_required_omitted_or_truncated_fields_are_incomplete(self): + omitted = self.account([start(input="omitted"), terminal()]) + truncated = self.account([start(), terminal(result="truncated")]) + self.assertEqual(omitted["status"], "incomplete") + self.assertEqual(truncated["status"], "incomplete") + self.assertEqual(omitted["coverage"]["required_fields_omitted"], 1) + self.assertEqual(truncated["coverage"]["required_fields_truncated"], 1) + + def test_required_unsupported_field_is_unsupported(self): + result = self.account([start(input="unsupported"), terminal()]) + self.assertEqual(result["status"], "unsupported") + self.assertEqual(assertion_status(result, ["native"]), "unsupported") + + def test_unsupported_boundary_only_affects_assertions_that_need_it(self): + result = account_runtime_evidence( + [start(), terminal()], + observation_closed=True, + supported_boundaries=("native",), + unsupported_boundaries=("code_mode_final",), + ) + self.assertEqual(result["status"], "unsupported") + self.assertEqual(assertion_status(result, ["native"]), "complete") + self.assertEqual(assertion_status(result, ["code_mode_final"]), "unsupported") + self.assertEqual(assertion_status(result, ["native", "code_mode_final"]), "unsupported") + + def test_unknown_required_boundary_is_unsupported_not_empty(self): + result = self.account([start(), terminal()]) + self.assertEqual(assertion_status(result, ["code_mode_final"]), "unsupported") + + def test_duplicate_invocation_identity_is_invalid(self): + result = self.account([start(sequence=1), start(sequence=2), terminal(sequence=3)]) + self.assertEqual(result["status"], "invalid") + self.assertEqual(result["coverage"]["duplicate_invocations"], 1) + + def test_orphan_duplicate_or_reversed_terminal_is_invalid(self): + cases = ( + [terminal(sequence=1)], + [start(sequence=1), terminal(sequence=2), terminal(sequence=3)], + [start(sequence=3), terminal(sequence=2)], + ) + for observations in cases: + with self.subTest(observations=observations): + result = self.account(observations) + self.assertEqual(result["status"], "invalid") + self.assertGreaterEqual(result["coverage"]["ambiguous_invocations"], 1) + + def test_duplicate_sequence_is_invalid(self): + result = self.account([start(sequence=1), terminal(sequence=1)]) + self.assertEqual(result["status"], "invalid") + self.assertEqual(result["coverage"]["duplicate_sequences"], 1) + + def test_concurrent_identical_calls_correlate_by_identity_not_fifo(self): + result = self.account([ + start("a", 1), + start("b", 2), + terminal("b", 3), + terminal("a", 4), + ]) + self.assertEqual(result["status"], "complete") + self.assertEqual(result["coverage"]["starts"], 2) + self.assertEqual(result["coverage"]["terminals"], 2) + + def test_product_failure_is_separate_from_capture_completeness(self): + failed_terminal = terminal() + failed_terminal["outcome"] = "failure" # Product data is intentionally irrelevant here. + result = self.account([start(), failed_terminal]) + self.assertEqual(result["status"], "complete") + + def test_global_capture_failure_affects_even_otherwise_supported_assertion(self): + result = account_runtime_evidence( + [start(), terminal()], + observation_closed=True, + supported_boundaries=("native",), + unsupported_boundaries=("code_mode_final",), + observer_failures=1, + ) + self.assertEqual(assertion_status(result, ["native"]), "incomplete") + + def test_invalid_accounting_inputs_fail_closed(self): + cases = ( + {"observer_failures": -1}, + {"callback_failures": -1}, + {"losses": True}, + {"process_state": "success"}, + {"observation_closed": 1}, + {"supported_boundaries": ("native",), "unsupported_boundaries": ("native",)}, + ) + for kwargs in cases: + with self.subTest(kwargs=kwargs): + kwargs.setdefault("observation_closed", True) + result = account_runtime_evidence([], **kwargs) + self.assertEqual(result["status"], "invalid") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_invoke.py b/tests/test_invoke.py index b1495e7..5622b59 100644 --- a/tests/test_invoke.py +++ b/tests/test_invoke.py @@ -706,11 +706,11 @@ def test_workflows_pin_external_actions_by_commit(self): ): self.assertNotIn(mutable, ci + publish) - def test_container_pins_opencode_2_0_18(self): + def test_container_pins_stock_opencode_2_0_23(self): containerfile = (Path(__file__).resolve().parents[1] / "Containerfile").read_text( encoding="utf-8" ) - self.assertIn("ARG OPENCODE_VERSION=2.0.18", containerfile) + self.assertIn("ARG OPENCODE_VERSION=2.0.23", containerfile) self.assertNotIn("ARG OPENCODE_VERSION=2.0.15", containerfile) def test_container_pins_base_images_and_copilot_release_asset(self):