diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 419945f..5331123 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -16,14 +16,14 @@ jobs: - name: Python checks run: | - python3 -m py_compile runner/cli.py container/invoke.py + python3 -m py_compile runner/cli.py container/invoke.py container/runtime_evidence.py python3 -m unittest discover -s tests -p 'test_*.py' - name: Build fake Copilot transport for action smoke test run: | cat > /tmp/Containerfile.fake-copilot <<'EOF' FROM alpine:3.22 - ENTRYPOINT ["/bin/sh", "-c", "if [ -z \"$GITHUB_TOKEN\" ] || [ -n \"$COPILOT_GITHUB_TOKEN\" ] || [ -n \"$GH_TOKEN\" ] || [ \"$EVAL_REASONING\" != \"medium\" ]; then exit 42; fi; printf '%s\\n' '{\"schema\":\"opencode-eval-runner/v1\",\"transport\":\"github-copilot-cli\",\"model\":\"fake\",\"reasoning\":\"medium\",\"reasoning_source\":\"explicit\",\"agent\":\"eval-runner\",\"skill\":null,\"exit_code\":0,\"session_id\":null,\"text\":\"fake action smoke\",\"tools\":[],\"actions\":[],\"skills_loaded\":[],\"stderr\":\"\",\"stdout\":\"\"}'"] + ENTRYPOINT ["/bin/sh", "-c", "if [ -z \"$GITHUB_TOKEN\" ] || [ -n \"$COPILOT_GITHUB_TOKEN\" ] || [ -n \"$GH_TOKEN\" ] || [ \"$EVAL_REASONING\" != \"medium\" ]; then exit 42; fi; printf '%s\\n' '{\"schema\":\"opencode-eval-runner/v1\",\"transport\":\"github-copilot-cli\",\"model\":\"fake\",\"reasoning\":\"medium\",\"reasoning_source\":\"explicit\",\"agent\":\"eval-runner\",\"skill\":null,\"exit_code\":0,\"session_id\":null,\"text\":\"fake action smoke\",\"tools\":[],\"actions\":[],\"skills_loaded\":[],\"stderr\":\"\",\"stdout\":\"\",\"runtime_evidence\":{\"schema\":\"opencode-eval-runner/runtime-evidence/v1\",\"status\":\"unsupported\",\"evidence_eligible\":false,\"observations\":[],\"coverage\":{\"starts\":{\"state\":\"unsupported\",\"reason\":\"observer_not_implemented\"},\"terminals\":{\"state\":\"unsupported\",\"reason\":\"observer_not_implemented\"},\"missing_terminals\":{\"state\":\"unsupported\",\"reason\":\"observer_not_implemented\"},\"losses\":[],\"unsupported\":[\"observer_not_implemented\"]}}}'"] EOF docker build -f /tmp/Containerfile.fake-copilot -t opencode-eval-runner:fake-copilot /tmp printf '%s\n' 'action smoke prompt' > /tmp/action-smoke-prompt.txt @@ -64,6 +64,8 @@ jobs: assert result["reasoning"] == "medium" assert result["reasoning_source"] == "explicit" assert result["text"] == "fake action smoke" + assert result["runtime_evidence"]["status"] == "unsupported" + assert result["runtime_evidence"]["evidence_eligible"] is False PY - name: Build OpenCode transport image diff --git a/Containerfile b/Containerfile index 97fb1f9..3ec8003 100644 --- a/Containerfile +++ b/Containerfile @@ -1,5 +1,5 @@ FROM node:24-bookworm-slim@sha256:0e0ff40c39bc087845bfb27465a0df4ea419520094bc35842ff83dd8cbe6f9b6 AS opencode-builder -ARG OPENCODE_VERSION=2.0.18 +ARG OPENCODE_VERSION=2.0.23 RUN npm install --global "@opencode/cli@${OPENCODE_VERSION}" \ && resolved="$(readlink -f "$(command -v opencode)")" \ && test -x "$resolved" \ diff --git a/README.md b/README.md index 0f5c5a7..6223f01 100644 --- a/README.md +++ b/README.md @@ -72,7 +72,7 @@ Known API-key environment variables are passed when present: Additional variables require explicit `--env NAME`. -Reasoning can be pinned explicitly with `--reasoning LEVEL`. The pinned OpenCode 2.0.18 CLI represents a model variant in the model reference, so the runner maps `--model provider/model --reasoning LEVEL` to `opencode run --model provider/model#LEVEL`. Supplying both a `#variant` in `--model` and `--reasoning` is rejected as ambiguous. If the model reference already contains a variant and `--reasoning` is omitted, the result records that variant with `"reasoning_source": "model-variant"`. If neither form supplies a level, the runner leaves OpenCode's provider/model default untouched and records `"reasoning": "provider-default"`. +Reasoning can be pinned explicitly with `--reasoning LEVEL`. The pinned stock OpenCode 2.0.23 CLI represents a model variant in the model reference, so the runner maps `--model provider/model --reasoning LEVEL` to `opencode run --model provider/model#LEVEL`. Supplying both a `#variant` in `--model` and `--reasoning` is rejected as ambiguous. If the model reference already contains a variant and `--reasoning` is omitted, the result records that variant with `"reasoning_source": "model-variant"`. If neither form supplies a level, the runner leaves OpenCode's provider/model default untouched and records `"reasoning": "provider-default"`. ### `github-copilot-cli` @@ -173,7 +173,7 @@ opencode-eval-runner invoke \ ... ``` -OpenCode 2.0.18 does not expose the old singular `debug agent ` command that returned a resolved tool map. The runner therefore performs the strongest supported zero-inference preflight: it requires the expected plugin entrypoint to be materialized in the isolated OpenCode config, runs `opencode debug agents` to prove the configured location starts successfully with plugins active, and requires the selected agent to resolve. Missing plugin materialization, plugin/startup failure, or missing agent is infrastructure/non-evidence, never a behavioral FAIL. Actual tool use remains a repository-owned behavioral assertion in the eval corpus. +Stock OpenCode 2.0.23 does not expose the old singular `debug agent ` command that returned a resolved tool map. The runner therefore performs the strongest supported zero-inference preflight: it requires the expected plugin entrypoint to be materialized in the isolated OpenCode config, runs `opencode debug agents` to prove the configured location starts successfully with plugins active, and requires the selected agent to resolve. Missing plugin materialization, plugin/startup failure, or missing agent is infrastructure/non-evidence, never a behavioral FAIL. Actual tool use remains a repository-owned behavioral assertion in the eval corpus. ### Evaluating a skill @@ -315,7 +315,7 @@ The eval repository decides whether that observed behavior is PASS, FAIL, or non The transport images currently pin: -- OpenCode CLI `2.0.18` +- OpenCode CLI `2.0.23` - GitHub Copilot CLI `1.0.83` The two CLIs are not bundled together. OpenCode's npm package is used only as a build-time native-binary selector; GitHub Copilot CLI is installed from its native release installer. Node/npm are absent from the final runtime images. @@ -340,6 +340,18 @@ OPENCODE_EVAL_RUNNER_COPILOT_IMAGE=... Tags matching `v*` are published with `opencode-` and `copilot-` prefixes. +## Evaluation trust model + +The normal evaluation profile is a **trusted-checkout** profile. It assumes the runner, pinned stock OpenCode runtime, reviewed instrumentation, and explicitly selected evaluated checkout/dependencies are trusted components of the evaluation environment. + +They are not trusted merely because they produce data that looks like evidence. Model prose, tool-returned collector-shaped JSON, target-writable files, requested actions, inferred identities, and reconstructed results do not establish that an event occurred. + +Authoritative runtime observations must come from reviewed instrumentation observing actual execution. Missing, partial, ambiguous, or unsupported required observations are non-evidence and must fail closed for the affected assertion. + +This profile does **not** claim resistance to an evaluated plugin that deliberately compromises the trusted runtime or instrumentation. Hostile-plugin isolation is a separate optional profile, not a prerequisite for normal Loom evaluation. + +See [Trusted-checkout runtime evidence](docs/trusted-checkout-evidence.md) and the [versioned runtime-evidence result contract](docs/runtime-evidence-contract.md). Until the observer lands, `runtime_evidence` is emitted as explicit `unsupported` non-evidence; existing `tools`, `actions`, `tool_result_evidence`, stdout, and similar fields remain convenience/diagnostic data only. + ## Security boundary The runner: diff --git a/container/invoke.py b/container/invoke.py index 1dc9111..62d334b 100644 --- a/container/invoke.py +++ b/container/invoke.py @@ -16,7 +16,13 @@ from pathlib import Path from typing import Any +try: + from .runtime_evidence import unsupported_runtime_evidence, validate_runtime_evidence +except ImportError: # direct container entrypoint + from runtime_evidence import unsupported_runtime_evidence, validate_runtime_evidence + RESULT_SCHEMA = "opencode-eval-runner/v1" +RUNTIME_EVIDENCE_UNSUPPORTED_REASON = "observer_not_implemented" OPENCODE_EVAL_TITLE = "opencode-eval-runner" COPILOT_AGENT_NAME = "eval-runner" COPILOT_AUTH_ENVS = ("COPILOT_GITHUB_TOKEN", "GH_TOKEN", "GITHUB_TOKEN") @@ -161,7 +167,10 @@ def _tool_result_text(value: Any, limit: int) -> tuple[str, bool]: def extract_tool_result_evidence(events: list[dict[str, Any]]) -> dict[str, Any]: - """Bound tool results from the full structured event stream before stdout clipping.""" + """Bound legacy tool-result diagnostics before stdout clipping. + + This historical field is not authoritative runtime_evidence. + """ evidence: dict[str, Any] = { "schema": "opencode-eval-runner/tool-results/v1", "source": "opencode.event-stream.full", @@ -755,6 +764,7 @@ def invoke_opencode( "stdout_truncated": len(stdout) > STDOUT_CAPTURE_LIMIT, "stdout_total_chars": len(stdout), "tool_result_evidence": extract_tool_result_evidence(events), + "runtime_evidence": unsupported_runtime_evidence(RUNTIME_EVIDENCE_UNSUPPORTED_REASON), "plugin_diagnostic": plugins, "plugin_preflight": plugin_preflight, } @@ -763,10 +773,11 @@ def invoke_opencode( events = parse_events(proc.stdout) sid = session_id(events) - # The structured `opencode run --format json` event stream is the - # authoritative evidence source. Starting a second OpenCode process to - # export the just-created session is redundant and can add a full timeout - # per invocation when export/session bootstrap fails. Keep eval latency + # The structured `opencode run --format json` event stream remains useful for + # product/convenience projections (`text`, `tools`, `actions`, and legacy + # `tool_result_evidence`). It is not authoritative `runtime_evidence`. Starting + # a second OpenCode process to export the just-created session is redundant and + # can add a full timeout per invocation when export/session bootstrap fails. Keep eval latency # bound to the requested target/judge execution only. text = extract_text(events) tools = extract_tools(events) @@ -802,6 +813,7 @@ def invoke_opencode( "stdout_truncated": len(proc.stdout) > STDOUT_CAPTURE_LIMIT, "stdout_total_chars": len(proc.stdout), "tool_result_evidence": extract_tool_result_evidence(events), + "runtime_evidence": unsupported_runtime_evidence(RUNTIME_EVIDENCE_UNSUPPORTED_REASON), "plugin_diagnostic": plugins, "plugin_preflight": plugin_preflight, } @@ -852,6 +864,7 @@ def invoke_copilot( "skills_loaded": [], "stderr": "github-copilot-cli requires COPILOT_GITHUB_TOKEN, GH_TOKEN, or GITHUB_TOKEN", "stdout": "", + "runtime_evidence": unsupported_runtime_evidence(RUNTIME_EVIDENCE_UNSUPPORTED_REASON), } root = Path("/tmp/copilot") @@ -906,10 +919,12 @@ def invoke_copilot( "stdout": proc.stdout[:STDOUT_CAPTURE_LIMIT], "stdout_truncated": len(proc.stdout) > STDOUT_CAPTURE_LIMIT, "stdout_total_chars": len(proc.stdout), + "runtime_evidence": unsupported_runtime_evidence(RUNTIME_EVIDENCE_UNSUPPORTED_REASON), } def emit_result(result: dict[str, Any]) -> None: + validate_runtime_evidence(result.get("runtime_evidence")) sys.stdout.write(json.dumps(result, separators=(",", ":")) + "\n") sys.stdout.flush() @@ -951,6 +966,7 @@ def main() -> int: "skills_loaded": [], "stderr": f"{type(exc).__name__}: {exc}", "stdout": "", + "runtime_evidence": unsupported_runtime_evidence(RUNTIME_EVIDENCE_UNSUPPORTED_REASON), "infrastructure_error": True, } emit_result(result) diff --git a/container/runtime_evidence.py b/container/runtime_evidence.py new file mode 100644 index 0000000..a73f98e --- /dev/null +++ b/container/runtime_evidence.py @@ -0,0 +1,251 @@ +from __future__ import annotations + +import json +from typing import Any + +RUNTIME_EVIDENCE_SCHEMA = "opencode-eval-runner/runtime-evidence/v1" + +STATUSES = {"complete", "incomplete", "unsupported", "invalid"} +FIELD_STATES = {"available", "redacted", "omitted", "unsupported"} +MODES = {"native", "code_mode"} +OUTCOMES = {"success", "error", "missing"} + +TOP_LEVEL_KEYS = {"schema", "status", "evidence_eligible", "observations", "coverage"} +OBSERVATION_KEYS = { + "invocation_id", + "tool", + "mode", + "actor", + "session_id", + "message_id", + "call_id", + "parent", + "input", + "outcome", + "result", + "error", + "start_sequence", + "terminal_sequence", +} +COVERAGE_KEYS = {"starts", "terminals", "missing_terminals", "losses", "unsupported"} + + +class RuntimeEvidenceError(ValueError): + """The runtime-evidence object is not safe to consume as contracted evidence.""" + + +def _require(condition: bool, message: str) -> None: + if not condition: + raise RuntimeEvidenceError(message) + + +def _exact_object(value: Any, keys: set[str], where: str) -> dict[str, Any]: + _require(type(value) is dict, f"{where} must be an object") + _require(set(value) == keys, f"{where} keys must be exactly {sorted(keys)}") + return value + + +def _string(value: Any, where: str) -> str: + _require(type(value) is str and bool(value.strip()), f"{where} must be a non-empty string") + return value + + +def _json_value(value: Any, where: str) -> None: + try: + json.dumps(value, ensure_ascii=False, allow_nan=False) + except (TypeError, ValueError, OverflowError) as exc: + raise RuntimeEvidenceError(f"{where} must be a finite JSON value") from exc + + +def field_available(value: Any) -> dict[str, Any]: + """Represent an exact projected runtime value. JSON null remains a real value.""" + _json_value(value, "field value") + return {"state": "available", "value": value} + + +def field_unavailable(state: str, reason: str) -> dict[str, str]: + """Represent a value that must not be replaced by an empty/default value.""" + _require(state in {"redacted", "omitted", "unsupported"}, "invalid unavailable field state") + _string(reason, "field reason") + return {"state": state, "reason": reason} + + +def _field( + raw: Any, + where: str, + *, + value_type: type | None = None, +) -> tuple[str, Any | None]: + _require(type(raw) is dict, f"{where} must be a field-state object") + state = raw.get("state") + _require(state in FIELD_STATES, f"{where}.state is invalid") + + if state == "available": + _require(set(raw) == {"state", "value"}, f"{where} available state requires state/value") + value = raw["value"] + _json_value(value, f"{where}.value") + if value_type is int: + _require(type(value) is int, f"{where}.value must be an integer") + elif value_type is str: + _string(value, f"{where}.value") + return state, value + + _require(set(raw) == {"state", "reason"}, f"{where} unavailable state requires state/reason") + _string(raw["reason"], f"{where}.reason") + return state, None + + +def _count(raw: Any, where: str) -> tuple[str, int | None]: + state, value = _field(raw, where, value_type=int) + _require(state in {"available", "unsupported"}, f"{where} must be available or unsupported") + if state == "available": + _require(value >= 0, f"{where}.value must be >= 0") + return state, value + + +def _codes(raw: Any, where: str) -> list[str]: + _require(type(raw) is list, f"{where} must be a list") + values = [_string(item, f"{where}[{index}]") for index, item in enumerate(raw)] + _require(len(values) == len(set(values)), f"{where} must not contain duplicates") + return values + + +def unsupported_runtime_evidence(reason: str) -> dict[str, Any]: + """Explicit non-evidence used until a reviewed observer can populate the contract.""" + _string(reason, "reason") + return { + "schema": RUNTIME_EVIDENCE_SCHEMA, + "status": "unsupported", + "evidence_eligible": False, + "observations": [], + "coverage": { + "starts": field_unavailable("unsupported", reason), + "terminals": field_unavailable("unsupported", reason), + "missing_terminals": field_unavailable("unsupported", reason), + "losses": [], + "unsupported": [reason], + }, + } + + +def validate_runtime_evidence(raw: Any) -> dict[str, Any]: + """Validate shape, sequencing, coverage, and fail-closed eligibility.""" + evidence = _exact_object(raw, TOP_LEVEL_KEYS, "runtime_evidence") + _require(evidence["schema"] == RUNTIME_EVIDENCE_SCHEMA, "unsupported runtime_evidence schema") + status = evidence["status"] + _require(status in STATUSES, "runtime_evidence.status is invalid") + _require(type(evidence["evidence_eligible"]) is bool, "runtime_evidence.evidence_eligible must be boolean") + _require(type(evidence["observations"]) is list, "runtime_evidence.observations must be a list") + + coverage = _exact_object(evidence["coverage"], COVERAGE_KEYS, "runtime_evidence.coverage") + starts_state, starts = _count(coverage["starts"], "runtime_evidence.coverage.starts") + terminals_state, terminals = _count(coverage["terminals"], "runtime_evidence.coverage.terminals") + missing_state, missing = _count(coverage["missing_terminals"], "runtime_evidence.coverage.missing_terminals") + losses = _codes(coverage["losses"], "runtime_evidence.coverage.losses") + unsupported = _codes(coverage["unsupported"], "runtime_evidence.coverage.unsupported") + + seen_invocations: set[str] = set() + seen_sequences: set[int] = set() + previous_start = -1 + terminal_observations = 0 + has_unsupported_field = False + + for index, raw_observation in enumerate(evidence["observations"]): + where = f"runtime_evidence.observations[{index}]" + observation = _exact_object(raw_observation, OBSERVATION_KEYS, where) + + invocation_id = _string(observation["invocation_id"], f"{where}.invocation_id") + _require(invocation_id not in seen_invocations, f"{where}.invocation_id must be unique") + seen_invocations.add(invocation_id) + + for name in ("tool", "actor", "session_id", "message_id", "call_id"): + state, _ = _field(observation[name], f"{where}.{name}", value_type=str) + has_unsupported_field |= state == "unsupported" + + _require(observation["mode"] in MODES, f"{where}.mode is invalid") + + parent_state, parent = _field(observation["parent"], f"{where}.parent") + has_unsupported_field |= parent_state == "unsupported" + if parent_state == "available": + parent = _exact_object(parent, {"kind", "id"}, f"{where}.parent.value") + _require(parent["kind"] in {"invocation", "session"}, f"{where}.parent.value.kind is invalid") + _string(parent["id"], f"{where}.parent.value.id") + + input_state, _ = _field(observation["input"], f"{where}.input") + has_unsupported_field |= input_state == "unsupported" + + outcome = observation["outcome"] + _require(outcome in OUTCOMES, f"{where}.outcome is invalid") + + result_state, _ = _field(observation["result"], f"{where}.result") + error_state, _ = _field(observation["error"], f"{where}.error") + has_unsupported_field |= result_state == "unsupported" or error_state == "unsupported" + + start_sequence = observation["start_sequence"] + _require(type(start_sequence) is int and start_sequence >= 0, f"{where}.start_sequence must be >= 0") + _require(start_sequence > previous_start, "observations must be ordered by increasing start_sequence") + _require(start_sequence not in seen_sequences, f"{where}.start_sequence must be unique") + previous_start = start_sequence + seen_sequences.add(start_sequence) + + terminal_state, terminal_sequence = _field( + observation["terminal_sequence"], f"{where}.terminal_sequence", value_type=int + ) + has_unsupported_field |= terminal_state == "unsupported" + + if outcome == "success": + _require(result_state in {"available", "redacted", "unsupported"}, f"{where}.result is invalid for success") + _require(error_state == "omitted", f"{where}.error must be omitted on success") + _require(terminal_state == "available", f"{where}.terminal_sequence must be available on success") + terminal_observations += 1 + elif outcome == "error": + _require(error_state in {"available", "redacted", "unsupported"}, f"{where}.error is invalid for error") + _require(result_state == "omitted", f"{where}.result must be omitted on error") + _require(terminal_state == "available", f"{where}.terminal_sequence must be available on error") + terminal_observations += 1 + else: + _require(result_state == error_state == "omitted", f"{where} missing outcome cannot carry result/error") + _require(terminal_state == "omitted", f"{where}.terminal_sequence must be omitted when terminal is missing") + + if terminal_state == "available": + _require(terminal_sequence >= 0, f"{where}.terminal_sequence must be >= 0") + _require(terminal_sequence > start_sequence, f"{where}.terminal_sequence must follow start_sequence") + _require(terminal_sequence not in seen_sequences, f"{where}.terminal_sequence must be unique") + seen_sequences.add(terminal_sequence) + + counts_available = starts_state == terminals_state == missing_state == "available" + coverage_complete = False + if counts_available: + _require(starts >= terminals, "coverage starts must be >= terminals") + _require(missing == starts - terminals, "coverage missing_terminals must equal starts - terminals") + _require(starts == len(evidence["observations"]), "coverage starts must equal observation count") + _require(terminals == terminal_observations, "coverage terminals must equal terminal observation count") + coverage_complete = missing == 0 and not losses and not unsupported + + if has_unsupported_field: + _require(unsupported, "unsupported observation fields must be reflected in coverage.unsupported") + + if status == "complete": + _require(counts_available and coverage_complete, "complete runtime evidence cannot contain coverage gaps") + _require(not has_unsupported_field, "complete runtime evidence cannot contain unsupported observation fields") + elif status == "unsupported": + _require(not evidence["observations"], "unsupported runtime evidence cannot contain observations") + _require(unsupported, "unsupported runtime evidence requires an unsupported reason") + _require( + starts_state == terminals_state == missing_state == "unsupported", + "unsupported runtime evidence requires unsupported coverage counts", + ) + elif status == "incomplete": + _require( + bool(losses or unsupported or not counts_available or not coverage_complete), + "incomplete runtime evidence must identify an incomplete condition", + ) + else: + _require(bool(losses or unsupported), "invalid runtime evidence must identify why capture was invalid") + + expected_eligible = status == "complete" and coverage_complete and not has_unsupported_field + _require( + evidence["evidence_eligible"] == expected_eligible, + "runtime_evidence.evidence_eligible does not match contract eligibility", + ) + return evidence diff --git a/docs/runtime-evidence-contract.md b/docs/runtime-evidence-contract.md new file mode 100644 index 0000000..b0047e9 --- /dev/null +++ b/docs/runtime-evidence-contract.md @@ -0,0 +1,173 @@ +# Runtime evidence result contract + +Schema: `opencode-eval-runner/runtime-evidence/v1` + +This is the authoritative runtime-observation contract for the trusted-checkout profile. It defines result shape and validation only. It does **not** implement the OpenCode observer. + +Until that observer is implemented and proven, the runner emits: + +```json +{ + "runtime_evidence": { + "schema": "opencode-eval-runner/runtime-evidence/v1", + "status": "unsupported", + "evidence_eligible": false, + "observations": [], + "coverage": { + "starts": {"state": "unsupported", "reason": "observer_not_implemented"}, + "terminals": {"state": "unsupported", "reason": "observer_not_implemented"}, + "missing_terminals": {"state": "unsupported", "reason": "observer_not_implemented"}, + "losses": [], + "unsupported": ["observer_not_implemented"] + } + } +} +``` + +The unsupported state is intentional. `0` would incorrectly claim that a complete observer saw zero calls. + +## Authority + +Only `runtime_evidence` may carry authoritative runtime observations for the trusted-checkout profile. + +Existing result fields remain useful product/diagnostic data, but are not substitutes: + +- `tools` and `actions` are projections from OpenCode product events; +- `tool_result_evidence` is a bounded legacy projection of tool-result-shaped product events despite its historical name; +- `stdout`, `stderr`, `text`, exported/session data, model prose, workspace files, and caller/tool-provided JSON are not runtime authority. + +Those fields may help debugging or presentation. They must not independently establish that a tool ran, which actor/session/call ran it, what final result/error the runtime returned, or that a call was absent. + +No HMAC, signing, protected channel, peer authentication, patched OpenCode, or hostile-plugin isolation is part of this schema. The trust boundary is the reviewed runner/runtime/instrumentation and explicitly trusted checkout described by TRUST-001. + +## Top-level fields + +| Field | Meaning | +| --- | --- | +| `schema` | Exact schema identifier. Consumers must reject unknown versions. | +| `status` | `complete`, `incomplete`, `unsupported`, or `invalid`. | +| `evidence_eligible` | Whether the captured **scope** is complete enough to be used as runtime evidence. Field-level availability must still be checked by each assertion. | +| `observations` | Tool invocations in strictly increasing `start_sequence` order. | +| `coverage` | Completeness accounting for starts, terminals, capture loss, and unsupported boundaries. | + +### Status + +- `complete`: coverage counts are known; no terminal is missing; no capture loss or unsupported observation boundary is reported. +- `incomplete`: some observations may be usable, but the captured scope is not complete. Assertions that require complete/absence evidence cannot PASS. +- `unsupported`: the runner cannot establish this runtime-evidence boundary. Observations are empty and coverage counts are explicitly `unsupported`, not zero. +- `invalid`: capture or projection was malformed, ambiguous, internally inconsistent, or otherwise unsafe to consume. + +`evidence_eligible` is `true` only for `status: complete` with complete coverage and no unsupported observation field. It does not make every observation field available. An assertion must also require `state: available` for every field value it depends on. + +This allows, for example, a call-existence assertion to remain usable when a result value had to be redacted, while a result-content assertion is ineligible. + +## Field states + +Fields whose value can be unavailable use one of these exact forms: + +```json +{"state": "available", "value": } +{"state": "redacted", "reason": ""} +{"state": "omitted", "reason": ""} +{"state": "unsupported", "reason": ""} +``` + +Meanings: + +- `available`: exact projected runtime value is present. JSON `null` is a real available value, not a missing value. +- `redacted`: the runtime value was observed but must not be exposed, for example because of credential protection. +- `omitted`: the field is intentionally absent or not applicable. It must never be interpreted as `""`, `[]`, `{}`, `0`, `false`, or `null`. +- `unsupported`: the reviewed observation boundary cannot establish the field. The corresponding boundary must also appear in `coverage.unsupported`. + +Missing keys are invalid. Unknown or unavailable values must never be converted to empty/default values. + +## Observation fields + +Each observation has exactly these fields: + +| Field | Meaning | +| --- | --- | +| `invocation_id` | Non-empty observer-assigned identity for one actual invocation. Unique within the capture. | +| `tool` | Field-state value containing the actual runtime tool name. | +| `mode` | `native` or `code_mode`. | +| `actor` | Field-state value containing the runtime-selected agent/actor identity. | +| `session_id` | Field-state value containing the runtime Session identity. | +| `message_id` | Field-state value containing the runtime message/step identity associated with the invocation. | +| `call_id` | Field-state value containing the runtime call identity. It is not assumed globally unique; `invocation_id` is the observation identity. | +| `parent` | Field-state value. When available it is `{"kind":"invocation"|"session","id":"..."}` and comes from runtime facts, never a child/result payload guess. Root calls use `omitted` with an explicit reason. | +| `input` | Field-state value containing the accepted/executable input at the observed execution boundary, not model prose or a requested action. | +| `outcome` | `success`, `error`, or `missing`. | +| `result` | Field-state final value returned to the caller for `success`; `omitted` for `error`/`missing`. | +| `error` | Field-state final error returned to the caller for `error`; `omitted` for `success`/`missing`. | +| `start_sequence` | Non-negative monotonic sequence assigned at actual invocation start. | +| `terminal_sequence` | Field-state non-negative sequence for the matching terminal. It must be later than `start_sequence`; it is `omitted` for `outcome: missing`. | + +Observations are ordered by `start_sequence`, never by completion order, input equality, FIFO matching, or result value. Concurrent identical calls therefore remain distinct. + +For Code Mode, `result`/`error` means the final value/error exposed to the Code Mode script. If stock OpenCode cannot expose and correlate that value exactly, the field/boundary is `unsupported`; it must not be reconstructed. + +## Coverage + +`coverage` has exactly: + +```text +starts +terminals +missing_terminals +losses +unsupported +``` + +`starts`, `terminals`, and `missing_terminals` are field-state integers. A known count is `available`; an unknown count is `unsupported`. Counts never default to zero. + +When counts are available: + +```text +missing_terminals == starts - terminals +starts == len(observations) +terminals == number of observations with outcome success|error +``` + +`losses` is a list of stable reason codes for capture loss or incompleteness, for example `capture_interrupted` or `record_limit`. + +`unsupported` is a list of stable reason codes for observation boundaries that the reviewed runtime cannot establish, for example `code_mode_final_result_unavailable`. + +A complete scope requires: + +```text +missing_terminals == 0 +losses == [] +unsupported == [] +``` + +Completeness is what permits absence assertions. An empty `observations` list is evidence of "no calls" only when status is `complete`, coverage counts are available and zero, and `evidence_eligible` is true. + +## Validation and fail-closed behavior + +The validator rejects: + +- missing or unknown keys; +- unknown schema/status/mode/outcome/field states; +- duplicate invocation IDs; +- out-of-order or reused sequences; +- terminal sequences that precede their starts; +- impossible result/error/outcome combinations; +- inconsistent coverage counts; +- `evidence_eligible: true` when completeness rules are not satisfied; +- unsupported observation fields that are not reflected in `coverage.unsupported`; +- unavailable values represented as empty/default values instead of an explicit field state. + +The container validates `runtime_evidence` immediately before emitting the result, and the host runner validates it again before persistence or printing. An alternate or stale image that omits or corrupts the contract is therefore rejected as infrastructure failure rather than accepted as behavioral evidence. + +## Relationship to PR #41 + +This contract reuses the useful fidelity ideas from PR #41: + +- unique invocation identity; +- start/terminal pairing; +- start-order preservation; +- explicit completeness accounting; +- field-level redaction/omission rather than fabricated defaults; +- fail-closed validation. + +It intentionally drops PR #41's hostile-runtime assumptions: no evidence signing, HMAC chain, protected channel, peer authentication, patched OpenCode, or separate hostile PluginHost is required for the trusted-checkout profile. diff --git a/docs/stock-opencode-2.0.23-observation.md b/docs/stock-opencode-2.0.23-observation.md new file mode 100644 index 0000000..e1653be --- /dev/null +++ b/docs/stock-opencode-2.0.23-observation.md @@ -0,0 +1,59 @@ +# Stock OpenCode 2.0.23 observation surface + +Purpose: implementation reference for the trusted-checkout evidence profile. + +Source checkpoint: stock OpenCode **v2.0.23** (`0fd7e2829449b052abf0078666669302923d77af`). This is distilled from the source assessment performed in superseded PR #43. + +OpenCode remains stock and immutable. A missing observation boundary is reported as unsupported; it is not a reason to patch OpenCode or add a hostile-runtime broker. + +## Useful stock surfaces + +| Observation need | Stock surface | Status | +| --- | --- | --- | +| Live runtime events | `ctx.event.subscribe()` | supported source; ordering/drain must be proven by integration test | +| Session creation / ancestry | `session.created` + Session API | supported | +| Agent for a step | Session step/message events | supported | +| Native tool call identity/input | Session tool input/called events | supported source | +| Native terminal success/failure | Session tool success/failed events | supported source | +| Tool pre-execution hook | `ctx.tool.hook("execute.before")` | supported; occurs before tool decode/execution | +| Tool post-handler hook | `ctx.tool.hook("execute.after")` | supported; occurs after handler result but before later core normalization | +| Tool registration wrapping | `ctx.tool.transform(...)` | supported candidate for reviewed same-process instrumentation | +| Code Mode inner name/input/status | Code Mode metadata + tool hooks | supported source | +| Code Mode unique inner invocation + exact final caller value/error | no single public final boundary demonstrated | **must be proven or marked unsupported** | + +## Native calls + +Stock Session events are the preferred source for native terminal facts because they represent the runtime's own Session lifecycle rather than model or tool payload claims. + +The observer must bind call identity, Session, agent/message context, input and terminal result/error without reconstructing them from prose or matching by value. + +## Code Mode + +Code Mode executes inner tools through the normal tool registry, so same-process reviewed instrumentation can observe real inner execution without isolating Loom. + +The difficult part is not security; it is exact correlation and finality: + +- inner calls share the outer `execute` context in stock OpenCode; +- public Code Mode metadata records name/input/status but not each inner returned value/error; +- `execute.after` is before later core normalization; +- concurrent identical inner calls must not be paired by FIFO, input equality, or completion order. + +The first implementation should test a runner-owned observer plugin using supported tool transforms/hooks and runtime events. It must allocate a unique observation identity at an actual execution boundary and prove how that identity reaches the final inner value/error. + +If that exact binding cannot be demonstrated for a case, the affected result field remains unavailable and the assertion cannot PASS. + +## Ordering and completeness + +A monotonic observer sequence is useful, but sequence alone is not completeness. The integration must also account for starts, terminals, observer loss, process interruption, and required descendant Sessions. + +Absence assertions are eligible only when the relevant scope is complete. Missing capture is never interpreted as "did not happen". + +## What is intentionally not required + +- plugin/process isolation from the trusted Loom checkout; +- remote PluginHost or capability broker; +- evidence-channel peer authentication against same-authority attackers; +- patched/forked OpenCode; +- cryptographic evidence authenticity after runtime compromise. + +Those belong only to a future optional untrusted-plugin profile. diff --git a/docs/trusted-checkout-evidence.md b/docs/trusted-checkout-evidence.md new file mode 100644 index 0000000..463218f --- /dev/null +++ b/docs/trusted-checkout-evidence.md @@ -0,0 +1,130 @@ +# Trusted-checkout runtime evidence + +Status: **replacement direction for TRUST-001**. + +This document supersedes the hostile-runtime direction explored in PR #41 and PR #43. Those PRs remain useful research/reference material, but normal Loom evaluation does not require the runner to defend itself from a deliberately malicious Loom checkout that shares its runtime authority. + +## Contract + +### TRUST-001 — authoritative runtime observation + +For evaluation of an explicitly trusted checkout, evidence used for scoring MUST originate from reviewed runtime instrumentation observing actual execution. + +The following MUST NOT independently establish that an event occurred: + +- model assertions or generated prose; +- tool payloads shaped like collector/evidence records; +- requested or intended actions; +- inferred actor, parent, or execution identity; +- reconstructed results; +- target-writable evidence files. + +Required observations MUST preserve enough runtime identity and ordering to evaluate the consumer contract, including actor/session/call identity, input, result or error, parent binding where applicable, and execution order. + +Missing, partial, ambiguous, lost, or unsupported required observations MUST make the affected assertion ineligible for PASS. The runner MUST NOT fill gaps from model text, stdout, workspace files, or guessed correlations. + +The concrete versioned result shape and fail-closed rules are defined in [Runtime evidence result contract](runtime-evidence-contract.md). Until the observer is implemented, the runner emits `runtime_evidence.status: unsupported` with unknown coverage counts explicitly marked `unsupported`; it does not turn unknown into zero. + +The trusted-checkout profile does not claim protection against malicious modification of the runner, stock OpenCode process, reviewed instrumentation, evaluated checkout, or their dependencies. + +## Trust model + +Trusted components: + +- the selected `opencode-eval-runner` revision; +- pinned **stock OpenCode 2.0.23**; +- reviewed runtime instrumentation; +- the explicitly selected Loom checkout and its reviewed dependencies; +- host-side evidence projection/persistence code. + +Not trusted as evidence authority: + +- legacy `tool_result_evidence`, `tools`, `actions`, stdout/stderr, and other product/diagnostic projections; +- model output; +- agent claims; +- tool-returned collector-shaped data; +- normal product/session/workspace files; +- caller-supplied identity or completeness claims. + +This is an evaluation-correctness boundary, not a hostile-code security boundary. + +## Required evidence behavior + +The target behavior remains strict even though the security scope is smaller: + +- **Native calls:** observe the actual runtime call, actor/session/call identity, accepted/executable input, and terminal result/error. +- **Code Mode inner calls:** assign a unique runtime observation identity per actual inner invocation, bind it to the real outer `execute` call, and observe the final value/error that Code Mode exposes to the script. +- **Delegation:** derive child Session identity and ancestry from runtime facts, not a parent result payload. +- **Ordering:** preserve runtime observation order; do not correlate concurrent calls by FIFO or input equality. +- **Completeness:** explicitly report missing starts/terminals, capture loss, unsupported boundaries, and incomplete scope. +- **Confidentiality:** redact or omit credentials before the runner first persists, clips, logs, or exports evidence. +- **Noninterference:** observation must not add product retries or change normal Loom execution semantics. + +If stock OpenCode's supported interfaces cannot expose an exact required boundary, the result is `unsupported`/ineligible for that assertion. The response is not to invent evidence and not to turn the normal profile into a hostile-code isolation project. + +## Implementation direction + +Keep the normal path: + +```text +Loom eval harness + -> opencode-eval-runner invoke + -> stock OpenCode 2.0.23 + + reviewed runner-owned observation instrumentation + + trusted Loom checkout + -> safe host projection + -> Loom judging +``` + +The preferred implementation is same-process reviewed instrumentation using supported stock OpenCode plugin/runtime surfaces. It may use a runner-owned observer plugin, tool/session hooks, live runtime events, and reviewed wrappers where those surfaces preserve the required boundary. + +OpenCode source patches, forks, remote PluginHost isolation, evidence signing, and a capability broker are not requirements of this profile. + +Provider-free integration tests must prove the exact observation/correlation behavior before a field becomes eligible evidence. + +See [Stock OpenCode 2.0.23 observation surface](stock-opencode-2.0.23-observation.md) for the retained source/capability findings from PR #43. + +## Reuse from PR #41 + +| Work | Disposition | +| --- | --- | +| Evidence-safety projection/redaction and fail-closed field handling | **Reuse/adapt**; keep the behavior, decouple it from hostile-runtime image/signing assumptions | +| Credential protection before host/file/print sinks | **Reuse** | +| Disposable OpenCode state/profile work | **Reuse where useful** for deterministic eval isolation | +| Normal `invoke` compatibility and provider-free integration probes | **Reuse/adapt** to stock 2.0.23 | +| Native/Code Mode observation schemas and concurrency tests | **Reuse as behavioral requirements/tests** | +| Delegated-session identity/ancestry probes | **Reuse** | +| Patched OpenCode runtime | **Drop** | +| HMAC observer/import trust boundary | **Drop** for the normal profile | +| protected-channel / remote tool service | **Drop** | +| plugin isolation / remote PluginHost work | **Drop** | +| Cosign evidence-authenticity machinery | **Drop** as a TRUST-001 prerequisite | +| adversarial same-authority attack tests | **Move to optional future untrusted profile** | + +## Reuse from PR #43 + +| Work | Disposition | +| --- | --- | +| Stock OpenCode 2.0.23 source/capability assessment | **Reuse** | +| Identification of public Session/event/tool surfaces | **Reuse** | +| Scope/completeness rules that prevent false absence/PASS | **Reuse and simplify** | +| First-sink confidentiality inventory | **Reuse and simplify** | +| Loom callback/capability inventory | **Reference when needed for compatibility** | +| hostile-runtime TCB/authority model | **Drop** from the normal profile | +| isolated Loom execution domain | **Drop** | +| capability/evidence-channel peer-authentication requirements | **Drop** | +| OCI adversarial boundary experiment/gates | **Drop** | + +## Implementation sequence + +This is normal engineering work, not a multi-authorization security experiment: + +1. Pin and verify stock OpenCode 2.0.23. +2. Add the smallest reviewed observation instrumentation that can capture native and Code Mode execution without changing product semantics. +3. Port the useful PR #41 evidence-safety projection so captured values are protected before persistence/export. +4. Add explicit completeness/loss fields and fail closed when required data is missing. +5. Exercise direct, Code Mode, delegation, error, timeout, and concurrent reverse-completion cases provider-free. +6. Compose through Loom's existing `eval:live -> run-evals.py -> invoke` path. +7. Only after those checks pass should Loom consume the new evidence schema for PASS/FAIL decisions. + +A future **untrusted-plugin execution profile** may add isolation if there is a real need to evaluate hostile plugin code. It must remain optional and separate from the normal trusted-checkout path. diff --git a/runner/cli.py b/runner/cli.py index dfe69e6..c68efc8 100644 --- a/runner/cli.py +++ b/runner/cli.py @@ -11,6 +11,8 @@ import tempfile from pathlib import Path +from container.runtime_evidence import RuntimeEvidenceError, validate_runtime_evidence + DEFAULT_IMAGES = { "opencode": "ghcr.io/bateau84/opencode-eval-runner:opencode-edge", "github-copilot-cli": "ghcr.io/bateau84/opencode-eval-runner:copilot-edge", @@ -34,6 +36,17 @@ class RunnerError(RuntimeError): pass +def validate_container_result(result: object) -> dict: + """Reject result objects that cannot carry contracted runtime evidence.""" + if not isinstance(result, dict): + raise RunnerError("container result must be a JSON object") + try: + validate_runtime_evidence(result.get("runtime_evidence")) + except RuntimeEvidenceError as exc: + raise RunnerError(f"container result has invalid runtime_evidence: {exc}") from exc + return result + + def default_auth_path() -> Path: base = Path(os.environ.get("XDG_DATA_HOME", Path.home() / ".local" / "share")) return base / "opencode" / "auth.json" @@ -381,8 +394,7 @@ def invoke(args: argparse.Namespace) -> int: + (f": {detail[:2000]}" if detail else "") ) from exc - if not isinstance(result, dict): - raise RunnerError("container result must be a JSON object") + result = validate_container_result(result) result_host.write_text(json.dumps(result, indent=2) + "\n", encoding="utf-8") if args.print_result: diff --git a/tests/test_invoke.py b/tests/test_invoke.py index b1495e7..5622b59 100644 --- a/tests/test_invoke.py +++ b/tests/test_invoke.py @@ -706,11 +706,11 @@ def test_workflows_pin_external_actions_by_commit(self): ): self.assertNotIn(mutable, ci + publish) - def test_container_pins_opencode_2_0_18(self): + def test_container_pins_stock_opencode_2_0_23(self): containerfile = (Path(__file__).resolve().parents[1] / "Containerfile").read_text( encoding="utf-8" ) - self.assertIn("ARG OPENCODE_VERSION=2.0.18", containerfile) + self.assertIn("ARG OPENCODE_VERSION=2.0.23", containerfile) self.assertNotIn("ARG OPENCODE_VERSION=2.0.15", containerfile) def test_container_pins_base_images_and_copilot_release_asset(self): diff --git a/tests/test_runtime_evidence.py b/tests/test_runtime_evidence.py new file mode 100644 index 0000000..a8d89b5 --- /dev/null +++ b/tests/test_runtime_evidence.py @@ -0,0 +1,151 @@ +from __future__ import annotations + +import io +import unittest +from contextlib import redirect_stdout + +from container.invoke import emit_result +from runner.cli import RunnerError, validate_container_result +from container.runtime_evidence import ( + RUNTIME_EVIDENCE_SCHEMA, + RuntimeEvidenceError, + field_available, + field_unavailable, + unsupported_runtime_evidence, + validate_runtime_evidence, +) + + +def complete_evidence() -> dict: + return { + "schema": RUNTIME_EVIDENCE_SCHEMA, + "status": "complete", + "evidence_eligible": True, + "observations": [ + { + "invocation_id": "obs-1", + "tool": field_available("read"), + "mode": "native", + "actor": field_available("general"), + "session_id": field_available("ses-1"), + "message_id": field_available("msg-1"), + "call_id": field_available("call-1"), + "parent": field_unavailable("omitted", "not_applicable"), + "input": field_available({"filePath": "/workspace/README.md"}), + "outcome": "success", + "result": field_available("contents"), + "error": field_unavailable("omitted", "not_applicable"), + "start_sequence": 4, + "terminal_sequence": field_available(5), + } + ], + "coverage": { + "starts": field_available(1), + "terminals": field_available(1), + "missing_terminals": field_available(0), + "losses": [], + "unsupported": [], + }, + } + + +class RuntimeEvidenceContractTests(unittest.TestCase): + def test_unsupported_contract_never_uses_zero_as_unknown(self): + evidence = unsupported_runtime_evidence("observer_not_implemented") + self.assertEqual(evidence["status"], "unsupported") + self.assertFalse(evidence["evidence_eligible"]) + self.assertEqual(evidence["observations"], []) + for name in ("starts", "terminals", "missing_terminals"): + self.assertEqual( + evidence["coverage"][name], + {"state": "unsupported", "reason": "observer_not_implemented"}, + ) + self.assertEqual(validate_runtime_evidence(evidence), evidence) + + def test_complete_evidence_is_eligible(self): + evidence = complete_evidence() + self.assertEqual(validate_runtime_evidence(evidence), evidence) + + def test_missing_field_fails_closed(self): + evidence = complete_evidence() + del evidence["observations"][0]["message_id"] + with self.assertRaisesRegex(RuntimeEvidenceError, "keys must be exactly"): + validate_runtime_evidence(evidence) + + def test_missing_terminal_cannot_be_eligible(self): + evidence = complete_evidence() + observation = evidence["observations"][0] + observation["outcome"] = "missing" + observation["result"] = field_unavailable("omitted", "missing_terminal") + observation["error"] = field_unavailable("omitted", "missing_terminal") + observation["terminal_sequence"] = field_unavailable("omitted", "missing_terminal") + evidence["coverage"]["terminals"] = field_available(0) + evidence["coverage"]["missing_terminals"] = field_available(1) + evidence["coverage"]["losses"] = ["missing_terminal"] + evidence["status"] = "incomplete" + evidence["evidence_eligible"] = True + with self.assertRaisesRegex(RuntimeEvidenceError, "evidence_eligible"): + validate_runtime_evidence(evidence) + + def test_unknown_coverage_is_not_replaced_with_zero(self): + evidence = complete_evidence() + evidence["status"] = "incomplete" + evidence["evidence_eligible"] = False + evidence["coverage"]["starts"] = field_unavailable("unsupported", "capture_interrupted") + evidence["coverage"]["terminals"] = field_unavailable("unsupported", "capture_interrupted") + evidence["coverage"]["missing_terminals"] = field_unavailable("unsupported", "capture_interrupted") + evidence["coverage"]["losses"] = ["capture_interrupted"] + self.assertEqual(validate_runtime_evidence(evidence), evidence) + + def test_unsupported_observation_field_requires_coverage_marker(self): + evidence = complete_evidence() + evidence["status"] = "incomplete" + evidence["evidence_eligible"] = False + evidence["observations"][0]["message_id"] = field_unavailable( + "unsupported", "message_identity_unavailable" + ) + with self.assertRaisesRegex(RuntimeEvidenceError, "coverage.unsupported"): + validate_runtime_evidence(evidence) + + def test_redaction_preserves_capture_eligibility_but_not_field_value(self): + evidence = complete_evidence() + evidence["observations"][0]["result"] = field_unavailable("redacted", "credential") + self.assertEqual(validate_runtime_evidence(evidence), evidence) + self.assertTrue(evidence["evidence_eligible"]) + self.assertNotIn("value", evidence["observations"][0]["result"]) + + def test_rejects_sequence_reuse(self): + evidence = complete_evidence() + evidence["observations"][0]["terminal_sequence"] = field_available(4) + with self.assertRaisesRegex(RuntimeEvidenceError, "must follow"): + validate_runtime_evidence(evidence) + + def test_emit_result_rejects_missing_runtime_evidence(self): + with redirect_stdout(io.StringIO()): + with self.assertRaises(RuntimeEvidenceError): + emit_result({"schema": "opencode-eval-runner/v1"}) + + def test_emit_result_accepts_explicit_unsupported_contract(self): + output = io.StringIO() + result = { + "schema": "opencode-eval-runner/v1", + "runtime_evidence": unsupported_runtime_evidence("observer_not_implemented"), + } + with redirect_stdout(output): + emit_result(result) + self.assertIn('"runtime_evidence"', output.getvalue()) + + def test_host_rejects_missing_runtime_evidence(self): + with self.assertRaisesRegex(RunnerError, "invalid runtime_evidence"): + validate_container_result({"schema": "opencode-eval-runner/v1"}) + + def test_host_accepts_explicit_unsupported_contract(self): + result = { + "schema": "opencode-eval-runner/v1", + "runtime_evidence": unsupported_runtime_evidence("observer_not_implemented"), + } + self.assertIs(validate_container_result(result), result) + + +if __name__ == "__main__": + unittest.main()