Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 4 additions & 2 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -16,14 +16,14 @@ jobs:

- name: Python checks
run: |
python3 -m py_compile runner/cli.py container/invoke.py
python3 -m py_compile runner/cli.py container/invoke.py container/runtime_evidence.py
python3 -m unittest discover -s tests -p 'test_*.py'

- name: Build fake Copilot transport for action smoke test
run: |
cat > /tmp/Containerfile.fake-copilot <<'EOF'
FROM alpine:3.22
ENTRYPOINT ["/bin/sh", "-c", "if [ -z \"$GITHUB_TOKEN\" ] || [ -n \"$COPILOT_GITHUB_TOKEN\" ] || [ -n \"$GH_TOKEN\" ] || [ \"$EVAL_REASONING\" != \"medium\" ]; then exit 42; fi; printf '%s\\n' '{\"schema\":\"opencode-eval-runner/v1\",\"transport\":\"github-copilot-cli\",\"model\":\"fake\",\"reasoning\":\"medium\",\"reasoning_source\":\"explicit\",\"agent\":\"eval-runner\",\"skill\":null,\"exit_code\":0,\"session_id\":null,\"text\":\"fake action smoke\",\"tools\":[],\"actions\":[],\"skills_loaded\":[],\"stderr\":\"\",\"stdout\":\"\"}'"]
ENTRYPOINT ["/bin/sh", "-c", "if [ -z \"$GITHUB_TOKEN\" ] || [ -n \"$COPILOT_GITHUB_TOKEN\" ] || [ -n \"$GH_TOKEN\" ] || [ \"$EVAL_REASONING\" != \"medium\" ]; then exit 42; fi; printf '%s\\n' '{\"schema\":\"opencode-eval-runner/v1\",\"transport\":\"github-copilot-cli\",\"model\":\"fake\",\"reasoning\":\"medium\",\"reasoning_source\":\"explicit\",\"agent\":\"eval-runner\",\"skill\":null,\"exit_code\":0,\"session_id\":null,\"text\":\"fake action smoke\",\"tools\":[],\"actions\":[],\"skills_loaded\":[],\"stderr\":\"\",\"stdout\":\"\",\"runtime_evidence\":{\"schema\":\"opencode-eval-runner/runtime-evidence/v1\",\"status\":\"unsupported\",\"evidence_eligible\":false,\"observations\":[],\"coverage\":{\"starts\":{\"state\":\"unsupported\",\"reason\":\"observer_not_implemented\"},\"terminals\":{\"state\":\"unsupported\",\"reason\":\"observer_not_implemented\"},\"missing_terminals\":{\"state\":\"unsupported\",\"reason\":\"observer_not_implemented\"},\"losses\":[],\"unsupported\":[\"observer_not_implemented\"]}}}'"]
EOF
docker build -f /tmp/Containerfile.fake-copilot -t opencode-eval-runner:fake-copilot /tmp
printf '%s\n' 'action smoke prompt' > /tmp/action-smoke-prompt.txt
Expand Down Expand Up @@ -64,6 +64,8 @@ jobs:
assert result["reasoning"] == "medium"
assert result["reasoning_source"] == "explicit"
assert result["text"] == "fake action smoke"
assert result["runtime_evidence"]["status"] == "unsupported"
assert result["runtime_evidence"]["evidence_eligible"] is False
PY

- name: Build OpenCode transport image
Expand Down
2 changes: 1 addition & 1 deletion Containerfile
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
FROM node:24-bookworm-slim@sha256:0e0ff40c39bc087845bfb27465a0df4ea419520094bc35842ff83dd8cbe6f9b6 AS opencode-builder
ARG OPENCODE_VERSION=2.0.18
ARG OPENCODE_VERSION=2.0.23
RUN npm install --global "@opencode/cli@${OPENCODE_VERSION}" \
&& resolved="$(readlink -f "$(command -v opencode)")" \
&& test -x "$resolved" \
Expand Down
18 changes: 15 additions & 3 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -72,7 +72,7 @@ Known API-key environment variables are passed when present:

Additional variables require explicit `--env NAME`.

Reasoning can be pinned explicitly with `--reasoning LEVEL`. The pinned OpenCode 2.0.18 CLI represents a model variant in the model reference, so the runner maps `--model provider/model --reasoning LEVEL` to `opencode run --model provider/model#LEVEL`. Supplying both a `#variant` in `--model` and `--reasoning` is rejected as ambiguous. If the model reference already contains a variant and `--reasoning` is omitted, the result records that variant with `"reasoning_source": "model-variant"`. If neither form supplies a level, the runner leaves OpenCode's provider/model default untouched and records `"reasoning": "provider-default"`.
Reasoning can be pinned explicitly with `--reasoning LEVEL`. The pinned stock OpenCode 2.0.23 CLI represents a model variant in the model reference, so the runner maps `--model provider/model --reasoning LEVEL` to `opencode run --model provider/model#LEVEL`. Supplying both a `#variant` in `--model` and `--reasoning` is rejected as ambiguous. If the model reference already contains a variant and `--reasoning` is omitted, the result records that variant with `"reasoning_source": "model-variant"`. If neither form supplies a level, the runner leaves OpenCode's provider/model default untouched and records `"reasoning": "provider-default"`.

### `github-copilot-cli`

Expand Down Expand Up @@ -173,7 +173,7 @@ opencode-eval-runner invoke \
...
```

OpenCode 2.0.18 does not expose the old singular `debug agent <id>` command that returned a resolved tool map. The runner therefore performs the strongest supported zero-inference preflight: it requires the expected plugin entrypoint to be materialized in the isolated OpenCode config, runs `opencode debug agents` to prove the configured location starts successfully with plugins active, and requires the selected agent to resolve. Missing plugin materialization, plugin/startup failure, or missing agent is infrastructure/non-evidence, never a behavioral FAIL. Actual tool use remains a repository-owned behavioral assertion in the eval corpus.
Stock OpenCode 2.0.23 does not expose the old singular `debug agent <id>` command that returned a resolved tool map. The runner therefore performs the strongest supported zero-inference preflight: it requires the expected plugin entrypoint to be materialized in the isolated OpenCode config, runs `opencode debug agents` to prove the configured location starts successfully with plugins active, and requires the selected agent to resolve. Missing plugin materialization, plugin/startup failure, or missing agent is infrastructure/non-evidence, never a behavioral FAIL. Actual tool use remains a repository-owned behavioral assertion in the eval corpus.

### Evaluating a skill

Expand Down Expand Up @@ -315,7 +315,7 @@ The eval repository decides whether that observed behavior is PASS, FAIL, or non

The transport images currently pin:

- OpenCode CLI `2.0.18`
- OpenCode CLI `2.0.23`
- GitHub Copilot CLI `1.0.83`

The two CLIs are not bundled together. OpenCode's npm package is used only as a build-time native-binary selector; GitHub Copilot CLI is installed from its native release installer. Node/npm are absent from the final runtime images.
Expand All @@ -340,6 +340,18 @@ OPENCODE_EVAL_RUNNER_COPILOT_IMAGE=...

Tags matching `v*` are published with `opencode-` and `copilot-` prefixes.

## Evaluation trust model

The normal evaluation profile is a **trusted-checkout** profile. It assumes the runner, pinned stock OpenCode runtime, reviewed instrumentation, and explicitly selected evaluated checkout/dependencies are trusted components of the evaluation environment.

They are not trusted merely because they produce data that looks like evidence. Model prose, tool-returned collector-shaped JSON, target-writable files, requested actions, inferred identities, and reconstructed results do not establish that an event occurred.

Authoritative runtime observations must come from reviewed instrumentation observing actual execution. Missing, partial, ambiguous, or unsupported required observations are non-evidence and must fail closed for the affected assertion.

This profile does **not** claim resistance to an evaluated plugin that deliberately compromises the trusted runtime or instrumentation. Hostile-plugin isolation is a separate optional profile, not a prerequisite for normal Loom evaluation.

See [Trusted-checkout runtime evidence](docs/trusted-checkout-evidence.md) and the [versioned runtime-evidence result contract](docs/runtime-evidence-contract.md). Until the observer lands, `runtime_evidence` is emitted as explicit `unsupported` non-evidence; existing `tools`, `actions`, `tool_result_evidence`, stdout, and similar fields remain convenience/diagnostic data only.

## Security boundary

The runner:
Expand Down
26 changes: 21 additions & 5 deletions container/invoke.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,13 @@
from pathlib import Path
from typing import Any

try:
from .runtime_evidence import unsupported_runtime_evidence, validate_runtime_evidence
except ImportError: # direct container entrypoint
from runtime_evidence import unsupported_runtime_evidence, validate_runtime_evidence

RESULT_SCHEMA = "opencode-eval-runner/v1"
RUNTIME_EVIDENCE_UNSUPPORTED_REASON = "observer_not_implemented"
OPENCODE_EVAL_TITLE = "opencode-eval-runner"
COPILOT_AGENT_NAME = "eval-runner"
COPILOT_AUTH_ENVS = ("COPILOT_GITHUB_TOKEN", "GH_TOKEN", "GITHUB_TOKEN")
Expand Down Expand Up @@ -161,7 +167,10 @@ def _tool_result_text(value: Any, limit: int) -> tuple[str, bool]:


def extract_tool_result_evidence(events: list[dict[str, Any]]) -> dict[str, Any]:
"""Bound tool results from the full structured event stream before stdout clipping."""
"""Bound legacy tool-result diagnostics before stdout clipping.

This historical field is not authoritative runtime_evidence.
"""
evidence: dict[str, Any] = {
"schema": "opencode-eval-runner/tool-results/v1",
"source": "opencode.event-stream.full",
Expand Down Expand Up @@ -755,6 +764,7 @@ def invoke_opencode(
"stdout_truncated": len(stdout) > STDOUT_CAPTURE_LIMIT,
"stdout_total_chars": len(stdout),
"tool_result_evidence": extract_tool_result_evidence(events),
"runtime_evidence": unsupported_runtime_evidence(RUNTIME_EVIDENCE_UNSUPPORTED_REASON),
"plugin_diagnostic": plugins,
"plugin_preflight": plugin_preflight,
}
Expand All @@ -763,10 +773,11 @@ def invoke_opencode(
events = parse_events(proc.stdout)
sid = session_id(events)

# The structured `opencode run --format json` event stream is the
# authoritative evidence source. Starting a second OpenCode process to
# export the just-created session is redundant and can add a full timeout
# per invocation when export/session bootstrap fails. Keep eval latency
# The structured `opencode run --format json` event stream remains useful for
# product/convenience projections (`text`, `tools`, `actions`, and legacy
# `tool_result_evidence`). It is not authoritative `runtime_evidence`. Starting
# a second OpenCode process to export the just-created session is redundant and
# can add a full timeout per invocation when export/session bootstrap fails. Keep eval latency
# bound to the requested target/judge execution only.
text = extract_text(events)
tools = extract_tools(events)
Expand Down Expand Up @@ -802,6 +813,7 @@ def invoke_opencode(
"stdout_truncated": len(proc.stdout) > STDOUT_CAPTURE_LIMIT,
"stdout_total_chars": len(proc.stdout),
"tool_result_evidence": extract_tool_result_evidence(events),
"runtime_evidence": unsupported_runtime_evidence(RUNTIME_EVIDENCE_UNSUPPORTED_REASON),
"plugin_diagnostic": plugins,
"plugin_preflight": plugin_preflight,
}
Expand Down Expand Up @@ -852,6 +864,7 @@ def invoke_copilot(
"skills_loaded": [],
"stderr": "github-copilot-cli requires COPILOT_GITHUB_TOKEN, GH_TOKEN, or GITHUB_TOKEN",
"stdout": "",
"runtime_evidence": unsupported_runtime_evidence(RUNTIME_EVIDENCE_UNSUPPORTED_REASON),
}

root = Path("/tmp/copilot")
Expand Down Expand Up @@ -906,10 +919,12 @@ def invoke_copilot(
"stdout": proc.stdout[:STDOUT_CAPTURE_LIMIT],
"stdout_truncated": len(proc.stdout) > STDOUT_CAPTURE_LIMIT,
"stdout_total_chars": len(proc.stdout),
"runtime_evidence": unsupported_runtime_evidence(RUNTIME_EVIDENCE_UNSUPPORTED_REASON),
}


def emit_result(result: dict[str, Any]) -> None:
validate_runtime_evidence(result.get("runtime_evidence"))
sys.stdout.write(json.dumps(result, separators=(",", ":")) + "\n")
sys.stdout.flush()

Expand Down Expand Up @@ -951,6 +966,7 @@ def main() -> int:
"skills_loaded": [],
"stderr": f"{type(exc).__name__}: {exc}",
"stdout": "",
"runtime_evidence": unsupported_runtime_evidence(RUNTIME_EVIDENCE_UNSUPPORTED_REASON),
"infrastructure_error": True,
}
emit_result(result)
Expand Down
Loading
Loading