Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
33 commits
Select commit Hold shift + click to select a range
d6d266e
chore: pin stock OpenCode 2.0.23
bateau84 Oct 6, 2026
f9dc9bf
docs: define trusted-checkout evaluation model
bateau84 Oct 6, 2026
97589bb
docs: replace TRUST-001 with trusted-checkout evidence contract
bateau84 Oct 6, 2026
b6e6f85
test: expect stock OpenCode 2.0.23
bateau84 Oct 6, 2026
bda7264
docs: retain stock 2.0.23 observation findings
bateau84 Oct 6, 2026
618b2e9
docs: link retained stock observation assessment
bateau84 Oct 6, 2026
0d916ee
feat(observer): add stock native tool capture plugin
bateau84 Oct 6, 2026
be715de
feat(observer): validate native tool observations
bateau84 Oct 6, 2026
b3a50e8
test(observer): cover fail-closed native projection
bateau84 Oct 6, 2026
9c15bd5
feat(observer): project stock native observations into results
bateau84 Oct 6, 2026
092a5b9
test(observer): add stock native probe tools
bateau84 Oct 6, 2026
c0a2b41
test(observer): exercise stock 2.0.23 provider-free
bateau84 Oct 6, 2026
e4b5eb2
ci(observer): prove stock native capture provider-free
bateau84 Oct 6, 2026
4adc0db
ci(observer): include observer parser checks
bateau84 Oct 6, 2026
8eb6b6d
docs(observer): distinguish product output from runtime evidence
bateau84 Oct 6, 2026
7f7b8b5
docs(observer): record selected stock native boundaries
bateau84 Oct 6, 2026
276f525
docs(observer): document native runtime contract
bateau84 Oct 6, 2026
d747e27
docs(observer): expose native observation result
bateau84 Oct 6, 2026
80b6156
test(observer): pin runner injection wiring
bateau84 Oct 6, 2026
20db262
fix(observer): use explicit stock plugin server entrypoint
bateau84 Oct 6, 2026
fa3ee45
test(observer): pin explicit plugin server entrypoint
bateau84 Oct 6, 2026
abbe23b
fix(observer): make native failure fixture unambiguous
bateau84 Oct 6, 2026
1d42bc9
test(observer): cross-check stock runtime identity and terminals
bateau84 Oct 6, 2026
ff57e64
ci(observer): keep stock probe command readable
bateau84 Oct 6, 2026
b3abb7a
fix(observer): reject non-JSON numeric constants
bateau84 Oct 6, 2026
ad9773e
fix(observer): support container script entrypoint import
bateau84 Oct 6, 2026
05cda88
test(observer): report incomplete terminals without crashing
bateau84 Oct 6, 2026
97f9d97
fix(observer): use canonical Session tool terminals
bateau84 Oct 6, 2026
0aa53da
fix(observer): validate canonical Session terminal boundaries
bateau84 Oct 6, 2026
fd8c52c
test(observer): cover Session terminal boundaries
bateau84 Oct 6, 2026
8ceddf2
docs(observer): document Session terminal boundary
bateau84 Oct 6, 2026
3feeb38
docs(observer): record stock Session terminal source
bateau84 Oct 6, 2026
026e0c5
docs(observer): describe Session-owned native terminals
bateau84 Oct 6, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,7 @@ jobs:

- name: Python checks
run: |
python3 -m py_compile runner/cli.py container/invoke.py
python3 -m py_compile runner/cli.py container/invoke.py container/native_observer.py
python3 -m unittest discover -s tests -p 'test_*.py'

- name: Build fake Copilot transport for action smoke test
Expand Down
53 changes: 53 additions & 0 deletions .github/workflows/native-observer-integration.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,53 @@
name: Stock native observer integration

on:
pull_request:
paths:
- 'container/invoke.py'
- 'container/native_observer.py'
- 'container/native_observer.ts'
- 'tests/integration/native_observer_probe.ts'
- 'tests/integration/run_native_observer_probe.py'
- 'tests/test_native_observer.py'
- '.github/workflows/native-observer-integration.yml'

permissions:
contents: read

jobs:
stock-native-observer:
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262
with:
ref: ${{ github.event.pull_request.head.sha }}
persist-credentials: false

- name: Verify probe syntax
run: python3 -m py_compile container/native_observer.py tests/integration/run_native_observer_probe.py

- name: Build runner with stock OpenCode 2.0.23
run: docker build -f Containerfile --target opencode -t opencode-eval-runner:native-observer-probe .

- name: Run provider-free native observer probe
run: |
docker run --rm \
--network none \
--tmpfs /tmp:rw,exec,nosuid,nodev,size=1g,mode=1777 \
--tmpfs /workspace:rw,nosuid,nodev,size=64m,mode=1777 \
--volume "$PWD:/probe-repo:ro" \
--entrypoint python3 \
opencode-eval-runner:native-observer-probe \
/probe-repo/tests/integration/run_native_observer_probe.py \
> native-observer-probe.json
cat native-observer-probe.json

- name: Preserve probe report
if: always()
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02
with:
name: native-observer-${{ github.event.pull_request.head.sha }}
path: native-observer-probe.json
if-no-files-found: warn
retention-days: 14
2 changes: 1 addition & 1 deletion Containerfile
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
FROM node:24-bookworm-slim@sha256:0e0ff40c39bc087845bfb27465a0df4ea419520094bc35842ff83dd8cbe6f9b6 AS opencode-builder
ARG OPENCODE_VERSION=2.0.18
ARG OPENCODE_VERSION=2.0.23
RUN npm install --global "@opencode/cli@${OPENCODE_VERSION}" \
&& resolved="$(readlink -f "$(command -v opencode)")" \
&& test -x "$resolved" \
Expand Down
26 changes: 23 additions & 3 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -72,7 +72,15 @@ Known API-key environment variables are passed when present:

Additional variables require explicit `--env NAME`.

Reasoning can be pinned explicitly with `--reasoning LEVEL`. The pinned OpenCode 2.0.18 CLI represents a model variant in the model reference, so the runner maps `--model provider/model --reasoning LEVEL` to `opencode run --model provider/model#LEVEL`. Supplying both a `#variant` in `--model` and `--reasoning` is rejected as ambiguous. If the model reference already contains a variant and `--reasoning` is omitted, the result records that variant with `"reasoning_source": "model-variant"`. If neither form supplies a level, the runner leaves OpenCode's provider/model default untouched and records `"reasoning": "provider-default"`.
Reasoning can be pinned explicitly with `--reasoning LEVEL`. The pinned stock OpenCode 2.0.23 CLI represents a model variant in the model reference, so the runner maps `--model provider/model --reasoning LEVEL` to `opencode run --model provider/model#LEVEL`. Supplying both a `#variant` in `--model` and `--reasoning` is rejected as ambiguous. If the model reference already contains a variant and `--reasoning` is omitted, the result records that variant with `"reasoning_source": "model-variant"`. If neither form supplies a level, the runner leaves OpenCode's provider/model default untouched and records `"reasoning": "provider-default"`.

### Native tool runtime observations

OpenCode results also include `native_tool_observations` for native/direct tools on the pinned stock 2.0.23 runtime. A runner-owned in-process plugin records decoded executable input at the tool implementation boundary and correlates it with stock `session.tool.success` / `session.tool.failed` terminals using the exact Session/message/call identity.

This field is fail-closed: missing capture, observer loss, missing terminals, sequence gaps, invalid fields, or incomplete shutdown make `evidence_eligible` false. Model text and tool-returned collector-shaped JSON are never parsed into this projection.

See `docs/native-tool-observer.md` for the exact boundary and limitations. Code Mode inner calls are not covered by this native observer.

### `github-copilot-cli`

Expand Down Expand Up @@ -173,7 +181,7 @@ opencode-eval-runner invoke \
...
```

OpenCode 2.0.18 does not expose the old singular `debug agent <id>` command that returned a resolved tool map. The runner therefore performs the strongest supported zero-inference preflight: it requires the expected plugin entrypoint to be materialized in the isolated OpenCode config, runs `opencode debug agents` to prove the configured location starts successfully with plugins active, and requires the selected agent to resolve. Missing plugin materialization, plugin/startup failure, or missing agent is infrastructure/non-evidence, never a behavioral FAIL. Actual tool use remains a repository-owned behavioral assertion in the eval corpus.
Stock OpenCode 2.0.23 does not expose the old singular `debug agent <id>` command that returned a resolved tool map. The runner therefore performs the strongest supported zero-inference preflight: it requires the expected plugin entrypoint to be materialized in the isolated OpenCode config, runs `opencode debug agents` to prove the configured location starts successfully with plugins active, and requires the selected agent to resolve. Missing plugin materialization, plugin/startup failure, or missing agent is infrastructure/non-evidence, never a behavioral FAIL. Actual tool use remains a repository-owned behavioral assertion in the eval corpus.

### Evaluating a skill

Expand Down Expand Up @@ -315,7 +323,7 @@ The eval repository decides whether that observed behavior is PASS, FAIL, or non

The transport images currently pin:

- OpenCode CLI `2.0.18`
- OpenCode CLI `2.0.23`
- GitHub Copilot CLI `1.0.83`

The two CLIs are not bundled together. OpenCode's npm package is used only as a build-time native-binary selector; GitHub Copilot CLI is installed from its native release installer. Node/npm are absent from the final runtime images.
Expand All @@ -340,6 +348,18 @@ OPENCODE_EVAL_RUNNER_COPILOT_IMAGE=...

Tags matching `v*` are published with `opencode-` and `copilot-` prefixes.

## Evaluation trust model

The normal evaluation profile is a **trusted-checkout** profile. It assumes the runner, pinned stock OpenCode runtime, reviewed instrumentation, and explicitly selected evaluated checkout/dependencies are trusted components of the evaluation environment.

They are not trusted merely because they produce data that looks like evidence. Model prose, tool-returned collector-shaped JSON, target-writable files, requested actions, inferred identities, and reconstructed results do not establish that an event occurred.

Authoritative runtime observations must come from reviewed instrumentation observing actual execution. Missing, partial, ambiguous, or unsupported required observations are non-evidence and must fail closed for the affected assertion.

This profile does **not** claim resistance to an evaluated plugin that deliberately compromises the trusted runtime or instrumentation. Hostile-plugin isolation is a separate optional profile, not a prerequisite for normal Loom evaluation.

See [Trusted-checkout runtime evidence](docs/trusted-checkout-evidence.md).

## Security boundary

The runner:
Expand Down
33 changes: 28 additions & 5 deletions container/invoke.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,11 @@
from pathlib import Path
from typing import Any

if __package__:
from container.native_observer import OBSERVATION_PATH, load_native_observations
else:
from native_observer import OBSERVATION_PATH, load_native_observations

RESULT_SCHEMA = "opencode-eval-runner/v1"
OPENCODE_EVAL_TITLE = "opencode-eval-runner"
COPILOT_AGENT_NAME = "eval-runner"
Expand Down Expand Up @@ -304,6 +309,10 @@ def prepare_opencode_env() -> dict[str, str]:
json.dumps({"$schema": "https://opencode.ai/config.json"}) + "\n",
encoding="utf-8",
)
observer_root = config / "eval-native-observer"
observer_root.mkdir(parents=True, exist_ok=True)
shutil.copyfile(Path(__file__).with_name("native_observer.ts"), observer_root / "server.ts")

if seed_config_root.is_dir():
# Loom itself can be an OpenCode global config root. OpenCode 2.0.11's
# packaged runtime can fail to register directory plugins even when
Expand Down Expand Up @@ -345,6 +354,10 @@ def prepare_opencode_env() -> dict[str, str]:
"XDG_CACHE_HOME": str(cache),
"XDG_STATE_HOME": str(state),
"OPENCODE_CONFIG_DIR": str(config),
# Inline config is the highest-priority local config in stock 2.0.23.
# Register the runner-owned observer there so its tool transform runs
# after project/global plugin transforms instead of depending on filename order.
"OPENCODE_CONFIG_CONTENT": json.dumps({"plugins": [observer_root.as_uri()]}),
"OPENCODE_DB": "opencode.db",
"OPENCODE_DISABLE_AUTOUPDATE": "1",
})
Expand Down Expand Up @@ -641,6 +654,7 @@ def plugin_diagnostic(env: dict[str, str]) -> dict[str, Any]:
loom_root = plugins_root / "loom"
loom_flat = plugins_root / "loom.ts"
loom_module_root = config_root / "loom-plugin"
observer_root = config_root / "eval-native-observer"
return {
"config_root": str(config_root),
"config_root_exists": config_root.exists(),
Expand All @@ -654,6 +668,8 @@ def plugin_diagnostic(env: dict[str, str]) -> dict[str, Any]:
"loom_index_exists": (loom_root / "index.ts").is_file(),
"loom_flat_exists": loom_flat.is_file(),
"loom_module_index_exists": (loom_module_root / "index.ts").is_file(),
"native_observer_exists": (observer_root / "server.ts").is_file(),
"native_observer_inline_configured": "eval-native-observer" in env.get("OPENCODE_CONFIG_CONTENT", ""),
}


Expand Down Expand Up @@ -710,6 +726,10 @@ def invoke_opencode(
if agent:
command += ["--agent", agent]
command += ["--model", invoked_model, prompt]

# A plugin-preflight process may have activated the observer already.
# Never let that prior process become evidence for the real invocation.
OBSERVATION_PATH.unlink(missing_ok=True)
run_started = time.perf_counter()
try:
proc = run(command, Path("/workspace"), env, timeout)
Expand All @@ -719,6 +739,7 @@ def invoke_opencode(
stderr = timeout_output(exc.stderr)
events = parse_events(stdout)
sid = session_id(events)
native_tool_observations = load_native_observations()
summary = last_event_summary(events)
detail = (
f"opencode run timed out after {timeout}s; "
Expand Down Expand Up @@ -755,19 +776,20 @@ def invoke_opencode(
"stdout_truncated": len(stdout) > STDOUT_CAPTURE_LIMIT,
"stdout_total_chars": len(stdout),
"tool_result_evidence": extract_tool_result_evidence(events),
"native_tool_observations": native_tool_observations,
"plugin_diagnostic": plugins,
"plugin_preflight": plugin_preflight,
}

run_seconds = time.perf_counter() - run_started
events = parse_events(proc.stdout)
sid = session_id(events)
native_tool_observations = load_native_observations()

# The structured `opencode run --format json` event stream is the
# authoritative evidence source. Starting a second OpenCode process to
# export the just-created session is redundant and can add a full timeout
# per invocation when export/session bootstrap fails. Keep eval latency
# bound to the requested target/judge execution only.
# The structured `opencode run --format json` stream remains the product
# output source. Reviewed native-tool evidence comes from the runner-owned
# observer projection above; stdout/tool payload JSON is never promoted into it.
# Starting a second process to export the Session remains unnecessary.
text = extract_text(events)
tools = extract_tools(events)
actions = extract_actions(events)
Expand Down Expand Up @@ -802,6 +824,7 @@ def invoke_opencode(
"stdout_truncated": len(proc.stdout) > STDOUT_CAPTURE_LIMIT,
"stdout_total_chars": len(proc.stdout),
"tool_result_evidence": extract_tool_result_evidence(events),
"native_tool_observations": native_tool_observations,
"plugin_diagnostic": plugins,
"plugin_preflight": plugin_preflight,
}
Expand Down
Loading
Loading