diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 419945f..4658b9f 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -16,7 +16,7 @@ jobs: - name: Python checks run: | - python3 -m py_compile runner/cli.py container/invoke.py + python3 -m py_compile runner/cli.py container/invoke.py tests/integration/run_code_mode_observer_probe.py python3 -m unittest discover -s tests -p 'test_*.py' - name: Build fake Copilot transport for action smoke test @@ -69,6 +69,12 @@ jobs: - name: Build OpenCode transport image run: docker build -f Containerfile --target opencode -t opencode-eval-runner:opencode-test . + - name: Probe stock Code Mode inner observation boundary + run: | + python3 tests/integration/run_code_mode_observer_probe.py \ + --image opencode-eval-runner:opencode-test \ + --output /tmp/code-mode-observer-probe + - name: Build Copilot transport image run: docker build -f Containerfile --target copilot -t opencode-eval-runner:copilot-test . diff --git a/Containerfile b/Containerfile index 97fb1f9..3ec8003 100644 --- a/Containerfile +++ b/Containerfile @@ -1,5 +1,5 @@ FROM node:24-bookworm-slim@sha256:0e0ff40c39bc087845bfb27465a0df4ea419520094bc35842ff83dd8cbe6f9b6 AS opencode-builder -ARG OPENCODE_VERSION=2.0.18 +ARG OPENCODE_VERSION=2.0.23 RUN npm install --global "@opencode/cli@${OPENCODE_VERSION}" \ && resolved="$(readlink -f "$(command -v opencode)")" \ && test -x "$resolved" \ diff --git a/README.md b/README.md index 0f5c5a7..faa29c5 100644 --- a/README.md +++ b/README.md @@ -72,7 +72,7 @@ Known API-key environment variables are passed when present: Additional variables require explicit `--env NAME`. -Reasoning can be pinned explicitly with `--reasoning LEVEL`. The pinned OpenCode 2.0.18 CLI represents a model variant in the model reference, so the runner maps `--model provider/model --reasoning LEVEL` to `opencode run --model provider/model#LEVEL`. Supplying both a `#variant` in `--model` and `--reasoning` is rejected as ambiguous. If the model reference already contains a variant and `--reasoning` is omitted, the result records that variant with `"reasoning_source": "model-variant"`. If neither form supplies a level, the runner leaves OpenCode's provider/model default untouched and records `"reasoning": "provider-default"`. +Reasoning can be pinned explicitly with `--reasoning LEVEL`. The pinned stock OpenCode 2.0.23 CLI represents a model variant in the model reference, so the runner maps `--model provider/model --reasoning LEVEL` to `opencode run --model provider/model#LEVEL`. Supplying both a `#variant` in `--model` and `--reasoning` is rejected as ambiguous. If the model reference already contains a variant and `--reasoning` is omitted, the result records that variant with `"reasoning_source": "model-variant"`. If neither form supplies a level, the runner leaves OpenCode's provider/model default untouched and records `"reasoning": "provider-default"`. ### `github-copilot-cli` @@ -173,7 +173,7 @@ opencode-eval-runner invoke \ ... ``` -OpenCode 2.0.18 does not expose the old singular `debug agent ` command that returned a resolved tool map. The runner therefore performs the strongest supported zero-inference preflight: it requires the expected plugin entrypoint to be materialized in the isolated OpenCode config, runs `opencode debug agents` to prove the configured location starts successfully with plugins active, and requires the selected agent to resolve. Missing plugin materialization, plugin/startup failure, or missing agent is infrastructure/non-evidence, never a behavioral FAIL. Actual tool use remains a repository-owned behavioral assertion in the eval corpus. +Stock OpenCode 2.0.23 does not expose the old singular `debug agent ` command that returned a resolved tool map. The runner therefore performs the strongest supported zero-inference preflight: it requires the expected plugin entrypoint to be materialized in the isolated OpenCode config, runs `opencode debug agents` to prove the configured location starts successfully with plugins active, and requires the selected agent to resolve. Missing plugin materialization, plugin/startup failure, or missing agent is infrastructure/non-evidence, never a behavioral FAIL. Actual tool use remains a repository-owned behavioral assertion in the eval corpus. ### Evaluating a skill @@ -315,7 +315,7 @@ The eval repository decides whether that observed behavior is PASS, FAIL, or non The transport images currently pin: -- OpenCode CLI `2.0.18` +- OpenCode CLI `2.0.23` - GitHub Copilot CLI `1.0.83` The two CLIs are not bundled together. OpenCode's npm package is used only as a build-time native-binary selector; GitHub Copilot CLI is installed from its native release installer. Node/npm are absent from the final runtime images. @@ -340,6 +340,18 @@ OPENCODE_EVAL_RUNNER_COPILOT_IMAGE=... Tags matching `v*` are published with `opencode-` and `copilot-` prefixes. +## Evaluation trust model + +The normal evaluation profile is a **trusted-checkout** profile. It assumes the runner, pinned stock OpenCode runtime, reviewed instrumentation, and explicitly selected evaluated checkout/dependencies are trusted components of the evaluation environment. + +They are not trusted merely because they produce data that looks like evidence. Model prose, tool-returned collector-shaped JSON, target-writable files, requested actions, inferred identities, and reconstructed results do not establish that an event occurred. + +Authoritative runtime observations must come from reviewed instrumentation observing actual execution. Missing, partial, ambiguous, or unsupported required observations are non-evidence and must fail closed for the affected assertion. + +This profile does **not** claim resistance to an evaluated plugin that deliberately compromises the trusted runtime or instrumentation. Hostile-plugin isolation is a separate optional profile, not a prerequisite for normal Loom evaluation. + +See [Trusted-checkout runtime evidence](docs/trusted-checkout-evidence.md). + ## Security boundary The runner: diff --git a/docs/code-mode-observer-experiment.md b/docs/code-mode-observer-experiment.md new file mode 100644 index 0000000..57b801d --- /dev/null +++ b/docs/code-mode-observer-experiment.md @@ -0,0 +1,188 @@ +# Code Mode inner-call observation experiment + +Status: **bounded negative result with partial stock support** + +Parent: PR #45 trusted-checkout runtime evidence. + +Runtime checkpoint: stock OpenCode **v2.0.23**, tag commit +`0fd7e2829449b052abf0078666669302923d77af`. + +This experiment asks one question only: how much of an actual inner Code Mode +tool invocation can reviewed same-process runner instrumentation observe without +patching OpenCode? + +## Result + +Stock OpenCode 2.0.23 supports a useful partial observation path: + +| Fact | Stock supported surface | Result | +| --- | --- | --- | +| unique inner invocation identity | runner-owned `tool.transform` wrapper allocates an ID when the decoded leaf handler is actually entered | **supported** | +| actual selected tool | wrapper is attached to the effective registered tool | **supported** | +| executable input | wrapper runs after core input decoding and receives the value passed to the leaf handler | **supported** | +| real outer `execute` binding | the real `Tool.Context` carries Session/message/outer CallID into each inner leaf | **supported** | +| start / handler-terminal ordering | runner observer sequence around the transformed leaf handler | **supported** | +| exact final value delivered to the Code Mode script | no supported public stock boundary exposes it with unique inner identity | **unsupported** | +| exact final error seen by the script catch path | no supported public stock boundary exposes it with unique inner identity | **unsupported** | + +The partial path is useful for proving that an inner call really entered a +particular tool with a particular decoded input. It is **not** sufficient for +assertions about the value/error ultimately observed by Code Mode. + +The experiment therefore reports the final-caller capability as: + +```json +{ + "status": "unsupported", + "reason": "stock_codemode_final_boundary_not_exposed" +} +``` + +No value is reconstructed from an earlier result. + +## Exact missing boundary + +There are three distinct boundaries in stock 2.0.23. + +### 1. Public transformed leaf handler + +`packages/core/src/tool/runtime.ts` decodes input and then calls the registered +tool's `execute(decoded, context)`. + +A runner-owned `ctx.tool.transform(...)` wrapper can therefore allocate a fresh +per-call ID at a real execution boundary and observe: + +- the effective tool registration; +- decoded/executable input; +- the real Session ID, agent, message ID and outer `execute` CallID carried in + `Tool.Context`; +- handler return or throw; +- start and handler completion order. + +This works for identical concurrent calls because correlation is carried by the +wrapper's own per-invocation state. It does not pair calls by input, FIFO order, +tool name, or completion order. + +But the handler result is still early. After it returns, core may: + +1. encode/normalize the tool output; +2. run public `tool.execute.after` hooks, which may mutate the result; +3. normalize content; +4. let the Code Mode adapter choose structured output vs text/null fallback; +5. validate Code Mode output; +6. JSON stringify/parse the value before it crosses into the confined script. + +So the transform wrapper cannot claim its handler terminal is the script-visible +terminal. + +### 2. Public core `tool.execute.after` + +`packages/core/src/tool.ts` exposes a later hook with the core result/error. + +This is also insufficient for exact Code Mode correlation: + +- every inner Code Mode call receives the same `Tool.Context.id` as the outer + model-visible `execute` call; +- therefore concurrent identical inner calls have the same public CallID; +- the hook runs before Code Mode's own final output decode/JSON round trip; +- failures that escape the core Tool.Error path need not produce this hook even + though Code Mode later converts the failure for the script. + +Using input equality, FIFO order, object identity, or completion order to join +this hook back to wrapper records would invent a correlation contract that stock +OpenCode does not provide. + +### 3. Private Code Mode terminal and catch materialization + +The last success-value boundary exists inside `@opencode/codemode`. + +In `packages/codemode/src/tool-runtime.ts`, `hooked(...)` runs +`hooks["tool.after"]` from `Effect.onExit` around the Code Mode execution body. +For a successful call, this happens after output decoding and the JSON +stringify/parse round trip, so its `CallResult.value` is the plain value that the +tool promise will deliver into the interpreter. + +However, `packages/core/src/codemode/tool.ts` constructs Code Mode with only +OpenCode's private `progressHooks(record)`. That hook uses the internal call +object only to update UI rows and publishes name/input/status. It does **not** +publish the success value, and stock plugin APIs provide no supported way to add +another Code Mode hook there. + +The error path is later still. A failed tool promise reaches the interpreter and +`packages/codemode/src/interpreter/interpreter.ts` materializes the failure into +the JavaScript error value bound by a `catch` clause. There is no plugin/runtime +hook at that materialization boundary either. The private Code Mode `tool.after` +can see the host-side failure before this conversion, but that is not the exact +JavaScript error object seen by the script. + +So stock exposes neither the final success value with public unique correlation +nor the final catch-path error representation. Those are the missing boundaries. + +## PR #41 reuse decision + +PR #41 correctly demonstrated the semantics needed at this boundary, but it did +so by patching OpenCode and adding an internal Code Mode observer seam. That +implementation is not reused here. + +Only the behavioral test ideas are retained: + +- overlapping identical calls; +- reverse completion; +- caught inner errors; +- mutation after an earlier observation point; +- real parent Session/call binding; +- script output that resembles an observer record. + +The stock experiment uses only supported plugin transforms/hooks. + +## Provider-free integration probe + +`tests/integration/run_code_mode_observer_probe.py` runs the actual +`opencode-eval-runner:opencode-test` image built from this checkout. + +It starts a loopback OpenAI-compatible fixture provider inside the container, so +there is no external provider call and no credential use. The fixture makes one +real model-visible `execute` call whose Code Mode script performs: + +1. one successful inner call; +2. one thrown inner error that the script catches; +3. two concurrent calls with identical input; +4. reverse completion of those two calls; +5. multiple inner calls under the same outer `execute`; +6. a returned collector-shaped fake record. + +The test plugin also changes one tool result from `BEFORE-MUTATION` to +`AFTER-MUTATION` in a public `execute.after` hook. The script asserts that it +receives `AFTER-MUTATION`. The transformed handler observer records +`BEFORE-MUTATION`. This is the concrete counterexample proving that the handler +terminal is not the caller-final value. + +The outer script finally returns JSON containing +`invocation_id: "fabricated-from-script"`. The runner event stream proves that +the outer script executed, while the same ID must be absent from observer records. +Script/model output therefore does not create an inner observation. + +## Evidence status + +The probe plugin writes diagnostics under `/tmp` only for the test. That file is +target-writable and is **not** an evidence authority. + +The experiment summary always reports: + +```json +{ + "status": "unsupported", + "evidence_eligible": false +} +``` + +This does not mean all inner facts are unavailable. It means the requested +end-to-end Code Mode result/error capability is incomplete on stock supported +surfaces, so the partial records must not be promoted as proof of the final +caller-visible value/error. + +A future stock OpenCode API could make this capability supported by exposing the +internal Code Mode per-call identity and post-conversion success value to plugins, +plus the materialized catch-path error value (or one supported terminal event that +carries both forms with the same invocation identity). Until then, the correct +runner result is `unsupported`. diff --git a/docs/stock-opencode-2.0.23-observation.md b/docs/stock-opencode-2.0.23-observation.md new file mode 100644 index 0000000..37df53c --- /dev/null +++ b/docs/stock-opencode-2.0.23-observation.md @@ -0,0 +1,73 @@ +# Stock OpenCode 2.0.23 observation surface + +Purpose: implementation reference for the trusted-checkout evidence profile. + +Source checkpoint: stock OpenCode **v2.0.23** (`0fd7e2829449b052abf0078666669302923d77af`). This is distilled from the source assessment performed in superseded PR #43 and the bounded Code Mode experiment in this branch. + +OpenCode remains stock and immutable. A missing observation boundary is reported as unsupported; it is not a reason to patch OpenCode or add a hostile-runtime broker. + +## Useful stock surfaces + +| Observation need | Stock surface | Status | +| --- | --- | --- | +| Live runtime events | `ctx.event.subscribe()` | supported source; ordering/drain must be proven by integration test | +| Session creation / ancestry | `session.created` + Session API | supported | +| Agent for a step | Session step/message events | supported | +| Native tool call identity/input | Session tool input/called events | supported source | +| Native terminal success/failure | Session tool success/failed events | supported source | +| Tool pre-execution hook | `ctx.tool.hook("execute.before")` | supported; occurs before tool decode/execution | +| Tool post-handler hook | `ctx.tool.hook("execute.after")` | supported; occurs after core tool execution but before Code Mode final conversion | +| Tool registration wrapping | `ctx.tool.transform(...)` | supported | +| Code Mode unique inner invocation/tool/input/outer binding | transformed leaf handler + real `Tool.Context` | **supported partial boundary** | +| Code Mode exact final caller value/error | internal `@opencode/codemode` `tool.after`; not exposed to plugins | **unsupported on stock public surfaces** | + +## Native calls + +Stock Session events are the preferred source for native terminal facts because they represent the runtime's own Session lifecycle rather than model or tool payload claims. + +The observer must bind call identity, Session, agent/message context, input and terminal result/error without reconstructing them from prose or matching by value. + +## Code Mode + +Code Mode executes inner tools through the normal tool registry, so same-process reviewed instrumentation can observe real inner execution without isolating Loom. + +The bounded stock-2.0.23 experiment establishes a useful partial path: + +- `ctx.tool.transform(...)` can wrap the actual effective leaf registration; +- core decodes input before entering that wrapper, so the wrapper sees executable input; +- the real outer `execute` `Tool.Context` reaches each inner handler, providing actual Session, agent, message and outer CallID; +- the wrapper can allocate a unique per-inner observation ID at handler entry, so identical concurrent calls and reverse completion do not require input/FIFO correlation. + +However, this wrapper is not the final Code Mode caller boundary. After it returns, core can encode the result, run mutating `tool.execute.after` hooks, normalize content, and Code Mode can select its return representation and perform output validation plus a JSON stringify/parse round trip. + +The later public `tool.execute.after` hook is also insufficient for exact correlation: every inner call reuses the outer `execute` `Tool.Context.id`. Concurrent identical inner calls therefore have the same public CallID. + +For successful calls, the last converted value exists internally: `packages/codemode/src/tool-runtime.ts` invokes Code Mode's `tool.after` after output validation and its JSON round trip. But `packages/core/src/codemode/tool.ts` supplies only private `progressHooks(record)` there. Those hooks expose name/input/status for UI progress and do not export the value. Stock plugin APIs do not provide a supported registration point for another Code Mode hook. + +Errors have an additional private step. After the tool promise fails, `packages/codemode/src/interpreter/interpreter.ts` materializes that host failure into the JavaScript error value used by a `catch` clause. No public plugin/runtime hook observes that materialized error with a unique inner invocation identity. + +Therefore: + +- unique inner identity, actual tool, executable input, outer binding, and start/handler-terminal ordering are observable; +- exact final value delivered to the script and exact final error seen by its catch path are **unsupported**; +- no earlier result may be promoted, paired, or reconstructed to fill that gap. + +See [Code Mode inner-call observation experiment](code-mode-observer-experiment.md) for the source trace and provider-free counterexamples. + +## Ordering and completeness + +A monotonic observer sequence is useful, but sequence alone is not completeness. The integration must also account for starts, terminals, observer loss, process interruption, and required descendant Sessions. + +Absence assertions are eligible only when the relevant scope is complete. Missing capture is never interpreted as "did not happen". + +For Code Mode specifically, a complete transformed-handler trace still does not make final caller-value assertions eligible: that capability is unsupported on the stock public surface. + +## What is intentionally not required + +- plugin/process isolation from the trusted Loom checkout; +- remote PluginHost or capability broker; +- evidence-channel peer authentication against same-authority attackers; +- patched/forked OpenCode; +- cryptographic evidence authenticity after runtime compromise. + +Those belong only to a future optional untrusted-plugin profile. diff --git a/docs/trusted-checkout-evidence.md b/docs/trusted-checkout-evidence.md new file mode 100644 index 0000000..e0a503c --- /dev/null +++ b/docs/trusted-checkout-evidence.md @@ -0,0 +1,127 @@ +# Trusted-checkout runtime evidence + +Status: **replacement direction for TRUST-001**. + +This document supersedes the hostile-runtime direction explored in PR #41 and PR #43. Those PRs remain useful research/reference material, but normal Loom evaluation does not require the runner to defend itself from a deliberately malicious Loom checkout that shares its runtime authority. + +## Contract + +### TRUST-001 — authoritative runtime observation + +For evaluation of an explicitly trusted checkout, evidence used for scoring MUST originate from reviewed runtime instrumentation observing actual execution. + +The following MUST NOT independently establish that an event occurred: + +- model assertions or generated prose; +- tool payloads shaped like collector/evidence records; +- requested or intended actions; +- inferred actor, parent, or execution identity; +- reconstructed results; +- target-writable evidence files. + +Required observations MUST preserve enough runtime identity and ordering to evaluate the consumer contract, including actor/session/call identity, input, result or error, parent binding where applicable, and execution order. + +Missing, partial, ambiguous, lost, or unsupported required observations MUST make the affected assertion ineligible for PASS. The runner MUST NOT fill gaps from model text, stdout, workspace files, or guessed correlations. + +The trusted-checkout profile does not claim protection against malicious modification of the runner, stock OpenCode process, reviewed instrumentation, evaluated checkout, or their dependencies. + +## Trust model + +Trusted components: + +- the selected `opencode-eval-runner` revision; +- pinned **stock OpenCode 2.0.23**; +- reviewed runtime instrumentation; +- the explicitly selected Loom checkout and its reviewed dependencies; +- host-side evidence projection/persistence code. + +Not trusted as evidence authority: + +- model output; +- agent claims; +- tool-returned collector-shaped data; +- normal product/session/workspace files; +- caller-supplied identity or completeness claims. + +This is an evaluation-correctness boundary, not a hostile-code security boundary. + +## Required evidence behavior + +The target behavior remains strict even though the security scope is smaller: + +- **Native calls:** observe the actual runtime call, actor/session/call identity, accepted/executable input, and terminal result/error. +- **Code Mode inner calls:** assign a unique runtime observation identity per actual inner invocation, bind it to the real outer `execute` call, and observe the final value/error that Code Mode exposes to the script. +- **Delegation:** derive child Session identity and ancestry from runtime facts, not a parent result payload. +- **Ordering:** preserve runtime observation order; do not correlate concurrent calls by FIFO or input equality. +- **Completeness:** explicitly report missing starts/terminals, capture loss, unsupported boundaries, and incomplete scope. +- **Confidentiality:** redact or omit credentials before the runner first persists, clips, logs, or exports evidence. +- **Noninterference:** observation must not add product retries or change normal Loom execution semantics. + +If stock OpenCode's supported interfaces cannot expose an exact required boundary, the result is `unsupported`/ineligible for that assertion. The response is not to invent evidence and not to turn the normal profile into a hostile-code isolation project. + +## Implementation direction + +Keep the normal path: + +```text +Loom eval harness + -> opencode-eval-runner invoke + -> stock OpenCode 2.0.23 + + reviewed runner-owned observation instrumentation + + trusted Loom checkout + -> safe host projection + -> Loom judging +``` + +The preferred implementation is same-process reviewed instrumentation using supported stock OpenCode plugin/runtime surfaces. It may use a runner-owned observer plugin, tool/session hooks, live runtime events, and reviewed wrappers where those surfaces preserve the required boundary. + +OpenCode source patches, forks, remote PluginHost isolation, evidence signing, and a capability broker are not requirements of this profile. + +Provider-free integration tests must prove the exact observation/correlation behavior before a field becomes eligible evidence. + +See [Stock OpenCode 2.0.23 observation surface](stock-opencode-2.0.23-observation.md) for the retained source/capability findings from PR #43. + +## Reuse from PR #41 + +| Work | Disposition | +| --- | --- | +| Evidence-safety projection/redaction and fail-closed field handling | **Reuse/adapt**; keep the behavior, decouple it from hostile-runtime image/signing assumptions | +| Credential protection before host/file/print sinks | **Reuse** | +| Disposable OpenCode state/profile work | **Reuse where useful** for deterministic eval isolation | +| Normal `invoke` compatibility and provider-free integration probes | **Reuse/adapt** to stock 2.0.23 | +| Native/Code Mode observation schemas and concurrency tests | **Reuse as behavioral requirements/tests** | +| Delegated-session identity/ancestry probes | **Reuse** | +| Patched OpenCode runtime | **Drop** | +| HMAC observer/import trust boundary | **Drop** for the normal profile | +| protected-channel / remote tool service | **Drop** | +| plugin isolation / remote PluginHost work | **Drop** | +| Cosign evidence-authenticity machinery | **Drop** as a TRUST-001 prerequisite | +| adversarial same-authority attack tests | **Move to optional future untrusted profile** | + +## Reuse from PR #43 + +| Work | Disposition | +| --- | --- | +| Stock OpenCode 2.0.23 source/capability assessment | **Reuse** | +| Identification of public Session/event/tool surfaces | **Reuse** | +| Scope/completeness rules that prevent false absence/PASS | **Reuse and simplify** | +| First-sink confidentiality inventory | **Reuse and simplify** | +| Loom callback/capability inventory | **Reference when needed for compatibility** | +| hostile-runtime TCB/authority model | **Drop** from the normal profile | +| isolated Loom execution domain | **Drop** | +| capability/evidence-channel peer-authentication requirements | **Drop** | +| OCI adversarial boundary experiment/gates | **Drop** | + +## Implementation sequence + +This is normal engineering work, not a multi-authorization security experiment: + +1. Pin and verify stock OpenCode 2.0.23. +2. Add the smallest reviewed observation instrumentation that can capture native and Code Mode execution without changing product semantics. +3. Port the useful PR #41 evidence-safety projection so captured values are protected before persistence/export. +4. Add explicit completeness/loss fields and fail closed when required data is missing. +5. Exercise direct, Code Mode, delegation, error, timeout, and concurrent reverse-completion cases provider-free. +6. Compose through Loom's existing `eval:live -> run-evals.py -> invoke` path. +7. Only after those checks pass should Loom consume the new evidence schema for PASS/FAIL decisions. + +A future **untrusted-plugin execution profile** may add isolation if there is a real need to evaluate hostile plugin code. It must remain optional and separate from the normal trusted-checkout path. diff --git a/tests/integration/code_mode_observer_probe.ts b/tests/integration/code_mode_observer_probe.ts new file mode 100644 index 0000000..b432b25 --- /dev/null +++ b/tests/integration/code_mode_observer_probe.ts @@ -0,0 +1,229 @@ +import { appendFileSync, writeFileSync } from "node:fs" +import { randomUUID } from "node:crypto" + +// Provider-free stock-runtime probe only. This file intentionally writes +// diagnostics to /tmp; those records are never authoritative runtime evidence. +const RECORDS = "/tmp/code-mode-observer-records.jsonl" +const LOADED = "/tmp/code-mode-observer-loaded" +const FABRICATED_ID = "fabricated-from-script" + +let sequence = 0 +let observerLosses = 0 +let toolOrdinal = 0 +let echoOrdinal = 0 +let releaseFirst!: () => void +const secondMayReleaseFirst = new Promise((resolve) => { + releaseFirst = resolve +}) + +const activeParents = new Map() +const wrapped = new WeakSet() + +function key(value: any) { + return [value.sessionID, value.messageID, value.id].join("\u0000") +} + +function snapshot(value: unknown): unknown { + if (value instanceof Error) return { name: value.name, message: value.message } + try { + return JSON.parse( + JSON.stringify(value, (_key, item) => { + if (item instanceof Error) return { name: item.name, message: item.message } + if (typeof item === "bigint") return String(item) + if (typeof item === "function") return "" + return item + }), + ) + } catch { + return { state: "omitted", reason: "not_json_serializable" } + } +} + +function emit(record: Record) { + const event = { + schema: "opencode-eval-runner/code-mode-observer-experiment/v1", + sequence: ++sequence, + observer_losses_before: observerLosses, + ...record, + } + try { + appendFileSync(RECORDS, JSON.stringify(event) + "\n") + } catch { + // Observation must not alter product execution. + observerLosses += 1 + } +} + +function resultView(event: any) { + if (event?.status === "completed") return { status: event.status, result: snapshot(event.result) } + if (event?.status === "error") return { status: event.status, error: snapshot(event.error) } + return { status: String(event?.status ?? "unknown") } +} + +export default { + id: "code-mode-observer-probe", + async setup(ctx: any) { + // Deterministic Code Mode fixtures. + await ctx.tool.transform((editor: any) => { + editor.namespace({ name: "codemodeprobe", description: "Stock Code Mode observer probe tools." }) + const add = (name: string, run: (input: any) => Promise) => + editor.add({ + name, + description: `Code Mode observer probe ${name}`, + input: { + type: "object", + properties: { tag: { type: "string" } }, + additionalProperties: false, + }, + options: { namespace: "codemodeprobe", codemode: true }, + execute: async (input: unknown) => { + toolOrdinal += 1 + return { content: await run(input) } + }, + }) + + add("success", async () => "SUCCESS-FINAL") + add("echo", async () => { + const n = ++echoOrdinal + if (n === 1) await secondMayReleaseFirst + if (n === 2) { + await new Promise((resolve) => setTimeout(resolve, 25)) + releaseFirst() + } + return `ECHO-${n}` + }) + add("throws", async () => { + throw new Error("THROW-RAW") + }) + add("mutate", async () => "BEFORE-MUTATION") + }) + + // Smallest stock-supported correlation path: wrap the actual registered + // Code Mode leaf handler. Core has already decoded input before this + // function runs. The same Tool.Context carries the real outer execute + // call identity, while this wrapper supplies a unique per-inner UUID. + await ctx.tool.transform((editor: any) => { + for (const item of editor.list()) { + if (item.options?.namespace !== "codemodeprobe" || item.options?.codemode === false) continue + if (wrapped.has(item.execute)) continue + editor.update(item.id, (tool: any) => { + const original = tool.execute + const observed = async (input: unknown, context: any) => { + const parent = activeParents.get(key(context)) + const invocationID = randomUUID() + emit({ + kind: "inner_start", + invocation_id: invocationID, + tool: item.id, + catalog_path: `codemodeprobe.${item.name}`, + parent: parent ?? { + state: "unsupported", + reason: "outer_execute_not_observed", + call_id: context.id, + session_id: context.sessionID, + message_id: context.messageID, + }, + actor: { + agent: context.agent, + session_id: context.sessionID, + message_id: context.messageID, + }, + input: snapshot(input), + boundary: "decoded-tool-handler-input", + }) + try { + const result = await original(input, context) + emit({ + kind: "inner_handler_terminal", + invocation_id: invocationID, + tool: item.id, + outcome: "returned", + handler_result: snapshot(result), + boundary: "tool-handler-return", + caller_terminal: { + status: "unsupported", + reason: "stock_codemode_final_boundary_not_exposed", + }, + }) + return result + } catch (error) { + emit({ + kind: "inner_handler_terminal", + invocation_id: invocationID, + tool: item.id, + outcome: "threw", + handler_error: snapshot(error), + boundary: "tool-handler-throw", + caller_terminal: { + status: "unsupported", + reason: "stock_codemode_final_boundary_not_exposed", + }, + }) + throw error + } + } + wrapped.add(observed) + tool.execute = observed + }) + } + }) + + await ctx.tool.hook("execute.before", (event: any) => { + if (event.tool !== "execute") return + const parent = { + tool: "execute", + call_id: event.id, + session_id: event.sessionID, + message_id: event.messageID, + agent: event.agent, + } + activeParents.set(key(event), parent) + emit({ kind: "parent_start", parent }) + }) + + // A later normal tool hook is allowed to mutate the result after the + // transformed leaf handler has returned. This is the counterexample that + // proves the wrapper terminal is not Code Mode's final caller boundary. + await ctx.tool.hook("execute.after", (event: any) => { + if (event.tool === "codemodeprobe_mutate" && event.status === "completed") { + event.result = { ...event.result, content: "AFTER-MUTATION" } + } + }) + + // This hook sees core's post-handler result/error, but all inner calls reuse + // the outer execute CallID. It therefore cannot correlate identical + // concurrent calls without an unsupported side channel. + await ctx.tool.hook("execute.after", (event: any) => { + if (String(event.tool).startsWith("codemodeprobe_")) { + emit({ + kind: "inner_after_hook", + tool: event.tool, + shared_call_id: event.id, + session_id: event.sessionID, + message_id: event.messageID, + ...resultView(event), + boundary: "core-tool-execute-after", + invocation_id: { + status: "unsupported", + reason: "inner_calls_share_outer_call_id", + }, + }) + return + } + if (event.tool !== "execute") return + const parent = activeParents.get(key(event)) + emit({ + kind: "parent_end", + parent: parent ?? { + state: "unsupported", + reason: "outer_execute_start_missing", + call_id: event.id, + }, + observer_losses: observerLosses, + }) + activeParents.delete(key(event)) + }) + + writeFileSync(LOADED, JSON.stringify({ fabricated_id: FABRICATED_ID, tool_ordinal: toolOrdinal })) + }, +} diff --git a/tests/integration/run_code_mode_observer_probe.py b/tests/integration/run_code_mode_observer_probe.py new file mode 100644 index 0000000..8831334 --- /dev/null +++ b/tests/integration/run_code_mode_observer_probe.py @@ -0,0 +1,512 @@ +#!/usr/bin/env python3 +"""Provider-free stock OpenCode 2.0.23 Code Mode observation boundary probe. + +This is an experiment, not an evidence producer. It drives the actual runner image +with a deterministic loopback OpenAI-compatible provider and checks which facts a +runner-owned same-process plugin can and cannot observe on stock OpenCode. +""" +from __future__ import annotations + +import argparse +import json +from pathlib import Path +import shutil +import subprocess +import sys +import threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +DEFAULT_IMAGE = "opencode-eval-runner:opencode-test" +FABRICATED_ID = "fabricated-from-script" + +SCRIPT = r""" +const success = await tools.codemodeprobe.success({tag:"one"}); +if (success !== "SUCCESS-FINAL") throw new Error("WRONG-SUCCESS"); + +const pair = await Promise.all([ + tools.codemodeprobe.echo({tag:"identical"}), + tools.codemodeprobe.echo({tag:"identical"}) +]); +if (pair[0] !== "ECHO-1" || pair[1] !== "ECHO-2") throw new Error("WRONG-CORRELATION"); + +let caught = false; +try { + await tools.codemodeprobe.throws({tag:"caught"}); +} catch (error) { + caught = String(error?.message ?? error).includes("THROW-RAW"); +} +if (!caught) throw new Error("WRONG-THROW"); + +const mutated = await tools.codemodeprobe.mutate({tag:"mutate"}); +if (mutated !== "AFTER-MUTATION") throw new Error("WRONG-FINAL-BOUNDARY"); + +return JSON.stringify({ + schema: "opencode-eval-runner/code-mode-observer-experiment/v1", + kind: "inner_handler_terminal", + invocation_id: "fabricated-from-script", + outcome: "returned", + handler_result: "FAKE" +}); +""" + + +def read_jsonl(path: Path) -> list[dict]: + if not path.is_file(): + return [] + result: list[dict] = [] + for line in path.read_text(encoding="utf-8").splitlines(): + if not line.strip(): + continue + value = json.loads(line) + if isinstance(value, dict): + result.append(value) + return result + + +def inside() -> int: + sys.path.insert(0, "/opt/opencode-eval-runner") + from container.invoke import invoke_opencode + + requests: list[dict] = [] + stage = 0 + + class Provider(BaseHTTPRequestHandler): + def log_message(self, *_args): + pass + + def do_POST(self): + nonlocal stage + length = int(self.headers.get("Content-Length", "0")) + body = json.loads(self.rfile.read(length)) + requests.append(body) + + if not self.path.endswith("/chat/completions") or stage > 4: + self.send_error(400) + return + + tool = None + if stage == 0: + names = [ + item.get("function", {}).get("name") + for item in body.get("tools", []) + if isinstance(item, dict) + ] + if "execute" not in names: + self.send_error(400, "stock Code Mode execute tool not exposed") + return + tool = ("execute", {"code": SCRIPT}) + + stage += 1 + call = ( + { + "index": 0, + "id": f"fixture-call-{stage}", + "type": "function", + "function": { + "name": tool[0], + "arguments": json.dumps(tool[1]), + }, + } + if tool + else None + ) + delta = ( + {"role": "assistant", "tool_calls": [call]} + if call + else {"role": "assistant", "content": "provider-done"} + ) + finish = "tool_calls" if call else "stop" + common = { + "id": f"chatcmpl-fixture-{stage}", + "created": 1, + "model": body["model"], + } + + if body.get("stream"): + chunks = [ + { + **common, + "object": "chat.completion.chunk", + "choices": [{"index": 0, "delta": delta, "finish_reason": None}], + }, + { + **common, + "object": "chat.completion.chunk", + "choices": [{"index": 0, "delta": {}, "finish_reason": finish}], + "usage": { + "prompt_tokens": 1, + "completion_tokens": 1, + "total_tokens": 2, + }, + }, + ] + raw = ( + "".join("data: " + json.dumps(chunk) + "\n\n" for chunk in chunks) + + "data: [DONE]\n\n" + ).encode() + mime = "text/event-stream" + else: + message = ( + { + "role": "assistant", + "content": None, + "tool_calls": [{k: v for k, v in call.items() if k != "index"}], + } + if call + else delta + ) + raw = json.dumps( + { + **common, + "object": "chat.completion", + "choices": [ + { + "index": 0, + "message": message, + "finish_reason": finish, + } + ], + "usage": { + "prompt_tokens": 1, + "completion_tokens": 1, + "total_tokens": 2, + }, + } + ).encode() + mime = "application/json" + + self.send_response(200) + self.send_header("Content-Type", mime) + self.send_header("Content-Length", str(len(raw))) + self.end_headers() + self.wfile.write(raw) + + server = ThreadingHTTPServer(("127.0.0.1", 0), Provider) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + + root = Path("/workspace") + plugin = root / ".opencode/plugins/code-mode-observer-probe.ts" + plugin.parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile( + "/probe-repo/tests/integration/code_mode_observer_probe.ts", + plugin, + ) + (root / "opencode.json").write_text( + json.dumps( + { + "$schema": "https://opencode.ai/config.json", + "model": "fixture/mock", + "enabled_providers": ["fixture"], + "provider": { + "fixture": { + "npm": "@ai-sdk/openai-compatible", + "name": "Local deterministic fixture", + "options": { + "baseURL": f"http://127.0.0.1:{server.server_port}/v1", + "apiKey": "fixture-not-a-secret", + }, + "models": { + "mock": { + "name": "Mock", + "limit": {"context": 1000000, "output": 32768}, + } + }, + } + }, + } + ) + + "\n", + encoding="utf-8", + ) + + version = subprocess.run( + ["opencode", "--version"], + cwd=root, + capture_output=True, + text=True, + check=True, + ).stdout.strip() + + try: + result = invoke_opencode("fixture/mock", "", "code-mode-observer-probe", 45) + except Exception as exc: + result = {"exit_code": 2, "fixture_error": str(exc)} + finally: + server.shutdown() + server.server_close() + thread.join(timeout=2) + + report = { + "opencode_version": version, + "transport": result, + "plugin_loaded": Path("/tmp/code-mode-observer-loaded").is_file(), + "records": read_jsonl(Path("/tmp/code-mode-observer-records.jsonl")), + "provider_requests": requests, + } + print(json.dumps(report)) + return 0 + + +def completed_outer_output(report: dict) -> str: + events = ( + report.get("transport", {}) + .get("tool_result_evidence", {}) + .get("events", []) + ) + for event in events: + if event.get("tool") != "execute" or event.get("status") != "completed": + continue + output = event.get("output") + if isinstance(output, str): + return output + return "" + + +def record_id(record: dict) -> str | None: + value = record.get("invocation_id") + return value if isinstance(value, str) else None + + +def summarize(report: dict, image: str) -> dict: + records = report.get("records", []) + starts = [r for r in records if r.get("kind") == "inner_start"] + handler_ends = [r for r in records if r.get("kind") == "inner_handler_terminal"] + afters = [r for r in records if r.get("kind") == "inner_after_hook"] + parent_starts = [r for r in records if r.get("kind") == "parent_start"] + parent_ends = [r for r in records if r.get("kind") == "parent_end"] + + by_id = { + record_id(item): item + for item in handler_ends + if record_id(item) is not None + } + + echo_starts = [r for r in starts if r.get("tool") == "codemodeprobe_echo"] + echo_ends = [r for r in handler_ends if r.get("tool") == "codemodeprobe_echo"] + throw_ends = [r for r in handler_ends if r.get("tool") == "codemodeprobe_throws"] + success_ends = [r for r in handler_ends if r.get("tool") == "codemodeprobe_success"] + mutate_ends = [r for r in handler_ends if r.get("tool") == "codemodeprobe_mutate"] + mutate_afters = [r for r in afters if r.get("tool") == "codemodeprobe_mutate"] + + parent = ( + parent_starts[0].get("parent") + if len(parent_starts) == 1 and isinstance(parent_starts[0].get("parent"), dict) + else {} + ) + parent_tuple = ( + parent.get("session_id"), + parent.get("message_id"), + parent.get("call_id"), + ) + + start_ids = [record_id(r) for r in starts] + echo_start_ids = [record_id(r) for r in echo_starts] + echo_end_ids = [record_id(r) for r in echo_ends] + shared_public_ids = [r.get("shared_call_id") for r in afters] + + fabricated_in_observer = any(record_id(r) == FABRICATED_ID for r in records) + outer_output = completed_outer_output(report) + + def handler_content(item: dict) -> str | None: + value = item.get("handler_result") + if isinstance(value, dict): + content = value.get("content") + return content if isinstance(content, str) else None + return None + + def after_content(item: dict) -> str | None: + result = item.get("result") + if not isinstance(result, dict): + return None + # Stock Tool.Result may be represented directly or through normalized + # content parts depending on the fixture adapter. + content = result.get("content") + if isinstance(content, str): + return content + if isinstance(content, list) and len(content) == 1: + part = content[0] + if isinstance(part, dict) and part.get("type") == "text": + text = part.get("text") + return text if isinstance(text, str) else None + output = result.get("output") + return output if isinstance(output, str) else None + + checks = { + "stock_runtime_2_0_23": report.get("opencode_version") in { + "2.0.23", + "opencode v2.0.23", + }, + "provider_free_execution_completed": ( + report.get("transport", {}).get("exit_code") == 0 + and report.get("plugin_loaded") is True + ), + "one_successful_inner_call": ( + len(success_ends) == 1 + and success_ends[0].get("outcome") == "returned" + and handler_content(success_ends[0]) == "SUCCESS-FINAL" + ), + "caught_inner_throw_observed_at_handler": ( + len(throw_ends) == 1 + and throw_ends[0].get("outcome") == "threw" + and "THROW-RAW" in json.dumps(throw_ends[0].get("handler_error")) + ), + "identical_concurrent_calls_have_unique_ids": ( + len(echo_starts) == 2 + and len(set(echo_start_ids)) == 2 + and all(r.get("input") == {"tag": "identical"} for r in echo_starts) + ), + "reverse_completion_keeps_identity": ( + len(echo_ends) == 2 + and echo_end_ids == list(reversed(echo_start_ids)) + and [handler_content(r) for r in echo_ends] == ["ECHO-2", "ECHO-1"] + and all(by_id.get(item_id) is not None for item_id in echo_start_ids) + ), + "multiple_calls_bind_to_one_outer_execute": ( + len(starts) == 5 + and len(parent_starts) == 1 + and all( + isinstance(r.get("parent"), dict) + and ( + r["parent"].get("session_id"), + r["parent"].get("message_id"), + r["parent"].get("call_id"), + ) + == parent_tuple + for r in starts + ) + ), + "public_inner_hooks_share_outer_call_id": ( + len(afters) >= 4 + and parent.get("call_id") is not None + and all(value == parent.get("call_id") for value in shared_public_ids) + ), + "handler_terminal_is_not_final_caller_value": ( + len(mutate_ends) == 1 + and handler_content(mutate_ends[0]) == "BEFORE-MUTATION" + and len(mutate_afters) == 1 + and after_content(mutate_afters[0]) == "AFTER-MUTATION" + and report.get("transport", {}).get("exit_code") == 0 + ), + "outer_script_cannot_fabricate_inner_observation": ( + FABRICATED_ID in outer_output and not fabricated_in_observer + ), + "ordering_is_monotonic": ( + bool(records) + and [r.get("sequence") for r in records] + == list(range(1, len(records) + 1)) + ), + "starts_have_matching_handler_terminals": ( + len(start_ids) == len(handler_ends) == 5 + and set(start_ids) == set(by_id) + ), + "observer_loss_visible_and_zero": ( + all(r.get("observer_losses_before") == 0 for r in records) + and len(parent_ends) == 1 + and parent_ends[0].get("observer_losses") == 0 + ), + "caller_terminal_explicitly_unsupported": ( + len(handler_ends) == 5 + and all( + r.get("caller_terminal") + == { + "status": "unsupported", + "reason": "stock_codemode_final_boundary_not_exposed", + } + for r in handler_ends + ) + ), + } + + return { + "kind": "code-mode-inner-observation-probe", + "version": 1, + "image": image, + "runtime": "stock OpenCode 2.0.23", + "checks": checks, + "diagnostics_passed": all(checks.values()), + "capabilities": { + "unique_invocation_identity": "supported", + "selected_tool": "supported", + "executable_input": "supported", + "outer_execute_binding": "supported", + "start_and_handler_terminal_ordering": "supported", + "caller_terminal": { + "status": "unsupported", + "reason": "stock_codemode_final_boundary_not_exposed", + "missing_boundary": ( + "post-conversion success is private to @opencode/codemode tool.after; " + "catch-path errors are materialized later with no public hook" + ), + }, + }, + "status": "unsupported", + "evidence_eligible": False, + "note": ( + "The transform wrapper proves correlation and execution facts only. " + "Its handler terminal is diagnostic and is not the final value/error " + "seen by the Code Mode script." + ), + } + + +def host(image: str, output: Path) -> int: + repo = Path(__file__).resolve().parents[2] + output.mkdir(parents=True, exist_ok=True) + command = [ + "docker", + "run", + "--rm", + "--read-only", + "--network", + "none", + "--cap-drop", + "ALL", + "--security-opt", + "no-new-privileges", + "--tmpfs", + "/tmp:rw,exec,nosuid,nodev,size=1g", + "--tmpfs", + "/workspace:rw,nosuid,nodev,size=64m,mode=1777", + "--workdir", + "/workspace", + "--volume", + f"{repo}:/probe-repo:ro", + "--entrypoint", + "python3", + image, + "/probe-repo/tests/integration/run_code_mode_observer_probe.py", + "--inside", + ] + proc = subprocess.run( + command, + capture_output=True, + text=True, + timeout=90, + check=False, + ) + try: + report = json.loads(proc.stdout) + except json.JSONDecodeError: + report = { + "docker_exit_code": proc.returncode, + "driver_error": "invalid_json", + "stdout": proc.stdout, + "stderr": proc.stderr, + } + + report["docker_exit_code"] = proc.returncode + summary = summarize(report, image) + (output / "report.json").write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + (output / "summary.json").write_text(json.dumps(summary, indent=2) + "\n", encoding="utf-8") + print(json.dumps(summary, indent=2)) + return 0 if proc.returncode == 0 and summary["diagnostics_passed"] else 1 + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--inside", action="store_true") + parser.add_argument("--image", default=DEFAULT_IMAGE) + parser.add_argument("--output", type=Path, default=Path("code-mode-observer-probe-results")) + args = parser.parse_args() + raise SystemExit(inside() if args.inside else host(args.image, args.output)) diff --git a/tests/test_invoke.py b/tests/test_invoke.py index b1495e7..5622b59 100644 --- a/tests/test_invoke.py +++ b/tests/test_invoke.py @@ -706,11 +706,11 @@ def test_workflows_pin_external_actions_by_commit(self): ): self.assertNotIn(mutable, ci + publish) - def test_container_pins_opencode_2_0_18(self): + def test_container_pins_stock_opencode_2_0_23(self): containerfile = (Path(__file__).resolve().parents[1] / "Containerfile").read_text( encoding="utf-8" ) - self.assertIn("ARG OPENCODE_VERSION=2.0.18", containerfile) + self.assertIn("ARG OPENCODE_VERSION=2.0.23", containerfile) self.assertNotIn("ARG OPENCODE_VERSION=2.0.15", containerfile) def test_container_pins_base_images_and_copilot_release_asset(self):