diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 419945f..c25d0a0 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -16,7 +16,7 @@ jobs: - name: Python checks run: | - python3 -m py_compile runner/cli.py container/invoke.py + python3 -m py_compile runner/cli.py container/invoke.py container/native_observer.py python3 -m unittest discover -s tests -p 'test_*.py' - name: Build fake Copilot transport for action smoke test diff --git a/.github/workflows/native-observer-integration.yml b/.github/workflows/native-observer-integration.yml new file mode 100644 index 0000000..9fc6bb5 --- /dev/null +++ b/.github/workflows/native-observer-integration.yml @@ -0,0 +1,53 @@ +name: Stock native observer integration + +on: + pull_request: + paths: + - 'container/invoke.py' + - 'container/native_observer.py' + - 'container/native_observer.ts' + - 'tests/integration/native_observer_probe.ts' + - 'tests/integration/run_native_observer_probe.py' + - 'tests/test_native_observer.py' + - '.github/workflows/native-observer-integration.yml' + +permissions: + contents: read + +jobs: + stock-native-observer: + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + with: + ref: ${{ github.event.pull_request.head.sha }} + persist-credentials: false + + - name: Verify probe syntax + run: python3 -m py_compile container/native_observer.py tests/integration/run_native_observer_probe.py + + - name: Build runner with stock OpenCode 2.0.23 + run: docker build -f Containerfile --target opencode -t opencode-eval-runner:native-observer-probe . + + - name: Run provider-free native observer probe + run: | + docker run --rm \ + --network none \ + --tmpfs /tmp:rw,exec,nosuid,nodev,size=1g,mode=1777 \ + --tmpfs /workspace:rw,nosuid,nodev,size=64m,mode=1777 \ + --volume "$PWD:/probe-repo:ro" \ + --entrypoint python3 \ + opencode-eval-runner:native-observer-probe \ + /probe-repo/tests/integration/run_native_observer_probe.py \ + > native-observer-probe.json + cat native-observer-probe.json + + - name: Preserve probe report + if: always() + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: native-observer-${{ github.event.pull_request.head.sha }} + path: native-observer-probe.json + if-no-files-found: warn + retention-days: 14 diff --git a/Containerfile b/Containerfile index 97fb1f9..3ec8003 100644 --- a/Containerfile +++ b/Containerfile @@ -1,5 +1,5 @@ FROM node:24-bookworm-slim@sha256:0e0ff40c39bc087845bfb27465a0df4ea419520094bc35842ff83dd8cbe6f9b6 AS opencode-builder -ARG OPENCODE_VERSION=2.0.18 +ARG OPENCODE_VERSION=2.0.23 RUN npm install --global "@opencode/cli@${OPENCODE_VERSION}" \ && resolved="$(readlink -f "$(command -v opencode)")" \ && test -x "$resolved" \ diff --git a/README.md b/README.md index 0f5c5a7..ff0d70f 100644 --- a/README.md +++ b/README.md @@ -72,7 +72,15 @@ Known API-key environment variables are passed when present: Additional variables require explicit `--env NAME`. -Reasoning can be pinned explicitly with `--reasoning LEVEL`. The pinned OpenCode 2.0.18 CLI represents a model variant in the model reference, so the runner maps `--model provider/model --reasoning LEVEL` to `opencode run --model provider/model#LEVEL`. Supplying both a `#variant` in `--model` and `--reasoning` is rejected as ambiguous. If the model reference already contains a variant and `--reasoning` is omitted, the result records that variant with `"reasoning_source": "model-variant"`. If neither form supplies a level, the runner leaves OpenCode's provider/model default untouched and records `"reasoning": "provider-default"`. +Reasoning can be pinned explicitly with `--reasoning LEVEL`. The pinned stock OpenCode 2.0.23 CLI represents a model variant in the model reference, so the runner maps `--model provider/model --reasoning LEVEL` to `opencode run --model provider/model#LEVEL`. Supplying both a `#variant` in `--model` and `--reasoning` is rejected as ambiguous. If the model reference already contains a variant and `--reasoning` is omitted, the result records that variant with `"reasoning_source": "model-variant"`. If neither form supplies a level, the runner leaves OpenCode's provider/model default untouched and records `"reasoning": "provider-default"`. + +### Native tool runtime observations + +OpenCode results also include `native_tool_observations` for native/direct tools on the pinned stock 2.0.23 runtime. A runner-owned in-process plugin records decoded executable input at the tool implementation boundary and correlates it with stock `session.tool.success` / `session.tool.failed` terminals using the exact Session/message/call identity. + +This field is fail-closed: missing capture, observer loss, missing terminals, sequence gaps, invalid fields, or incomplete shutdown make `evidence_eligible` false. Model text and tool-returned collector-shaped JSON are never parsed into this projection. + +See `docs/native-tool-observer.md` for the exact boundary and limitations. Code Mode inner calls are not covered by this native observer. ### `github-copilot-cli` @@ -173,7 +181,7 @@ opencode-eval-runner invoke \ ... ``` -OpenCode 2.0.18 does not expose the old singular `debug agent ` command that returned a resolved tool map. The runner therefore performs the strongest supported zero-inference preflight: it requires the expected plugin entrypoint to be materialized in the isolated OpenCode config, runs `opencode debug agents` to prove the configured location starts successfully with plugins active, and requires the selected agent to resolve. Missing plugin materialization, plugin/startup failure, or missing agent is infrastructure/non-evidence, never a behavioral FAIL. Actual tool use remains a repository-owned behavioral assertion in the eval corpus. +Stock OpenCode 2.0.23 does not expose the old singular `debug agent ` command that returned a resolved tool map. The runner therefore performs the strongest supported zero-inference preflight: it requires the expected plugin entrypoint to be materialized in the isolated OpenCode config, runs `opencode debug agents` to prove the configured location starts successfully with plugins active, and requires the selected agent to resolve. Missing plugin materialization, plugin/startup failure, or missing agent is infrastructure/non-evidence, never a behavioral FAIL. Actual tool use remains a repository-owned behavioral assertion in the eval corpus. ### Evaluating a skill @@ -315,7 +323,7 @@ The eval repository decides whether that observed behavior is PASS, FAIL, or non The transport images currently pin: -- OpenCode CLI `2.0.18` +- OpenCode CLI `2.0.23` - GitHub Copilot CLI `1.0.83` The two CLIs are not bundled together. OpenCode's npm package is used only as a build-time native-binary selector; GitHub Copilot CLI is installed from its native release installer. Node/npm are absent from the final runtime images. @@ -340,6 +348,18 @@ OPENCODE_EVAL_RUNNER_COPILOT_IMAGE=... Tags matching `v*` are published with `opencode-` and `copilot-` prefixes. +## Evaluation trust model + +The normal evaluation profile is a **trusted-checkout** profile. It assumes the runner, pinned stock OpenCode runtime, reviewed instrumentation, and explicitly selected evaluated checkout/dependencies are trusted components of the evaluation environment. + +They are not trusted merely because they produce data that looks like evidence. Model prose, tool-returned collector-shaped JSON, target-writable files, requested actions, inferred identities, and reconstructed results do not establish that an event occurred. + +Authoritative runtime observations must come from reviewed instrumentation observing actual execution. Missing, partial, ambiguous, or unsupported required observations are non-evidence and must fail closed for the affected assertion. + +This profile does **not** claim resistance to an evaluated plugin that deliberately compromises the trusted runtime or instrumentation. Hostile-plugin isolation is a separate optional profile, not a prerequisite for normal Loom evaluation. + +See [Trusted-checkout runtime evidence](docs/trusted-checkout-evidence.md). + ## Security boundary The runner: diff --git a/container/invoke.py b/container/invoke.py index 1dc9111..2525d27 100644 --- a/container/invoke.py +++ b/container/invoke.py @@ -16,6 +16,11 @@ from pathlib import Path from typing import Any +if __package__: + from container.native_observer import OBSERVATION_PATH, load_native_observations +else: + from native_observer import OBSERVATION_PATH, load_native_observations + RESULT_SCHEMA = "opencode-eval-runner/v1" OPENCODE_EVAL_TITLE = "opencode-eval-runner" COPILOT_AGENT_NAME = "eval-runner" @@ -304,6 +309,10 @@ def prepare_opencode_env() -> dict[str, str]: json.dumps({"$schema": "https://opencode.ai/config.json"}) + "\n", encoding="utf-8", ) + observer_root = config / "eval-native-observer" + observer_root.mkdir(parents=True, exist_ok=True) + shutil.copyfile(Path(__file__).with_name("native_observer.ts"), observer_root / "server.ts") + if seed_config_root.is_dir(): # Loom itself can be an OpenCode global config root. OpenCode 2.0.11's # packaged runtime can fail to register directory plugins even when @@ -345,6 +354,10 @@ def prepare_opencode_env() -> dict[str, str]: "XDG_CACHE_HOME": str(cache), "XDG_STATE_HOME": str(state), "OPENCODE_CONFIG_DIR": str(config), + # Inline config is the highest-priority local config in stock 2.0.23. + # Register the runner-owned observer there so its tool transform runs + # after project/global plugin transforms instead of depending on filename order. + "OPENCODE_CONFIG_CONTENT": json.dumps({"plugins": [observer_root.as_uri()]}), "OPENCODE_DB": "opencode.db", "OPENCODE_DISABLE_AUTOUPDATE": "1", }) @@ -641,6 +654,7 @@ def plugin_diagnostic(env: dict[str, str]) -> dict[str, Any]: loom_root = plugins_root / "loom" loom_flat = plugins_root / "loom.ts" loom_module_root = config_root / "loom-plugin" + observer_root = config_root / "eval-native-observer" return { "config_root": str(config_root), "config_root_exists": config_root.exists(), @@ -654,6 +668,8 @@ def plugin_diagnostic(env: dict[str, str]) -> dict[str, Any]: "loom_index_exists": (loom_root / "index.ts").is_file(), "loom_flat_exists": loom_flat.is_file(), "loom_module_index_exists": (loom_module_root / "index.ts").is_file(), + "native_observer_exists": (observer_root / "server.ts").is_file(), + "native_observer_inline_configured": "eval-native-observer" in env.get("OPENCODE_CONFIG_CONTENT", ""), } @@ -710,6 +726,10 @@ def invoke_opencode( if agent: command += ["--agent", agent] command += ["--model", invoked_model, prompt] + + # A plugin-preflight process may have activated the observer already. + # Never let that prior process become evidence for the real invocation. + OBSERVATION_PATH.unlink(missing_ok=True) run_started = time.perf_counter() try: proc = run(command, Path("/workspace"), env, timeout) @@ -719,6 +739,7 @@ def invoke_opencode( stderr = timeout_output(exc.stderr) events = parse_events(stdout) sid = session_id(events) + native_tool_observations = load_native_observations() summary = last_event_summary(events) detail = ( f"opencode run timed out after {timeout}s; " @@ -755,6 +776,7 @@ def invoke_opencode( "stdout_truncated": len(stdout) > STDOUT_CAPTURE_LIMIT, "stdout_total_chars": len(stdout), "tool_result_evidence": extract_tool_result_evidence(events), + "native_tool_observations": native_tool_observations, "plugin_diagnostic": plugins, "plugin_preflight": plugin_preflight, } @@ -762,12 +784,12 @@ def invoke_opencode( run_seconds = time.perf_counter() - run_started events = parse_events(proc.stdout) sid = session_id(events) + native_tool_observations = load_native_observations() - # The structured `opencode run --format json` event stream is the - # authoritative evidence source. Starting a second OpenCode process to - # export the just-created session is redundant and can add a full timeout - # per invocation when export/session bootstrap fails. Keep eval latency - # bound to the requested target/judge execution only. + # The structured `opencode run --format json` stream remains the product + # output source. Reviewed native-tool evidence comes from the runner-owned + # observer projection above; stdout/tool payload JSON is never promoted into it. + # Starting a second process to export the Session remains unnecessary. text = extract_text(events) tools = extract_tools(events) actions = extract_actions(events) @@ -802,6 +824,7 @@ def invoke_opencode( "stdout_truncated": len(proc.stdout) > STDOUT_CAPTURE_LIMIT, "stdout_total_chars": len(proc.stdout), "tool_result_evidence": extract_tool_result_evidence(events), + "native_tool_observations": native_tool_observations, "plugin_diagnostic": plugins, "plugin_preflight": plugin_preflight, } diff --git a/container/native_observer.py b/container/native_observer.py new file mode 100644 index 0000000..938bd8c --- /dev/null +++ b/container/native_observer.py @@ -0,0 +1,243 @@ +from __future__ import annotations + +import json +import os +import stat +from pathlib import Path +from typing import Any + +OBSERVATION_PATH = Path("/tmp/runtime/native-tool-observer.jsonl") +SCHEMA = "opencode-eval-runner/native-tool-observer-event/v1" +RESULT_SCHEMA = "opencode-eval-runner/native-tool-observations/v1" +MAX_CAPTURE_BYTES = 8 * 1024 * 1024 +MAX_RECORDS = 10001 + + +class InvalidObservation(ValueError): + pass + + +def projection() -> dict[str, Any]: + return { + "schema": RESULT_SCHEMA, + "status": "unavailable", + "evidence_eligible": False, + "records": [], + "issues": [], + "coverage": { + "capture_started": False, + "capture_ended": False, + "observed_starts": 0, + "observed_terminals": 0, + "missing_terminals": 0, + "observer_failures": None, + "unavailable_fields": None, + }, + } + + +def unavailable(reason: str) -> dict[str, Any]: + result = projection() + result["issues"] = [reason] + return result + + +def _constant(_: str) -> Any: + raise InvalidObservation("invalid_json_constant") + + +def _object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + result: dict[str, Any] = {} + for key, value in pairs: + if key in result: + raise InvalidObservation("duplicate_json_key") + result[key] = value + return result + + +def _read(path: Path) -> bytes: + fd = os.open(path, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK) + try: + info = os.fstat(fd) + if not stat.S_ISREG(info.st_mode) or info.st_nlink != 1: + raise InvalidObservation("unsafe_capture_file") + if info.st_size > MAX_CAPTURE_BYTES: + raise InvalidObservation("capture_limit") + raw = os.read(fd, MAX_CAPTURE_BYTES + 1) + if len(raw) > MAX_CAPTURE_BYTES: + raise InvalidObservation("capture_limit") + return raw + finally: + os.close(fd) + + +def _identity(value: Any) -> str: + if not isinstance(value, str) or not value or len(value.encode("utf-8")) > 1024 or "\x00" in value: + raise InvalidObservation("invalid_identity") + return value + + +def _field(value: Any) -> dict[str, Any]: + if not isinstance(value, dict) or set(value) not in ({"state", "value"}, {"state", "reason"}): + raise InvalidObservation("invalid_field") + if value.get("state") == "available": + return {"state": "available", "value": value["value"]} + if value.get("state") == "omitted" and isinstance(value.get("reason"), str): + return {"state": "omitted", "reason": value["reason"]} + raise InvalidObservation("invalid_field") + + +def load_native_observations(path: Path = OBSERVATION_PATH) -> dict[str, Any]: + result = projection() + coverage = result["coverage"] + calls: dict[tuple[str, str, str], dict[str, Any]] = {} + ordered: list[dict[str, Any]] = [] + ended = False + try: + raw = _read(path) + if not raw: + return unavailable("empty_capture") + if not raw.endswith(b"\n"): + raise InvalidObservation("unterminated_capture") + lines = raw.splitlines() + if len(lines) > MAX_RECORDS: + raise InvalidObservation("record_limit") + + for expected_sequence, line in enumerate(lines): + if ended: + raise InvalidObservation("records_after_capture_end") + try: + event = json.loads( + line.decode("utf-8"), + object_pairs_hook=_object, + parse_constant=_constant, + ) + except (json.JSONDecodeError, UnicodeDecodeError) as exc: + raise InvalidObservation("malformed_capture") from exc + if not isinstance(event, dict) or event.get("schema") != SCHEMA: + raise InvalidObservation("wrong_schema") + if event.get("sequence") != expected_sequence: + raise InvalidObservation("ambiguous_order") + if not isinstance(event.get("observer_failures"), int) or event["observer_failures"] < 0: + raise InvalidObservation("invalid_failure_count") + + kind = event.get("kind") + if expected_sequence == 0: + if kind != "capture_start" or event.get("version") != 1: + raise InvalidObservation("missing_capture_start") + expected = { + "source": "stock-opencode-2.0.23-plugin", + "input_boundary": "decoded-tool-execute", + "terminal_boundary": "session.tool.success+session.tool.failed", + "correlation": "session-message-call-id", + "ordering": "observer-monotonic-sequence", + } + if any(event.get(key) != value for key, value in expected.items()): + raise InvalidObservation("unsupported_capture_boundary") + coverage["capture_started"] = True + continue + + if kind == "call_start": + if event.get("boundary") != "decoded-tool-execute": + raise InvalidObservation("unsupported_start_boundary") + identity = { + "tool": _identity(event.get("tool")), + "session_id": _identity(event.get("session_id")), + "agent": _identity(event.get("agent")), + "message_id": _identity(event.get("message_id")), + "call_id": _identity(event.get("call_id")), + } + key = (identity["session_id"], identity["message_id"], identity["call_id"]) + if key in calls: + raise InvalidObservation("duplicate_call_start") + record = { + **identity, + "input": _field(event.get("input")), + "start_sequence": expected_sequence, + "terminal_sequence": None, + "outcome": "missing", + } + calls[key] = record + ordered.append(record) + coverage["observed_starts"] += 1 + continue + + if kind == "call_terminal": + outcome = event.get("outcome") + expected_boundary = { + "success": "session.tool.success", + "failure": "session.tool.failed", + }.get(outcome) + if expected_boundary is None or event.get("boundary") != expected_boundary: + raise InvalidObservation("unsupported_terminal_boundary") + session_id = _identity(event.get("session_id")) + message_id = _identity(event.get("message_id")) + call_id = _identity(event.get("call_id")) + key = (session_id, message_id, call_id) + record = calls.get(key) + if record is None or record["terminal_sequence"] is not None: + raise InvalidObservation("ambiguous_terminal") + if record["tool"] != _identity(event.get("tool")) or record["agent"] != _identity(event.get("agent")): + raise InvalidObservation("identity_changed") + if outcome == "success" and "result" in event and "error" not in event: + record["result"] = _field(event["result"]) + elif outcome == "failure" and "error" in event and "result" not in event: + record["error"] = _field(event["error"]) + else: + raise InvalidObservation("invalid_terminal") + record["outcome"] = outcome + record["terminal_sequence"] = expected_sequence + coverage["observed_terminals"] += 1 + continue + + if kind == "capture_end": + for field in ("calls_started", "calls_terminal", "outstanding_calls", "observer_failures", "unavailable_fields"): + if not isinstance(event.get(field), int) or event[field] < 0: + raise InvalidObservation("invalid_completeness") + if event["calls_started"] != coverage["observed_starts"]: + raise InvalidObservation("start_count_mismatch") + if event["calls_terminal"] != coverage["observed_terminals"]: + raise InvalidObservation("terminal_count_mismatch") + missing = sum(record["outcome"] == "missing" for record in ordered) + if event["outstanding_calls"] != missing: + raise InvalidObservation("outstanding_count_mismatch") + coverage["capture_ended"] = True + coverage["observer_failures"] = event["observer_failures"] + coverage["unavailable_fields"] = event["unavailable_fields"] + ended = True + continue + + raise InvalidObservation("unknown_record_kind") + + result["records"] = ordered + coverage["missing_terminals"] = sum(record["outcome"] == "missing" for record in ordered) + if not ended: + result["issues"].append("missing_capture_end") + if coverage["missing_terminals"]: + result["issues"].append("missing_terminals") + if coverage["observer_failures"]: + result["issues"].append("observer_failures") + if coverage["unavailable_fields"]: + result["issues"].append("unavailable_fields") + if any( + value.get("state") != "available" + for record in ordered + for value in (record.get("input"), record.get("result"), record.get("error")) + if isinstance(value, dict) + ) and "unavailable_fields" not in result["issues"]: + result["issues"].append("unavailable_fields") + result["evidence_eligible"] = not result["issues"] + result["status"] = "complete" if result["evidence_eligible"] else "incomplete" + return result + except FileNotFoundError: + return unavailable("missing_capture") + except InvalidObservation as exc: + result["status"] = "invalid" + result["issues"] = [str(exc)] + result["records"] = [] + return result + except OSError: + result["status"] = "invalid" + result["issues"] = ["capture_io_error"] + result["records"] = [] + return result diff --git a/container/native_observer.ts b/container/native_observer.ts new file mode 100644 index 0000000..b96faf7 --- /dev/null +++ b/container/native_observer.ts @@ -0,0 +1,202 @@ +import { appendFileSync, mkdirSync, writeFileSync } from "node:fs" +import { dirname } from "node:path" + +const PATH = "/tmp/runtime/native-tool-observer.jsonl" +const SCHEMA = "opencode-eval-runner/native-tool-observer-event/v1" +const MAX_FIELD_BYTES = 256 * 1024 + +let sequence = 0 +let starts = 0 +let terminals = 0 +let observerFailures = 0 +let unavailableFields = 0 + +type Identity = { + tool: string + sessionID: string + agent: string + messageID: string + callID: string +} + +const active = new Map() + +function key(sessionID: string, messageID: string, callID: string) { + return [sessionID, messageID, callID].join("\u0000") +} + +function snapshot(value: unknown): { state: "available"; value: unknown } | { state: "omitted"; reason: string } { + const seen = new Set() + let nodes = 0 + const copy = (item: unknown, depth: number): unknown => { + if (++nodes > 10000 || depth > 32) throw new Error("snapshot_limit") + if (item === null || typeof item === "string" || typeof item === "boolean") return item + if (typeof item === "number" && Number.isFinite(item)) return item + if (!item || typeof item !== "object") throw new Error("non_json_value") + if (seen.has(item)) throw new Error("cyclic_value") + const proto = Object.getPrototypeOf(item) + if (!Array.isArray(item) && proto !== Object.prototype && proto !== null) throw new Error("non_plain_value") + seen.add(item) + const descriptors = Object.getOwnPropertyDescriptors(item) + const result: Record | unknown[] = Array.isArray(item) ? [] : Object.create(null) + for (const property of Reflect.ownKeys(descriptors)) { + if (Array.isArray(item) && property === "length") continue + if (typeof property !== "string") throw new Error("symbol_key") + const descriptor = descriptors[property] + if (!descriptor.enumerable || !("value" in descriptor)) throw new Error("non_data_property") + Object.defineProperty(result, property, { enumerable: true, value: copy(descriptor.value, depth + 1) }) + } + if (Array.isArray(item) && Object.keys(result).length !== item.length) throw new Error("sparse_array") + seen.delete(item) + return result + } + + try { + const copied = copy(value, 0) + if (Buffer.byteLength(JSON.stringify(copied), "utf8") > MAX_FIELD_BYTES) { + return { state: "omitted", reason: "field_limit" } + } + return { state: "available", value: copied } + } catch { + return { state: "omitted", reason: "unsupported_snapshot" } + } +} + +function field(value: unknown) { + const result = snapshot(value) + if (result.state !== "available") unavailableFields++ + return result +} + +function write(record: Record) { + const event = { schema: SCHEMA, sequence: sequence++, observer_failures: observerFailures, ...record } + try { + appendFileSync(PATH, JSON.stringify(event) + "\n", { encoding: "utf8" }) + } catch { + observerFailures++ + } +} + +function terminal(event: any) { + if (event?.type !== "session.tool.success" && event?.type !== "session.tool.failed") return + const data = event.data + if ( + !data || + typeof data.sessionID !== "string" || + typeof data.assistantMessageID !== "string" || + typeof data.id !== "string" + ) { + observerFailures++ + return + } + + const current = active.get(key(data.sessionID, data.assistantMessageID, data.id)) + if (!current) return + active.delete(key(data.sessionID, data.assistantMessageID, data.id)) + terminals++ + + const common = { + kind: "call_terminal", + tool: current.tool, + session_id: data.sessionID, + agent: current.agent, + message_id: data.assistantMessageID, + call_id: data.id, + boundary: event.type, + } + if (event.type === "session.tool.success") { + write({ + ...common, + outcome: "success", + result: field({ + content: data.content, + ...(data.metadata === undefined ? {} : { metadata: data.metadata }), + executed: data.executed, + ...(data.resultState === undefined ? {} : { result_state: data.resultState }), + }), + }) + return + } + write({ ...common, outcome: "failure", error: field(data.error) }) +} + +export default { + id: "eval-native-observer", + async setup(ctx: any) { + try { + mkdirSync(dirname(PATH), { recursive: true }) + writeFileSync(PATH, "", { encoding: "utf8" }) + } catch { + observerFailures++ + } + + write({ + kind: "capture_start", + version: 1, + source: "stock-opencode-2.0.23-plugin", + input_boundary: "decoded-tool-execute", + terminal_boundary: "session.tool.success+session.tool.failed", + correlation: "session-message-call-id", + ordering: "observer-monotonic-sequence", + }) + + const controller = new AbortController() + const eventTask = (async () => { + try { + for await (const event of ctx.event.subscribe({ signal: controller.signal })) terminal(event) + } catch { + if (!controller.signal.aborted) observerFailures++ + } + })() + + await ctx.tool.transform((editor: any) => { + for (const item of editor.list()) { + if (item.options?.codemode !== false) continue + editor.update(item.id, (tool: any) => { + const execute = tool.execute + tool.execute = async (input: unknown, context: any) => { + const identity: Identity = { + tool: item.id, + sessionID: context.sessionID, + agent: context.agent, + messageID: context.messageID, + callID: context.id, + } + const identityKey = key(identity.sessionID, identity.messageID, identity.callID) + starts++ + if (active.has(identityKey)) observerFailures++ + active.set(identityKey, identity) + write({ + kind: "call_start", + tool: identity.tool, + session_id: identity.sessionID, + agent: identity.agent, + message_id: identity.messageID, + call_id: identity.callID, + input: field(input), + boundary: "decoded-tool-execute", + }) + return await execute(input, context) + } + }) + } + }) + + return async () => { + // Session terminal events are published before the step can finish. Give + // the already-running local subscriber one turn to consume queued events, + // then close it without letting observer failure affect product behavior. + await new Promise((resolve) => setTimeout(resolve, 0)) + controller.abort() + await eventTask + write({ + kind: "capture_end", + calls_started: starts, + calls_terminal: terminals, + outstanding_calls: active.size, + observer_failures: observerFailures, + unavailable_fields: unavailableFields, + }) + } + }, +} diff --git a/docs/native-tool-observer.md b/docs/native-tool-observer.md new file mode 100644 index 0000000..2c81830 --- /dev/null +++ b/docs/native-tool-observer.md @@ -0,0 +1,88 @@ +# Stock native-tool observer + +This is the reviewed runtime observer for **native/direct tool calls** on stock OpenCode **2.0.23**. + +Source checkpoint: `0fd7e2829449b052abf0078666669302923d77af`. + +It does not patch or fork OpenCode. Loom and the evaluated checkout remain trusted in this profile. + +## Boundaries + +The observer is a runner-owned in-process Promise plugin. + +For tools whose effective definition has `options.codemode === false`: + +1. `ctx.tool.transform(...)` wraps the effective tool definition. +2. Stock OpenCode decodes the accepted input and then calls that wrapped `tool.execute`. +3. The wrapper records a **start** with: + - resolved/effective tool name; + - Session ID; + - agent; + - assistant message ID; + - real tool call ID; + - the decoded input actually passed to the tool. +4. A runner-owned `ctx.event.subscribe()` consumer records the stock Session terminal: + - `session.tool.success` with the canonical post-truncation content/metadata; or + - `session.tool.failed` with the canonical Session error representation. + +Stock 2.0.23 loads inline `OPENCODE_CONFIG_CONTENT` after project/global config sources. The runner registers the observer there so its tool transform is applied after evaluated local plugin transforms rather than depending on filename order. + +The terminal boundary is deliberately Session-owned. This matters for failures: a Promise-plugin rejection can escape before `tool.execute.after`, while the stock Session runner still settles the real call with `session.tool.failed`. Success is likewise taken from `session.tool.success` after stock output truncation, so the observer does not reconstruct a later result from an earlier hook. + +## Correlation and ordering + +Start and terminal records correlate only on: + +`(sessionID, messageID, callID)` + +The observer never pairs calls by input equality, FIFO position, tool name, or completion order. + +Each attempted observer write receives a monotonic sequence. The host projection requires a contiguous sequence and preserves both start and terminal sequence numbers. A gap is invalid evidence. + +## Completeness and loss + +A normal plugin lifetime writes: + +- `capture_start`; +- zero or more start/terminal records; +- `capture_end` with start, terminal, outstanding, observer-failure, and unavailable-field counts. + +The projection is evidence-eligible only when the capture is structurally valid, ended cleanly, has no missing terminals, no observer failures, and no unavailable required fields. + +A missing file, missing end marker, sequence gap, count mismatch, duplicate call, terminal without a start, oversized/unsupported field, or observer I/O loss fails closed. + +Observer failures do **not** change a tool's return value/error and do not trigger product retries. Loss is reflected only in observation eligibility. + +## Payload separation + +The projection reads only `/tmp/runtime/native-tool-observer.jsonl`, produced by the runner-owned plugin. + +It never scans: + +- model text; +- `opencode run --format json` prose/content fields; +- tool-returned strings; +- collector-shaped JSON embedded in any of those. + +The provider-free integration test explicitly places collector-shaped JSON in both a native tool result and model output and proves that the observation count does not change. + +## Scope + +This implementation covers native/direct calls only. + +Code Mode inner-call finality remains a separate problem because stock inner calls have different identity/finality constraints. This observer does not claim Code Mode coverage. + +## Verification + +`tests/test_native_observer.py` checks the parser and fail-closed rules. + +`tests/integration/run_native_observer_probe.py` builds/runs the actual repository image with stock OpenCode 2.0.23 and a loopback fake OpenAI-compatible provider under `--network none`. It proves: + +- native success; +- native failure; +- accepted/executable input after a pre-hook rewrite; +- resolved tool + Session + agent/message + real call ID; +- start/terminal correlation; +- ordering across multiple calls; +- collector-shaped payload separation; +- no observer-induced model retry. diff --git a/docs/stock-opencode-2.0.23-observation.md b/docs/stock-opencode-2.0.23-observation.md new file mode 100644 index 0000000..2860311 --- /dev/null +++ b/docs/stock-opencode-2.0.23-observation.md @@ -0,0 +1,66 @@ +# Stock OpenCode 2.0.23 observation surface + +Purpose: implementation reference for the trusted-checkout evidence profile. + +Source checkpoint: stock OpenCode **v2.0.23** (`0fd7e2829449b052abf0078666669302923d77af`). This is distilled from the source assessment performed in superseded PR #43. + +OpenCode remains stock and immutable. A missing observation boundary is reported as unsupported; it is not a reason to patch OpenCode or add a hostile-runtime broker. + +## Useful stock surfaces + +| Observation need | Stock surface | Status | +| --- | --- | --- | +| Live runtime events | `ctx.event.subscribe()` | supported source; ordering/drain must be proven by integration test | +| Session creation / ancestry | `session.created` + Session API | supported | +| Agent for a step | Session step/message events | supported | +| Native tool call identity/input | Tool transform wrapping the decoded executable boundary | implemented and integration-tested | +| Native terminal success/failure | `session.tool.success` / `session.tool.failed` from `ctx.event.subscribe()` | implemented and integration-tested | +| Tool pre-execution hook | `ctx.tool.hook("execute.before")` | supported; occurs before tool decode/execution | +| Tool post-handler hook | `ctx.tool.hook("execute.after")` | supported; occurs after handler result but before later core normalization | +| Tool registration wrapping | `ctx.tool.transform(...)` | supported candidate for reviewed same-process instrumentation | +| Code Mode inner name/input/status | Code Mode metadata + tool hooks | supported source | +| Code Mode unique inner invocation + exact final caller value/error | no single public final boundary demonstrated | **must be proven or marked unsupported** | + +## Native calls + +The implemented native observer uses two supported stock boundaries: + +- a final tool transform wraps direct tools (`options.codemode === false`) and records the value passed to `tool.execute`, after stock input decoding; +- a runner-owned live event subscriber records canonical `session.tool.success` / `session.tool.failed` terminals and correlates them to the start by exact Session/message/call identity. + +The observer is injected through `OPENCODE_CONFIG_CONTENT`, which stock 2.0.23 loads as the final local config source. That makes its transform/hook later than discovered global/project plugins rather than relying on filename ordering. + +Correlation is only by `(sessionID, messageID, callID)`. Input equality, FIFO pairing, model text, and tool-returned JSON are not correlation sources. + +The terminal snapshot is Session-owned: success is the post-truncation Session content/metadata and failure is the canonical Session error. This also covers defects that bypass `tool.execute.after`; no terminal is invented when the Session itself has not settled the call. + +## Code Mode + +Code Mode executes inner tools through the normal tool registry, so same-process reviewed instrumentation can observe real inner execution without isolating Loom. + +The difficult part is not security; it is exact correlation and finality: + +- inner calls share the outer `execute` context in stock OpenCode; +- public Code Mode metadata records name/input/status but not each inner returned value/error; +- `execute.after` is before later core normalization; +- concurrent identical inner calls must not be paired by FIFO, input equality, or completion order. + +The first implementation should test a runner-owned observer plugin using supported tool transforms/hooks and runtime events. It must allocate a unique observation identity at an actual execution boundary and prove how that identity reaches the final inner value/error. + +If that exact binding cannot be demonstrated for a case, the affected result field remains unavailable and the assertion cannot PASS. + +## Ordering and completeness + +A monotonic observer sequence is useful, but sequence alone is not completeness. The integration must also account for starts, terminals, observer loss, process interruption, and required descendant Sessions. + +Absence assertions are eligible only when the relevant scope is complete. Missing capture is never interpreted as "did not happen". + +## What is intentionally not required + +- plugin/process isolation from the trusted Loom checkout; +- remote PluginHost or capability broker; +- evidence-channel peer authentication against same-authority attackers; +- patched/forked OpenCode; +- cryptographic evidence authenticity after runtime compromise. + +Those belong only to a future optional untrusted-plugin profile. diff --git a/docs/trusted-checkout-evidence.md b/docs/trusted-checkout-evidence.md new file mode 100644 index 0000000..e0a503c --- /dev/null +++ b/docs/trusted-checkout-evidence.md @@ -0,0 +1,127 @@ +# Trusted-checkout runtime evidence + +Status: **replacement direction for TRUST-001**. + +This document supersedes the hostile-runtime direction explored in PR #41 and PR #43. Those PRs remain useful research/reference material, but normal Loom evaluation does not require the runner to defend itself from a deliberately malicious Loom checkout that shares its runtime authority. + +## Contract + +### TRUST-001 — authoritative runtime observation + +For evaluation of an explicitly trusted checkout, evidence used for scoring MUST originate from reviewed runtime instrumentation observing actual execution. + +The following MUST NOT independently establish that an event occurred: + +- model assertions or generated prose; +- tool payloads shaped like collector/evidence records; +- requested or intended actions; +- inferred actor, parent, or execution identity; +- reconstructed results; +- target-writable evidence files. + +Required observations MUST preserve enough runtime identity and ordering to evaluate the consumer contract, including actor/session/call identity, input, result or error, parent binding where applicable, and execution order. + +Missing, partial, ambiguous, lost, or unsupported required observations MUST make the affected assertion ineligible for PASS. The runner MUST NOT fill gaps from model text, stdout, workspace files, or guessed correlations. + +The trusted-checkout profile does not claim protection against malicious modification of the runner, stock OpenCode process, reviewed instrumentation, evaluated checkout, or their dependencies. + +## Trust model + +Trusted components: + +- the selected `opencode-eval-runner` revision; +- pinned **stock OpenCode 2.0.23**; +- reviewed runtime instrumentation; +- the explicitly selected Loom checkout and its reviewed dependencies; +- host-side evidence projection/persistence code. + +Not trusted as evidence authority: + +- model output; +- agent claims; +- tool-returned collector-shaped data; +- normal product/session/workspace files; +- caller-supplied identity or completeness claims. + +This is an evaluation-correctness boundary, not a hostile-code security boundary. + +## Required evidence behavior + +The target behavior remains strict even though the security scope is smaller: + +- **Native calls:** observe the actual runtime call, actor/session/call identity, accepted/executable input, and terminal result/error. +- **Code Mode inner calls:** assign a unique runtime observation identity per actual inner invocation, bind it to the real outer `execute` call, and observe the final value/error that Code Mode exposes to the script. +- **Delegation:** derive child Session identity and ancestry from runtime facts, not a parent result payload. +- **Ordering:** preserve runtime observation order; do not correlate concurrent calls by FIFO or input equality. +- **Completeness:** explicitly report missing starts/terminals, capture loss, unsupported boundaries, and incomplete scope. +- **Confidentiality:** redact or omit credentials before the runner first persists, clips, logs, or exports evidence. +- **Noninterference:** observation must not add product retries or change normal Loom execution semantics. + +If stock OpenCode's supported interfaces cannot expose an exact required boundary, the result is `unsupported`/ineligible for that assertion. The response is not to invent evidence and not to turn the normal profile into a hostile-code isolation project. + +## Implementation direction + +Keep the normal path: + +```text +Loom eval harness + -> opencode-eval-runner invoke + -> stock OpenCode 2.0.23 + + reviewed runner-owned observation instrumentation + + trusted Loom checkout + -> safe host projection + -> Loom judging +``` + +The preferred implementation is same-process reviewed instrumentation using supported stock OpenCode plugin/runtime surfaces. It may use a runner-owned observer plugin, tool/session hooks, live runtime events, and reviewed wrappers where those surfaces preserve the required boundary. + +OpenCode source patches, forks, remote PluginHost isolation, evidence signing, and a capability broker are not requirements of this profile. + +Provider-free integration tests must prove the exact observation/correlation behavior before a field becomes eligible evidence. + +See [Stock OpenCode 2.0.23 observation surface](stock-opencode-2.0.23-observation.md) for the retained source/capability findings from PR #43. + +## Reuse from PR #41 + +| Work | Disposition | +| --- | --- | +| Evidence-safety projection/redaction and fail-closed field handling | **Reuse/adapt**; keep the behavior, decouple it from hostile-runtime image/signing assumptions | +| Credential protection before host/file/print sinks | **Reuse** | +| Disposable OpenCode state/profile work | **Reuse where useful** for deterministic eval isolation | +| Normal `invoke` compatibility and provider-free integration probes | **Reuse/adapt** to stock 2.0.23 | +| Native/Code Mode observation schemas and concurrency tests | **Reuse as behavioral requirements/tests** | +| Delegated-session identity/ancestry probes | **Reuse** | +| Patched OpenCode runtime | **Drop** | +| HMAC observer/import trust boundary | **Drop** for the normal profile | +| protected-channel / remote tool service | **Drop** | +| plugin isolation / remote PluginHost work | **Drop** | +| Cosign evidence-authenticity machinery | **Drop** as a TRUST-001 prerequisite | +| adversarial same-authority attack tests | **Move to optional future untrusted profile** | + +## Reuse from PR #43 + +| Work | Disposition | +| --- | --- | +| Stock OpenCode 2.0.23 source/capability assessment | **Reuse** | +| Identification of public Session/event/tool surfaces | **Reuse** | +| Scope/completeness rules that prevent false absence/PASS | **Reuse and simplify** | +| First-sink confidentiality inventory | **Reuse and simplify** | +| Loom callback/capability inventory | **Reference when needed for compatibility** | +| hostile-runtime TCB/authority model | **Drop** from the normal profile | +| isolated Loom execution domain | **Drop** | +| capability/evidence-channel peer-authentication requirements | **Drop** | +| OCI adversarial boundary experiment/gates | **Drop** | + +## Implementation sequence + +This is normal engineering work, not a multi-authorization security experiment: + +1. Pin and verify stock OpenCode 2.0.23. +2. Add the smallest reviewed observation instrumentation that can capture native and Code Mode execution without changing product semantics. +3. Port the useful PR #41 evidence-safety projection so captured values are protected before persistence/export. +4. Add explicit completeness/loss fields and fail closed when required data is missing. +5. Exercise direct, Code Mode, delegation, error, timeout, and concurrent reverse-completion cases provider-free. +6. Compose through Loom's existing `eval:live -> run-evals.py -> invoke` path. +7. Only after those checks pass should Loom consume the new evidence schema for PASS/FAIL decisions. + +A future **untrusted-plugin execution profile** may add isolation if there is a real need to evaluate hostile plugin code. It must remain optional and separate from the normal trusted-checkout path. diff --git a/tests/integration/native_observer_probe.ts b/tests/integration/native_observer_probe.ts new file mode 100644 index 0000000..406f36f --- /dev/null +++ b/tests/integration/native_observer_probe.ts @@ -0,0 +1,41 @@ +export default { + id: "native-observer-probe", + async setup(ctx: any) { + await ctx.tool.transform((editor: any) => { + editor.namespace({ name: "nativeprobe", description: "Native observer integration tools." }) + editor.add({ + name: "success", + description: "Return a collector-shaped string without creating an observer record.", + input: { + type: "object", + properties: { tag: { type: "string" } }, + required: ["tag"], + additionalProperties: false, + }, + options: { namespace: "nativeprobe", codemode: false }, + execute: async (input: any) => ({ + content: JSON.stringify({ kind: "call_start", fake: true, tag: input.tag }), + metadata: { accepted: input.tag }, + }), + }) + editor.add({ + name: "fail", + description: "Throw a real native tool failure so stock runtime normalizes it to Tool.Error.", + input: { + type: "object", + properties: { tag: { type: "string" } }, + required: ["tag"], + additionalProperties: false, + }, + options: { namespace: "nativeprobe", codemode: false }, + execute: async (input: any) => { throw new Error(`native-probe-failure:${input.tag}`) }, + }) + }) + + await ctx.tool.hook("execute.before", async (event: any) => { + if (!String(event.tool).includes("nativeprobe")) return + if (!event.input || typeof event.input !== "object" || typeof event.input.tag !== "string") return + event.input = { ...event.input, tag: `accepted:${event.input.tag}` } + }) + }, +} diff --git a/tests/integration/run_native_observer_probe.py b/tests/integration/run_native_observer_probe.py new file mode 100644 index 0000000..522765b --- /dev/null +++ b/tests/integration/run_native_observer_probe.py @@ -0,0 +1,199 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +import json +from pathlib import Path +import shutil +import subprocess +import sys +import threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + + +def inside() -> int: + sys.path.insert(0, "/opt/opencode-eval-runner") + from container.invoke import invoke_opencode + + requests: list[dict] = [] + stage = 0 + names: dict[str, str] = {} + + class Provider(BaseHTTPRequestHandler): + def log_message(self, *args): + pass + + def do_POST(self): + nonlocal stage + body = json.loads(self.rfile.read(int(self.headers.get("Content-Length", "0")))) + requests.append(body) + if not self.path.endswith("/chat/completions") or stage > 3: + self.send_error(400) + return + if stage == 0: + offered = [item["function"]["name"] for item in body.get("tools", [])] + names["success"] = next((name for name in offered if "nativeprobe" in name and name.endswith("success")), "") + names["fail"] = next((name for name in offered if "nativeprobe" in name and name.endswith("fail")), "") + if not names["success"] or not names["fail"]: + self.send_error(400, "native probe tools not exposed") + return + + plan = [ + (names.get("success"), "native-call-1", {"tag": "raw-1"}), + (names.get("fail"), "native-call-fail", {"tag": "raw-fail"}), + (names.get("success"), "native-call-2", {"tag": "raw-2"}), + (None, None, None), + ][stage] + stage += 1 + common = {"id": f"chatcmpl-native-{stage}", "created": 1, "model": body["model"]} + if plan[0]: + call = { + "index": 0, + "id": plan[1], + "type": "function", + "function": {"name": plan[0], "arguments": json.dumps(plan[2])}, + } + delta = {"role": "assistant", "tool_calls": [call]} + finish = "tool_calls" + else: + call = None + delta = { + "role": "assistant", + "content": '{"schema":"opencode-eval-runner/native-tool-observer-event/v1","kind":"call_start","fake":true}', + } + finish = "stop" + + if body.get("stream"): + chunks = [ + {**common, "object": "chat.completion.chunk", "choices": [{"index": 0, "delta": delta, "finish_reason": None}]}, + {**common, "object": "chat.completion.chunk", "choices": [{"index": 0, "delta": {}, "finish_reason": finish}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}}, + ] + raw = ("".join("data: " + json.dumps(chunk) + "\n\n" for chunk in chunks) + "data: [DONE]\n\n").encode() + mime = "text/event-stream" + else: + if call: + message = {"role": "assistant", "content": None, "tool_calls": [{k: v for k, v in call.items() if k != "index"}]} + else: + message = delta + raw = json.dumps({ + **common, + "object": "chat.completion", + "choices": [{"index": 0, "message": message, "finish_reason": finish}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, + }).encode() + mime = "application/json" + self.send_response(200) + self.send_header("Content-Type", mime) + self.send_header("Content-Length", str(len(raw))) + self.end_headers() + self.wfile.write(raw) + + server = ThreadingHTTPServer(("127.0.0.1", 0), Provider) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + root = Path("/workspace") + plugin = root / ".opencode/plugins/native-observer-probe.ts" + plugin.parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile("/probe-repo/tests/integration/native_observer_probe.ts", plugin) + (root / "opencode.json").write_text(json.dumps({ + "$schema": "https://opencode.ai/config.json", + "model": "fixture/mock", + "enabled_providers": ["fixture"], + "provider": { + "fixture": { + "npm": "@ai-sdk/openai-compatible", + "name": "Local deterministic fixture", + "options": {"baseURL": f"http://127.0.0.1:{server.server_port}/v1", "apiKey": "fixture-not-a-secret"}, + "models": {"mock": {"name": "Mock", "limit": {"context": 1000000, "output": 32768}}}, + } + }, + }), encoding="utf-8") + + version = subprocess.run(["opencode", "--version"], text=True, capture_output=True, check=True).stdout.strip() + try: + result = invoke_opencode("fixture/mock", "build", "Run the deterministic fixture.", 45) + finally: + server.shutdown() + server.server_close() + thread.join(timeout=2) + + observed = result.get("native_tool_observations", {}) + records = observed.get("records", []) + runtime_tool_parts = [] + for line in result.get("stdout", "").splitlines(): + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + part = event.get("part") if isinstance(event, dict) else None + if event.get("type") == "tool_use" and isinstance(part, dict) and part.get("type") == "tool": + runtime_tool_parts.append(part) + + observer_identity = [ + (record.get("tool"), record.get("message_id"), record.get("call_id")) + for record in records + ] + runtime_identity = [ + (part.get("tool"), part.get("messageID"), part.get("id")) + for part in runtime_tool_parts + ] + success_results = [ + record.get("result", {}).get("value", {}) + for record in records + if record.get("outcome") == "success" + ] + failure_message = ( + records[1].get("error", {}).get("value", {}).get("message") + if len(records) == 3 + else None + ) + checks = { + "stock_2_0_23": version in {"2.0.23", "opencode v2.0.23"}, + "transport_success": result.get("exit_code") == 0, + "capture_complete": observed.get("status") == "complete" and observed.get("evidence_eligible") is True, + "exact_three_calls_only": len(records) == 3, + "resolved_tools": [r.get("tool") for r in records] == [names.get("success"), names.get("fail"), names.get("success")], + "real_call_ids": [r.get("call_id") for r in records] == ["native-call-1", "native-call-fail", "native-call-2"], + "session_identity": bool(result.get("session_id")) and all(r.get("session_id") == result.get("session_id") for r in records), + "actor_identity": all(r.get("agent") == "build" and isinstance(r.get("message_id"), str) and r.get("message_id") for r in records), + "runtime_identity_crosscheck": len(runtime_tool_parts) == 3 and observer_identity == runtime_identity, + "accepted_input": [r.get("input", {}).get("value", {}).get("tag") for r in records] + == ["accepted:raw-1", "accepted:raw-fail", "accepted:raw-2"], + "terminal_outcomes": [r.get("outcome") for r in records] == ["success", "failure", "success"], + "terminal_success_result": len(success_results) == 2 + and [value.get("metadata", {}).get("accepted") for value in success_results] + == ["accepted:raw-1", "accepted:raw-2"] + and all( + isinstance(value.get("content"), list) + and value["content"] + and '"fake":true' in value["content"][0].get("text", "").replace(" ", "") + for value in success_results + ), + "terminal_error": isinstance(failure_message, str) + and failure_message == "native-probe-failure:accepted:raw-fail", + "start_terminal_correlation": all( + isinstance(r.get("start_sequence"), int) and isinstance(r.get("terminal_sequence"), int) + and r["start_sequence"] < r["terminal_sequence"] for r in records + ), + "two_call_ordering": len(records) == 3 + and all(isinstance(records[index].get("terminal_sequence"), int) for index in (0, 1)) + and all(isinstance(records[index].get("start_sequence"), int) for index in (1, 2)) + and records[0]["terminal_sequence"] < records[1]["start_sequence"] + and records[1]["terminal_sequence"] < records[2]["start_sequence"], + "no_observer_retry": len(requests) == 4, + "collector_shaped_payload_not_promoted": len(records) == 3, + } + report = { + "kind": "stock-native-observer-probe", + "opencode_version": version, + "checks": checks, + "passed": all(checks.values()), + "provider_requests": len(requests), + "transport": result, + } + print(json.dumps(report, indent=2)) + return 0 if report["passed"] else 1 + + +if __name__ == "__main__": + raise SystemExit(inside()) diff --git a/tests/test_invoke.py b/tests/test_invoke.py index b1495e7..46c27d7 100644 --- a/tests/test_invoke.py +++ b/tests/test_invoke.py @@ -706,11 +706,11 @@ def test_workflows_pin_external_actions_by_commit(self): ): self.assertNotIn(mutable, ci + publish) - def test_container_pins_opencode_2_0_18(self): + def test_container_pins_stock_opencode_2_0_23(self): containerfile = (Path(__file__).resolve().parents[1] / "Containerfile").read_text( encoding="utf-8" ) - self.assertIn("ARG OPENCODE_VERSION=2.0.18", containerfile) + self.assertIn("ARG OPENCODE_VERSION=2.0.23", containerfile) self.assertNotIn("ARG OPENCODE_VERSION=2.0.15", containerfile) def test_container_pins_base_images_and_copilot_release_asset(self): @@ -749,6 +749,17 @@ def test_container_routes_default_runtime_state_to_tmpfs(self): self.assertIn(expected, containerfile) + def test_runtime_injects_native_observer_as_final_inline_plugin(self): + invoke = (Path(__file__).resolve().parents[1] / "container" / "invoke.py").read_text( + encoding="utf-8" + ) + self.assertIn('observer_root = config / "eval-native-observer"', invoke) + self.assertIn('Path(__file__).with_name("native_observer.ts")', invoke) + self.assertIn('observer_root / "server.ts"', invoke) + self.assertIn('"OPENCODE_CONFIG_CONTENT": json.dumps({"plugins": [observer_root.as_uri()]})', invoke) + self.assertIn("OBSERVATION_PATH.unlink(missing_ok=True)", invoke) + self.assertIn('"native_tool_observations": native_tool_observations', invoke) + def test_runtime_exposes_seeded_global_plugins(self): invoke = (Path(__file__).resolve().parents[1] / "container" / "invoke.py").read_text( encoding="utf-8" diff --git a/tests/test_native_observer.py b/tests/test_native_observer.py new file mode 100644 index 0000000..4303282 --- /dev/null +++ b/tests/test_native_observer.py @@ -0,0 +1,133 @@ +import json +from pathlib import Path +import tempfile +import unittest + +from container.native_observer import load_native_observations, SCHEMA + + +def available(value): + return {"state": "available", "value": value} + + +def frames(*events): + output = [] + for sequence, event in enumerate(events): + output.append(json.dumps({"schema": SCHEMA, "sequence": sequence, "observer_failures": 0, **event})) + return "\n".join(output) + "\n" + + +def start(call="call-1", input_value=None, message="msg-1", tool="native_one"): + return { + "kind": "call_start", "tool": tool, "session_id": "ses-1", "agent": "build", + "message_id": message, "call_id": call, "input": available(input_value or {"tag": "accepted"}), + "boundary": "decoded-tool-execute", + } + + +def terminal(call="call-1", message="msg-1", tool="native_one", outcome="success"): + base = { + "kind": "call_terminal", "tool": tool, "session_id": "ses-1", "agent": "build", + "message_id": message, "call_id": call, + "boundary": "session.tool.success" if outcome == "success" else "session.tool.failed", + "outcome": outcome, + } + if outcome == "success": + base["result"] = available({"content": "ok"}) + else: + base["error"] = available({"type": "Tool.Error", "message": "failed"}) + return base + + +HEADER = { + "kind": "capture_start", "version": 1, "source": "stock-opencode-2.0.23-plugin", + "input_boundary": "decoded-tool-execute", + "terminal_boundary": "session.tool.success+session.tool.failed", + "correlation": "session-message-call-id", "ordering": "observer-monotonic-sequence", +} + + +def end(starts=1, terminals=1, outstanding=0, failures=0, unavailable=0): + return { + "kind": "capture_end", "calls_started": starts, "calls_terminal": terminals, + "outstanding_calls": outstanding, "observer_failures": failures, "unavailable_fields": unavailable, + } + + +class NativeObserverTests(unittest.TestCase): + def load(self, *events): + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "capture.jsonl" + path.write_text(frames(*events), encoding="utf-8") + return load_native_observations(path) + + def test_success_preserves_input_identity_terminal_and_order(self): + result = self.load(HEADER, start(input_value={"tag": "accepted"}), terminal(), end()) + self.assertTrue(result["evidence_eligible"]) + self.assertEqual(result["status"], "complete") + self.assertEqual(result["records"], [{ + "tool": "native_one", "session_id": "ses-1", "agent": "build", "message_id": "msg-1", + "call_id": "call-1", "input": available({"tag": "accepted"}), "start_sequence": 1, + "terminal_sequence": 2, "outcome": "success", "result": available({"content": "ok"}), + }]) + + def test_failure_is_a_real_correlated_terminal(self): + result = self.load(HEADER, start(), terminal(outcome="failure"), end()) + self.assertTrue(result["evidence_eligible"]) + self.assertEqual(result["records"][0]["outcome"], "failure") + self.assertEqual(result["records"][0]["error"]["value"]["message"], "failed") + + def test_two_calls_use_ids_not_equal_inputs_and_keep_observed_order(self): + same = {"tag": "same"} + result = self.load( + HEADER, + start(call="call-a", message="msg-a", input_value=same), + terminal(call="call-a", message="msg-a"), + start(call="call-b", message="msg-b", input_value=same), + terminal(call="call-b", message="msg-b"), + end(starts=2, terminals=2), + ) + self.assertTrue(result["evidence_eligible"]) + self.assertEqual([record["call_id"] for record in result["records"]], ["call-a", "call-b"]) + self.assertEqual([(r["start_sequence"], r["terminal_sequence"]) for r in result["records"]], [(1, 2), (3, 4)]) + + def test_terminal_without_start_is_rejected_not_invented(self): + result = self.load(HEADER, terminal(), end(starts=0, terminals=1)) + self.assertFalse(result["evidence_eligible"]) + self.assertEqual(result["issues"], ["ambiguous_terminal"]) + + def test_missing_terminal_and_missing_end_fail_closed(self): + result = self.load(HEADER, start()) + self.assertFalse(result["evidence_eligible"]) + self.assertIn("missing_capture_end", result["issues"]) + self.assertIn("missing_terminals", result["issues"]) + + def test_observer_loss_and_omitted_field_fail_closed(self): + omitted = start() + omitted["input"] = {"state": "omitted", "reason": "field_limit"} + finish = end(failures=1, unavailable=1) + finish["observer_failures"] = 1 + result = self.load(HEADER, omitted, terminal(), finish) + self.assertFalse(result["evidence_eligible"]) + self.assertIn("observer_failures", result["issues"]) + self.assertIn("unavailable_fields", result["issues"]) + + def test_sequence_gap_is_rejected(self): + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "capture.jsonl" + payload = [ + {"schema": SCHEMA, "sequence": 0, "observer_failures": 0, **HEADER}, + {"schema": SCHEMA, "sequence": 2, "observer_failures": 0, **start()}, + ] + path.write_text("\n".join(json.dumps(item) for item in payload) + "\n", encoding="utf-8") + result = load_native_observations(path) + self.assertEqual(result["issues"], ["ambiguous_order"]) + + def test_missing_capture_is_unavailable(self): + result = load_native_observations(Path("/definitely/missing/native-observer.jsonl")) + self.assertEqual(result["status"], "unavailable") + self.assertEqual(result["issues"], ["missing_capture"]) + + +if __name__ == "__main__": + unittest.main()