diff --git a/README.md b/README.md index 22923fc..08b8629 100644 --- a/README.md +++ b/README.md @@ -352,10 +352,12 @@ This profile does **not** claim resistance to an evaluated plugin that deliberat See [Trusted-checkout runtime evidence](docs/trusted-checkout-evidence.md) and the [versioned runtime-evidence result contract](docs/runtime-evidence-contract.md). -OpenCode results now expose `opencode-eval-runner/runtime-evidence/v1` as the single authoritative runtime-evidence object. Native calls are observed through the stock-2.0.23 decoded-execution and Session terminal boundaries. Code Mode inner identity/input/ordering is observable, while exact final script-visible value/error remains explicitly `unsupported`. Existing `tools`, `actions`, `tool_result_evidence`, stdout/stderr, and model text remain convenience/diagnostic data only. +OpenCode results now expose `opencode-eval-runner/runtime-evidence/v1` as the single authoritative runtime-evidence object. Native calls are observed through the stock-2.0.23 decoded-execution and Session terminal boundaries. Code Mode inner identity/input/ordering is observable, while exact final script-visible value/error remains explicitly `unsupported`. Existing `tools`, `actions`, `tool_result_evidence`, stdout/stderr, and model text remain convenience/diagnostic data only; they do not expose runtime-evidence eligibility and are never substitutes for `runtime_evidence`. Overall evidence eligibility is separate from assertion eligibility: an unsupported Code Mode finality boundary does not invalidate an unrelated complete native assertion, and redacted/omitted fields only block assertions that require those exact values. +The `github-copilot-cli` transport has no OpenCode runtime observer. It still emits the canonical `runtime_evidence` object, but with status `unsupported`. + > Stock OpenCode 2.0.23 does not expose a supported boundary that proves the exact final value/error seen by a Code Mode script for each inner call. That assertion is reported as unsupported. ## Security boundary diff --git a/container/evidence_safety.py b/container/evidence_safety.py index 6a3ed4c..5089f3d 100644 --- a/container/evidence_safety.py +++ b/container/evidence_safety.py @@ -400,7 +400,6 @@ def summary(self) -> dict[str, Any]: return { "schema": SCHEMA, "inventory_complete": self.sanitizer.inventory_complete, - "evidence_eligible": self.sanitizer.inventory_complete and not losses, "fields": self.fields, "losses": losses, } diff --git a/container/invoke.py b/container/invoke.py index 6ba7203..95dadfe 100644 --- a/container/invoke.py +++ b/container/invoke.py @@ -167,16 +167,6 @@ def extract_actions(events: list[dict[str, Any]]) -> list[dict[str, Any]]: STDERR_CAPTURE_LIMIT = 20000 -def _tool_result_text(value: Any, limit: int) -> tuple[str, bool]: - text = value if isinstance(value, str) else json.dumps(value, ensure_ascii=False, sort_keys=True) - if len(text) <= limit: - return text, False - marker = "\n[... tool-result field truncated ...]\n" - retained = limit - len(marker) - head = retained // 2 - return text[:head] + marker + text[-(retained - head):], True - - def extract_tool_result_evidence( events: list[dict[str, Any]], sanitizer: Sanitizer | None = None, @@ -316,7 +306,6 @@ def drop_event_fields(sequence: int) -> None: evidence["events"] = recent evidence["safety"] = projection.summary() - evidence["evidence_eligible"] = evidence["safety"]["evidence_eligible"] while ( len(json.dumps(evidence, ensure_ascii=False, separators=(",", ":"))) @@ -328,7 +317,6 @@ def drop_event_fields(sequence: int) -> None: evidence["omitted_events"] += 1 projection.loss("size_limit") evidence["safety"] = projection.summary() - evidence["evidence_eligible"] = False return evidence diff --git a/container/native_observer.py b/container/native_observer.py index 370373e..cce6b1a 100644 --- a/container/native_observer.py +++ b/container/native_observer.py @@ -132,7 +132,7 @@ def load_runtime_observations(path: Path = OBSERVATION_PATH) -> dict[str, Any]: "kind": "capture_start", "version": 1, "source": "stock-opencode-2.0.23-plugin", - "native_input_boundary": "decoded-tool-execute", + "native_input_boundary": "decoded-tool-execute+outer-execute-before", "native_terminal_boundary": "session.tool.success+session.tool.failed", "code_input_boundary": "decoded-code-tool-handler", "code_terminal_boundary": "tool-handler-return+tool-handler-throw", diff --git a/container/native_observer.ts b/container/native_observer.ts index f9be967..6e6a934 100644 --- a/container/native_observer.ts +++ b/container/native_observer.ts @@ -258,7 +258,7 @@ export default { kind: "capture_start", version: 1, source: "stock-opencode-2.0.23-plugin", - native_input_boundary: "decoded-tool-execute", + native_input_boundary: "decoded-tool-execute+outer-execute-before", native_terminal_boundary: "session.tool.success+session.tool.failed", code_input_boundary: "decoded-code-tool-handler", code_terminal_boundary: "tool-handler-return+tool-handler-throw", @@ -395,25 +395,53 @@ export default { } }) - // Public hooks are used only to bind inner calls to the actual outer - // execute CallID. They are not Code Mode final-result evidence. - await ctx.tool.hook("execute.before", (event: any) => { + // The synthetic Code Mode execute tool is created inside Tool.snapshot, + // after registration transforms have run. Observe that real model-facing + // invocation at the stock execute.before runtime hook so it cannot vanish + // from the native boundary. This is an exact observed effective input, but + // unlike transformed registered tools it is before CodeMode.Input decode. + await ctx.tool.hook("execute.before", async (event: any) => { if (event?.tool !== "execute") return if ( - typeof event.sessionID === "string" && - typeof event.messageID === "string" && - typeof event.id === "string" + typeof event.sessionID !== "string" || + typeof event.agent !== "string" || + typeof event.messageID !== "string" || + typeof event.id !== "string" ) { - const key = identityKey({ - sessionID: event.sessionID, - messageID: event.messageID, - callID: event.id, - }) - activeOuter.set( - key, - nativeActive.get(key)?.invocationID ?? "native-outer-unobserved:" + randomUUID(), - ) + callbackFailures += 1 + return + } + + const identity: Identity = { + invocationID: nativeInvocationID(), + tool: "execute", + sessionID: event.sessionID, + agent: event.agent, + messageID: event.messageID, + callID: event.id, + } + const key = identityKey(identity) + const existing = nativeActive.get(key) + if (existing) { + activeOuter.set(key, existing.invocationID) + return } + + nativeStarts += 1 + nativeActive.set(key, identity) + activeOuter.set(key, identity.invocationID) + write({ + kind: "native_start", + invocation_id: identity.invocationID, + tool: project(identity.tool), + session_id: project(identity.sessionID), + agent: project(identity.agent), + message_id: project(identity.messageID), + call_id: project(identity.callID), + parent_session_id: await parentSession(ctx, identity.sessionID), + input: project(event.input), + boundary: "tool-execute-before", + }) }) await ctx.tool.hook("execute.after", (event: any) => { if (event?.tool !== "execute") return diff --git a/container/runtime_evidence.py b/container/runtime_evidence.py index 805f37b..9352791 100644 --- a/container/runtime_evidence.py +++ b/container/runtime_evidence.py @@ -6,11 +6,12 @@ RUNTIME_EVIDENCE_SCHEMA = "opencode-eval-runner/runtime-evidence/v1" -STATUSES = {"complete", "incomplete", "unsupported", "invalid"} -FIELD_STATES = {"available", "redacted", "omitted", "unsupported"} -MODES = {"native", "code_mode"} -OUTCOMES = {"success", "error", "missing"} -PROCESS_STATES = {"completed", "timeout", "interrupted", "unsupported"} +STATUSES = frozenset({"complete", "incomplete", "unsupported", "invalid"}) +FIELD_STATES = frozenset({"available", "redacted", "omitted", "unsupported"}) +MODES = frozenset({"native", "code_mode"}) +OUTCOMES = frozenset({"success", "error", "missing"}) +PROCESS_STATES = frozenset({"completed", "timeout", "interrupted", "unsupported"}) +EVENT_KINDS = frozenset({"start", "terminal"}) BOUNDARY_NATIVE = "native" BOUNDARY_CODE_MODE_EXECUTION = "code_mode_execution" @@ -136,11 +137,6 @@ def _codes(raw: Any, where: str) -> list[str]: return values -STATUSES = ("complete", "incomplete", "unsupported", "invalid") -FIELD_STATES = frozenset({"available", "redacted", "omitted", "unsupported"}) -PROCESS_STATES = frozenset({"completed", "timeout", "interrupted", "unsupported"}) -EVENT_KINDS = frozenset({"start", "terminal"}) - _GLOBAL_INVALID = frozenset({ "invalid_accounting_input", "malformed_observation", @@ -178,7 +174,6 @@ def _new_boundary(*, declared_supported: bool = False, declared_unsupported: boo "terminals": 0, "missing_terminals": 0, "required_fields_omitted": 0, - "required_fields_truncated": 0, "required_fields_unsupported": 0, "issues": ["unsupported_boundary"] if declared_unsupported else [], } @@ -221,8 +216,8 @@ def account_runtime_evidence( Each observation must contain kind, sequence, invocation_id, boundary, and required_fields. required_fields maps semantic - field names to available, redacted, omitted, truncated, or - unsupported. Adapters decide which fields are required; this layer only + field names to available, redacted, omitted, or unsupported. + Adapters decide which fields are required; this layer only accounts for their explicit states. observation_closed is an ordinary correctness signal from the capture @@ -242,7 +237,6 @@ def account_runtime_evidence( "ambiguous_invocations": 0, "duplicate_sequences": 0, "required_fields_omitted": 0, - "required_fields_truncated": 0, "required_fields_unsupported": 0, "process_state": process_state if process_state in PROCESS_STATES else "invalid", "supported_boundaries": [], @@ -416,8 +410,6 @@ def account_runtime_evidence( _add_issue(issues, "missing_terminal") if coverage["required_fields_omitted"]: _add_issue(issues, "required_field_omitted") - if coverage["required_fields_truncated"]: - _add_issue(issues, "required_field_truncated") if coverage["required_fields_unsupported"]: _add_issue(issues, "required_field_unsupported") if unsupported: @@ -559,6 +551,43 @@ def _parent(start: Mapping[str, Any]) -> dict[str, Any]: return field_unavailable("omitted", "session_parent_unavailable") +def _available_string(raw: Any) -> str | None: + if ( + isinstance(raw, Mapping) + and raw.get("state") == "available" + and set(raw) == {"state", "value"} + and type(raw.get("value")) is str + and raw["value"] + ): + return raw["value"] + return None + + +def _code_parent_is_observed( + start: Mapping[str, Any], + starts: Mapping[str, Mapping[str, Any]], +) -> bool: + parent_id = start.get("parent_invocation_id") + if type(parent_id) is not str or not parent_id: + return False + outer = starts.get(parent_id) + if not isinstance(outer, Mapping) or outer.get("kind") != "native_start": + return False + if outer.get("boundary") != "tool-execute-before": + return False + outer_sequence = outer.get("sequence") + child_sequence = start.get("sequence") + if type(outer_sequence) is not int or type(child_sequence) is not int or outer_sequence >= child_sequence: + return False + outer_tool = _available_string(outer.get("tool")) + if outer_tool is not None and outer_tool != "execute": + return False + return all( + outer.get(name) == start.get(name) + for name in ("session_id", "agent", "message_id", "call_id") + ) + + def _capture_issue_kind(code: str) -> str: if code in { "malformed_capture", @@ -643,6 +672,11 @@ def build_runtime_evidence( for invocation_id, start in sorted(starts.items(), key=lambda item: item[1]["sequence"]): mode = "native" if start.get("kind") == "native_start" else "code_mode" boundary = BOUNDARY_NATIVE if mode == "native" else BOUNDARY_CODE_MODE_EXECUTION + if mode == "code_mode" and not _code_parent_is_observed(start, starts): + # A Code Mode parent is authoritative only when it resolves to the + # observed model-facing outer execute invocation with the same + # runtime identity. A dangling opaque token is invalid evidence. + accounting_events.append({}) terminal = terminals.get(invocation_id) expected_terminal = "native_terminal" if mode == "native" else "code_terminal" if terminal is not None and terminal.get("kind") != expected_terminal: @@ -904,6 +938,7 @@ def validate_runtime_evidence(raw: Any) -> dict[str, Any]: previous_start = -1 terminal_count = 0 reconstructed: list[dict[str, Any]] = [] + validated_by_id: dict[str, Mapping[str, Any]] = {} for index, observation in enumerate(evidence["observations"]): where = f"runtime_evidence.observations[{index}]" @@ -924,6 +959,26 @@ def validate_runtime_evidence(raw: Any) -> dict[str, Any]: _string(parent["id"], f"{where}.parent.value.id") if mode == "code_mode": required["parent"] = parent_state + if status != "invalid": + _require( + parent_state == "available" + and isinstance(parent, dict) + and parent.get("kind") == "invocation", + f"{where}.parent must identify an observed outer invocation", + ) + outer = validated_by_id.get(parent["id"]) + _require( + isinstance(outer, Mapping) and outer.get("mode") == "native", + f"{where}.parent must reference an earlier native observation", + ) + outer_tool_state, outer_tool = _field(outer.get("tool"), f"{where}.parent.outer.tool") + if outer_tool_state == "available": + _require(outer_tool == "execute", f"{where}.parent must reference outer execute") + for identity_name in ("actor", "session_id", "message_id", "call_id"): + _require( + outer.get(identity_name) == observation.get(identity_name), + f"{where}.parent outer identity does not match {identity_name}", + ) start_sequence = observation["start_sequence"] _require(type(start_sequence) is int and start_sequence >= 0, f"{where}.start_sequence must be >= 0") @@ -948,6 +1003,7 @@ def validate_runtime_evidence(raw: Any) -> dict[str, Any]: ) if outcome == "missing": _require(terminal_state == result_state == error_state == "omitted", f"{where} missing terminal must be explicit") + validated_by_id[invocation_id] = observation continue _require(terminal_state == "available" and terminal_sequence > start_sequence, f"{where}.terminal_sequence must follow start") _require(terminal_sequence not in seen_sequences, f"{where}.terminal_sequence must be unique") @@ -972,6 +1028,7 @@ def validate_runtime_evidence(raw: Any) -> dict[str, Any]: "boundary": boundary, "required_fields": terminal_required, }) + validated_by_id[invocation_id] = observation if aggregate_available: _require(starts == len(evidence["observations"]), "coverage starts does not match observations") diff --git a/docs/code-mode-observer-experiment.md b/docs/code-mode-observer-experiment.md index 57b801d..9417db4 100644 --- a/docs/code-mode-observer-experiment.md +++ b/docs/code-mode-observer-experiment.md @@ -20,7 +20,7 @@ Stock OpenCode 2.0.23 supports a useful partial observation path: | unique inner invocation identity | runner-owned `tool.transform` wrapper allocates an ID when the decoded leaf handler is actually entered | **supported** | | actual selected tool | wrapper is attached to the effective registered tool | **supported** | | executable input | wrapper runs after core input decoding and receives the value passed to the leaf handler | **supported** | -| real outer `execute` binding | the real `Tool.Context` carries Session/message/outer CallID into each inner leaf | **supported** | +| real outer `execute` binding | the real `Tool.Context` carries Session/message/outer CallID into each inner leaf; production evidence also records the outer call at `execute.before` | **supported** | | start / handler-terminal ordering | runner observer sequence around the transformed leaf handler | **supported** | | exact final value delivered to the Code Mode script | no supported public stock boundary exposes it with unique inner identity | **unsupported** | | exact final error seen by the script catch path | no supported public stock boundary exposes it with unique inner identity | **unsupported** | @@ -56,6 +56,8 @@ per-call ID at a real execution boundary and observe: - decoded/executable input; - the real Session ID, agent, message ID and outer `execute` CallID carried in `Tool.Context`; +- a parent ID that resolves to the separately observed model-facing outer `execute` + invocation rather than to an internal correlation-only token; - handler return or throw; - start and handler completion order. diff --git a/docs/native-tool-observer.md b/docs/native-tool-observer.md index 4601cc8..f58aafa 100644 --- a/docs/native-tool-observer.md +++ b/docs/native-tool-observer.md @@ -8,13 +8,15 @@ This observer is the production native/direct-call adapter for the trusted-check The runner injects a reviewed Promise plugin through stock \`OPENCODE_CONFIG_CONTENT\` so its transform is applied after project/global transforms. -For native/direct tools it observes: +For registered native/direct tools it observes: 1. decoded/executable input at the transformed \`tool.execute\` boundary; 2. Session-owned terminal events: - \`session.tool.success\` - \`session.tool.failed\`. +Stock Code Mode's model-facing `execute` tool is synthetic: OpenCode creates it inside `Tool.snapshot` after registration transforms have run. The observer records that outer invocation at the stock `execute.before` hook. Its input is the exact effective value seen by that hook, before `CodeMode.Input` decode. The same Session terminal events settle it. + The Session terminal is intentional. A rejected transformed handler can bypass \`tool.execute.after\` while stock OpenCode still settles the invocation through \`session.tool.failed\`. ## Identity and ordering @@ -25,7 +27,7 @@ The adapter correlates using the real runtime identity tuple: It never correlates by FIFO, input equality, tool name, or completion order. -The public invocation ID is opaque. Dynamic identity/input/result/error fields are sanitized before the internal capture file is written. +The public invocation ID is opaque. Code Mode inner records reference the invocation ID of the observed outer `execute` record; a dangling or identity-mismatched parent is invalid evidence. Dynamic identity/input/result/error fields are sanitized before the internal capture file is written. The canonical \`runtime_evidence\` builder then validates: diff --git a/docs/runtime-evidence-contract.md b/docs/runtime-evidence-contract.md index b5c18dc..b293269 100644 --- a/docs/runtime-evidence-contract.md +++ b/docs/runtime-evidence-contract.md @@ -4,7 +4,7 @@ Public schema: \`opencode-eval-runner/runtime-evidence/v1\`. \`runtime_evidence\` is the only authoritative runtime-evidence object in the runner result. Raw observer records are internal adapter input and are not serialized as a competing public result. -The existing \`tools\`, \`actions\`, \`tool_result_evidence\`, stdout/stderr, Session/model text, and workspace files are convenience or diagnostic data only. +The existing \`tools\`, \`actions\`, \`tool_result_evidence\`, stdout/stderr, Session/model text, and workspace files are convenience or diagnostic data only. They do not publish runtime-evidence eligibility and must not be used as substitutes even when their contents look like runtime-evidence JSON. ## Top-level meaning @@ -68,7 +68,7 @@ The runner-owned stock OpenCode 2.0.23 observer records: - terminal success/error and ordering; - Session ancestry where applicable. -The start boundary is the decoded \`tool.execute\` wrapper. +The normal registered-tool start boundary is the decoded `tool.execute` wrapper. The synthetic Code Mode `execute` registration is created after transforms, so its start is observed at the stock `execute.before` hook instead. For that one tool, `input` is the exact effective hook input before `CodeMode.Input` decode. The terminal boundary is Session-owned: @@ -88,7 +88,7 @@ For Code Mode inner calls the runner can observe, on stock 2.0.23: - decoded/executable input; - Session/message/agent; - the actual outer \`execute\` CallID; -- parent binding to that outer invocation; +- parent binding to the authoritative observed outer `execute` invocation; - start and handler-terminal ordering; - success-vs-error outcome at the handler boundary. @@ -123,6 +123,7 @@ It accounts for: - duplicate sequence IDs; - terminal-without-start; - identity changes between start and terminal; +- missing, dangling, or identity-mismatched Code Mode outer-parent observations; - unsupported boundaries; - assertion-scoped field availability. @@ -150,6 +151,12 @@ Product outcome remains independent: - a successful product result can have incomplete evidence; - \`exit_code\` and timeout status do not become evidence eligibility. +## Unsupported areas + +- \`code_mode_finality\`: stock OpenCode 2.0.23 does not expose the exact final value/error seen by each Code Mode script call. +- \`github-copilot-cli\`: this transport has no OpenCode runtime observer, so its \`runtime_evidence\` object is explicitly \`unsupported\`. +- hostile evaluated plugins: same-process instrumentation is not protected from a plugin that deliberately compromises the trusted runtime. + ## Trust scope This is the normal trusted-checkout Loom eval profile. It uses stock OpenCode 2.0.23 and reviewed same-process instrumentation. diff --git a/docs/stock-opencode-2.0.23-observation.md b/docs/stock-opencode-2.0.23-observation.md index 37df53c..05c1c0a 100644 --- a/docs/stock-opencode-2.0.23-observation.md +++ b/docs/stock-opencode-2.0.23-observation.md @@ -13,7 +13,8 @@ OpenCode remains stock and immutable. A missing observation boundary is reported | Live runtime events | `ctx.event.subscribe()` | supported source; ordering/drain must be proven by integration test | | Session creation / ancestry | `session.created` + Session API | supported | | Agent for a step | Session step/message events | supported | -| Native tool call identity/input | Session tool input/called events | supported source | +| Native/direct decoded input | transformed registered `tool.execute` | supported | +| Synthetic outer Code Mode `execute` identity/effective input | `execute.before` + Session terminal | supported partial input boundary; pre-decode | | Native terminal success/failure | Session tool success/failed events | supported source | | Tool pre-execution hook | `ctx.tool.hook("execute.before")` | supported; occurs before tool decode/execution | | Tool post-handler hook | `ctx.tool.hook("execute.after")` | supported; occurs after core tool execution but before Code Mode final conversion | @@ -27,6 +28,8 @@ Stock Session events are the preferred source for native terminal facts because The observer must bind call identity, Session, agent/message context, input and terminal result/error without reconstructing them from prose or matching by value. +The model-facing Code Mode `execute` tool is not in the transformed registry: stock OpenCode creates it later inside `Tool.snapshot`. Its real invocation is therefore recorded from `execute.before` and settled by the same Session terminal event. This prevents a Code Mode run from disappearing from the native boundary. The hook input is exact at that stock surface but is before `CodeMode.Input` decode. + ## Code Mode Code Mode executes inner tools through the normal tool registry, so same-process reviewed instrumentation can observe real inner execution without isolating Loom. @@ -36,6 +39,7 @@ The bounded stock-2.0.23 experiment establishes a useful partial path: - `ctx.tool.transform(...)` can wrap the actual effective leaf registration; - core decodes input before entering that wrapper, so the wrapper sees executable input; - the real outer `execute` `Tool.Context` reaches each inner handler, providing actual Session, agent, message and outer CallID; +- that same outer call is independently present as an authoritative native observation, and each inner parent ID must resolve to it; - the wrapper can allocate a unique per-inner observation ID at handler entry, so identical concurrent calls and reverse completion do not require input/FIFO correlation. However, this wrapper is not the final Code Mode caller boundary. After it returns, core can encode the result, run mutating `tool.execute.after` hooks, normalize content, and Code Mode can select its return representation and perform output validation plus a JSON stringify/parse round trip. diff --git a/docs/trusted-checkout-evidence.md b/docs/trusted-checkout-evidence.md index 31819f9..8a3464e 100644 --- a/docs/trusted-checkout-evidence.md +++ b/docs/trusted-checkout-evidence.md @@ -40,7 +40,7 @@ The public authority is only: \`opencode-eval-runner/runtime-evidence/v1\` -There is no public \`native_tool_observations\` or \`evidence_accounting\` authority. Existing \`tools\`, \`actions\`, \`tool_result_evidence\`, stdout/stderr, model text, and similar fields remain convenience/diagnostic data. +There is no public \`native_tool_observations\` or \`evidence_accounting\` authority. Existing \`tools\`, \`actions\`, \`tool_result_evidence\`, stdout/stderr, model text, and similar fields remain convenience/diagnostic data. They expose no runtime-evidence eligibility signal and are never promoted into \`runtime_evidence\`. ## Native boundary @@ -153,6 +153,12 @@ PR #45 carries provider-free tests for: The diagnostic Code Mode probe remains as the stock-runtime proof for the unsupported final boundary. +## Explicit unsupported areas + +- Exact Code Mode caller-final value/error is unsupported on stock OpenCode 2.0.23. +- GitHub Copilot CLI invocations expose a canonical \`runtime_evidence\` object with status \`unsupported\`; they do not have the OpenCode runtime observer. +- The trusted-checkout profile does not defend the observer from a deliberately hostile plugin sharing the OpenCode process. + ## Out of scope The integrated normal profile does not introduce: diff --git a/tests/integration/run_runtime_evidence_acceptance.py b/tests/integration/run_runtime_evidence_acceptance.py index 69f4932..88773b1 100644 --- a/tests/integration/run_runtime_evidence_acceptance.py +++ b/tests/integration/run_runtime_evidence_acceptance.py @@ -126,6 +126,25 @@ def matching(evidence: dict[str, Any], suffix: str) -> list[dict[str, Any]]: return [item for item in aggregate_invocations(evidence) if tool_matches(item, suffix)] +def outer_for_inner(evidence: dict[str, Any], item: dict[str, Any]) -> dict[str, Any]: + parent = unwrap(item.get("parent")) + if not isinstance(parent, dict) or parent.get("kind") != "invocation": + return {} + parent_id = parent.get("id") + if not isinstance(parent_id, str) or not parent_id: + return {} + return next( + ( + candidate + for candidate in aggregate_invocations(evidence) + if candidate.get("invocation_id") == parent_id + and candidate.get("mode") == "native" + and tool_name(candidate) == "execute" + ), + {}, + ) + + def sequence(item: dict[str, Any], key: str) -> int | None: value = unwrap(item.get(key)) return value if type(value) is int else None @@ -376,6 +395,22 @@ def scenario_outcomes(result: dict[str, Any]) -> dict[str, Any]: } +def validate_authority_surfaces(result: dict[str, Any]) -> dict[str, bool]: + diagnostic = result.get("tool_result_evidence") + safety = diagnostic.get("safety") if isinstance(diagnostic, dict) else None + return { + "runtime_evidence_is_only_eligibility_surface": ( + isinstance(result.get("runtime_evidence"), dict) + and "evidence_eligible" not in result + and (not isinstance(diagnostic, dict) or "evidence_eligible" not in diagnostic) + and (not isinstance(safety, dict) or "evidence_eligible" not in safety) + and "native_tool_observations" not in result + and "evidence_accounting" not in result + ), + } + + + def validate_contract(evidence: dict[str, Any]) -> dict[str, bool]: try: validate_runtime_evidence(evidence) @@ -454,6 +489,9 @@ def validate_code(result: dict[str, Any], suffix: str, expected: str, *, error: final_field = item.get("error") if error else item.get("result") other_field = item.get("result") if error else item.get("error") parent = unwrap(item.get("parent")) + outer = outer_for_inner(evidence, item) + native = boundaries.get(BOUNDARY_NATIVE) if isinstance(boundaries, dict) else None + native_starts = unwrap(native.get("starts")) if isinstance(native, dict) else None outcome = item.get("outcome") checks.update({ "product_success": result.get("exit_code") == 0 and result.get("text") == expected, @@ -470,7 +508,15 @@ def validate_code(result: dict[str, Any], suffix: str, expected: str, *, error: "inner_parent_bound_to_outer_invocation": isinstance(parent, dict) and parent.get("kind") == "invocation" and isinstance(parent.get("id"), str) - and bool(parent.get("id")), + and bool(parent.get("id")) + and outer.get("invocation_id") == parent.get("id"), + "outer_execute_observed_authoritatively": bool(outer) + and terminal_present(outer) + and unwrap(outer.get("call_id")) == unwrap(item.get("call_id")) + and unwrap(outer.get("session_id")) == unwrap(item.get("session_id")) + and unwrap(outer.get("message_id")) == unwrap(item.get("message_id")) + and unwrap(outer.get("actor")) == unwrap(item.get("actor")), + "native_absence_cannot_hide_outer_execute": type(native_starts) is int and native_starts >= 1, "inner_terminal_observed": terminal_present(item), "inner_outcome_observed": outcome == ("error" if error else "success"), "final_value_or_error_explicitly_unsupported": field_state(final_field) == "unsupported" @@ -488,6 +534,7 @@ def validate_concurrent(result: dict[str, Any]) -> dict[str, bool]: starts = [sequence(item, "start_sequence") for item in ordered] terminals = [sequence(item, "terminal_sequence") for item in ordered] parents = [json.dumps(unwrap(item.get("parent")), sort_keys=True) for item in ordered] + outers = [outer_for_inner(evidence, item) for item in ordered] checks.update({ "product_success": result.get("exit_code") == 0 and result.get("text") == "PRODUCT-CONCURRENT_REVERSE", "two_inner_observations": len(ordered) == 2, @@ -499,7 +546,14 @@ def validate_concurrent(result: dict[str, Any]) -> dict[str, bool]: and starts[0] < starts[1] < terminals[1] < terminals[0], "same_actual_outer_parent": len(parents) == 2 and parents[0] == parents[1] - and parents[0] not in {"null", "{}"}, + and parents[0] not in {"null", "{}"} + and len(outers) == 2 + and bool(outers[0]) + and outers[0].get("invocation_id") == outers[1].get("invocation_id") + and all( + unwrap(outer.get("call_id")) == unwrap(item.get("call_id")) + for outer, item in zip(outers, ordered) + ), "finality_remains_unsupported_for_both": len(ordered) == 2 and all(field_state(item.get("result")) == "unsupported" for item in ordered), "overall_evidence_stays_eligible": evidence.get("status") == "complete" @@ -671,6 +725,7 @@ def run_scenario(image: str, scenario: str, output: Path) -> dict[str, Any]: result = json.loads(result_path.read_text(encoding="utf-8")) if result_path.is_file() else {} oracle = read_oracle(workspace / "runtime-evidence-oracle.jsonl") checks = VALIDATORS[scenario](result, oracle, provider.requests) + checks.update(validate_authority_surfaces(result)) report = { "scenario": scenario, "runner_exit_code": proc.returncode, diff --git a/tests/test_evidence_safety.py b/tests/test_evidence_safety.py index c386af6..27b5bc0 100644 --- a/tests/test_evidence_safety.py +++ b/tests/test_evidence_safety.py @@ -155,7 +155,7 @@ def test_malformed_selected_inventory_fails_closed(self): disposition(projection.summary(), "output", 0)["reason"], "credential_inventory_unavailable", ) - self.assertFalse(projection.summary()["evidence_eligible"]) + self.assertNotIn("evidence_eligible", projection.summary()) def test_success_failure_and_running_events_are_projected_without_invention(self): secret = "fixture-secret" @@ -213,10 +213,11 @@ def test_success_failure_and_running_events_are_projected_without_invention(self self.assertEqual(evidence["events"][2]["status"], "running") self.assertNotIn("output", evidence["events"][2]) self.assertNotIn("error", evidence["events"][2]) - self.assertFalse(evidence["evidence_eligible"]) + self.assertNotIn("evidence_eligible", evidence) + self.assertNotIn("evidence_eligible", evidence["safety"]) self.assertNotIn(secret, json.dumps(evidence)) - def test_oversize_runtime_evidence_field_is_explicitly_omitted(self): + def test_oversize_tool_result_field_is_explicitly_omitted(self): raw = "safe-prefix-" + ("x" * 7000) events = [{ "type": "tool_use", @@ -240,7 +241,8 @@ def test_oversize_runtime_evidence_field_is_explicitly_omitted(self): disposition(evidence["safety"], "output", 0)["reason"], "size_limit", ) - self.assertFalse(evidence["evidence_eligible"]) + self.assertNotIn("evidence_eligible", evidence) + self.assertNotIn("evidence_eligible", evidence["safety"]) self.assertNotIn(raw[:64], json.dumps(evidence)) def test_actual_invoke_path_never_exports_secret_and_keeps_product_failure(self): @@ -300,7 +302,8 @@ class Result: parsed["tool_result_evidence"]["events"][0]["output"], {"answer": REDACTED}, ) - self.assertFalse(parsed["tool_result_evidence"]["evidence_eligible"]) + self.assertNotIn("evidence_eligible", parsed["tool_result_evidence"]) + self.assertNotIn("evidence_eligible", parsed["tool_result_evidence"]["safety"]) def test_timeout_path_sanitizes_before_clipping_and_keeps_timeout_status(self): secret = "TIMEOUT-SECRET" @@ -349,7 +352,8 @@ def fail(command, cwd, env, timeout): self.assertNotIn(secret, wire) self.assertEqual(result["exit_code"], 124) self.assertTrue(result["timed_out"]) - self.assertFalse(result["tool_result_evidence"]["evidence_eligible"]) + self.assertNotIn("evidence_eligible", result["tool_result_evidence"]) + self.assertNotIn("evidence_eligible", result["tool_result_evidence"]["safety"]) if __name__ == "__main__": diff --git a/tests/test_native_observer.py b/tests/test_native_observer.py index b8727c1..31fe99c 100644 --- a/tests/test_native_observer.py +++ b/tests/test_native_observer.py @@ -16,7 +16,7 @@ def available(value): "kind": "capture_start", "version": 1, "source": "stock-opencode-2.0.23-plugin", - "native_input_boundary": "decoded-tool-execute", + "native_input_boundary": "decoded-tool-execute+outer-execute-before", "native_terminal_boundary": "session.tool.success+session.tool.failed", "code_input_boundary": "decoded-code-tool-handler", "code_terminal_boundary": "tool-handler-return+tool-handler-throw", diff --git a/tests/test_runtime_evidence.py b/tests/test_runtime_evidence.py index 2768d84..8ee0847 100644 --- a/tests/test_runtime_evidence.py +++ b/tests/test_runtime_evidence.py @@ -34,28 +34,28 @@ def capture(*records, ended=True, failures=0, callbacks=0, issues=()): } -def native_start(iid="n1", seq=1, call="call-1", input_value=None): +def native_start(iid="n1", seq=1, call="call-1", input_value=None, tool="runtimeevidence_nativeSuccess"): return { "kind": "native_start", "sequence": seq, "invocation_id": iid, - "tool": available("runtimeevidence_nativeSuccess"), + "tool": available(tool), "session_id": available("ses-1"), "agent": available("general"), "message_id": available("msg-1"), "call_id": available(call), "parent_session_id": available(None), "input": available(input_value or {"value": "accepted"}), - "boundary": "decoded-tool-execute", + "boundary": "tool-execute-before" if tool == "execute" else "decoded-tool-execute", } -def native_terminal(iid="n1", seq=2, call="call-1", outcome="success", value=None): +def native_terminal(iid="n1", seq=2, call="call-1", outcome="success", value=None, tool="runtimeevidence_nativeSuccess"): item = { "kind": "native_terminal", "sequence": seq, "invocation_id": iid, - "tool": available("runtimeevidence_nativeSuccess"), + "tool": available(tool), "session_id": available("ses-1"), "agent": available("general"), "message_id": available("msg-1"), @@ -127,10 +127,10 @@ def test_native_complete_while_code_finality_remains_unsupported(self): def test_code_execution_facts_are_eligible_but_final_value_is_not(self): evidence = build_runtime_evidence(capture( - native_start(), + native_start(call="outer-call", input_value={"code": "return 1"}, tool="execute"), code_start(), code_terminal(), - native_terminal(seq=5), + native_terminal(seq=5, call="outer-call", tool="execute"), )) code = next(item for item in evidence["observations"] if item["mode"] == "code_mode") self.assertEqual(evidence["coverage"]["boundaries"][BOUNDARY_CODE_MODE_EXECUTION]["status"], "complete") @@ -236,12 +236,12 @@ def test_duplicate_sequence_is_invalid(self): def test_concurrent_identical_code_calls_correlate_by_identity_not_fifo(self): evidence = build_runtime_evidence(capture( - native_start(seq=1, call="outer-call"), + native_start(seq=1, call="outer-call", input_value={"code": "return 1"}, tool="execute"), code_start("a", 2), code_start("b", 3), code_terminal("b", 4), code_terminal("a", 5), - native_terminal(seq=6, call="outer-call"), + native_terminal(seq=6, call="outer-call", tool="execute"), )) code = [item for item in evidence["observations"] if item["mode"] == "code_mode"] self.assertEqual([item["invocation_id"] for item in code], ["a", "b"]) @@ -252,6 +252,23 @@ def test_concurrent_identical_code_calls_correlate_by_identity_not_fifo(self): ) self.assertEqual(assertion_status(evidence, [BOUNDARY_CODE_MODE_EXECUTION]), "complete") + def test_code_parent_must_resolve_to_observed_outer_execute(self): + evidence = build_runtime_evidence(capture( + code_start(seq=1), + code_terminal(seq=2), + )) + self.assertEqual(evidence["status"], "invalid") + self.assertFalse(evidence["evidence_eligible"]) + + evidence = build_runtime_evidence(capture( + native_start(seq=1, call="outer-call", tool="runtimeevidence_nativeSuccess"), + code_start(seq=2), + code_terminal(seq=3), + native_terminal(seq=4, call="outer-call"), + )) + self.assertEqual(evidence["status"], "invalid") + self.assertFalse(evidence["evidence_eligible"]) + def test_validator_rejects_forged_global_eligibility(self): evidence = build_runtime_evidence(capture(native_start(), native_terminal())) forged = copy.deepcopy(evidence) diff --git a/tests/test_runtime_evidence_acceptance.py b/tests/test_runtime_evidence_acceptance.py index 8341428..ab6438b 100644 --- a/tests/test_runtime_evidence_acceptance.py +++ b/tests/test_runtime_evidence_acceptance.py @@ -49,8 +49,19 @@ def native_capture(): def code_capture(): value = native_capture() + outer_start = { + **value["records"][0], + "tool": available("execute"), + "input": available({"code": "return 1"}), + "boundary": "tool-execute-before", + } + outer_terminal = { + **value["records"][1], + "tool": available("execute"), + "result": available({"content": "CODE-MODE-DONE"}), + } value["records"] = [ - value["records"][0], + outer_start, { "kind": "code_start", "sequence": 2, "invocation_id": "c1", "tool": available("runtimeevidence_innerEcho"), @@ -68,7 +79,7 @@ def code_capture(): "outcome": "success", "boundary": "tool-handler-return", "finality": {"state": "unsupported", "reason": CODE_MODE_FINALITY_REASON}, }, - {**value["records"][1], "sequence": 4}, + {**outer_terminal, "sequence": 4}, ] return value @@ -124,5 +135,20 @@ def test_collector_payload_is_not_authoritative_observation(self): self.assertTrue(checks["model_payload_not_promoted"]) + def test_convenience_surfaces_do_not_publish_runtime_eligibility(self): + evidence = build_runtime_evidence(native_capture()) + result = { + "runtime_evidence": evidence, + "tools": ["demo"], + "actions": [{"tool": "demo", "args": {}}], + "stdout": '{"evidence_eligible":true}', + "tool_result_evidence": { + "schema": "opencode-eval-runner/tool-results/v1", + "safety": {"schema": "opencode-eval-runner/evidence-safety/v1"}, + }, + } + checks = A.validate_authority_surfaces(result) + self.assertTrue(all(checks.values()), checks) + if __name__ == "__main__": unittest.main()