Skip to content
4 changes: 3 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -352,10 +352,12 @@ This profile does **not** claim resistance to an evaluated plugin that deliberat

See [Trusted-checkout runtime evidence](docs/trusted-checkout-evidence.md) and the [versioned runtime-evidence result contract](docs/runtime-evidence-contract.md).

OpenCode results now expose `opencode-eval-runner/runtime-evidence/v1` as the single authoritative runtime-evidence object. Native calls are observed through the stock-2.0.23 decoded-execution and Session terminal boundaries. Code Mode inner identity/input/ordering is observable, while exact final script-visible value/error remains explicitly `unsupported`. Existing `tools`, `actions`, `tool_result_evidence`, stdout/stderr, and model text remain convenience/diagnostic data only.
OpenCode results now expose `opencode-eval-runner/runtime-evidence/v1` as the single authoritative runtime-evidence object. Native calls are observed through the stock-2.0.23 decoded-execution and Session terminal boundaries. Code Mode inner identity/input/ordering is observable, while exact final script-visible value/error remains explicitly `unsupported`. Existing `tools`, `actions`, `tool_result_evidence`, stdout/stderr, and model text remain convenience/diagnostic data only; they do not expose runtime-evidence eligibility and are never substitutes for `runtime_evidence`.

Overall evidence eligibility is separate from assertion eligibility: an unsupported Code Mode finality boundary does not invalidate an unrelated complete native assertion, and redacted/omitted fields only block assertions that require those exact values.

The `github-copilot-cli` transport has no OpenCode runtime observer. It still emits the canonical `runtime_evidence` object, but with status `unsupported`.

> Stock OpenCode 2.0.23 does not expose a supported boundary that proves the exact final value/error seen by a Code Mode script for each inner call. That assertion is reported as unsupported.

## Security boundary
Expand Down
1 change: 0 additions & 1 deletion container/evidence_safety.py
Original file line number Diff line number Diff line change
Expand Up @@ -400,7 +400,6 @@ def summary(self) -> dict[str, Any]:
return {
"schema": SCHEMA,
"inventory_complete": self.sanitizer.inventory_complete,
"evidence_eligible": self.sanitizer.inventory_complete and not losses,
"fields": self.fields,
"losses": losses,
}
12 changes: 0 additions & 12 deletions container/invoke.py
Original file line number Diff line number Diff line change
Expand Up @@ -167,16 +167,6 @@ def extract_actions(events: list[dict[str, Any]]) -> list[dict[str, Any]]:
STDERR_CAPTURE_LIMIT = 20000


def _tool_result_text(value: Any, limit: int) -> tuple[str, bool]:
text = value if isinstance(value, str) else json.dumps(value, ensure_ascii=False, sort_keys=True)
if len(text) <= limit:
return text, False
marker = "\n[... tool-result field truncated ...]\n"
retained = limit - len(marker)
head = retained // 2
return text[:head] + marker + text[-(retained - head):], True


def extract_tool_result_evidence(
events: list[dict[str, Any]],
sanitizer: Sanitizer | None = None,
Expand Down Expand Up @@ -316,7 +306,6 @@ def drop_event_fields(sequence: int) -> None:

evidence["events"] = recent
evidence["safety"] = projection.summary()
evidence["evidence_eligible"] = evidence["safety"]["evidence_eligible"]

while (
len(json.dumps(evidence, ensure_ascii=False, separators=(",", ":")))
Expand All @@ -328,7 +317,6 @@ def drop_event_fields(sequence: int) -> None:
evidence["omitted_events"] += 1
projection.loss("size_limit")
evidence["safety"] = projection.summary()
evidence["evidence_eligible"] = False

return evidence

Expand Down
24 changes: 8 additions & 16 deletions container/runtime_evidence.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,11 +6,12 @@

RUNTIME_EVIDENCE_SCHEMA = "opencode-eval-runner/runtime-evidence/v1"

STATUSES = {"complete", "incomplete", "unsupported", "invalid"}
FIELD_STATES = {"available", "redacted", "omitted", "unsupported"}
MODES = {"native", "code_mode"}
OUTCOMES = {"success", "error", "missing"}
PROCESS_STATES = {"completed", "timeout", "interrupted", "unsupported"}
STATUSES = frozenset({"complete", "incomplete", "unsupported", "invalid"})
FIELD_STATES = frozenset({"available", "redacted", "omitted", "unsupported"})
MODES = frozenset({"native", "code_mode"})
OUTCOMES = frozenset({"success", "error", "missing"})
PROCESS_STATES = frozenset({"completed", "timeout", "interrupted", "unsupported"})
EVENT_KINDS = frozenset({"start", "terminal"})

BOUNDARY_NATIVE = "native"
BOUNDARY_CODE_MODE_EXECUTION = "code_mode_execution"
Expand Down Expand Up @@ -136,11 +137,6 @@ def _codes(raw: Any, where: str) -> list[str]:
return values


STATUSES = ("complete", "incomplete", "unsupported", "invalid")
FIELD_STATES = frozenset({"available", "redacted", "omitted", "unsupported"})
PROCESS_STATES = frozenset({"completed", "timeout", "interrupted", "unsupported"})
EVENT_KINDS = frozenset({"start", "terminal"})

_GLOBAL_INVALID = frozenset({
"invalid_accounting_input",
"malformed_observation",
Expand Down Expand Up @@ -178,7 +174,6 @@ def _new_boundary(*, declared_supported: bool = False, declared_unsupported: boo
"terminals": 0,
"missing_terminals": 0,
"required_fields_omitted": 0,
"required_fields_truncated": 0,
"required_fields_unsupported": 0,
"issues": ["unsupported_boundary"] if declared_unsupported else [],
}
Expand Down Expand Up @@ -221,8 +216,8 @@ def account_runtime_evidence(

Each observation must contain kind, sequence, invocation_id,
boundary, and required_fields. required_fields maps semantic
field names to available, redacted, omitted, truncated, or
unsupported. Adapters decide which fields are required; this layer only
field names to available, redacted, omitted, or unsupported.
Adapters decide which fields are required; this layer only
accounts for their explicit states.

observation_closed is an ordinary correctness signal from the capture
Expand All @@ -242,7 +237,6 @@ def account_runtime_evidence(
"ambiguous_invocations": 0,
"duplicate_sequences": 0,
"required_fields_omitted": 0,
"required_fields_truncated": 0,
"required_fields_unsupported": 0,
"process_state": process_state if process_state in PROCESS_STATES else "invalid",
"supported_boundaries": [],
Expand Down Expand Up @@ -416,8 +410,6 @@ def account_runtime_evidence(
_add_issue(issues, "missing_terminal")
if coverage["required_fields_omitted"]:
_add_issue(issues, "required_field_omitted")
if coverage["required_fields_truncated"]:
_add_issue(issues, "required_field_truncated")
if coverage["required_fields_unsupported"]:
_add_issue(issues, "required_field_unsupported")
if unsupported:
Expand Down
8 changes: 7 additions & 1 deletion docs/runtime-evidence-contract.md
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@ Public schema: \`opencode-eval-runner/runtime-evidence/v1\`.

\`runtime_evidence\` is the only authoritative runtime-evidence object in the runner result. Raw observer records are internal adapter input and are not serialized as a competing public result.

The existing \`tools\`, \`actions\`, \`tool_result_evidence\`, stdout/stderr, Session/model text, and workspace files are convenience or diagnostic data only.
The existing \`tools\`, \`actions\`, \`tool_result_evidence\`, stdout/stderr, Session/model text, and workspace files are convenience or diagnostic data only. They do not publish runtime-evidence eligibility and must not be used as substitutes even when their contents look like runtime-evidence JSON.

## Top-level meaning

Expand Down Expand Up @@ -150,6 +150,12 @@ Product outcome remains independent:
- a successful product result can have incomplete evidence;
- \`exit_code\` and timeout status do not become evidence eligibility.

## Unsupported areas

- \`code_mode_finality\`: stock OpenCode 2.0.23 does not expose the exact final value/error seen by each Code Mode script call.
- \`github-copilot-cli\`: this transport has no OpenCode runtime observer, so its \`runtime_evidence\` object is explicitly \`unsupported\`.
- hostile evaluated plugins: same-process instrumentation is not protected from a plugin that deliberately compromises the trusted runtime.

## Trust scope

This is the normal trusted-checkout Loom eval profile. It uses stock OpenCode 2.0.23 and reviewed same-process instrumentation.
Expand Down
8 changes: 7 additions & 1 deletion docs/trusted-checkout-evidence.md
Original file line number Diff line number Diff line change
Expand Up @@ -40,7 +40,7 @@ The public authority is only:

\`opencode-eval-runner/runtime-evidence/v1\`

There is no public \`native_tool_observations\` or \`evidence_accounting\` authority. Existing \`tools\`, \`actions\`, \`tool_result_evidence\`, stdout/stderr, model text, and similar fields remain convenience/diagnostic data.
There is no public \`native_tool_observations\` or \`evidence_accounting\` authority. Existing \`tools\`, \`actions\`, \`tool_result_evidence\`, stdout/stderr, model text, and similar fields remain convenience/diagnostic data. They expose no runtime-evidence eligibility signal and are never promoted into \`runtime_evidence\`.

## Native boundary

Expand Down Expand Up @@ -153,6 +153,12 @@ PR #45 carries provider-free tests for:

The diagnostic Code Mode probe remains as the stock-runtime proof for the unsupported final boundary.

## Explicit unsupported areas

- Exact Code Mode caller-final value/error is unsupported on stock OpenCode 2.0.23.
- GitHub Copilot CLI invocations expose a canonical \`runtime_evidence\` object with status \`unsupported\`; they do not have the OpenCode runtime observer.
- The trusted-checkout profile does not defend the observer from a deliberately hostile plugin sharing the OpenCode process.

## Out of scope

The integrated normal profile does not introduce:
Expand Down
16 changes: 16 additions & 0 deletions tests/integration/run_runtime_evidence_acceptance.py
Original file line number Diff line number Diff line change
Expand Up @@ -376,6 +376,21 @@ def scenario_outcomes(result: dict[str, Any]) -> dict[str, Any]:
}


def validate_authority_surfaces(result: dict[str, Any]) -> dict[str, bool]:
diagnostic = result.get("tool_result_evidence")
safety = diagnostic.get("safety") if isinstance(diagnostic, dict) else None
return {
"runtime_evidence_is_only_eligibility_surface": (
isinstance(result.get("runtime_evidence"), dict)
and "evidence_eligible" not in result
and (not isinstance(diagnostic, dict) or "evidence_eligible" not in diagnostic)
and (not isinstance(safety, dict) or "evidence_eligible" not in safety)
and "native_tool_observations" not in result
and "evidence_accounting" not in result
),
}


def validate_contract(evidence: dict[str, Any]) -> dict[str, bool]:
try:
validate_runtime_evidence(evidence)
Expand Down Expand Up @@ -671,6 +686,7 @@ def run_scenario(image: str, scenario: str, output: Path) -> dict[str, Any]:
result = json.loads(result_path.read_text(encoding="utf-8")) if result_path.is_file() else {}
oracle = read_oracle(workspace / "runtime-evidence-oracle.jsonl")
checks = VALIDATORS[scenario](result, oracle, provider.requests)
checks.update(validate_authority_surfaces(result))
report = {
"scenario": scenario,
"runner_exit_code": proc.returncode,
Expand Down
16 changes: 10 additions & 6 deletions tests/test_evidence_safety.py
Original file line number Diff line number Diff line change
Expand Up @@ -155,7 +155,7 @@ def test_malformed_selected_inventory_fails_closed(self):
disposition(projection.summary(), "output", 0)["reason"],
"credential_inventory_unavailable",
)
self.assertFalse(projection.summary()["evidence_eligible"])
self.assertNotIn("evidence_eligible", projection.summary())

def test_success_failure_and_running_events_are_projected_without_invention(self):
secret = "fixture-secret"
Expand Down Expand Up @@ -213,10 +213,11 @@ def test_success_failure_and_running_events_are_projected_without_invention(self
self.assertEqual(evidence["events"][2]["status"], "running")
self.assertNotIn("output", evidence["events"][2])
self.assertNotIn("error", evidence["events"][2])
self.assertFalse(evidence["evidence_eligible"])
self.assertNotIn("evidence_eligible", evidence)
self.assertNotIn("evidence_eligible", evidence["safety"])
self.assertNotIn(secret, json.dumps(evidence))

def test_oversize_runtime_evidence_field_is_explicitly_omitted(self):
def test_oversize_tool_result_field_is_explicitly_omitted(self):
raw = "safe-prefix-" + ("x" * 7000)
events = [{
"type": "tool_use",
Expand All @@ -240,7 +241,8 @@ def test_oversize_runtime_evidence_field_is_explicitly_omitted(self):
disposition(evidence["safety"], "output", 0)["reason"],
"size_limit",
)
self.assertFalse(evidence["evidence_eligible"])
self.assertNotIn("evidence_eligible", evidence)
self.assertNotIn("evidence_eligible", evidence["safety"])
self.assertNotIn(raw[:64], json.dumps(evidence))

def test_actual_invoke_path_never_exports_secret_and_keeps_product_failure(self):
Expand Down Expand Up @@ -300,7 +302,8 @@ class Result:
parsed["tool_result_evidence"]["events"][0]["output"],
{"answer": REDACTED},
)
self.assertFalse(parsed["tool_result_evidence"]["evidence_eligible"])
self.assertNotIn("evidence_eligible", parsed["tool_result_evidence"])
self.assertNotIn("evidence_eligible", parsed["tool_result_evidence"]["safety"])

def test_timeout_path_sanitizes_before_clipping_and_keeps_timeout_status(self):
secret = "TIMEOUT-SECRET"
Expand Down Expand Up @@ -349,7 +352,8 @@ def fail(command, cwd, env, timeout):
self.assertNotIn(secret, wire)
self.assertEqual(result["exit_code"], 124)
self.assertTrue(result["timed_out"])
self.assertFalse(result["tool_result_evidence"]["evidence_eligible"])
self.assertNotIn("evidence_eligible", result["tool_result_evidence"])
self.assertNotIn("evidence_eligible", result["tool_result_evidence"]["safety"])


if __name__ == "__main__":
Expand Down
15 changes: 15 additions & 0 deletions tests/test_runtime_evidence_acceptance.py
Original file line number Diff line number Diff line change
Expand Up @@ -123,6 +123,21 @@ def test_collector_payload_is_not_authoritative_observation(self):
self.assertTrue(checks["tool_payload_not_promoted"])
self.assertTrue(checks["model_payload_not_promoted"])

def test_convenience_surfaces_do_not_publish_runtime_eligibility(self):
evidence = build_runtime_evidence(native_capture())
result = {
"runtime_evidence": evidence,
"tools": ["demo"],
"actions": [{"tool": "demo", "args": {}}],
"stdout": '{"evidence_eligible":true}',
"tool_result_evidence": {
"schema": "opencode-eval-runner/tool-results/v1",
"safety": {"schema": "opencode-eval-runner/evidence-safety/v1"},
},
}
checks = A.validate_authority_surfaces(result)
self.assertTrue(all(checks.values()), checks)


if __name__ == "__main__":
unittest.main()
Loading