diff --git a/README.md b/README.md index 22923fc..08b8629 100644 --- a/README.md +++ b/README.md @@ -352,10 +352,12 @@ This profile does **not** claim resistance to an evaluated plugin that deliberat See [Trusted-checkout runtime evidence](docs/trusted-checkout-evidence.md) and the [versioned runtime-evidence result contract](docs/runtime-evidence-contract.md). -OpenCode results now expose `opencode-eval-runner/runtime-evidence/v1` as the single authoritative runtime-evidence object. Native calls are observed through the stock-2.0.23 decoded-execution and Session terminal boundaries. Code Mode inner identity/input/ordering is observable, while exact final script-visible value/error remains explicitly `unsupported`. Existing `tools`, `actions`, `tool_result_evidence`, stdout/stderr, and model text remain convenience/diagnostic data only. +OpenCode results now expose `opencode-eval-runner/runtime-evidence/v1` as the single authoritative runtime-evidence object. Native calls are observed through the stock-2.0.23 decoded-execution and Session terminal boundaries. Code Mode inner identity/input/ordering is observable, while exact final script-visible value/error remains explicitly `unsupported`. Existing `tools`, `actions`, `tool_result_evidence`, stdout/stderr, and model text remain convenience/diagnostic data only; they do not expose runtime-evidence eligibility and are never substitutes for `runtime_evidence`. Overall evidence eligibility is separate from assertion eligibility: an unsupported Code Mode finality boundary does not invalidate an unrelated complete native assertion, and redacted/omitted fields only block assertions that require those exact values. +The `github-copilot-cli` transport has no OpenCode runtime observer. It still emits the canonical `runtime_evidence` object, but with status `unsupported`. + > Stock OpenCode 2.0.23 does not expose a supported boundary that proves the exact final value/error seen by a Code Mode script for each inner call. That assertion is reported as unsupported. ## Security boundary diff --git a/container/evidence_safety.py b/container/evidence_safety.py index 6a3ed4c..5089f3d 100644 --- a/container/evidence_safety.py +++ b/container/evidence_safety.py @@ -400,7 +400,6 @@ def summary(self) -> dict[str, Any]: return { "schema": SCHEMA, "inventory_complete": self.sanitizer.inventory_complete, - "evidence_eligible": self.sanitizer.inventory_complete and not losses, "fields": self.fields, "losses": losses, } diff --git a/container/invoke.py b/container/invoke.py index 6ba7203..95dadfe 100644 --- a/container/invoke.py +++ b/container/invoke.py @@ -167,16 +167,6 @@ def extract_actions(events: list[dict[str, Any]]) -> list[dict[str, Any]]: STDERR_CAPTURE_LIMIT = 20000 -def _tool_result_text(value: Any, limit: int) -> tuple[str, bool]: - text = value if isinstance(value, str) else json.dumps(value, ensure_ascii=False, sort_keys=True) - if len(text) <= limit: - return text, False - marker = "\n[... tool-result field truncated ...]\n" - retained = limit - len(marker) - head = retained // 2 - return text[:head] + marker + text[-(retained - head):], True - - def extract_tool_result_evidence( events: list[dict[str, Any]], sanitizer: Sanitizer | None = None, @@ -316,7 +306,6 @@ def drop_event_fields(sequence: int) -> None: evidence["events"] = recent evidence["safety"] = projection.summary() - evidence["evidence_eligible"] = evidence["safety"]["evidence_eligible"] while ( len(json.dumps(evidence, ensure_ascii=False, separators=(",", ":"))) @@ -328,7 +317,6 @@ def drop_event_fields(sequence: int) -> None: evidence["omitted_events"] += 1 projection.loss("size_limit") evidence["safety"] = projection.summary() - evidence["evidence_eligible"] = False return evidence diff --git a/container/runtime_evidence.py b/container/runtime_evidence.py index 805f37b..b2cb3a6 100644 --- a/container/runtime_evidence.py +++ b/container/runtime_evidence.py @@ -6,11 +6,12 @@ RUNTIME_EVIDENCE_SCHEMA = "opencode-eval-runner/runtime-evidence/v1" -STATUSES = {"complete", "incomplete", "unsupported", "invalid"} -FIELD_STATES = {"available", "redacted", "omitted", "unsupported"} -MODES = {"native", "code_mode"} -OUTCOMES = {"success", "error", "missing"} -PROCESS_STATES = {"completed", "timeout", "interrupted", "unsupported"} +STATUSES = frozenset({"complete", "incomplete", "unsupported", "invalid"}) +FIELD_STATES = frozenset({"available", "redacted", "omitted", "unsupported"}) +MODES = frozenset({"native", "code_mode"}) +OUTCOMES = frozenset({"success", "error", "missing"}) +PROCESS_STATES = frozenset({"completed", "timeout", "interrupted", "unsupported"}) +EVENT_KINDS = frozenset({"start", "terminal"}) BOUNDARY_NATIVE = "native" BOUNDARY_CODE_MODE_EXECUTION = "code_mode_execution" @@ -136,11 +137,6 @@ def _codes(raw: Any, where: str) -> list[str]: return values -STATUSES = ("complete", "incomplete", "unsupported", "invalid") -FIELD_STATES = frozenset({"available", "redacted", "omitted", "unsupported"}) -PROCESS_STATES = frozenset({"completed", "timeout", "interrupted", "unsupported"}) -EVENT_KINDS = frozenset({"start", "terminal"}) - _GLOBAL_INVALID = frozenset({ "invalid_accounting_input", "malformed_observation", @@ -178,7 +174,6 @@ def _new_boundary(*, declared_supported: bool = False, declared_unsupported: boo "terminals": 0, "missing_terminals": 0, "required_fields_omitted": 0, - "required_fields_truncated": 0, "required_fields_unsupported": 0, "issues": ["unsupported_boundary"] if declared_unsupported else [], } @@ -221,8 +216,8 @@ def account_runtime_evidence( Each observation must contain kind, sequence, invocation_id, boundary, and required_fields. required_fields maps semantic - field names to available, redacted, omitted, truncated, or - unsupported. Adapters decide which fields are required; this layer only + field names to available, redacted, omitted, or unsupported. + Adapters decide which fields are required; this layer only accounts for their explicit states. observation_closed is an ordinary correctness signal from the capture @@ -242,7 +237,6 @@ def account_runtime_evidence( "ambiguous_invocations": 0, "duplicate_sequences": 0, "required_fields_omitted": 0, - "required_fields_truncated": 0, "required_fields_unsupported": 0, "process_state": process_state if process_state in PROCESS_STATES else "invalid", "supported_boundaries": [], @@ -416,8 +410,6 @@ def account_runtime_evidence( _add_issue(issues, "missing_terminal") if coverage["required_fields_omitted"]: _add_issue(issues, "required_field_omitted") - if coverage["required_fields_truncated"]: - _add_issue(issues, "required_field_truncated") if coverage["required_fields_unsupported"]: _add_issue(issues, "required_field_unsupported") if unsupported: diff --git a/docs/runtime-evidence-contract.md b/docs/runtime-evidence-contract.md index b5c18dc..19c800b 100644 --- a/docs/runtime-evidence-contract.md +++ b/docs/runtime-evidence-contract.md @@ -4,7 +4,7 @@ Public schema: \`opencode-eval-runner/runtime-evidence/v1\`. \`runtime_evidence\` is the only authoritative runtime-evidence object in the runner result. Raw observer records are internal adapter input and are not serialized as a competing public result. -The existing \`tools\`, \`actions\`, \`tool_result_evidence\`, stdout/stderr, Session/model text, and workspace files are convenience or diagnostic data only. +The existing \`tools\`, \`actions\`, \`tool_result_evidence\`, stdout/stderr, Session/model text, and workspace files are convenience or diagnostic data only. They do not publish runtime-evidence eligibility and must not be used as substitutes even when their contents look like runtime-evidence JSON. ## Top-level meaning @@ -150,6 +150,12 @@ Product outcome remains independent: - a successful product result can have incomplete evidence; - \`exit_code\` and timeout status do not become evidence eligibility. +## Unsupported areas + +- \`code_mode_finality\`: stock OpenCode 2.0.23 does not expose the exact final value/error seen by each Code Mode script call. +- \`github-copilot-cli\`: this transport has no OpenCode runtime observer, so its \`runtime_evidence\` object is explicitly \`unsupported\`. +- hostile evaluated plugins: same-process instrumentation is not protected from a plugin that deliberately compromises the trusted runtime. + ## Trust scope This is the normal trusted-checkout Loom eval profile. It uses stock OpenCode 2.0.23 and reviewed same-process instrumentation. diff --git a/docs/trusted-checkout-evidence.md b/docs/trusted-checkout-evidence.md index 31819f9..8a3464e 100644 --- a/docs/trusted-checkout-evidence.md +++ b/docs/trusted-checkout-evidence.md @@ -40,7 +40,7 @@ The public authority is only: \`opencode-eval-runner/runtime-evidence/v1\` -There is no public \`native_tool_observations\` or \`evidence_accounting\` authority. Existing \`tools\`, \`actions\`, \`tool_result_evidence\`, stdout/stderr, model text, and similar fields remain convenience/diagnostic data. +There is no public \`native_tool_observations\` or \`evidence_accounting\` authority. Existing \`tools\`, \`actions\`, \`tool_result_evidence\`, stdout/stderr, model text, and similar fields remain convenience/diagnostic data. They expose no runtime-evidence eligibility signal and are never promoted into \`runtime_evidence\`. ## Native boundary @@ -153,6 +153,12 @@ PR #45 carries provider-free tests for: The diagnostic Code Mode probe remains as the stock-runtime proof for the unsupported final boundary. +## Explicit unsupported areas + +- Exact Code Mode caller-final value/error is unsupported on stock OpenCode 2.0.23. +- GitHub Copilot CLI invocations expose a canonical \`runtime_evidence\` object with status \`unsupported\`; they do not have the OpenCode runtime observer. +- The trusted-checkout profile does not defend the observer from a deliberately hostile plugin sharing the OpenCode process. + ## Out of scope The integrated normal profile does not introduce: diff --git a/tests/integration/run_runtime_evidence_acceptance.py b/tests/integration/run_runtime_evidence_acceptance.py index 69f4932..8266587 100644 --- a/tests/integration/run_runtime_evidence_acceptance.py +++ b/tests/integration/run_runtime_evidence_acceptance.py @@ -376,6 +376,21 @@ def scenario_outcomes(result: dict[str, Any]) -> dict[str, Any]: } +def validate_authority_surfaces(result: dict[str, Any]) -> dict[str, bool]: + diagnostic = result.get("tool_result_evidence") + safety = diagnostic.get("safety") if isinstance(diagnostic, dict) else None + return { + "runtime_evidence_is_only_eligibility_surface": ( + isinstance(result.get("runtime_evidence"), dict) + and "evidence_eligible" not in result + and (not isinstance(diagnostic, dict) or "evidence_eligible" not in diagnostic) + and (not isinstance(safety, dict) or "evidence_eligible" not in safety) + and "native_tool_observations" not in result + and "evidence_accounting" not in result + ), + } + + def validate_contract(evidence: dict[str, Any]) -> dict[str, bool]: try: validate_runtime_evidence(evidence) @@ -671,6 +686,7 @@ def run_scenario(image: str, scenario: str, output: Path) -> dict[str, Any]: result = json.loads(result_path.read_text(encoding="utf-8")) if result_path.is_file() else {} oracle = read_oracle(workspace / "runtime-evidence-oracle.jsonl") checks = VALIDATORS[scenario](result, oracle, provider.requests) + checks.update(validate_authority_surfaces(result)) report = { "scenario": scenario, "runner_exit_code": proc.returncode, diff --git a/tests/test_evidence_safety.py b/tests/test_evidence_safety.py index c386af6..27b5bc0 100644 --- a/tests/test_evidence_safety.py +++ b/tests/test_evidence_safety.py @@ -155,7 +155,7 @@ def test_malformed_selected_inventory_fails_closed(self): disposition(projection.summary(), "output", 0)["reason"], "credential_inventory_unavailable", ) - self.assertFalse(projection.summary()["evidence_eligible"]) + self.assertNotIn("evidence_eligible", projection.summary()) def test_success_failure_and_running_events_are_projected_without_invention(self): secret = "fixture-secret" @@ -213,10 +213,11 @@ def test_success_failure_and_running_events_are_projected_without_invention(self self.assertEqual(evidence["events"][2]["status"], "running") self.assertNotIn("output", evidence["events"][2]) self.assertNotIn("error", evidence["events"][2]) - self.assertFalse(evidence["evidence_eligible"]) + self.assertNotIn("evidence_eligible", evidence) + self.assertNotIn("evidence_eligible", evidence["safety"]) self.assertNotIn(secret, json.dumps(evidence)) - def test_oversize_runtime_evidence_field_is_explicitly_omitted(self): + def test_oversize_tool_result_field_is_explicitly_omitted(self): raw = "safe-prefix-" + ("x" * 7000) events = [{ "type": "tool_use", @@ -240,7 +241,8 @@ def test_oversize_runtime_evidence_field_is_explicitly_omitted(self): disposition(evidence["safety"], "output", 0)["reason"], "size_limit", ) - self.assertFalse(evidence["evidence_eligible"]) + self.assertNotIn("evidence_eligible", evidence) + self.assertNotIn("evidence_eligible", evidence["safety"]) self.assertNotIn(raw[:64], json.dumps(evidence)) def test_actual_invoke_path_never_exports_secret_and_keeps_product_failure(self): @@ -300,7 +302,8 @@ class Result: parsed["tool_result_evidence"]["events"][0]["output"], {"answer": REDACTED}, ) - self.assertFalse(parsed["tool_result_evidence"]["evidence_eligible"]) + self.assertNotIn("evidence_eligible", parsed["tool_result_evidence"]) + self.assertNotIn("evidence_eligible", parsed["tool_result_evidence"]["safety"]) def test_timeout_path_sanitizes_before_clipping_and_keeps_timeout_status(self): secret = "TIMEOUT-SECRET" @@ -349,7 +352,8 @@ def fail(command, cwd, env, timeout): self.assertNotIn(secret, wire) self.assertEqual(result["exit_code"], 124) self.assertTrue(result["timed_out"]) - self.assertFalse(result["tool_result_evidence"]["evidence_eligible"]) + self.assertNotIn("evidence_eligible", result["tool_result_evidence"]) + self.assertNotIn("evidence_eligible", result["tool_result_evidence"]["safety"]) if __name__ == "__main__": diff --git a/tests/test_runtime_evidence_acceptance.py b/tests/test_runtime_evidence_acceptance.py index 8341428..44de4e7 100644 --- a/tests/test_runtime_evidence_acceptance.py +++ b/tests/test_runtime_evidence_acceptance.py @@ -123,6 +123,21 @@ def test_collector_payload_is_not_authoritative_observation(self): self.assertTrue(checks["tool_payload_not_promoted"]) self.assertTrue(checks["model_payload_not_promoted"]) + def test_convenience_surfaces_do_not_publish_runtime_eligibility(self): + evidence = build_runtime_evidence(native_capture()) + result = { + "runtime_evidence": evidence, + "tools": ["demo"], + "actions": [{"tool": "demo", "args": {}}], + "stdout": '{"evidence_eligible":true}', + "tool_result_evidence": { + "schema": "opencode-eval-runner/tool-results/v1", + "safety": {"schema": "opencode-eval-runner/evidence-safety/v1"}, + }, + } + checks = A.validate_authority_surfaces(result) + self.assertTrue(all(checks.values()), checks) + if __name__ == "__main__": unittest.main()