Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion container/native_observer.py
Original file line number Diff line number Diff line change
Expand Up @@ -132,7 +132,7 @@ def load_runtime_observations(path: Path = OBSERVATION_PATH) -> dict[str, Any]:
"kind": "capture_start",
"version": 1,
"source": "stock-opencode-2.0.23-plugin",
"native_input_boundary": "decoded-tool-execute",
"native_input_boundary": "decoded-tool-execute+outer-execute-before",
"native_terminal_boundary": "session.tool.success+session.tool.failed",
"code_input_boundary": "decoded-code-tool-handler",
"code_terminal_boundary": "tool-handler-return+tool-handler-throw",
Expand Down
60 changes: 44 additions & 16 deletions container/native_observer.ts
Original file line number Diff line number Diff line change
Expand Up @@ -258,7 +258,7 @@ export default {
kind: "capture_start",
version: 1,
source: "stock-opencode-2.0.23-plugin",
native_input_boundary: "decoded-tool-execute",
native_input_boundary: "decoded-tool-execute+outer-execute-before",
native_terminal_boundary: "session.tool.success+session.tool.failed",
code_input_boundary: "decoded-code-tool-handler",
code_terminal_boundary: "tool-handler-return+tool-handler-throw",
Expand Down Expand Up @@ -395,25 +395,53 @@ export default {
}
})

// Public hooks are used only to bind inner calls to the actual outer
// execute CallID. They are not Code Mode final-result evidence.
await ctx.tool.hook("execute.before", (event: any) => {
// The synthetic Code Mode execute tool is created inside Tool.snapshot,
// after registration transforms have run. Observe that real model-facing
// invocation at the stock execute.before runtime hook so it cannot vanish
// from the native boundary. This is an exact observed effective input, but
// unlike transformed registered tools it is before CodeMode.Input decode.
await ctx.tool.hook("execute.before", async (event: any) => {
if (event?.tool !== "execute") return
if (
typeof event.sessionID === "string" &&
typeof event.messageID === "string" &&
typeof event.id === "string"
typeof event.sessionID !== "string" ||
typeof event.agent !== "string" ||
typeof event.messageID !== "string" ||
typeof event.id !== "string"
) {
const key = identityKey({
sessionID: event.sessionID,
messageID: event.messageID,
callID: event.id,
})
activeOuter.set(
key,
nativeActive.get(key)?.invocationID ?? "native-outer-unobserved:" + randomUUID(),
)
callbackFailures += 1
return
}

const identity: Identity = {
invocationID: nativeInvocationID(),
tool: "execute",
sessionID: event.sessionID,
agent: event.agent,
messageID: event.messageID,
callID: event.id,
}
const key = identityKey(identity)
const existing = nativeActive.get(key)
if (existing) {
activeOuter.set(key, existing.invocationID)
return
}

nativeStarts += 1
nativeActive.set(key, identity)
activeOuter.set(key, identity.invocationID)
write({
kind: "native_start",
invocation_id: identity.invocationID,
tool: project(identity.tool),
session_id: project(identity.sessionID),
agent: project(identity.agent),
message_id: project(identity.messageID),
call_id: project(identity.callID),
parent_session_id: await parentSession(ctx, identity.sessionID),
input: project(event.input),
boundary: "tool-execute-before",
})
})
await ctx.tool.hook("execute.after", (event: any) => {
if (event?.tool !== "execute") return
Expand Down
65 changes: 65 additions & 0 deletions container/runtime_evidence.py
Original file line number Diff line number Diff line change
Expand Up @@ -551,6 +551,43 @@ def _parent(start: Mapping[str, Any]) -> dict[str, Any]:
return field_unavailable("omitted", "session_parent_unavailable")


def _available_string(raw: Any) -> str | None:
if (
isinstance(raw, Mapping)
and raw.get("state") == "available"
and set(raw) == {"state", "value"}
and type(raw.get("value")) is str
and raw["value"]
):
return raw["value"]
return None


def _code_parent_is_observed(
start: Mapping[str, Any],
starts: Mapping[str, Mapping[str, Any]],
) -> bool:
parent_id = start.get("parent_invocation_id")
if type(parent_id) is not str or not parent_id:
return False
outer = starts.get(parent_id)
if not isinstance(outer, Mapping) or outer.get("kind") != "native_start":
return False
if outer.get("boundary") != "tool-execute-before":
return False
outer_sequence = outer.get("sequence")
child_sequence = start.get("sequence")
if type(outer_sequence) is not int or type(child_sequence) is not int or outer_sequence >= child_sequence:
return False
outer_tool = _available_string(outer.get("tool"))
if outer_tool is not None and outer_tool != "execute":
return False
return all(
outer.get(name) == start.get(name)
for name in ("session_id", "agent", "message_id", "call_id")
)


def _capture_issue_kind(code: str) -> str:
if code in {
"malformed_capture",
Expand Down Expand Up @@ -635,6 +672,11 @@ def build_runtime_evidence(
for invocation_id, start in sorted(starts.items(), key=lambda item: item[1]["sequence"]):
mode = "native" if start.get("kind") == "native_start" else "code_mode"
boundary = BOUNDARY_NATIVE if mode == "native" else BOUNDARY_CODE_MODE_EXECUTION
if mode == "code_mode" and not _code_parent_is_observed(start, starts):
# A Code Mode parent is authoritative only when it resolves to the
# observed model-facing outer execute invocation with the same
# runtime identity. A dangling opaque token is invalid evidence.
accounting_events.append({})
terminal = terminals.get(invocation_id)
expected_terminal = "native_terminal" if mode == "native" else "code_terminal"
if terminal is not None and terminal.get("kind") != expected_terminal:
Expand Down Expand Up @@ -896,6 +938,7 @@ def validate_runtime_evidence(raw: Any) -> dict[str, Any]:
previous_start = -1
terminal_count = 0
reconstructed: list[dict[str, Any]] = []
validated_by_id: dict[str, Mapping[str, Any]] = {}

for index, observation in enumerate(evidence["observations"]):
where = f"runtime_evidence.observations[{index}]"
Expand All @@ -916,6 +959,26 @@ def validate_runtime_evidence(raw: Any) -> dict[str, Any]:
_string(parent["id"], f"{where}.parent.value.id")
if mode == "code_mode":
required["parent"] = parent_state
if status != "invalid":
_require(
parent_state == "available"
and isinstance(parent, dict)
and parent.get("kind") == "invocation",
f"{where}.parent must identify an observed outer invocation",
)
outer = validated_by_id.get(parent["id"])
_require(
isinstance(outer, Mapping) and outer.get("mode") == "native",
f"{where}.parent must reference an earlier native observation",
)
outer_tool_state, outer_tool = _field(outer.get("tool"), f"{where}.parent.outer.tool")
if outer_tool_state == "available":
_require(outer_tool == "execute", f"{where}.parent must reference outer execute")
for identity_name in ("actor", "session_id", "message_id", "call_id"):
_require(
outer.get(identity_name) == observation.get(identity_name),
f"{where}.parent outer identity does not match {identity_name}",
)

start_sequence = observation["start_sequence"]
_require(type(start_sequence) is int and start_sequence >= 0, f"{where}.start_sequence must be >= 0")
Expand All @@ -940,6 +1003,7 @@ def validate_runtime_evidence(raw: Any) -> dict[str, Any]:
)
if outcome == "missing":
_require(terminal_state == result_state == error_state == "omitted", f"{where} missing terminal must be explicit")
validated_by_id[invocation_id] = observation
continue
_require(terminal_state == "available" and terminal_sequence > start_sequence, f"{where}.terminal_sequence must follow start")
_require(terminal_sequence not in seen_sequences, f"{where}.terminal_sequence must be unique")
Expand All @@ -964,6 +1028,7 @@ def validate_runtime_evidence(raw: Any) -> dict[str, Any]:
"boundary": boundary,
"required_fields": terminal_required,
})
validated_by_id[invocation_id] = observation

if aggregate_available:
_require(starts == len(evidence["observations"]), "coverage starts does not match observations")
Expand Down
4 changes: 3 additions & 1 deletion docs/code-mode-observer-experiment.md
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,7 @@ Stock OpenCode 2.0.23 supports a useful partial observation path:
| unique inner invocation identity | runner-owned `tool.transform` wrapper allocates an ID when the decoded leaf handler is actually entered | **supported** |
| actual selected tool | wrapper is attached to the effective registered tool | **supported** |
| executable input | wrapper runs after core input decoding and receives the value passed to the leaf handler | **supported** |
| real outer `execute` binding | the real `Tool.Context` carries Session/message/outer CallID into each inner leaf | **supported** |
| real outer `execute` binding | the real `Tool.Context` carries Session/message/outer CallID into each inner leaf; production evidence also records the outer call at `execute.before` | **supported** |
| start / handler-terminal ordering | runner observer sequence around the transformed leaf handler | **supported** |
| exact final value delivered to the Code Mode script | no supported public stock boundary exposes it with unique inner identity | **unsupported** |
| exact final error seen by the script catch path | no supported public stock boundary exposes it with unique inner identity | **unsupported** |
Expand Down Expand Up @@ -56,6 +56,8 @@ per-call ID at a real execution boundary and observe:
- decoded/executable input;
- the real Session ID, agent, message ID and outer `execute` CallID carried in
`Tool.Context`;
- a parent ID that resolves to the separately observed model-facing outer `execute`
invocation rather than to an internal correlation-only token;
- handler return or throw;
- start and handler completion order.

Expand Down
6 changes: 4 additions & 2 deletions docs/native-tool-observer.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,13 +8,15 @@ This observer is the production native/direct-call adapter for the trusted-check

The runner injects a reviewed Promise plugin through stock \`OPENCODE_CONFIG_CONTENT\` so its transform is applied after project/global transforms.

For native/direct tools it observes:
For registered native/direct tools it observes:

1. decoded/executable input at the transformed \`tool.execute\` boundary;
2. Session-owned terminal events:
- \`session.tool.success\`
- \`session.tool.failed\`.

Stock Code Mode's model-facing `execute` tool is synthetic: OpenCode creates it inside `Tool.snapshot` after registration transforms have run. The observer records that outer invocation at the stock `execute.before` hook. Its input is the exact effective value seen by that hook, before `CodeMode.Input` decode. The same Session terminal events settle it.

The Session terminal is intentional. A rejected transformed handler can bypass \`tool.execute.after\` while stock OpenCode still settles the invocation through \`session.tool.failed\`.

## Identity and ordering
Expand All @@ -25,7 +27,7 @@ The adapter correlates using the real runtime identity tuple:

It never correlates by FIFO, input equality, tool name, or completion order.

The public invocation ID is opaque. Dynamic identity/input/result/error fields are sanitized before the internal capture file is written.
The public invocation ID is opaque. Code Mode inner records reference the invocation ID of the observed outer `execute` record; a dangling or identity-mismatched parent is invalid evidence. Dynamic identity/input/result/error fields are sanitized before the internal capture file is written.

The canonical \`runtime_evidence\` builder then validates:

Expand Down
5 changes: 3 additions & 2 deletions docs/runtime-evidence-contract.md
Original file line number Diff line number Diff line change
Expand Up @@ -68,7 +68,7 @@ The runner-owned stock OpenCode 2.0.23 observer records:
- terminal success/error and ordering;
- Session ancestry where applicable.

The start boundary is the decoded \`tool.execute\` wrapper.
The normal registered-tool start boundary is the decoded `tool.execute` wrapper. The synthetic Code Mode `execute` registration is created after transforms, so its start is observed at the stock `execute.before` hook instead. For that one tool, `input` is the exact effective hook input before `CodeMode.Input` decode.

The terminal boundary is Session-owned:

Expand All @@ -88,7 +88,7 @@ For Code Mode inner calls the runner can observe, on stock 2.0.23:
- decoded/executable input;
- Session/message/agent;
- the actual outer \`execute\` CallID;
- parent binding to that outer invocation;
- parent binding to the authoritative observed outer `execute` invocation;
- start and handler-terminal ordering;
- success-vs-error outcome at the handler boundary.

Expand Down Expand Up @@ -123,6 +123,7 @@ It accounts for:
- duplicate sequence IDs;
- terminal-without-start;
- identity changes between start and terminal;
- missing, dangling, or identity-mismatched Code Mode outer-parent observations;
- unsupported boundaries;
- assertion-scoped field availability.

Expand Down
6 changes: 5 additions & 1 deletion docs/stock-opencode-2.0.23-observation.md
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,8 @@ OpenCode remains stock and immutable. A missing observation boundary is reported
| Live runtime events | `ctx.event.subscribe()` | supported source; ordering/drain must be proven by integration test |
| Session creation / ancestry | `session.created` + Session API | supported |
| Agent for a step | Session step/message events | supported |
| Native tool call identity/input | Session tool input/called events | supported source |
| Native/direct decoded input | transformed registered `tool.execute` | supported |
| Synthetic outer Code Mode `execute` identity/effective input | `execute.before` + Session terminal | supported partial input boundary; pre-decode |
| Native terminal success/failure | Session tool success/failed events | supported source |
| Tool pre-execution hook | `ctx.tool.hook("execute.before")` | supported; occurs before tool decode/execution |
| Tool post-handler hook | `ctx.tool.hook("execute.after")` | supported; occurs after core tool execution but before Code Mode final conversion |
Expand All @@ -27,6 +28,8 @@ Stock Session events are the preferred source for native terminal facts because

The observer must bind call identity, Session, agent/message context, input and terminal result/error without reconstructing them from prose or matching by value.

The model-facing Code Mode `execute` tool is not in the transformed registry: stock OpenCode creates it later inside `Tool.snapshot`. Its real invocation is therefore recorded from `execute.before` and settled by the same Session terminal event. This prevents a Code Mode run from disappearing from the native boundary. The hook input is exact at that stock surface but is before `CodeMode.Input` decode.

## Code Mode

Code Mode executes inner tools through the normal tool registry, so same-process reviewed instrumentation can observe real inner execution without isolating Loom.
Expand All @@ -36,6 +39,7 @@ The bounded stock-2.0.23 experiment establishes a useful partial path:
- `ctx.tool.transform(...)` can wrap the actual effective leaf registration;
- core decodes input before entering that wrapper, so the wrapper sees executable input;
- the real outer `execute` `Tool.Context` reaches each inner handler, providing actual Session, agent, message and outer CallID;
- that same outer call is independently present as an authoritative native observation, and each inner parent ID must resolve to it;
- the wrapper can allocate a unique per-inner observation ID at handler entry, so identical concurrent calls and reverse completion do not require input/FIFO correlation.

However, this wrapper is not the final Code Mode caller boundary. After it returns, core can encode the result, run mutating `tool.execute.after` hooks, normalize content, and Code Mode can select its return representation and perform output validation plus a JSON stringify/parse round trip.
Expand Down
Loading
Loading