diff --git a/.github/workflows/evidence-safety.yml b/.github/workflows/evidence-safety.yml new file mode 100644 index 0000000..676024e --- /dev/null +++ b/.github/workflows/evidence-safety.yml @@ -0,0 +1,130 @@ +name: Evidence safety boundaries + +on: + pull_request: + types: [opened, synchronize, reopened] + paths: + - container/** + - runner/** + - bin/opencode-eval-runner + - evidence-safety/** + - tests/test_evidence_safety*.py + - tests/integration/run_evidence_safety_probe.py + - tests/integration/run_podman_preflight_probe.py + - .github/workflows/evidence-safety.yml + +permissions: + contents: read + +defaults: + run: + shell: bash + +jobs: + build: + if: github.event.pull_request.head.repo.full_name == github.repository + runs-on: ubuntu-latest + timeout-minutes: 10 + permissions: + contents: read + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + with: + ref: ${{ github.event.pull_request.head.sha }} + persist-credentials: false + - name: Test source and build a separate opt-in image + run: | + mkdir evidence-safety-results + python3 -m unittest discover -s tests -p 'test_*.py' 2>&1 | tee evidence-safety-results/tests.txt + python3 -m py_compile container/*.py runner/*.py tests/integration/run_evidence_safety_probe.py tests/integration/run_podman_preflight_probe.py + git rev-parse HEAD > evidence-safety-results/source-commit.txt + sha256sum container/__init__.py container/evidence_safety.py container/invoke.py runner/safe_invoke.py runner/cli.py bin/opencode-eval-runner > evidence-safety-results/code.sha256 + docker build -f evidence-safety/Containerfile \ + --build-arg RUNNER_REVISION="$(git rev-parse HEAD)" \ + --build-arg PACKAGE_INIT_SHA256="$(sha256sum container/__init__.py | cut -d' ' -f1)" \ + --build-arg SAFETY_MODULE_SHA256="$(sha256sum container/evidence_safety.py | cut -d' ' -f1)" \ + --build-arg INVOKE_SHA256="$(sha256sum container/invoke.py | cut -d' ' -f1)" \ + -t evidence-safety:test . + docker save -o image.tar evidence-safety:test + git archive --format=tar HEAD > evidence-safety-results/source.tar + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: safety-image-${{ github.run_id }}-${{ github.run_attempt }} + path: image.tar + compression-level: 0 + if-no-files-found: error + retention-days: 1 + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + if: always() + with: + name: safety-source-${{ github.run_id }}-${{ github.run_attempt }} + path: evidence-safety-results/ + if-no-files-found: error + retention-days: 14 + + publish: + needs: build + runs-on: ubuntu-latest + timeout-minutes: 10 + permissions: + contents: read + packages: write + outputs: + reference: ${{ steps.image.outputs.reference }} + steps: + - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 + with: + name: safety-image-${{ github.run_id }}-${{ github.run_attempt }} + path: image-data + - name: Publish image bytes without executing the workload + id: image + env: + REGISTRY_TOKEN: ${{ secrets.GITHUB_TOKEN }} + HEAD_SHA: ${{ github.event.pull_request.head.sha }} + run: | + [[ "$HEAD_SHA" =~ ^[0-9a-f]{40}$ ]] + mkdir publication + docker load --input image-data/image.tar + printf '%s' "$REGISTRY_TOKEN" | docker login ghcr.io --username "$GITHUB_ACTOR" --password-stdin + trap 'docker logout ghcr.io' EXIT + tag="ghcr.io/bateau84/opencode-eval-runner:evidence-safety-${HEAD_SHA:0:12}-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}" + docker tag evidence-safety:test "$tag" + docker push "$tag" + image="$(docker image inspect "$tag" --format '{{index .RepoDigests 0}}')" + [[ "$image" =~ ^ghcr\.io/bateau84/opencode-eval-runner@sha256:[0-9a-f]{64}$ ]] + printf '%s\n' "$image" > publication/image.txt + printf 'reference=%s\n' "$image" >> "$GITHUB_OUTPUT" + docker image inspect "$image" > publication/image-inspect.json + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: safety-publication-${{ github.run_id }}-${{ github.run_attempt }} + path: publication/ + if-no-files-found: error + retention-days: 14 + + verify: + needs: [build, publish] + runs-on: ubuntu-latest + timeout-minutes: 15 + permissions: + contents: read + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + with: + ref: ${{ github.event.pull_request.head.sha }} + persist-credentials: false + - name: Exercise actual pre-output boundaries without real inference + env: + IMAGE: ${{ needs.publish.outputs.reference }} + run: | + docker pull "$IMAGE" + python3 tests/integration/run_evidence_safety_probe.py --image "$IMAGE" --output evidence-safety-results + podman --version + python3 tests/integration/run_podman_preflight_probe.py --image "$IMAGE" --output evidence-safety-results + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + if: always() + with: + name: safety-connection-${{ github.event.pull_request.head.sha }}-${{ github.run_attempt }} + path: evidence-safety-results/ + if-no-files-found: error + retention-days: 14 diff --git a/.github/workflows/local-runtime.yml b/.github/workflows/local-runtime.yml new file mode 100644 index 0000000..d7e44e0 --- /dev/null +++ b/.github/workflows/local-runtime.yml @@ -0,0 +1,186 @@ +name: Normal-invoke runtime observation seam + +on: + pull_request: + types: [opened, synchronize, reopened] + paths: + - runtime-patches/** + - runner/cli.py + - runner/observer.py + - container/** + - bin/opencode-eval-runner + - tests/integration/run_capture_probe.py + - tests/integration/capture_probe.ts + - tests/test_workflow_boundaries.py + - .github/workflows/local-runtime.yml + +permissions: + contents: read + +defaults: + run: + shell: bash + +jobs: + build: + if: github.event.pull_request.head.repo.full_name == github.repository + runs-on: ubuntu-latest + timeout-minutes: 30 + permissions: + contents: read + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + with: + ref: ${{ github.event.pull_request.head.sha }} + persist-credentials: false + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + with: + repository: anomalyco/opencode + ref: cd9a14a6b688d4021bee381dfd39d2cef9c0f862 + path: .runtime-source + persist-credentials: false + - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 + with: + bun-version: 1.4.2 + - name: Apply exact downstream source patch + run: | + mkdir -p local-runtime-results + python3 -m py_compile runtime-patches/*.py + python3 runtime-patches/apply.py .runtime-source + git -C .runtime-source add -N packages/core/src/codemode/local-observation.ts packages/core/test/local-observation.test.ts + git -C .runtime-source diff --binary > local-runtime-results/runtime.patch + git rev-parse HEAD > local-runtime-results/runner-revision.txt + sha256sum runtime-patches/*.py runtime-patches/*.ts runtime-patches/Containerfile > local-runtime-results/patch-inputs.sha256 + - name: Install pinned runtime dependencies + working-directory: .runtime-source + run: bun install --frozen-lockfile + - name: Test normal host-semantics observation source + run: | + root="$(mktemp -d "$RUNNER_TEMP/local-runtime-test.XXXXXX")" + mkdir -p "$root"/{home,config,data,state,cache,run} + HOME="$root/home" XDG_CONFIG_HOME="$root/config" XDG_DATA_HOME="$root/data" \ + XDG_STATE_HOME="$root/state" XDG_CACHE_HOME="$root/cache" XDG_RUNTIME_DIR="$root/run" \ + OPENCODE_DISABLE_AUTOUPDATE=1 bun test --cwd .runtime-source/packages/core --timeout 20000 test/local-observation.test.ts \ + 2>&1 | tee local-runtime-results/source-tests.txt + rm -rf "$root" + - name: Build the existing CLI with the local patch + working-directory: .runtime-source + env: + OPENCODE_VERSION: 2.0.18-eval.5 + OPENCODE_CHANNEL: latest + run: | + bun run packages/cli/script/build.ts --skip-install --target=opencode-linux-x64 \ + 2>&1 | tee "$GITHUB_WORKSPACE/local-runtime-results/build.txt" + cp packages/cli/dist/cli-linux-x64/bin/opencode "$GITHUB_WORKSPACE/local-runtime-results/opencode" + sha256sum "$GITHUB_WORKSPACE/local-runtime-results/opencode" > "$GITHUB_WORKSPACE/local-runtime-results/binary.sha256" + - name: Build experimental image without replacing defaults + run: | + docker build -f runtime-patches/Containerfile \ + --build-arg RUNNER_REVISION="$(git rev-parse HEAD)" \ + -t local-observation:test local-runtime-results + - name: Save image data without publication credentials + run: docker save -o images.tar local-observation:test + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + if: always() + with: + name: runtime-build-evidence-${{ github.run_id }}-${{ github.run_attempt }} + path: | + local-runtime-results/** + !local-runtime-results/opencode + if-no-files-found: error + retention-days: 14 + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: runtime-image-data-${{ github.run_id }}-${{ github.run_attempt }} + path: images.tar + compression-level: 0 + if-no-files-found: error + retention-days: 1 + + publish: + needs: build + runs-on: ubuntu-latest + timeout-minutes: 10 + permissions: + contents: read + packages: write + outputs: + reference: ${{ steps.image.outputs.reference }} + steps: + - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 + with: + name: runtime-image-data-${{ github.run_id }}-${{ github.run_attempt }} + path: image-data + # No repository checkout or executable workload in the publication job. + - name: Publish fixed destination from image archive data + id: image + env: + REGISTRY_TOKEN: ${{ secrets.GITHUB_TOKEN }} + HEAD_SHA: ${{ github.event.pull_request.head.sha }} + run: | + [[ "$HEAD_SHA" =~ ^[0-9a-f]{40}$ ]] + mkdir publication + docker load --input image-data/images.tar + printf '%s' "$REGISTRY_TOKEN" | docker login ghcr.io --username "$GITHUB_ACTOR" --password-stdin + trap 'docker logout ghcr.io' EXIT + tag="ghcr.io/bateau84/opencode-eval-runner:observer-pr41-${HEAD_SHA:0:12}-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}" + docker tag local-observation:test "$tag" + docker push "$tag" | tee publication/publish.txt + image="$(docker image inspect "$tag" --format '{{index .RepoDigests 0}}')" + [[ "$image" =~ ^ghcr\.io/bateau84/opencode-eval-runner@sha256:[0-9a-f]{64}$ ]] + printf '%s\n' "$image" > publication/image.txt + printf 'reference=%s\n' "$image" >> "$GITHUB_OUTPUT" + docker image inspect "$image" > publication/image-inspect.json + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: runtime-publication-${{ github.run_id }}-${{ github.run_attempt }} + path: publication/ + retention-days: 14 + if-no-files-found: error + + verify: + needs: [build, publish] + runs-on: ubuntu-latest + timeout-minutes: 10 + permissions: + contents: read + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + with: + ref: ${{ github.event.pull_request.head.sha }} + persist-credentials: false + - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 + with: + name: runtime-build-evidence-${{ github.run_id }}-${{ github.run_attempt }} + path: local-runtime-results + - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 + with: + name: runtime-publication-${{ github.run_id }}-${{ github.run_attempt }} + path: local-runtime-results + - name: Probe published runtime without publication credentials + env: + LOCAL_IMAGE: ${{ needs.publish.outputs.reference }} + run: | + docker pull "$LOCAL_IMAGE" + python3 runtime-patches/run_image_probe.py --image "$LOCAL_IMAGE" --output local-runtime-results/probe + - name: Preserve Loom eval:live invoke contract + env: + LOCAL_IMAGE: ${{ needs.publish.outputs.reference }} + run: | + python3 runtime-patches/run_eval_live_compat_probe.py \ + --image "$LOCAL_IMAGE" \ + --output local-runtime-results/eval-live-compat + - name: Exercise real delegated Session identity and permission enforcement + env: + LOCAL_IMAGE: ${{ needs.publish.outputs.reference }} + run: | + python3 runtime-patches/run_delegated_session_probe.py \ + --image "$LOCAL_IMAGE" \ + --output local-runtime-results/delegated-session + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + if: always() + with: + name: local-runtime-${{ github.event.pull_request.head.sha }}-${{ github.run_attempt }} + path: local-runtime-results/ + if-no-files-found: error + retention-days: 14 diff --git a/.github/workflows/observer-integration.yml b/.github/workflows/observer-integration.yml new file mode 100644 index 0000000..38d6633 --- /dev/null +++ b/.github/workflows/observer-integration.yml @@ -0,0 +1,45 @@ +name: Observer boundary integration + +on: + pull_request: + paths: + - 'tests/integration/**' + - 'runner/observer.py' + - 'tests/test_capture_probe.py' + - '.github/workflows/observer-integration.yml' + +permissions: + contents: read + +jobs: + capture-boundary: + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + with: + ref: ${{ github.event.pull_request.head.sha }} + persist-credentials: false + + - name: Verify probe syntax and record checkout + run: | + python3 -m py_compile tests/integration/run_capture_probe.py + git rev-parse HEAD + + - name: Pull unchanged immutable OpenCode 2.0.18 image + run: docker pull ghcr.io/bateau84/opencode-eval-runner@sha256:68ef7322c75aede0e8cc76d0e3531e8b82dd417bbb5e5100264a89eab7fe8627 + + - name: Verify old-image rejection (negative control) + run: python3 tests/integration/run_capture_probe.py --expect-unsupported-baseline --output capture-probe-results + # All 12 checks must pass on the pinned old image; unexpected eligibility + # or any diagnostic failure is still a job failure. Capture stays BLOCKED. + # Positive capture is tested separately by Protected runtime channel. + + - name: Preserve diagnostic evidence even on blocked capture + if: always() + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: capture-boundary-${{ github.event.pull_request.head.sha }} + path: capture-probe-results/ + if-no-files-found: error + retention-days: 14 diff --git a/.github/workflows/protected-channel.yml b/.github/workflows/protected-channel.yml new file mode 100644 index 0000000..97d7487 --- /dev/null +++ b/.github/workflows/protected-channel.yml @@ -0,0 +1,144 @@ +name: Protected runtime channel + +on: + pull_request: + types: [opened, synchronize, reopened] + paths: + - protected-runtime/** + - runner/protected.py + - runner/protected_launch.py + - bin/opencode-eval-runner + - tests/integration/protected_tools.py + - tests/integration/test_protected_connection.py + - tests/test_review_accounting.py + - tests/test_workflow_boundaries.py + - tests/test_protected.py + - .github/workflows/protected-channel.yml + +permissions: + contents: read + +defaults: + run: + shell: bash + +jobs: + build: + if: github.event.pull_request.head.repo.full_name == github.repository + runs-on: ubuntu-latest + permissions: + contents: read + timeout-minutes: 15 + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + with: + ref: ${{ github.event.pull_request.head.sha }} + persist-credentials: false + - name: Python tests and provenance + run: | + mkdir -p protected-results + python3 -m unittest discover -s tests -p 'test_*.py' 2>&1 | tee protected-results/unit-tests.txt + python3 -m py_compile protected-runtime/invoke.py tests/integration/test_protected_connection.py + git rev-parse HEAD > protected-results/checkout.txt + sha256sum runner/protected.py runner/protected_launch.py protected-runtime/*.py protected-runtime/*.ts protected-runtime/Containerfile tests/integration/protected_tools.py tests/integration/test_protected_connection.py > protected-results/sources.sha256 + - name: Build images without registry credentials + run: | + for target in protected fixture-tools; do + docker build -f protected-runtime/Containerfile --target "$target" --build-arg RUNNER_REVISION="$(git rev-parse HEAD)" -t "pr41-${target}:build" . + done + docker save -o images.tar pr41-protected:build pr41-fixture-tools:build + - name: Preserve source and build evidence + if: always() + run: git archive --format=tar HEAD > protected-results/tested-source.tar + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + if: always() + with: + name: protected-build-evidence-${{ github.run_id }}-${{ github.run_attempt }} + path: protected-results/ + retention-days: 14 + if-no-files-found: error + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: protected-image-data-${{ github.run_id }}-${{ github.run_attempt }} + path: images.tar + compression-level: 0 + retention-days: 1 + if-no-files-found: error + + publish: + needs: build + runs-on: ubuntu-latest + timeout-minutes: 10 + permissions: + contents: read + packages: write + outputs: + runtime: ${{ steps.images.outputs.protected }} + tools: ${{ steps.images.outputs.fixture-tools }} + steps: + - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 + with: + name: protected-image-data-${{ github.run_id }}-${{ github.run_attempt }} + path: image-data + # No checkout, build, tests, or image execution on this privileged runner. + # Publication identifies experimental bytes; it does not approve their safety. + - name: Publish fixed image destinations from archive data + id: images + env: + REGISTRY_TOKEN: ${{ secrets.GITHUB_TOKEN }} + HEAD_SHA: ${{ github.event.pull_request.head.sha }} + run: | + [[ "$HEAD_SHA" =~ ^[0-9a-f]{40}$ ]] + mkdir publication + docker load --input image-data/images.tar + printf '%s' "$REGISTRY_TOKEN" | docker login ghcr.io --username "$GITHUB_ACTOR" --password-stdin + trap 'docker logout ghcr.io' EXIT + for target in protected fixture-tools; do + tag="ghcr.io/bateau84/opencode-eval-runner:pr41-${target}-${HEAD_SHA:0:12}-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}" + docker tag "pr41-${target}:build" "$tag" + docker push "$tag" | tee "publication/${target}-publish.txt" + image="$(docker image inspect "$tag" --format '{{index .RepoDigests 0}}')" + [[ "$image" =~ ^ghcr\.io/bateau84/opencode-eval-runner@sha256:[0-9a-f]{64}$ ]] + printf '%s\n' "$image" > "publication/${target}-image.txt" + printf '%s=%s\n' "$target" "$image" >> "$GITHUB_OUTPUT" + docker image inspect "$image" > "publication/${target}-inspect.json" + done + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: protected-publication-${{ github.run_id }}-${{ github.run_attempt }} + path: publication/ + retention-days: 14 + if-no-files-found: error + + verify: + needs: [build, publish] + runs-on: ubuntu-latest + timeout-minutes: 15 + permissions: + contents: read + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + with: + ref: ${{ github.event.pull_request.head.sha }} + persist-credentials: false + - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 + with: + name: protected-build-evidence-${{ github.run_id }}-${{ github.run_attempt }} + path: protected-results + - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 + with: + name: protected-publication-${{ github.run_id }}-${{ github.run_attempt }} + path: protected-results + - name: Real runtime to host importer, target attacks and transport faults + env: + RUNTIME_IMAGE: ${{ needs.publish.outputs.runtime }} + TOOL_IMAGE: ${{ needs.publish.outputs.tools }} + run: | + python3 tests/integration/test_protected_connection.py --image "$RUNTIME_IMAGE" --tool-image "$TOOL_IMAGE" --output protected-results/connection + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + if: always() + with: + name: protected-channel-${{ github.event.pull_request.head.sha }}-${{ github.run_attempt }} + path: protected-results/ + retention-days: 14 + if-no-files-found: error diff --git a/.github/workflows/sign-normal-invoke-evidence.yml b/.github/workflows/sign-normal-invoke-evidence.yml new file mode 100644 index 0000000..90cdfae --- /dev/null +++ b/.github/workflows/sign-normal-invoke-evidence.yml @@ -0,0 +1,284 @@ +name: Sign approved normal-invoke evidence + +# Trusted release/provenance workflow. Invoke the default-branch definition for +# an explicitly reviewed source commit. PR workflows receive no OIDC authority. +on: + workflow_dispatch: + inputs: + source_commit: + description: Reviewed 40-hex source commit to rebuild and sign + required: true + type: string + pull_request: + description: Pull request number containing source_commit + required: true + type: string + +permissions: + contents: read + +defaults: + run: + shell: bash + +jobs: + build: + if: github.ref == 'refs/heads/main' + runs-on: ubuntu-latest + timeout-minutes: 35 + permissions: + contents: read + outputs: + source_commit: ${{ steps.source.outputs.sha }} + steps: + - name: Validate reviewed source selection + id: source + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + SOURCE_COMMIT: ${{ inputs.source_commit }} + PR_NUMBER: ${{ inputs.pull_request }} + run: | + set -euo pipefail + [[ "$SOURCE_COMMIT" =~ ^[0-9a-f]{40}$ ]] + [[ "$PR_NUMBER" =~ ^[1-9][0-9]*$ ]] + test "$(gh api "repos/$GITHUB_REPOSITORY/pulls/$PR_NUMBER" --jq .head.sha)" = "$SOURCE_COMMIT" + printf 'sha=%s\n' "$SOURCE_COMMIT" >> "$GITHUB_OUTPUT" + + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + with: + ref: ${{ steps.source.outputs.sha }} + persist-credentials: false + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + with: + repository: anomalyco/opencode + ref: cd9a14a6b688d4021bee381dfd39d2cef9c0f862 + path: .runtime-source + persist-credentials: false + - uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 + with: + bun-version: 1.4.2 + + - name: Rebuild reviewed runtime from source + run: | + set -euo pipefail + mkdir -p approved-results + python3 -m py_compile runtime-patches/*.py + python3 runtime-patches/apply.py .runtime-source + bun install --cwd .runtime-source --frozen-lockfile + root="$(mktemp -d "$RUNNER_TEMP/approved-runtime-test.XXXXXX")" + mkdir -p "$root"/{home,config,data,state,cache,run} + HOME="$root/home" XDG_CONFIG_HOME="$root/config" XDG_DATA_HOME="$root/data" \ + XDG_STATE_HOME="$root/state" XDG_CACHE_HOME="$root/cache" XDG_RUNTIME_DIR="$root/run" \ + OPENCODE_DISABLE_AUTOUPDATE=1 bun test --cwd .runtime-source/packages/core --timeout 20000 test/local-observation.test.ts \ + 2>&1 | tee approved-results/source-tests.txt + rm -rf "$root" + OPENCODE_VERSION=2.0.18-eval.4 OPENCODE_CHANNEL=latest \ + bun run --cwd .runtime-source packages/cli/script/build.ts --skip-install --target=opencode-linux-x64 \ + 2>&1 | tee approved-results/build.txt + cp .runtime-source/packages/cli/dist/cli-linux-x64/bin/opencode approved-results/opencode + sha256sum approved-results/opencode > approved-results/binary.sha256 + docker build -f runtime-patches/Containerfile \ + --build-arg RUNNER_REVISION="${{ steps.source.outputs.sha }}" \ + -t approved-normal-invoke:build approved-results + docker save -o approved-image.tar approved-normal-invoke:build + git rev-parse HEAD > approved-results/source-commit.txt + git archive --format=tar HEAD > approved-results/tested-source.tar + + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: approved-build-${{ steps.source.outputs.sha }}-${{ github.run_id }} + path: | + approved-image.tar + approved-results/** + !approved-results/opencode + compression-level: 0 + if-no-files-found: error + retention-days: 7 + + publish: + needs: build + runs-on: ubuntu-latest + timeout-minutes: 10 + permissions: + contents: read + packages: write + outputs: + image: ${{ steps.publish.outputs.image }} + steps: + - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 + with: + name: approved-build-${{ needs.build.outputs.source_commit }}-${{ github.run_id }} + path: approved + - name: Publish reviewed image bytes without executing them + id: publish + env: + REGISTRY_TOKEN: ${{ secrets.GITHUB_TOKEN }} + SOURCE_COMMIT: ${{ needs.build.outputs.source_commit }} + run: | + set -euo pipefail + docker load --input approved/approved-image.tar + printf '%s' "$REGISTRY_TOKEN" | docker login ghcr.io --username "$GITHUB_ACTOR" --password-stdin + trap 'docker logout ghcr.io' EXIT + tag="ghcr.io/bateau84/opencode-eval-runner:approved-${SOURCE_COMMIT:0:12}-${GITHUB_RUN_ID}" + docker tag approved-normal-invoke:build "$tag" + docker push "$tag" + image="$(docker image inspect "$tag" --format '{{index .RepoDigests 0}}')" + [[ "$image" =~ ^ghcr\.io/bateau84/opencode-eval-runner@sha256:[0-9a-f]{64}$ ]] + printf 'image=%s\n' "$image" >> "$GITHUB_OUTPUT" + mkdir publication + printf '%s\n' "$image" > publication/image.txt + docker image inspect "$image" > publication/image-inspect.json + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: approved-publication-${{ needs.build.outputs.source_commit }}-${{ github.run_id }} + path: publication/ + if-no-files-found: error + retention-days: 30 + + verify: + needs: [build, publish] + runs-on: ubuntu-latest + timeout-minutes: 15 + permissions: + contents: read + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + with: + ref: ${{ needs.build.outputs.source_commit }} + persist-credentials: false + - name: Verify rebuilt image through normal invoke + env: + IMAGE: ${{ needs.publish.outputs.image }} + SOURCE_COMMIT: ${{ needs.build.outputs.source_commit }} + run: | + set -euo pipefail + docker pull "$IMAGE" + test "$(docker image inspect "$IMAGE" --format '{{ index .Config.Labels "org.opencontainers.image.revision" }}')" = "$SOURCE_COMMIT" + test "$(docker image inspect "$IMAGE" --format '{{ index .Config.Labels "io.opencode.runtime.version" }}')" = "2.0.18-eval.4" + mkdir -p approved-evidence + python3 runtime-patches/run_image_probe.py --image "$IMAGE" --output approved-evidence/probe + python3 runtime-patches/run_eval_live_compat_probe.py --image "$IMAGE" --output approved-evidence/eval-live-compat + python3 runtime-patches/run_delegated_session_probe.py --image "$IMAGE" --output approved-evidence/delegated-session + jq -e '.runtime_seam_passed == true and .handoff_acceptance == "BLOCKED"' approved-evidence/probe/summary.json >/dev/null + jq -e '.passed == true and .evidence_status == "diagnostic_non_evidence"' approved-evidence/eval-live-compat/summary.json >/dev/null + jq -e '.passed == true' approved-evidence/delegated-session/summary.json >/dev/null + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: approved-evidence-${{ needs.build.outputs.source_commit }}-${{ github.run_id }} + path: approved-evidence/ + if-no-files-found: error + retention-days: 30 + + sign: + needs: [build, publish, verify] + # Repository configuration must protect this environment with required + # reviewers. Green PR CI alone cannot authorize a signature. + environment: release-signing + runs-on: ubuntu-latest + timeout-minutes: 10 + permissions: + contents: read + packages: write + id-token: write + env: + COSIGN_OIDC_ISSUER: https://token.actions.githubusercontent.com + steps: + - uses: sigstore/cosign-installer@6f9f17788090df1f26f669e9d70d6ae9567deba6 + - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 + with: + name: approved-evidence-${{ needs.build.outputs.source_commit }}-${{ github.run_id }} + path: evidence + - name: Finalize reviewed result without executing source code + env: + SOURCE_COMMIT: ${{ needs.build.outputs.source_commit }} + IMAGE: ${{ needs.publish.outputs.image }} + PR_NUMBER: ${{ inputs.pull_request }} + run: | + set -euo pipefail + test -z "$(find evidence -type l -print -quit)" + test "$(du -sb evidence | cut -f1)" -le 25000000 + jq -e --arg image "$IMAGE" ' + .kind == "eval-live-invoke-compatibility" and + .version == 2 and + .image == $image and + .passed == true and + .evidence_status == "diagnostic_non_evidence" + ' evidence/eval-live-compat/summary.json >/dev/null + jq -e --arg image "$IMAGE" ' + .kind == "delegated-session-normal-invoke" and + .version == 1 and + .image == $image and + .passed == true + ' evidence/delegated-session/summary.json >/dev/null + jq -e --arg image "$IMAGE" --arg source "$SOURCE_COMMIT" ' + .kind == "normal-invoke-runtime-seam-probe" and + .version == 2 and + .image == $image and + .runner_revision == $source and + .runtime_seam_passed == true and + .handoff_acceptance == "BLOCKED" and + .independent_code_approval == false + ' evidence/probe/summary.json >/dev/null + mkdir signed + export SOURCE_COMMIT IMAGE PR_NUMBER + python3 - <<'PY' + import hashlib, json, os + from pathlib import Path + files = { + "eval_live_compat": Path("evidence/eval-live-compat/summary.json"), + "delegated_session": Path("evidence/delegated-session/summary.json"), + "runtime_probe": Path("evidence/probe/summary.json"), + } + parsed = {name: json.loads(path.read_text()) for name, path in files.items()} + payload = { + "schema": "opencode-eval-runner/signed-result/v2", + "source_commit": os.environ["SOURCE_COMMIT"], + "pull_request": int(os.environ["PR_NUMBER"]), + "workflow_run_id": int(os.environ["GITHUB_RUN_ID"]), + "image_digest": os.environ["IMAGE"], + "results": { + "eval_live_compat": bool(parsed["eval_live_compat"]["passed"]), + "delegated_session": bool(parsed["delegated_session"]["passed"]), + "runtime_seam": bool(parsed["runtime_probe"]["runtime_seam_passed"]), + }, + "evidence_sha256": { + name: hashlib.sha256(path.read_bytes()).hexdigest() + for name, path in files.items() + }, + "evidence_scope": "diagnostic normal-invoke observations", + "protected_capture_accepted": False, + "in_process_plugin_protection": "unsupported", + } + Path("signed/eval-result.json").write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n") + PY + printf '%s\n' "$IMAGE" > signed/image.txt + - name: Sign and verify reviewed image and finalized result + env: + SOURCE_COMMIT: ${{ needs.build.outputs.source_commit }} + IMAGE: ${{ needs.publish.outputs.image }} + REGISTRY_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + set -euo pipefail + identity="https://github.com/$GITHUB_WORKFLOW_REF" + printf '%s' "$REGISTRY_TOKEN" | docker login ghcr.io --username "$GITHUB_ACTOR" --password-stdin + trap 'docker logout ghcr.io' EXIT + cosign sign --yes -a "source_commit=$SOURCE_COMMIT" -a "workflow_run=$GITHUB_RUN_ID" "$IMAGE" + cosign verify "$IMAGE" \ + --certificate-identity "$identity" \ + --certificate-oidc-issuer "$COSIGN_OIDC_ISSUER" \ + -a "source_commit=$SOURCE_COMMIT" \ + -a "workflow_run=$GITHUB_RUN_ID" \ + > signed/image-verification.json + cosign sign-blob --yes --bundle signed/eval-result.sigstore.json signed/eval-result.json + cosign verify-blob signed/eval-result.json \ + --bundle signed/eval-result.sigstore.json \ + --certificate-identity "$identity" \ + --certificate-oidc-issuer "$COSIGN_OIDC_ISSUER" \ + > signed/result-verification.txt + printf '%s\n' "$identity" > signed/expected-signer-identity.txt + - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: signed-approved-${{ needs.build.outputs.source_commit }}-${{ github.run_id }} + path: signed/ + if-no-files-found: error + retention-days: 90 diff --git a/bin/opencode-eval-runner b/bin/opencode-eval-runner index 7f99289..7cfbfd8 100755 --- a/bin/opencode-eval-runner +++ b/bin/opencode-eval-runner @@ -4,7 +4,11 @@ from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parents[1])) -from runner.cli import main +if len(sys.argv) > 1 and sys.argv[1] == "observe": + sys.argv.pop(1) + from runner.protected import main +else: + from runner.cli import main if __name__ == "__main__": raise SystemExit(main()) diff --git a/container/evidence_safety.py b/container/evidence_safety.py new file mode 100644 index 0000000..fad7b79 --- /dev/null +++ b/container/evidence_safety.py @@ -0,0 +1,675 @@ +"""RSP v1: typed safety projection, not an observation authenticator. + +Consume Loom's private_policy() shape without rediscovering credentials. The +same module is used before container emission and by the host verifier. No +policy values, source locations, or policy hashes are public evidence. +""" +from __future__ import annotations + +import hashlib +import hmac +import json +import math +import re +from pathlib import Path +from typing import Any + +POLICY = 'loom-eval-credential-inventory/v1' +VERSION = 'source-path-roles/v1' +SAFETY = 'loom-eval-evidence-safety/v1' +REQUEST = 'opencode-eval-runner/evidence-safety-request/v1' +ACK = 'opencode-eval-runner/evidence-safety-ack/v1' +RESULT = 'opencode-eval-runner/safe-result/v1' +EVENTS = 'opencode-eval-runner/safe-tool-results/v1' +CONSUMER = 'runner-evidence-safety/v1' +RUNTIME_STATE = 'opencode-eval-runner/runtime-state/v1' +EXPECTED_MIGRATION_COUNT = 48 +FIRST_MIGRATION = '20260127222353_familiar_lady_ursula' +LAST_MIGRATION = '20260923013825_project_time_active' +POLICY_LIMIT = 128_000 +WIRE_LIMIT = 132_096 +RESULT_LIMIT = 1_000_000 +SUPPORTED_JSON_ESCAPE_LAYERS = 3 +MAX_JSON_ESCAPE_RECOVERY_LAYERS = 32 +SOURCE_NAMES = frozenset(('env', 'auth', 'config', 'models', 'credential_seed', 'config_root')) +STAGES = ['container.before_clip', 'container.before_output'] +REASONS = ('credential_match', 'sensitive_key', 'inventory_incomplete', 'upstream_clipped', + 'unsupported_schema', 'unsupported_representation', 'opaque_payload_unverified', + 'size_limit', 'missing', 'invalid', 'write_failed') +MISSING = object() + + +class Invalid(ValueError): + pass + + +def require(ok: bool, reason: str = 'invalid') -> None: + if not ok: + raise Invalid(reason) + + +def _pairs(pairs): + obj = {} + for k, v in pairs: + require(k not in obj) + obj[k] = v + return obj + + +def _bad_number(_): + raise Invalid('unsupported_representation') + + +def strict_loads(raw: bytes | str): + if isinstance(raw, bytes): + raw = raw.decode('utf-8', errors='strict') + value = json.loads(raw, object_pairs_hook=_pairs, parse_constant=_bad_number) + owned(value) + return value + + +def owned(value: Any, depth=0, budget=None) -> Any: + """Bounded JSON-only copy; never call user serialization hooks.""" + budget = [20_000] if budget is None else budget + budget[0] -= 1 + require(depth <= 32 and budget[0] >= 0, 'unsupported_representation') + if value is None or type(value) in (bool, int): + return value + if type(value) is float: + require(math.isfinite(value), 'unsupported_representation') + return value + if type(value) is str: + value.encode('utf-8', errors='strict') + return value + if type(value) is list: + return [owned(v, depth + 1, budget) for v in value] + if type(value) is dict: + require(all(type(k) is str for k in value), 'unsupported_representation') + return {owned(k, depth + 1, budget): owned(v, depth + 1, budget) for k, v in value.items()} + raise Invalid('unsupported_representation') + + +def encode(value) -> bytes: + return json.dumps(value, ensure_ascii=False, allow_nan=False, separators=(',', ':')).encode('utf-8') + + +def module_sha() -> str: + return hashlib.sha256(Path(__file__).read_bytes()).hexdigest() + + +def sensitive_key(key: str) -> bool: + key = re.sub(r'([A-Z]+)([A-Z][a-z])', r'\1_\2', key) + key = re.sub(r'([a-z0-9])([A-Z])', r'\1_\2', key) + key = '_'.join(re.findall(r'[a-z0-9]+', key.lower())) + return (key in {'key', 'apikey', 'access', 'refresh', 'token'} or + key.endswith(('_key', '_token')) or any( + key == x or key.endswith('_' + x) for x in + ('secret', 'password', 'credential', 'authorization', 'cookie'))) + + +_JSON_ESCAPE_SIMPLE = { + '"': '"', '\\': '\\', '/': '/', 'b': '\b', 'f': '\f', + 'n': '\n', 'r': '\r', 't': '\t', +} + + +def _json_unescape_layer(value: str) -> tuple[str, bool]: + """Decode one JSON-string escaping layer without requiring a whole JSON document.""" + out: list[str] = [] + changed = False + i = 0 + while i < len(value): + if value[i] != '\\' or i + 1 >= len(value): + out.append(value[i]) + i += 1 + continue + kind = value[i + 1] + simple = _JSON_ESCAPE_SIMPLE.get(kind) + if simple is not None: + out.append(simple) + changed = True + i += 2 + continue + if kind == 'u' and i + 6 <= len(value): + digits = value[i + 2:i + 6] + if re.fullmatch(r'[0-9a-fA-F]{4}', digits): + code = int(digits, 16) + consumed = 6 + if 0xD800 <= code <= 0xDBFF and i + 12 <= len(value) and value[i + 6:i + 8] == '\\u': + low_digits = value[i + 8:i + 12] + if re.fullmatch(r'[0-9a-fA-F]{4}', low_digits): + low = int(low_digits, 16) + if 0xDC00 <= low <= 0xDFFF: + code = 0x10000 + ((code - 0xD800) << 10) + (low - 0xDC00) + consumed = 12 + if not (0xD800 <= code <= 0xDFFF): + out.append(chr(code)) + changed = True + i += consumed + continue + out.append(value[i]) + i += 1 + return ''.join(out), changed + + +class Policy: + def __init__(self, value=None): + self.valid = False + self.complete = False + self.values: tuple[str, ...] = () + self.private = None + self.matcher = None + try: + require(type(value) is dict and set(value) == {'schema', 'policy_version', 'complete', 'sources', 'values'}) + require(value['schema'] == POLICY and value['policy_version'] == VERSION) + require(type(value['complete']) is bool and type(value['sources']) is dict) + require(SOURCE_NAMES == value['sources'].keys()) + require(all(type(k) is str and type(v) is str and v in {'complete', 'incomplete', 'not_selected'} + for k, v in value['sources'].items())) + require(type(value['values']) is list and len(value['values']) <= 4096) + require(all(type(v) is str and v for v in value['values'])) + require(len(encode(value)) <= POLICY_LIMIT) + complete = all(s in {'complete', 'not_selected'} for s in value['sources'].values()) + require(not value['complete'] or complete) + self.private = owned(value) + self.valid = True + self.complete = value['complete'] and complete + # An incomplete inventory is never upgraded using its partial matcher. + if self.complete: + self.values = tuple(sorted(set(value['values']), key=lambda v: (-len(v), v))) + variants = set(self.values) + frontier = variants.copy() + for _ in range(SUPPORTED_JSON_ESCAPE_LAYERS): + new = {json.dumps(v, ensure_ascii=ascii_only)[1:-1] + for v in frontier for ascii_only in (False, True)} - variants + variants.update(new) + frontier = new + # One substitution pass: generated markers are never re-scanned. + if variants: + self.matcher = re.compile('|'.join(re.escape(v) for v in sorted(variants, key=lambda v: (-len(v), v)))) + except (Invalid, ValueError, TypeError, UnicodeError, RecursionError, OverflowError): + self.valid = self.complete = False + self.values = () + self.private = None + self.matcher = None + + def matches(self, value: str) -> bool: + return self.matcher is not None and self.matcher.search(value) is not None + + def unsupported_recoverable(self, value: str) -> bool: + """Detect recoverable JSON-escape representations beyond the supported profile.""" + if self.matcher is None: + return False + current = value + for _ in range(MAX_JSON_ESCAPE_RECOVERY_LAYERS): + current, changed = _json_unescape_layer(current) + if not changed: + return False + if self.matches(current): + return True + # More than the bounded recovery profile is itself unsupported. Never + # retain a partially understood deeply escaped representation. + _, changed = _json_unescape_layer(current) + return changed + + def payload(self, value): + if type(value) is str: + # An opaque string is not a declared nested JSON boundary. Never + # mask its syntax into invalid JSON; a typed json_string role must + # explicitly opt in to parsing/re-encoding instead. + direct = self.matches(value) + if value.lstrip().startswith(('{', '[')) and direct: + raise Invalid('opaque_payload_unverified') + if not direct and self.unsupported_recoverable(value): + raise Invalid('unsupported_representation') + safe = self.matcher.sub('***REDACTED***', value) if self.matcher else value + return safe, safe != value + if type(value) is list: + parts = [self.payload(v) for v in value] + return [v for v, _ in parts], any(changed for _, changed in parts) + if type(value) is dict: + # Mapping keys are payload data too, but rewriting them can change + # protocol meaning or collide. Unsupported deeper JSON-escape + # representations therefore omit the enclosing payload. + if any(sensitive_key(k) or self.matches(k) for k in value): + raise Invalid('sensitive_key') + if any(self.unsupported_recoverable(k) for k in value): + raise Invalid('unsupported_representation') + parts = {k: self.payload(v) for k, v in value.items()} + return {k: v for k, (v, _) in parts.items()}, any(changed for _, changed in parts.values()) + # Arbitrary scalar payloads are not protocol counters. Do not make an + # invalid JSON token or change its type to hide a credential. + if self.matches(encode(value).decode()): + raise Invalid('credential_match') + return value, False + + +class Projection: + def __init__(self, policy: Policy, stage='runner'): + self.policy = policy + self.stage = stage + self.fields = [] + self.loss = {reason: 0 for reason in REASONS} + + def omit(self, field, reason, event=None): + require(reason in REASONS) + self.fields.append({'event': event, 'field': field, 'state': 'omitted', 'reason': reason, 'stage': self.stage}) + self.loss[reason] += 1 + return MISSING + + def field(self, field, value=MISSING, *, role='payload', event=None, limit=6000, clipped=False): + if value is MISSING: + return self.omit(field, 'missing', event) + if clipped: + return self.omit(field, 'upstream_clipped', event) + if not self.policy.complete: + return self.omit(field, 'inventory_incomplete', event) + try: + value = owned(value) + if role == 'identity': + require(value is None or type(value) is str, 'invalid') + if value is not None and self.policy.matches(value): + return self.omit(field, 'credential_match', event) + if value is not None and self.policy.unsupported_recoverable(value): + return self.omit(field, 'unsupported_representation', event) + safe, changed = value, False + elif role == 'identities': + require(type(value) is list and all(type(v) is str for v in value), 'invalid') + if any(self.policy.matches(v) for v in value): + return self.omit(field, 'credential_match', event) + if any(self.policy.unsupported_recoverable(v) for v in value): + return self.omit(field, 'unsupported_representation', event) + safe, changed = value, False + elif role == 'json_string': + require(type(value) is str, 'unsupported_representation') + safe_value, changed = self.policy.payload(strict_loads(value)) + safe = encode(safe_value).decode('utf-8') if changed else value + else: + safe, changed = self.policy.payload(value) + # Sanitation always precedes the size decision. Never retain a prefix. + if len(encode(safe)) > limit: + return self.omit(field, 'size_limit', event) + disposition = {'event': event, 'field': field, 'state': 'redacted' if changed else 'exact'} + if changed: + disposition.update(reason='credential_match', stage=self.stage) + self.loss['credential_match'] += 1 + self.fields.append(disposition) + return safe + except (Invalid, ValueError, UnicodeError, TypeError, RecursionError, OverflowError) as exc: + return self.omit(field, str(exc) if type(exc) is Invalid else 'unsupported_representation', event) + + def summary(self): + return {'schema': SAFETY, 'policy_version': VERSION, 'inventory_complete': self.policy.complete, + 'coverage_complete': self.policy.complete and not any(self.loss.values()), + 'fields': self.fields, 'loss_counts': self.loss} + + +# Only these generated fields are protocol. Dynamic identifiers remain payload. +PUBLIC_COUNTERS = {'exit_code', 'stdout_total_chars', 'stderr_total_chars'} +PUBLIC_FLAGS = {'timed_out', 'infrastructure_error', 'stdout_truncated', 'stderr_truncated'} +DYNAMIC = {'model', 'reasoning', 'agent', 'skill', 'session_id', 'credential_source'} +OPAQUE = {'stdout', 'stderr', 'plugin_diagnostic', 'plugin_preflight'} +TOP_LEVEL_REQUIRED = DYNAMIC | OPAQUE | { + 'transport', 'reasoning_source', 'text', 'tools', 'actions', 'skills_loaded', + 'timing', 'tool_result_evidence' +} +TOP_LEVEL_PROTOCOL = {'transport', 'reasoning_source', 'timing', 'tool_result_evidence', 'runtime_state'} +ALLOWED = PUBLIC_COUNTERS | PUBLIC_FLAGS | TOP_LEVEL_REQUIRED | {'schema', 'runtime_state'} + + +def validated_runtime_state(value: Any) -> dict[str, Any]: + require(type(value) is dict and set(value) == { + 'schema', 'profile', 'database_source', 'database_created', 'database_seed_present', + 'auth_source', 'session_rows_before_inference', 'credential_rows_before_inference', + 'migration_count', 'first_migration', 'last_migration', + }, 'unsupported_schema') + require(value['schema'] == RUNTIME_STATE and value['profile'] == 'disposable', 'invalid') + require(value['database_source'] == 'runtime-bootstrap', 'invalid') + require(value['database_created'] is True and value['database_seed_present'] is False, 'invalid') + require(value['auth_source'] in {'none', 'explicit'}, 'invalid') + for name in ('session_rows_before_inference', 'credential_rows_before_inference', 'migration_count'): + require(type(value[name]) is int and 0 <= value[name] <= 2**53 - 1, 'invalid') + require(value['session_rows_before_inference'] == 0 and value['credential_rows_before_inference'] == 0, 'invalid') + require(value['migration_count'] == EXPECTED_MIGRATION_COUNT, 'invalid') + require(value['first_migration'] == FIRST_MIGRATION, 'invalid') + require(value['last_migration'] == LAST_MIGRATION, 'invalid') + return owned(value) + + +def project_result(raw: Any, policy: Policy, stage='runner') -> dict: + p = Projection(policy, stage) + result = {'schema': RESULT} + if type(raw) is not dict or set(raw) - ALLOWED or raw.get('schema') != 'opencode-eval-runner/v1': + for name in TOP_LEVEL_REQUIRED: + p.omit(name, 'unsupported_schema') + result['evidence_safety'] = p.summary() + return result + for name in ('transport', 'reasoning_source'): + options = {'transport': {'opencode', 'github-copilot-cli'}, + 'reasoning_source': {'explicit', 'model-variant', 'provider-default'}}[name] + if raw.get(name) in options: + result[name] = raw[name] + p.fields.append({'event': None, 'field': name, 'state': 'exact'}) + else: + p.omit(name, 'invalid') + for name in PUBLIC_COUNTERS: + if name in raw and type(raw[name]) is int and abs(raw[name]) <= 2**53 - 1: + result[name] = raw[name] + for name in PUBLIC_FLAGS: + if name in raw and type(raw[name]) is bool: + result[name] = raw[name] + for name in DYNAMIC | {'text', 'tools', 'skills_loaded'}: + role = 'identity' if name in DYNAMIC else 'identities' if name in {'tools', 'skills_loaded'} else 'payload' + value = p.field(name, raw.get(name, MISSING), role=role, limit=200_000 if name == 'text' else 6000) + if value is not MISSING: + result[name] = value + # Unknown/raw streams can contain arbitrary encodings, malformed/partial + # JSON, and generated runtime secrets. They have no safe typed adapter yet. + for name in OPAQUE: + p.omit(name, 'upstream_clipped' if raw.get(name + '_truncated') else 'opaque_payload_unverified') + if 'runtime_state' in raw: + try: + result['runtime_state'] = validated_runtime_state(raw['runtime_state']) + p.fields.append({'event': None, 'field': 'runtime_state', 'state': 'exact'}) + except (Invalid, ValueError, UnicodeError, TypeError, RecursionError, OverflowError) as exc: + p.omit('runtime_state', str(exc) if type(exc) is Invalid else 'unsupported_representation') + + timing = raw.get('timing') + timing_keys = {'run_seconds', 'export_seconds', 'export_exit_code', 'total_seconds'} + if type(timing) is dict and not (set(timing) - timing_keys) and all( + v is None or (type(v) in (int, float) and math.isfinite(v)) for v in timing.values()): + result['timing'] = dict(timing) + p.fields.append({'event': None, 'field': 'timing', 'state': 'exact'}) + else: + p.omit('timing', 'unsupported_schema') + # Actions have fixed structure but dynamic selector values. If any action + # loses identity/input fidelity, the whole action list is unavailable. + actions = raw.get('actions') + action_safe = [] + try: + require(policy.complete, 'inventory_incomplete') + require(type(actions) is list and len(actions) <= 64, 'size_limit') + for action in actions: + require(type(action) is dict and set(action) == {'tool', 'args'}, 'unsupported_schema') + require(type(action['tool']) is str, 'invalid') + require(not policy.matches(action['tool']), 'credential_match') + require(not policy.unsupported_recoverable(action['tool']), 'unsupported_representation') + require(type(action['args']) is dict, 'invalid') + args, changed = policy.payload(owned(action['args'])) + require(not changed, 'credential_match') + action_safe.append({'tool': action['tool'], 'args': args}) + require(len(encode(action_safe)) <= 48000, 'size_limit') + result['actions'] = action_safe + p.fields.append({'event': None, 'field': 'actions', 'state': 'exact'}) + except (Invalid, ValueError, UnicodeError, TypeError, RecursionError, OverflowError) as exc: + p.omit('actions', str(exc) if type(exc) is Invalid else 'unsupported_representation') + evidence = raw.get('tool_result_evidence') + if type(evidence) is dict and evidence.get('schema') == 'runner-unclipped-events/internal-v1': + events = evidence['events'] + rows = [] + count = 0 + for ev in events: + if ev.get('type') != 'tool_use': + continue + count += 1 + part = ev.get('part') + state = part.get('state') if type(part) is dict else None + if (type(state) is not dict or part.get('type') != 'tool' or + set(ev) - {'type', 'timestamp', 'sessionID', 'part'} or + set(part) - {'id', 'partID', 'sessionID', 'messageID', 'type', 'callID', 'tool', 'state', 'time'} or + set(state) - {'status', 'input', 'output', 'error', 'title', 'metadata', 'time', 'attachments'}): + p.omit('event', 'unsupported_schema', count - 1) + continue + if len(rows) >= 64: + p.omit('event', 'size_limit', count - 1) + continue + row = {'sequence': count} + if any(name in state for name in ('metadata', 'attachments', 'title')): + p.omit('metadata', 'opaque_payload_unverified', count - 1) + status = state.get('status') + if status in {'pending', 'running', 'completed', 'error'}: + row['status'] = status + p.fields.append({'event': count - 1, 'field': 'status', 'state': 'exact'}) + else: + p.omit('status', 'invalid', count - 1) + fields = {'tool': part.get('tool', MISSING), 'call_id': part.get('callID', part.get('id', MISSING)), + 'session_id': ev.get('sessionID', MISSING), 'input': state.get('input', MISSING)} + for name in ('output', 'error'): + if name in state: + fields[name] = state[name] + if not any(name in state for name in ('output', 'error')): + fields['output'] = MISSING + metadata = state.get('metadata') + # Pinned OpenCode declares Session truncation in metadata. Missing + # or unfamiliar declarations cannot attest a complete native output. + meta = metadata.get('metadata', metadata) if type(metadata) is dict else {} + upstream_complete = type(meta) is dict and meta.get('truncated') is False + upstream_clipped = type(meta) is dict and meta.get('truncated') is True + for name, original in fields.items(): + if name == 'output' and not upstream_complete: + p.omit(name, 'upstream_clipped' if upstream_clipped else 'opaque_payload_unverified', count-1) + continue + value = p.field(name, original, event=count-1, + role='identity' if name in {'tool', 'call_id', 'session_id'} else 'payload', + limit=2000 if name == 'input' else 256 if name.endswith('_id') or name == 'tool' else 6000) + if value is not MISSING: + row[name] = value + rows.append(row) + result['tool_result_evidence'] = {'schema': EVENTS, 'source': 'opencode.event-stream.full', + 'observed_events': count, 'omitted_events': count - len(rows), 'events': rows} + p.fields.append({'event': None, 'field': 'tool_result_evidence', 'state': 'exact'}) + else: + p.omit('tool_result_evidence', 'upstream_clipped' if evidence else 'missing') + result['evidence_safety'] = p.summary() + return result + + +def receipt(policy: Policy, run_id: str, revision: str, binding_key: bytes = b'') -> dict: + return {'schema': ACK, 'consumer': CONSUMER, 'run_id': run_id, + 'policy_schema': POLICY, 'policy_version': VERSION, 'projection_schema': SAFETY, + 'stages': STAGES, 'policy_valid': policy.valid, 'inventory_complete': policy.complete, + 'module_sha256': module_sha(), 'image_source_revision': revision, + 'policy_receipt': hmac.new(binding_key, encode(policy.private), hashlib.sha256).hexdigest()} + + +def fallback(reason='invalid', stage='transport', policy=None): + p = Projection(policy or Policy(), stage) + for name in sorted(TOP_LEVEL_REQUIRED): + p.omit(name, reason) + return {'schema': RESULT, 'exit_code': 2, 'infrastructure_error': True, 'evidence_safety': p.summary()} + + +def read_request(stream): + """Private stdin is consumed before product startup and is never forwarded.""" + try: + raw = stream.read(WIRE_LIMIT + 1) + require(len(raw) <= WIRE_LIMIT) + data = strict_loads(raw) + require(type(data) is dict and set(data) == {'schema', 'run_id', 'policy', 'binding_key'}) + require(data['schema'] == REQUEST and type(data['run_id']) is str and + re.fullmatch('[0-9a-f]{64}', data['run_id']) is not None) + require(type(data['binding_key']) is str and re.fullmatch('[0-9a-f]{64}', data['binding_key']) is not None) + return Policy(data['policy']), data['run_id'], bytes.fromhex(data['binding_key']) + except (ValueError, Invalid, TypeError, UnicodeError, OSError, RecursionError): + return Policy(), '0' * 64, b'' + + +def assert_safe_preview(value, policy): + """Validate sanitized payloads without rescanning generated display markers.""" + if type(value) is str: + pieces = value.split('***REDACTED***') + require(all(not policy.matches(v) and not policy.unsupported_recoverable(v) for v in pieces)) + elif type(value) is dict: + require(all(not sensitive_key(k) and not policy.matches(k) and + not policy.unsupported_recoverable(k) for k in value)) + for v in value.values(): assert_safe_preview(v, policy) + elif type(value) is list: + for v in value: assert_safe_preview(v, policy) + else: + require(not policy.matches(encode(value).decode())) + + +def validate_reply(raw: bytes | str, policy: Policy, run_id: str, revision: str, binding_key: bytes = b'') -> dict: + """Schema/availability acknowledgement, not authentication of hostile code.""" + require(len(raw) <= RESULT_LIMIT) + reply = strict_loads(raw) + require(type(reply) is dict and reply.get('schema') == RESULT) + ack = reply.get('evidence_safety_ack') + require(type(ack) is dict and type(ack.get('policy_valid')) is bool and type(ack.get('inventory_complete')) is bool) + require(ack == receipt(policy, run_id, revision, binding_key)) + allowed = (PUBLIC_COUNTERS | PUBLIC_FLAGS | DYNAMIC | { + 'schema', 'transport', 'reasoning_source', 'text', 'tools', 'actions', 'skills_loaded', + 'timing', 'tool_result_evidence', 'runtime_state', 'evidence_safety', 'evidence_safety_ack'}) + require(not (set(reply) - allowed)) + summary = reply['evidence_safety'] + require(type(summary) is dict and set(summary) == { + 'schema', 'policy_version', 'inventory_complete', 'coverage_complete', 'fields', 'loss_counts'}) + require(summary['schema'] == SAFETY and summary['policy_version'] == VERSION) + require(type(summary['inventory_complete']) is bool and summary['inventory_complete'] == policy.complete) + require(type(summary['coverage_complete']) is bool) + require(type(summary['loss_counts']) is dict and set(summary['loss_counts']) == set(REASONS)) + require(all(type(v) is int and 0 <= v <= 2**53 - 1 for v in summary['loss_counts'].values())) + require(type(summary['fields']) is list and len(summary['fields']) <= 10_000) + dispositions = {} + counts = {r: 0 for r in REASONS} + top_names = TOP_LEVEL_REQUIRED | {'runtime_state'} + event_names = {'event', 'tool', 'call_id', 'session_id', 'input', 'output', 'error', 'status', 'metadata'} + for item in summary['fields']: + require(type(item) is dict) + event, name, state = item.get('event'), item.get('field'), item.get('state') + require(event is None or type(event) is int and 0 <= event <= 2**53 - 1) + require(type(name) is str and name in (top_names if event is None else event_names)) + require(state in {'exact', 'redacted', 'omitted'}) + required = {'event', 'field', 'state'} | ({'reason', 'stage'} if state != 'exact' else set()) + require(set(item) == required and (event, name) not in dispositions) + if state != 'exact': + require(item['reason'] in REASONS and item['stage'] == 'runner') + counts[item['reason']] += 1 + require(state != 'redacted' or item['reason'] == 'credential_match') + require( + policy.complete or + state == 'omitted' or + (event is None and name in TOP_LEVEL_PROTOCOL and state == 'exact') or + (event is not None and name == 'status' and state == 'exact') + ) + dispositions[(event, name)] = item + require(counts == summary['loss_counts']) + require(summary['coverage_complete'] == (policy.complete and not any(counts.values()))) + for name in TOP_LEVEL_REQUIRED: + item = dispositions.get((None, name)) + require(item is not None) + require((name in reply) == (item['state'] != 'omitted')) + if name in TOP_LEVEL_PROTOCOL and name in reply: + require(item['state'] == 'exact') + runtime_item = dispositions.get((None, 'runtime_state')) + if 'runtime_state' in reply: + require(runtime_item is not None and runtime_item['state'] == 'exact') + if runtime_item is not None and runtime_item['state'] == 'omitted': + require('runtime_state' not in reply) + for name in PUBLIC_COUNTERS: + require(name not in reply or type(reply[name]) is int and abs(reply[name]) <= 2**53 - 1) + for name in PUBLIC_FLAGS: + require(name not in reply or type(reply[name]) is bool) + require('transport' not in reply or reply['transport'] in {'opencode', 'github-copilot-cli'}) + require('reasoning_source' not in reply or reply['reasoning_source'] in {'explicit', 'model-variant', 'provider-default'}) + if 'runtime_state' in reply: + require(validated_runtime_state(reply['runtime_state']) == reply['runtime_state']) + require(dispositions.get((None, 'runtime_state'), {}).get('state') == 'exact') + if 'tool_result_evidence' in reply: + evidence = reply['tool_result_evidence'] + require(type(evidence) is dict and set(evidence) == {'schema', 'source', 'observed_events', 'omitted_events', 'events'}) + require(evidence['schema'] == EVENTS and evidence['source'] == 'opencode.event-stream.full') + require(type(evidence['events']) is list and len(evidence['events']) <= 64) + require(all(type(evidence[n]) is int and evidence[n] >= 0 for n in ('observed_events', 'omitted_events'))) + require(evidence['observed_events'] - evidence['omitted_events'] == len(evidence['events'])) + seen = set() + for event in evidence['events']: + require(type(event) is dict and not (set(event) - {'sequence','status','tool','call_id','session_id','input','output','error'})) + seq = event.get('sequence') + require(type(seq) is int and 1 <= seq <= evidence['observed_events'] and seq not in seen) + seen.add(seq) + event_id = seq - 1 + for required_name in ('tool', 'call_id', 'session_id', 'input', 'status'): + require((event_id, required_name) in dispositions) + require(any((event_id, name) in dispositions for name in ('output', 'error'))) + for name in event_names - {'event', 'metadata'}: + item = dispositions.get((event_id, name)) + if name in event: + require(item is not None and item['state'] != 'omitted') + if name in {'tool', 'call_id', 'session_id'}: + require(item['state'] == 'exact' and type(event[name]) is str) + if name == 'status': + require(item['state'] == 'exact') + if item and item['state'] == 'omitted': + require(name not in event) + status = event.get('status') + require(status is None or status in {'pending', 'running', 'completed', 'error'}) + status_item = dispositions[(event_id, 'status')] + require((status is None) == (status_item['state'] == 'omitted')) + if status == 'completed': + require((event_id, 'output') in dispositions) + if status == 'error': + require((event_id, 'error') in dispositions) + retained_ids = {event['sequence'] - 1 for event in evidence['events']} + omitted_ids = { + event for (event, name), item in dispositions.items() + if event is not None and name == 'event' and item['state'] == 'omitted' + } + require(len(omitted_ids) == evidence['omitted_events']) + require(all(0 <= event < evidence['observed_events'] for event in omitted_ids)) + require(not (retained_ids & omitted_ids)) + require(retained_ids | omitted_ids == set(range(evidence['observed_events']))) + for (event, name), item in dispositions.items(): + if event is None: + continue + if name == 'event': + require(item['state'] == 'omitted' and event in omitted_ids) + else: + require(event in retained_ids) + timing = reply.get('timing') + if timing is not None: + require(type(timing) is dict and not (set(timing) - {'run_seconds','export_seconds','export_exit_code','total_seconds'})) + require(all(v is None or type(v) in (int,float) and math.isfinite(v) for v in timing.values())) + # A changed field may never pass as exact; no duplicate/missing metadata or + # value attached to an omission may be admitted by a consumer. + for (event, name), item in dispositions.items(): + if event is None: + container = reply + else: + container = next((x for x in reply.get('tool_result_evidence', {}).get('events', []) + if x['sequence'] == event + 1), {}) + if item['state'] == 'omitted': + require(name not in container) + else: + require(name in container) + val = container[name] + if name in DYNAMIC or name in {'tools','actions','skills_loaded','tool','call_id','session_id'}: + require(item['state'] == 'exact') + if name == 'text': + require(type(val) is str) + if item['state'] == 'redacted': + assert_safe_preview(val, policy) + if item['state'] == 'exact': + check = Projection(policy) + role = 'identity' if name in DYNAMIC or name in {'tool','call_id'} else 'identities' if name in {'tools','skills_loaded'} else 'payload' + # Fixed action keys are protocol; nested tool/args were separately validated. + if name == 'actions': + require(type(val) is list and len(val) <= 64) + for action in val: + require(type(action) is dict and set(action) == {'tool','args'}) + require(type(action['tool']) is str and not policy.matches(action['tool']) + and not policy.unsupported_recoverable(action['tool'])) + require(type(action['args']) is dict and policy.payload(action['args']) == (action['args'], False)) + elif name == 'runtime_state': + # Runtime-state counters/discriminators are reviewed protocol + # structure. Credentials such as "0", "1", "text", or "low" + # must not reclassify those fixed values as payload. + require(validated_runtime_state(val) == val) + elif name == 'status': + require(val in {'pending', 'running', 'completed', 'error'}) + elif event is None and name in TOP_LEVEL_PROTOCOL: + pass + else: + projected = check.field(name, val, role=role, limit=200_000 if name == 'text' else 6000) + require(projected is not MISSING and check.fields[0]['state'] == 'exact') + return reply diff --git a/container/invoke.py b/container/invoke.py index 1dc9111..831c8e0 100644 --- a/container/invoke.py +++ b/container/invoke.py @@ -8,17 +8,48 @@ import secrets import selectors import shutil +import sqlite3 import subprocess import sys +import tempfile import time import urllib.error import urllib.request from pathlib import Path from typing import Any +# Only the opt-in safety image/host mode enables this adapter. +RSP_ACTIVE = False +RSP_POLICY = None +RSP_RUN_ID = None +RSP_STREAM_INVALID = False +RSP_BINDING_KEY = b'' + + +def evidence_slice(text: str, limit: int) -> str: + # Omit opaque streams before any runner clipping. Typed values are projected + # separately from the complete in-memory event stream. Legacy behavior stays unchanged. + return "" if RSP_ACTIVE else text[:limit] + + +def rsp_module(): + import importlib.util + name = "runner_evidence_safety" + if name not in sys.modules: + spec = importlib.util.spec_from_file_location(name, Path(__file__).with_name("evidence_safety.py")) + module = importlib.util.module_from_spec(spec) + sys.modules[name] = module + spec.loader.exec_module(module) + return sys.modules[name] + RESULT_SCHEMA = "opencode-eval-runner/v1" OPENCODE_EVAL_TITLE = "opencode-eval-runner" COPILOT_AGENT_NAME = "eval-runner" +RUNTIME_STATE_SCHEMA = "opencode-eval-runner/runtime-state/v1" +DISPOSABLE_STATE_PROFILE = "disposable" +EXPECTED_MIGRATION_COUNT = 48 +FIRST_MIGRATION = "20260127222353_familiar_lady_ursula" +LAST_MIGRATION = "20260923013825_project_time_active" COPILOT_AUTH_ENVS = ("COPILOT_GITHUB_TOKEN", "GH_TOKEN", "GITHUB_TOKEN") COPILOT_EXCLUDED_TOOLS = ( "bash", "powershell", "list_bash", "list_powershell", "read_bash", @@ -36,6 +67,7 @@ def run(command: list[str], cwd: Path, env: dict[str, str], timeout: int) -> sub cwd=cwd, env=env, capture_output=True, + stdin=subprocess.DEVNULL if RSP_ACTIVE else None, text=True, timeout=max(timeout, 1), check=False, @@ -43,9 +75,16 @@ def run(command: list[str], cwd: Path, env: dict[str, str], timeout: int) -> sub def timeout_output(value: str | bytes | None) -> str: + global RSP_STREAM_INVALID if value is None: return "" if isinstance(value, bytes): + if RSP_ACTIVE: + try: + return value.decode("utf-8", errors="strict") + except UnicodeError: + RSP_STREAM_INVALID = True + return "" return value.decode("utf-8", errors="replace") return value @@ -72,14 +111,19 @@ def last_event_summary(events: list[dict[str, Any]]) -> dict[str, Any] | None: def parse_events(text: str) -> list[dict[str, Any]]: + global RSP_STREAM_INVALID events: list[dict[str, Any]] = [] for line in text.splitlines(): try: - value = json.loads(line) - except json.JSONDecodeError: + value = rsp_module().strict_loads(line) if RSP_ACTIVE else json.loads(line) + except (ValueError, TypeError, RecursionError): + if RSP_ACTIVE and line.strip(): + RSP_STREAM_INVALID = True continue if isinstance(value, dict): events.append(value) + elif RSP_ACTIVE: + RSP_STREAM_INVALID = True return events @@ -127,7 +171,7 @@ def tool_action(part: dict[str, Any]) -> dict[str, Any] | None: if part.get("type") != "tool" or not isinstance(tool, str): return None state = part.get("state") - args = state.get("input") if isinstance(state, dict) and isinstance(state.get("input"), dict) else {} + args = state.get("input") if isinstance(state, dict) and isinstance(state.get("input"), dict) else (None if RSP_ACTIVE else {}) return {"tool": tool, "args": args} @@ -152,7 +196,7 @@ def extract_actions(events: list[dict[str, Any]]) -> list[dict[str, Any]]: def _tool_result_text(value: Any, limit: int) -> tuple[str, bool]: text = value if isinstance(value, str) else json.dumps(value, ensure_ascii=False, sort_keys=True) - if len(text) <= limit: + if RSP_ACTIVE or len(text) <= limit: return text, False marker = "\n[... tool-result field truncated ...]\n" retained = limit - len(marker) @@ -162,6 +206,10 @@ def _tool_result_text(value: Any, limit: int) -> tuple[str, bool]: def extract_tool_result_evidence(events: list[dict[str, Any]]) -> dict[str, Any]: """Bound tool results from the full structured event stream before stdout clipping.""" + if RSP_ACTIVE: + # Never serialize/clip an unprotected field. This marker is private to + # this function and emit_result; it is never sent to the host. + return {"schema": "runner-unclipped-events/internal-v1", "events": events} evidence: dict[str, Any] = { "schema": "opencode-eval-runner/tool-results/v1", "source": "opencode.event-stream.full", @@ -297,6 +345,11 @@ def prepare_opencode_env() -> dict[str, str]: seed_models = Path("/seed/models.json") seed_database = Path("/seed/opencode.db") seed_config_root = Path("/seed/opencode-config") + state_profile = os.environ.get("EVAL_OPENCODE_STATE_PROFILE", "default") + if state_profile not in {"default", DISPOSABLE_STATE_PROFILE}: + raise RuntimeError("unsupported OpenCode state profile") + if state_profile == DISPOSABLE_STATE_PROFILE and seed_database.is_file(): + raise RuntimeError("disposable profile forbids database seeds") if seed_config.is_file(): shutil.copyfile(seed_config, config / "opencode.json") else: @@ -351,6 +404,75 @@ def prepare_opencode_env() -> dict[str, str]: return env +def attest_disposable_runtime_state(env: dict[str, str]) -> dict[str, Any]: + """Attest the production disposable database without mutating it.""" + data = Path(env["XDG_DATA_HOME"]) / "opencode" + database = data / "opencode.db" + if not database.is_file(): + raise RuntimeError("disposable database bootstrap produced no database") + auth = data / "auth.json" + auth_source = env.get("EVAL_OPENCODE_AUTH_SOURCE") + if auth_source == "none" and auth.exists(): + raise RuntimeError("disposable profile unexpectedly loaded auth state") + if auth_source == "explicit" and not auth.is_file(): + raise RuntimeError("explicit disposable auth seed was not loaded") + if auth_source not in {"none", "explicit"}: + raise RuntimeError("invalid disposable auth source") + try: + with sqlite3.connect(f"file:{database}?mode=ro", uri=True) as db: + tables = {row[0] for row in db.execute( + "SELECT name FROM sqlite_master WHERE type='table' AND name NOT LIKE 'sqlite_%'" + )} + migrations = [row[0] for row in db.execute("SELECT id FROM migration ORDER BY id")] + session_rows = db.execute("SELECT COUNT(*) FROM session_v2").fetchone()[0] + credential_rows = db.execute("SELECT COUNT(*) FROM credential").fetchone()[0] + except sqlite3.Error as exc: + raise RuntimeError("disposable database attestation failed") from exc + if not {"session_v2", "credential", "migration"} <= tables: + raise RuntimeError("disposable database schema incomplete") + if (len(migrations) != EXPECTED_MIGRATION_COUNT or + migrations[0] != FIRST_MIGRATION or migrations[-1] != LAST_MIGRATION): + raise RuntimeError("disposable database migration journal mismatch") + if session_rows != 0 or credential_rows != 0: + raise RuntimeError("disposable database was mutated before model inference") + return { + "schema": RUNTIME_STATE_SCHEMA, + "profile": DISPOSABLE_STATE_PROFILE, + "database_source": "runtime-bootstrap", + "database_created": True, + "database_seed_present": False, + "auth_source": auth_source, + "session_rows_before_inference": session_rows, + "credential_rows_before_inference": credential_rows, + "migration_count": len(migrations), + "first_migration": migrations[0], + "last_migration": migrations[-1], + } + + +def disposable_runtime_state(env: dict[str, str], timeout: int) -> dict[str, Any] | None: + """Bootstrap and attest fresh OpenCode state before model inference.""" + if env.get("EVAL_OPENCODE_STATE_PROFILE") != DISPOSABLE_STATE_PROFILE: + return None + if env.get("EVAL_OPENCODE_DATABASE_SOURCE") != "runtime-bootstrap": + raise RuntimeError("invalid disposable database source") + data = Path(env["XDG_DATA_HOME"]) / "opencode" + database = data / "opencode.db" + if Path("/seed/opencode.db").is_file() or database.exists(): + raise RuntimeError("disposable database must start absent") + + # This command initializes the normal OpenCode storage/session stack but + # does not invoke a model provider. The runtime owns schema/bootstrap and + # its migration journal; the runner does not fabricate either. + bootstrap = run( + ["opencode", "session", "list", "--standalone", "--format", "json", "--max-count", "1"], + Path("/workspace"), env, min(timeout, 30), + ) + if bootstrap.returncode != 0: + raise RuntimeError("disposable database bootstrap failed") + return attest_disposable_runtime_state(env) + + def ensure_model_config(env: dict[str, str], model: str) -> None: config_root = Path( env.get("OPENCODE_CONFIG_DIR") @@ -404,7 +526,7 @@ def _standalone_json_request( except urllib.error.HTTPError as exc: body = exc.read().decode("utf-8", errors="replace") raise RuntimeError( - f"OpenCode preflight request {method} {path} failed: HTTP {exc.code}: {body[:2000]}" + f"OpenCode preflight request {method} {path} failed: HTTP {exc.code}: {evidence_slice(body, 2000)}" ) from exc except urllib.error.URLError as exc: raise RuntimeError( @@ -416,7 +538,7 @@ def _standalone_json_request( return json.loads(body) except json.JSONDecodeError as exc: raise RuntimeError( - f"OpenCode preflight request {method} {path} returned non-JSON: {body[:2000]}" + f"OpenCode preflight request {method} {path} returned non-JSON: {evidence_slice(body, 2000)}" ) from exc @@ -453,7 +575,7 @@ def _start_preflight_server( stderr = proc.stderr.read() if proc.stderr is not None else "" raise RuntimeError( "OpenCode preflight server did not report readiness" - + (f": {stderr[:3000]}" if stderr.strip() else "") + + (f": {evidence_slice(stderr, 3000)}" if stderr.strip() else "") ) line = proc.stdout.readline() try: @@ -463,7 +585,7 @@ def _start_preflight_server( stderr = proc.stderr.read() if proc.stderr is not None else "" raise RuntimeError( f"OpenCode preflight server returned invalid readiness JSON: {line!r}" - + (f"; logs: {stderr[:3000]}" if stderr.strip() else "") + + (f"; logs: {evidence_slice(stderr, 3000)}" if stderr.strip() else "") ) from exc url = ready.get("url") if isinstance(ready, dict) else None if not isinstance(url, str) or not url: @@ -526,7 +648,7 @@ def verify_expected_plugin( ) raise RuntimeError( f"expected plugin {expected_plugin!r} is not materialized in {plugins_root}; " - f"available entries: {available[:80]}" + f"available entries: {([] if RSP_ACTIVE else available[:80])}" ) # Agent resolution is intentionally left to the real `opencode run --agent` @@ -538,12 +660,32 @@ def verify_expected_plugin( # the plugin inventory on that same server. preflight_timeout = min(timeout, 30) started = time.monotonic() - server, base_url, authorization = _start_preflight_server( - inventory_env := dict(env), - preflight_timeout, - ) + inventory_env = dict(env) + preflight_root: Path | None = None + if env.get("EVAL_OPENCODE_STATE_PROFILE") == DISPOSABLE_STATE_PROFILE: + # Plugin activation uses Session APIs, so it must not share the + # production disposable database whose zero-session state is attested + # for the actual model run. Keep the reviewed config/plugin tree but + # give the preflight its own disposable HOME/data/cache/state roots. + preflight_parent = Path(env.get("XDG_DATA_HOME", "/tmp/runtime/data")).parent + preflight_parent.mkdir(parents=True, exist_ok=True) + preflight_root = Path(tempfile.mkdtemp(prefix="plugin-preflight-", dir=preflight_parent)) + for name in ("home", "data", "cache", "state", "config"): + (preflight_root / name).mkdir(parents=True, exist_ok=True) + inventory_env.update({ + "HOME": str(preflight_root / "home"), + "XDG_DATA_HOME": str(preflight_root / "data"), + "XDG_CACHE_HOME": str(preflight_root / "cache"), + "XDG_STATE_HOME": str(preflight_root / "state"), + "XDG_CONFIG_HOME": str(preflight_root / "config"), + }) + server = None logs = "" try: + server, base_url, authorization = _start_preflight_server( + inventory_env, + preflight_timeout, + ) remaining = lambda: max(0.5, preflight_timeout - (time.monotonic() - started)) created = _standalone_json_request( base_url, @@ -602,7 +744,7 @@ def verify_expected_plugin( ) raise RuntimeError( f"expected plugin {expected_plugin!r} is not present in OpenCode plugin inventory " - f"after activation; available plugins: {available[:120]}" + f"after activation; available plugins: {([] if RSP_ACTIVE else available[:120])}" ) state = plugin.get("state") status = state.get("status") if isinstance(state, dict) else None @@ -615,13 +757,18 @@ def verify_expected_plugin( + (f" ({ref})" if ref else "") ) except Exception as exc: - logs = _stop_preflight_server(server) + if server is not None: + logs = _stop_preflight_server(server) raise RuntimeError( str(exc) - + (f"; server logs: {logs.strip()[:4000]}" if logs.strip() else "") + + (f"; server logs: {evidence_slice(logs.strip(), 4000)}" if logs.strip() else "") ) from exc else: - logs = _stop_preflight_server(server) + if server is not None: + logs = _stop_preflight_server(server) + finally: + if preflight_root is not None: + shutil.rmtree(preflight_root, ignore_errors=True) return { "expected": expected_plugin, @@ -683,9 +830,15 @@ def invoke_opencode( reasoning: str = "", ) -> dict[str, Any]: env = prepare_opencode_env() + runtime_state = disposable_runtime_state(env, timeout) plugins = plugin_diagnostic(env) expected_plugin = os.environ.get("EVAL_EXPECT_PLUGIN", "").strip() plugin_preflight = verify_expected_plugin(env, agent, model, expected_plugin, timeout) + if runtime_state is not None: + # The activation preflight must not leave a Session or credential in the + # production disposable database. Re-attest after preflight and before + # the actual model/provider request. + runtime_state = attest_disposable_runtime_state(env) invoked_model, reasoning_label, reasoning_source = resolve_opencode_reasoning(model, reasoning) # OpenCode V2 has no documented force-refresh command for the model @@ -727,7 +880,7 @@ def invoke_opencode( ) if stderr.strip(): detail += "\n" + stderr.strip() - return { + result = { "schema": RESULT_SCHEMA, "transport": "opencode", "model": model, @@ -748,16 +901,19 @@ def invoke_opencode( "export_exit_code": None, "total_seconds": round(run_seconds, 3), }, - "stderr": detail[:STDERR_CAPTURE_LIMIT], - "stderr_truncated": len(detail) > STDERR_CAPTURE_LIMIT, + "stderr": evidence_slice(detail, STDERR_CAPTURE_LIMIT), + "stderr_truncated": False if RSP_ACTIVE else len(detail) > STDERR_CAPTURE_LIMIT, "stderr_total_chars": len(detail), - "stdout": stdout[:STDOUT_CAPTURE_LIMIT], - "stdout_truncated": len(stdout) > STDOUT_CAPTURE_LIMIT, + "stdout": evidence_slice(stdout, STDOUT_CAPTURE_LIMIT), + "stdout_truncated": False if RSP_ACTIVE else len(stdout) > STDOUT_CAPTURE_LIMIT, "stdout_total_chars": len(stdout), "tool_result_evidence": extract_tool_result_evidence(events), "plugin_diagnostic": plugins, "plugin_preflight": plugin_preflight, } + if runtime_state is not None: + result["runtime_state"] = runtime_state + return result run_seconds = time.perf_counter() - run_started events = parse_events(proc.stdout) @@ -775,7 +931,7 @@ def invoke_opencode( export_seconds = 0.0 export_exit_code: int | None = None - return { + result = { "schema": RESULT_SCHEMA, "transport": "opencode", "model": model, @@ -795,16 +951,19 @@ def invoke_opencode( "export_exit_code": export_exit_code, "total_seconds": round(run_seconds + export_seconds, 3), }, - "stderr": proc.stderr[:STDERR_CAPTURE_LIMIT], - "stderr_truncated": len(proc.stderr) > STDERR_CAPTURE_LIMIT, + "stderr": evidence_slice(proc.stderr, STDERR_CAPTURE_LIMIT), + "stderr_truncated": False if RSP_ACTIVE else len(proc.stderr) > STDERR_CAPTURE_LIMIT, "stderr_total_chars": len(proc.stderr), - "stdout": proc.stdout[:STDOUT_CAPTURE_LIMIT], - "stdout_truncated": len(proc.stdout) > STDOUT_CAPTURE_LIMIT, + "stdout": evidence_slice(proc.stdout, STDOUT_CAPTURE_LIMIT), + "stdout_truncated": False if RSP_ACTIVE else len(proc.stdout) > STDOUT_CAPTURE_LIMIT, "stdout_total_chars": len(proc.stdout), "tool_result_evidence": extract_tool_result_evidence(events), "plugin_diagnostic": plugins, "plugin_preflight": plugin_preflight, } + if runtime_state is not None: + result["runtime_state"] = runtime_state + return result def copilot_auth_source(env: dict[str, str]) -> str | None: @@ -900,21 +1059,52 @@ def invoke_copilot( "tools": [], "actions": [], "skills_loaded": [], - "stderr": proc.stderr[:STDERR_CAPTURE_LIMIT], - "stderr_truncated": len(proc.stderr) > STDERR_CAPTURE_LIMIT, + "stderr": evidence_slice(proc.stderr, STDERR_CAPTURE_LIMIT), + "stderr_truncated": False if RSP_ACTIVE else len(proc.stderr) > STDERR_CAPTURE_LIMIT, "stderr_total_chars": len(proc.stderr), - "stdout": proc.stdout[:STDOUT_CAPTURE_LIMIT], - "stdout_truncated": len(proc.stdout) > STDOUT_CAPTURE_LIMIT, + "stdout": evidence_slice(proc.stdout, STDOUT_CAPTURE_LIMIT), + "stdout_truncated": False if RSP_ACTIVE else len(proc.stdout) > STDOUT_CAPTURE_LIMIT, "stdout_total_chars": len(proc.stdout), } def emit_result(result: dict[str, Any]) -> None: - sys.stdout.write(json.dumps(result, separators=(",", ":")) + "\n") + if RSP_ACTIVE: + safety = rsp_module() + try: + projected = safety.fallback('invalid', 'runner', RSP_POLICY) if RSP_STREAM_INVALID else safety.project_result(result, RSP_POLICY) + if RSP_STREAM_INVALID: + if type(result.get('exit_code')) is int: + projected['exit_code'] = result['exit_code'] + if type(result.get('timed_out')) is bool: + projected['timed_out'] = result['timed_out'] + revision_file = Path(__file__).with_name('evidence-safety-revision.txt') + revision = revision_file.read_text().strip() if revision_file.is_file() else 'unavailable' + projected['evidence_safety_ack'] = safety.receipt(RSP_POLICY, RSP_RUN_ID, revision, RSP_BINDING_KEY) + encoded = safety.encode(projected) + if len(encoded) > safety.RESULT_LIMIT: + projected = safety.fallback('size_limit', 'runner', RSP_POLICY) + if type(result.get('exit_code')) is int: + projected['exit_code'] = result['exit_code'] + if type(result.get('timed_out')) is bool: + projected['timed_out'] = result['timed_out'] + projected['evidence_safety_ack'] = safety.receipt(RSP_POLICY, RSP_RUN_ID, revision, RSP_BINDING_KEY) + encoded = safety.encode(projected) + except Exception: + # No exception text, policy value or partial serialization may escape. + encoded = safety.encode(safety.fallback('invalid', 'runner')) + sys.stdout.write(encoded.decode('utf-8') + "\n") + else: + sys.stdout.write(json.dumps(result, separators=(",", ":")) + "\n") sys.stdout.flush() def main() -> int: + global RSP_ACTIVE, RSP_POLICY, RSP_RUN_ID, RSP_STREAM_INVALID, RSP_BINDING_KEY + RSP_ACTIVE = os.environ.get('EVAL_EVIDENCE_SAFETY') == '1' + RSP_STREAM_INVALID = False + if RSP_ACTIVE: + RSP_POLICY, RSP_RUN_ID, RSP_BINDING_KEY = rsp_module().read_request(sys.stdin.buffer) try: transport = os.environ.get("EVAL_TRANSPORT", "opencode") model = os.environ["EVAL_MODEL"] @@ -949,7 +1139,7 @@ def main() -> int: "tools": [], "actions": [], "skills_loaded": [], - "stderr": f"{type(exc).__name__}: {exc}", + "stderr": "runner_exception" if RSP_ACTIVE else f"{type(exc).__name__}: {exc}", "stdout": "", "infrastructure_error": True, } diff --git a/docs/disposable-opencode-state.md b/docs/disposable-opencode-state.md new file mode 100644 index 0000000..7690d7a --- /dev/null +++ b/docs/disposable-opencode-state.md @@ -0,0 +1,170 @@ +# Disposable OpenCode state profile + +## Purpose + +`--opencode-state-profile disposable` is an opt-in lifecycle for provider-free +composition tests that need the **real `opencode-eval-runner invoke` path** but +must not read the caller's installed OpenCode auth or database state. + +It does not replace normal invoke behavior and does not change the default image +or the default state-selection rules. + +## Exact invocation + +For the RSP composition profile: + +```bash +opencode-eval-runner invoke \ + --engine podman \ + --image '' \ + --transport opencode \ + --opencode-state-profile disposable \ + --require-evidence-safety \ + --evidence-policy-file /private/disposable/inventory.json \ + --workspace /path/to/disposable/workspace \ + --workspace-mode rw \ + --config /path/to/synthetic-provider-config.json \ + --model fixture/mock \ + --prompt-file /path/to/prompt.txt \ + --output /path/to/result.json +``` + +Loom should add the state-profile option underneath its existing: + +```text +bun run eval:live -> scripts/run-evals.py -> opencode-eval-runner invoke +``` + +No alternate `observe` entrypoint is involved. + +## Selection contract + +The profile is deliberately explicit: + +- `--database` is rejected. +- Ambient `OPENCODE_EVAL_RUNNER_AUTH`, `..._DB`, `..._CONFIG`, + `..._CONFIG_ROOT`, or `..._MODELS` overrides are rejected before container + execution. +- Installed default `auth.json`, model catalog, and database paths are not + selected. +- Default provider credential environment variables are not forwarded merely + because they exist on the host. Only names explicitly requested with `--env` + are forwarded. +- `--auth`, `--config`, `--models-catalog`, and `--config-root` remain available + as explicit synthetic inputs. The RSP policy source states must describe those + explicit selections exactly. +- No `/seed/opencode.db` mount is present. + +The ordinary `default` profile retains the historical fallback behavior. + +## Database lifecycle + +Inside the disposable container, OpenCode receives a fresh XDG tree under +`/tmp/runtime`. Before any model invocation, the runner executes a provider-free +OpenCode session-list startup against that same environment. + +The database must be absent before this startup. OpenCode then owns bootstrap: + +1. the pinned OpenCode database layer sees an empty database; +2. it creates the current schema using its generated schema bootstrap; +3. it creates the normal `migration` journal; +4. it records the pinned migration IDs itself. + +The runner does **not** create tables or fake migration-journal rows. + +For the pinned OpenCode source used by the eval image, the runner then opens the +new database read-only and attests: + +- `session_v2`, `credential`, and `migration` tables exist; +- the migration journal contains exactly 48 migrations; +- first migration is `20260127222353_familiar_lady_ursula`; +- last migration is `20260923013825_project_time_active`; +- there are zero Session rows before inference; +- there are zero credential rows before inference. + +A mismatch fails before the model/provider request. + +If `--expected-plugin` is used, its activation barrier creates a temporary +Session. Under the disposable profile that preflight runs against a **separate +temporary OpenCode HOME/data/cache/state tree** while reusing only the reviewed +config/plugin root. Its temporary state is deleted afterward. The production +disposable database is re-attested after plugin preflight and before the actual +model/provider request, so its reported zero Session/credential counts remain +true at the inference boundary. + +This avoids the unsupported hand-built-schema path where a pre-created +`session` table with no matching migration journal makes OpenCode replay the +first migration and fail with `table session already exists`. + +## Result acknowledgement + +Successful disposable execution adds: + +```json +{ + "runtime_state": { + "schema": "opencode-eval-runner/runtime-state/v1", + "profile": "disposable", + "database_source": "runtime-bootstrap", + "database_created": true, + "database_seed_present": false, + "auth_source": "none", + "session_rows_before_inference": 0, + "credential_rows_before_inference": 0, + "migration_count": 48, + "first_migration": "20260127222353_familiar_lady_ursula", + "last_migration": "20260923013825_project_time_active" + } +} +``` + +`auth_source` may be `explicit` only when an explicit synthetic `--auth` seed +was selected. + +In RSP mode `runtime_state` is a reviewed protocol field and receives an +`exact` disposition. The host safety adapter rejects disposable replies that do +not carry the matching attestation. + +## Inventory agreement + +For `source-path-roles/v1`, the disposable profile requires policy source states +to match actual runner selection: + +- `credential_seed`: `not_selected`; +- `auth`: `not_selected` unless explicit `--auth`, then `complete`; +- `config`: `not_selected` unless explicit `--config`, then `complete`; +- `models`: `not_selected` unless explicit `--models-catalog`, then `complete`; +- `config_root`: `not_selected` unless explicit `--config-root`, then `complete`. + +An incomplete policy or a contradictory source declaration is rejected before +container/model execution. The runner does not upgrade incomplete inventory by +inference. + +## Scope + +This profile proves disposable OpenCode host-state bootstrap for composition +tests. It does not establish protected-capture acceptance, same-process plugin +isolation, signing, or safety for arbitrary tool-created files. + + +## Docker and Podman image-ID compatibility + +The host preflight accepts both valid engine renderings of a local image config +ID: + +- Docker: `sha256:<64 lowercase hex>` +- Podman: `<64 lowercase hex>` (some Podman versions may also render the + Docker-style prefix) + +`evidence_load.image_config` is always canonicalized to +`sha256:<64 lowercase hex>`. + +Execution remains content-addressed: after immutable RepoDigest/label/source +validation, the runner replaces the requested image reference with the exact +config ID representation returned by that engine. It never falls back to a +mutable tag. Malformed, uppercase, truncated, overlong, or non-SHA256 IDs are +rejected. + +The evidence-safety workflow includes unit regressions for both representations +and an actual Podman preflight against the published immutable safety image. +No provider inference is used by that preflight. diff --git a/docs/execution-observer.md b/docs/execution-observer.md new file mode 100644 index 0000000..159f21d --- /dev/null +++ b/docs/execution-observer.md @@ -0,0 +1,342 @@ +# Execution observer export v1 — integration candidate + +> **Normal-invoke note (PR #41 current direction):** this document describes the older +> `--observer-key-file` / HMAC integration candidate. It is **not** the observation +> interface Loom should consume for `bun run eval:live -> runner invoke`, and its +> signing candidate does not establish protection from arbitrary same-process +> evaluated plugins. See +> [loom-normal-invoke-observation.md](loom-normal-invoke-observation.md) for the +> current diagnostic runtime contract and +> [plugin-isolation-feasibility.md](plugin-isolation-feasibility.md) for the +> remaining protection boundary. + +## Status and ownership + +This is the **runner-side export implementation and proposed producer contract** +for the Loom feature handoff `Export trustworthy Code Mode inner-call results for +evals`. It is not a claim that Loom's observer already emits this contract. + +Inspected upstream baseline: + +- Runner: `ad4d6a26fc137202e4f35f2a14503f53b81991ca`. +- Loom branch `functionality-anchor-requirements-coherance`: + `6e255092388a57f141e609fee954cb7ae5977d4b`. No published observer wire contract was + found in the inspected eval harness and source tree. +- OpenCode image from the handoff: + `ghcr.io/bateau84/opencode-eval-runner@sha256:68ef7322c75aede0e8cc76d0e3531e8b82dd417bbb5e5100264a89eab7fe8627`. + +**Loom owns observation**, including actual hook results/errors, exact actor and +session, unique invocation correlation, concurrency-safe sequencing, safe +redaction, and proof that no later hook mutates the returned value. The runner +owns authenticated transport admission, bounded parsing, completeness validation, +sanitized projection, and the evidence exit gate. No production hook, tool wrapper, +provider, or second observer is installed here. + +The PR stays draft until the Loom Worker agrees/adapts the producer contract and +the real deterministic-provider OpenCode integration tests below pass. Authentication +must not be used to paper over an unsupported observation boundary. + +## No image rebuild for this export path + +The old host runner accepts arbitrary additional mounts/environment names, but +only copies the container's outer result JSON to the final artifact. It neither +imports a sidecar nor authenticates/completeness-checks one. + +The new host runner uses that existing OCI mount/environment mechanism, imports +the sidecar **after invocation**, and adds `observed_execution` alongside existing +fields such as `observed_tool_results`. `container/invoke.py`, `Containerfile`, the +OpenCode executable, and image pins are unchanged. `Containerfile` copies only +`container/`, not `runner/`, into the runtime image. This implementation therefore +needs an updated **host checkout**, not a new image digest. It does not deploy or +claim to update an already published image. + +A future upstream hook/runtime change may still require new image bytes. That +change must publish and report its actual immutable digest separately. + +## CLI boundary + +```sh +bin/opencode-eval-runner invoke \ + --transport opencode \ + --image ghcr.io/bateau84/opencode-eval-runner@sha256:68ef7322c75aede0e8cc76d0e3531e8b82dd417bbb5e5100264a89eab7fe8627 \ + --workspace /disposable/eval-workspace \ + --model fixture/model \ + --prompt-file /disposable/prompt.txt \ + --observer-key-file /trusted-host/observer.key \ + --output /disposable/artifacts/result.json +``` + +This is an integration example, **not a working fixture provider configuration**. +The observer and deterministic provider must be supplied by the trusted harness. + +`--observer-key-file` is OpenCode-only and opts into **required** capture. The +runner creates a fresh private-lifetime capture directory, mounts it read/write at +`/eval-observer`, and supplies only these public values to the target: + +| Variable | Meaning | +| --- | --- | +| `EVAL_OBSERVER_PROTOCOL` | `1` | +| `EVAL_OBSERVER_RUN_ID` | Fresh random 256-bit nonce for this invocation | +| `EVAL_OBSERVER_PATH` | `/eval-observer/records.jsonl` | + +The corresponding path on the host is a per-run temporary directory. The trusted +observer integration may materialize the file through the mount. The directory is +not a trust boundary; target code may see, overwrite, or delete its contents. +Authenticity comes from the independent producer described below. No key path or +key bytes are added to container arguments or forwarded environment variables. + +The key must be a private, regular, single-link file with at least 32 bytes and no +more than 4096 bytes, outside **every** container mount. Do not place it in the +workspace, config root, seeds, or any extra mounted tree. Symlinks/hardlinks, +nonprivate permissions, a directly forwarded raw/hex/base64 key, and colliding +observer environment/mount options are rejected. These checks cannot discover +arbitrary copies of a key: keeping the signer and all key copies outside the +untrusted execution environment remains a deployment obligation. + +The container's original fields and semantic results are not reinterpreted. An +otherwise successful invocation exits **4** when required observer evidence is +unavailable, incomplete, unsupported, or invalid. Existing nonzero transport exit +codes are preserved. Timeout/invalid outer JSON writes a non-evidence artifact and +exits 2. This is an **eval evidence gate**, not a change to tool behavior. + +Without the flag, no observer mount/environment is added and the old invocation +exit behavior remains. The new field is explicitly unavailable with reason +`capture_not_requested`. Any same-named field supplied by the container is always +replaced: stdout cannot self-attest a trustworthy observer projection. + +## Trust boundary — a key in Code Mode is NOT authentication + +The key is shared only by the trusted producer and the host verifier. HMAC-SHA256 +is used to authenticate a stream, not to establish whether its author observed +reality. The deployment must ensure that evaluated scripts, shell tools, other +untrusted child processes, and plugins under evaluation cannot read/use the key +or ask a signer to authenticate arbitrary claimed results. Merely hiding a key +in a closure, environment variable, workspace file, or same-user sidecar is not a +proof of that isolation. + +An execution-hook observer in OpenCode needs a reviewed, protected connection to +the signing/export boundary. **This runner PR does not implement or claim that +connection exists.** If the current hook implementation cannot establish it, do +not enable positive evidence: report unsupported capture or allow missing capture +to fail closed. There is deliberately no `trust-unsigned` escape hatch. + +The trust model includes the host launcher, host verifier, and final host artifact +storage. It does not claim protection from a compromised host or a dishonest +trusted producer. A final artifact is not a standalone signed attestation; keep it +outside evaluated workloads. Do not retrofit this projection into historical +artifacts. + +## Authenticated JSONL wire contract + +Each newline-terminated frame is a JSON object with exactly two string keys: + +```json +{"payload":"","mac":""} +``` + +For frame `i`, compute: + +```text +mac_i = HMAC-SHA256( + key, + UTF8("opencode-eval-observer/v1") || 0x00 || ASCII(run_id) || 0x00 || + previous_mac_bytes || payload_bytes +) +``` + +`previous_mac_bytes` is 32 zero bytes for frame zero; subsequently it is the +previous frame's 32-byte MAC. The payload is the **exact original bytes**, not +re-serialized JSON, so producers need not share Python's number or key-order +formatting. `payload` uses standard base64. Each payload contains `run_id`, integer +`seq` starting at zero with no gaps, and `kind`. + +The nonce rejects cross-run replay; chaining and sequence validation reject +removed/reordered/duplicated records; a mandatory authenticated footer makes +interrupted suffixes incomplete. Appending anything after the footer invalidates +the capture. Signatures must cover redacted records, not raw secrets. + +### Header (`seq: 0`) + +```json +{ + "kind":"capture_start", "version":1, "run_id":"", "seq":0, + "source":"loom-execution-hook", + "boundary":"tool-return-to-caller", + "correlation":"execution-invocation-id", + "ordering":"monotonic-sequence" +} +``` + +These are producer commitments, not magic strings that grant trust. The producer +must demonstrate that `tool-return-to-caller` is the **final** native/Code Mode +return or throw boundary, independent of script return values and after any +output-changing hook. A `tool.execute.after` observation is insufficient if a +later hook can still change the result. Unknown versions/boundaries are rejected. + +### Call start + +```json +{ + "kind":"call_start", "run_id":"", "seq":1, + "invocation_id":"observer-unique-inner-1", "tool":"sentinel", + "mode":"code_mode", + "actor":{"agent":"worker","session_id":"actual-child-session"}, + "parent":{"session_id":"actual-parent-session","call_id":"execute-call"}, + "input":{"state":"available","redaction":"safe","value":{"n":42}} +} +``` + +`mode` is `native` or `code_mode`. Native calls may use `parent: null`; Code Mode +calls require a parent. Parent identity is the pair `(session_id, call_id)`. The +producer-assigned `invocation_id` must be unique **across the entire capture** and +must be carried through the actual execution context to its terminal callback. +Do not derive it by pairing tool name/input, arrival order, parent call ID, or +requested actor. All identifiers are nonempty bounded ASCII tokens (up to 256 +characters; letters, digits, `_ . : / @ -`). Unsafe/unsupported identity is not +silently guessed, normalized, clipped, or repaired. + +### Call terminal + +```json +{ + "kind":"call_end", "run_id":"", "seq":2, + "invocation_id":"observer-unique-inner-1", "outcome":"returned", + "result":{"state":"available","redaction":"safe","value":"raw sentinel"} +} +``` + +A throw uses `outcome: "threw"` and `error` instead of `result`. Exactly one terminal +is allowed per invocation. A terminal references the authenticated start binding, +not a new actor/input supplied by the script. Completion may occur in any order. + +Result values stay their observed JSON types. In particular, a string containing +JSON stays a string. JSON `null` is an actual value. Non-JSON values/JS `undefined` +must be explicitly omitted by the producer, not coerced into invented JSON. +Domain-denial JSON is still `returned`, **not** a thrown failure or a successful +domain operation. The consumer must inspect the actual value. + +### Field availability and redaction + +`input`, `result`, and `error` each use a field descriptor: + +| State | Meaning | +| --- | --- | +| `available` | Exact representable JSON value with authenticated `redaction: "safe"` | +| `redacted` | Safe value remains but it is no longer exact original evidence | +| `omitted` | No trustworthy/safely serializable value is available | +| `truncated` | The complete value was not retained | + +The producer must establish safe redaction **before persistence/signing**. A +missing safe-redaction claim causes omission, with no preview or raw reason text. +The runner additionally scrubs known host secrets (including encoded verifier +key forms) and sensitive structured keys recursively, before checking field size. +It does not pretend that generic regexes can discover every unknown secret in +free text; that is why authenticated producer redaction is mandatory. Oversized +fields are marked truncated with no misleading partial value. Producer-supplied +omission reasons/previews are not copied. Redacted, missing, or truncated fields +make this v1 capture ineligible for positive deterministic assertions. + +### Footer + +```json +{ + "kind":"capture_end", "run_id":"", "seq":3, + "calls_started":1, "calls_ended":1, + "omitted_records":0, "truncated":false, "unsupported":[] +} +``` + +Counts must match the records actually emitted. The producer must set omissions, +truncation, and unsupported coverage truthfully and must not sign a complete +footer while any invocation/flush is still pending. `unsupported` contains bounded +non-sensitive reason identifiers. A missing footer, missing terminal, omitted +record, unsupported segment, or truncated field/capture cannot be evidence for +PASS. Merely having some valid calls does not make a partial capture complete. + +Limits: 8 MiB total capture, 256 KiB per frame, 10,001 frames, and 16 KiB per +sanitized field. Limit breaches explicitly reject/mark the capture; no clipped +prefix is promoted to complete evidence. Duplicate JSON keys, non-finite numbers, +invalid encoding, unsupported record shapes, and unsafe filesystem objects fail +closed. The importer reads only its fresh capture path, never a Loom database. + +## Exported projection and consumer rule + +```json +{ + "observed_execution": { + "kind":"execution-observer-projection", "version":1, + "run_id":"", "status":"complete", "evidence_eligible":true, + "records":[{ + "invocation_id":"observer-unique-inner-1", "tool":"sentinel", + "actor":{"agent":"worker","session_id":"actual-child-session"}, + "parent":{"session_id":"actual-parent-session","call_id":"execute-call"}, + "mode":"code_mode", "start_sequence":1, "terminal_sequence":2, + "input":{"state":"available","value":{"n":42}}, + "outcome":"returned", "result":{"state":"available","value":"raw sentinel"}, + "evidence_eligible":true + }], + "issues":[], + "coverage":{ + "capture_started":true, "capture_ended":true, + "observed_starts":1, "observed_terminals":1, "missing_terminals":0, + "omitted_records":0, "truncated":false, "unsupported":[] + } + } +} +``` + +Records are in **start order**. `terminal_sequence` captures completion order; +there is no claim that overlapping calls executed serially. Missing terminals +use `outcome: "missing"` and no result/error. Valid authenticated prefixes may be +retained for diagnosis, but every record remains ineligible when the overall +capture is incomplete. Invalid/authentication-failed captures export no records. + +The consumer must require version 1, `status == "complete"`, +`evidence_eligible is true`, no issues, complete coverage, and eligible matching +records **before** asserting the requested tool outcome. Eligibility is necessary, +not sufficient for PASS: an empty capture does not satisfy an expected call; a +completed/returned call does not imply domain success. Never fall back to execute +source, parent output, CLI status, or old metadata when inner evidence is absent. +The Loom consumer still needs to adopt this rule; this PR does not silently change +Loom's existing judge or historical artifacts. + +## Verification and acceptance gaps + +Run the provider-free runner tests with: + +```sh +python -m unittest discover -s tests -p 'test_observer.py' -v +``` + +The test-only producer independently encodes authenticated frames. Real local +subprocesses emulate OCI transport and exercise the actual host CLI, mounts, +environment forwarding, result-file export, and exit codes. Tests use disposable +HOME/XDG/workspace/key state, no real provider, and no installation-wide database. +The fake engine's access to a fixture-only signing key is **not evidence of signer +isolation in OpenCode**. + +Covered exporter properties include native/nested-shaped sentinel preservation, +script-controlled outer-output independence, denial strings vs throws, identical +inputs/shared parents/reverse completion, no guessed pairing, missing footer and +terminal, replay/forgery/reordering, malformed input, symlinks/FIFOs/hardlinks, +redaction before clipping, explicit unsafe-field omission, fresh nonces, unchanged +product arguments/results, key non-forwarding, and nonzero indeterminate gates. + +Before marking CAP-AC-001 through CAP-AC-004 end-to-end complete, Loom must supply: + +1. Its real observer implementation/revision and the agreed/adapted wire contract, + with a protected producer/signing boundary. No duplicate observer should be + added to this repository. +2. An actual supported execution-context identity shared by start/end callbacks, + including overlapping identical calls and actual subagent session/actor. +3. Proof that the hook sees the final returned value or thrown error, not a value + subsequently changed by another plugin or the Code Mode runtime. +4. A deterministic provider fixture running OpenCode 2.0.18 native and Code Mode + tools: discard/transform returns, catch exceptions, return denial JSON, + reverse completion order, and inject interrupted/forged capture. Run with + observer on/off and verify unchanged tool behavior/permissions. Consume the + exported projection through Loom's real deterministic assertion path. + +Runner fixture tests support the export contract. They do **not** discharge those +real-runtime capture, signer-isolation, or Loom-consumer acceptance obligations. +No live provider run or rebuilt/published image is claimed. diff --git a/docs/loom-host-semantics-feasibility.md b/docs/loom-host-semantics-feasibility.md new file mode 100644 index 0000000..b77a5c5 --- /dev/null +++ b/docs/loom-host-semantics-feasibility.md @@ -0,0 +1,217 @@ +# Loom host-semantics capture feasibility + +## Decision + +The protected-capture work must sit **under Loom's existing eval entrypoint**: + +```text +bun run eval:live ... + -> python3 scripts/run-evals.py + -> opencode-eval-runner invoke + -> normal OpenCode agent/session/tool execution +``` + +PR #41 must not require Loom to replace that path with `observe`, a synthetic +session loop, or a remote-tool reconstruction of normal host behavior. Existing +case selection, prompts, agents, permissions, workspace setup, retries, judges, +artifacts, and pass/fail behavior remain owned by Loom's current harness. + +The restricted `observe` / `codemode-inner/direct-session/v1` profile remains a +supplemental security experiment only. It is not the migration target for normal +Loom evals. + +## Feasibility result + +There are two separate questions. + +### 1. Can the real runtime expose the required semantics without replacing them? + +**Yes, for the core tool boundary.** + +OpenCode 2.0.18 already routes normal direct/native tools through the central +Tool service. The same service is used when a real child Session executes tools. +Its context contains the actual executing session, agent, assistant message and +runtime call ID. Code Mode inner calls ultimately dispatch through that same Tool +service, while the Code Mode interpreter owns the final JSON value/error seen by +the script. + +The downstream runtime patch therefore keeps normal execution intact and exposes +two diagnostic hook streams: + +- `tool.execute.native-observed` — direct/native executable-input starts plus + final Session-owned success/failure terminals, including the outer `execute` + call. The terminal is emitted after `ToolOutput.truncate` and successful + Session event publication. +- `tool.execute.observed` — Code Mode inner starts and the final value/error + delivered to the Code Mode caller. + +Both use one process-wide monotonic sequence. When host-semantics observation is +enabled, the actual outer native `execute` invocation ID is bound to its inner +Code Mode records without adding metadata to tool inputs or public Tool.Context. + +The experimental normal-invoke image enables those seams but otherwise uses the +same `opencode-eval-runner invoke` entrypoint and normal OpenCode configuration. +It does **not** apply the restricted profile's removal of Code Mode networking or +remote-tool substitution. + +These hook records remain diagnostic until a protected collection path exists. + +### 2. Can those observations be protected from the current in-process Loom plugin +without changing Loom's host semantics? + +**Not with the current OpenCode plugin isolation model.** + +The evaluated Loom plugin and the observation seam execute in the same OpenCode +process. An arbitrary in-process plugin therefore shares the process/OS authority +needed to reach same-process files, descriptors, sockets, environment, libraries, +or a signing/collector API made available to that process. + +A hidden path, random file descriptor, same-process HMAC key, localhost socket, +closure-only signer, or second hook does not create a supported security boundary +between two pieces of code with the same process authority. + +Moving Loom tool closures behind `/call` protects the collector only by replacing +the host environment those tools normally depend on. That is useful as a restricted +smoke, but it cannot prove normal Loom session, permission, grant, OQ, cancellation, +gate or continuation behavior. + +Therefore the smallest honest conclusion is: + +> semantic observation under normal `invoke` is feasible in PR #41; a +> target-unforgeable collector while arbitrary Loom plugin code remains in-process +> requires a broader OpenCode plugin-isolation/capability project. + +PR #41 must not silently build that broader platform. + +## Supported / unsupported matrix + +| Capability | Normal-invoke runtime seam | Protected positive evidence | +| --- | --- | --- | +| Existing `bun run eval:live` case behavior | **Preserved** — same runner `invoke` contract | N/A | +| Direct/native dispatched input | **Supported diagnostic seam** | **Blocked** for in-process plugin | +| Direct/native final Tool result/error | **Supported diagnostic seam** | **Blocked** for in-process plugin | +| Code Mode exact dispatched input | **Supported diagnostic seam** | Restricted profile already proves a subset | +| Code Mode final caller value/error | **Supported diagnostic seam** | Restricted profile already proves a subset | +| Genuine agent/session/message/call identity | **From actual Tool.Context** | Protection blocked in normal in-process profile | +| Unique child invocation correlation | **Runtime-generated IDs** | Protection blocked in normal in-process profile | +| Separate start/completion ordering | **Shared runtime sequence** | Protection blocked in normal in-process profile | +| Inner -> outer `execute` parent binding | **Runtime-bound, not FIFO/input inference** | Protection blocked in normal in-process profile | +| Delegated foreground child Session identity | **Exercised through real built-in subagent** | Protection blocked in normal in-process profile | +| Child Session ancestry / parentID | **Verified by querying the actual child Session in the delegated probe** | Protection blocked in normal in-process profile | +| Background delegation / cancellation / OQ lifecycle | **Not yet exercised by this focused runner probe** | Unsupported | +| Normal permission/grant/plugin behavior | **Not replaced by the observation patch** | Loom composition proof still required | +| Cancellation/lifecycle/OQ/gate continuation | **Normal host path retained** | Loom composition proof still required | +| Collector unforgeable by arbitrary in-process Loom plugin | No | **Unsupported** | +| Native/delegated full handoff acceptance | No | **Open** | + +Only demonstrated coverage may become eligible evidence. + +## Exact Loom-facing interface + +### Entrypoint + +No new Loom-facing command is required. + +```bash +OPENCODE_EVAL_RUNNER_OPENCODE_IMAGE="" \ + bun run eval:live -- --model +``` + +Loom continues to construct cases and call: + +```text +opencode-eval-runner invoke ... +``` + +with its existing arguments. The runner/image implementation changes underneath +that contract. + +### Diagnostic runtime hooks + +A Loom feasibility observer may subscribe to both hook streams: + +```ts +await ctx.tool.hook("execute.native-observed", event => { /* diagnostic only */ }) +await ctx.tool.hook("execute.observed", event => { /* diagnostic only */ }) +``` + +`execute.native-observed` uses schema `opencode-native-observation/v2`. +`execute.observed` retains `opencode-local-observation/v1`. + +The two streams share the runtime sequence counter. Native records carry +`mode: "native"`, actual `call_id`, actor identity, a unique invocation ID and +`parent: null`. Code Mode inner records retain the existing actual actor and +parent structure; when the normal host observation mode is active their parent +invocation ID is the actual observed outer `execute` invocation. + +Neither hook is an authenticated producer. Loom must not promote these diagnostic +records into PASS evidence merely because their shape validates. + +### Result compatibility + +The normal runner result contract remains the existing `invoke` result. Existing +eval cases and judge behavior must not require `observed_execution` to be present +or eligible unless a case is explicitly migrated after the protected normal-host +path exists. + +The normal-invoke compatibility probe verifies the public runner CLI with a +deterministic local provider, not a second eval entrypoint. + +## Broader-project boundary + +To make normal in-process Loom observations target-unforgeable, OpenCode needs an +actual isolation boundary between evaluated plugin code and the collector +capability. Examples that could be researched separately include a process-isolated +plugin host with a narrow capability broker or another runtime-supported sandbox +that preserves every OpenCode API Loom requires. + +That project must prove permission, session, cancellation, child/delegation, +continuation and lifecycle equivalence. It must not be introduced implicitly in +this exporter PR. + +Until such a boundary exists, the runner should report normal-host observation +protection as unsupported rather than signing or authenticating same-process claims. + +## Loom composition work after a compatible protected path exists + +Loom then owns only the adaptation above its stable entrypoint: + +1. pin the runner image/revision and the committed observer/parser/smoke checkpoint; +2. consume the agreed versioned projection without falling back to stdout/source; +3. run the existing eval cases unchanged through `bun run eval:live`; +4. add focused assertions for native, inner and delegated observations; +5. preserve disposable HOME/XDG/workspace state and avoid the installation-wide DB; +6. keep unsupported coverage non-evidence. + +The handoff identifies Loom checkpoint `f8439e4`. PR #41 does not rewrite those +tests or historical results. + +## Contract erratum + +The historical `c1629415...` restricted-profile proof used runtime events +`opencode-local-observation/v1`, wire `opencode-protected-observation/v2`, and +projection version **3**. Any text describing that implementation as wire v1 / +projection v2 is stale documentation, not the tested contract. Historical +artifacts keep their original bytes and meaning. + +## Current verification rule + +The reviewed source commit and its immutable image digest must come from the same +completed **Normal-invoke runtime observation seam** workflow. Because the image +records the runner revision in OCI metadata, a source commit after that build +requires a newly published digest before the pair can be claimed as matching. + +The compatibility workflow must enter through the public +`opencode-eval-runner invoke` CLI and prove, provider-free, that: + +- ordinary result text and tool outcomes remain intact; +- native success is observed after session truncation/publication; +- native failure is the canonical Session failure representation; +- inner Code Mode values/errors remain final caller values; +- inner parent identity matches the actual outer `execute` observation; +- native/inner ordering remains truthful. + +Runner fixture evidence does not replace Loom checkpoint `f8439e4` composition. + +See [the exact normal-invoke interface](loom-normal-invoke-observation.md) and +[the bounded plugin-isolation feasibility proposal](plugin-isolation-feasibility.md). diff --git a/docs/loom-normal-invoke-observation.md b/docs/loom-normal-invoke-observation.md new file mode 100644 index 0000000..0e01c25 --- /dev/null +++ b/docs/loom-normal-invoke-observation.md @@ -0,0 +1,395 @@ +# Loom normal-invoke observation interface + +## Status + +This is the exact **diagnostic runtime interface** supplied underneath Loom's +existing live-eval path: + +```text +bun run eval:live ... + -> scripts/run-evals.py + -> opencode-eval-runner invoke + -> normal OpenCode/Loom execution +``` + +It does not replace `invoke` with `observe`, does not move Loom tools behind the +restricted remote-tool service, and does not create positive protected evidence. + +Two different claims must remain separate: + +1. **Trusted installation observation:** a reviewed OpenCode runtime and reviewed + installed plugin can receive truthful runtime-owned observation events. +2. **Protection from arbitrary evaluated in-process plugins:** not supported. + An arbitrary plugin in the same OpenCode process shares process authority with + the observation callbacks and any same-process collector/signing capability. + +Signing an artifact does not turn claim 1 into claim 2. + +## Enablement + +The reviewed experimental normal-invoke image enables both seams: + +```text +OPENCODE_EVAL_HOST_OBSERVATIONS=1 +OPENCODE_EVAL_OBSERVATIONS=1 +``` + +- `OPENCODE_EVAL_HOST_OBSERVATIONS=1` enables native/direct starts and final + session terminals. +- `OPENCODE_EVAL_OBSERVATIONS=1` enables Code Mode inner observations. + +The flags expose events only. They do not authorize evidence eligibility. + +Loom does not need a new CLI verb. With a reviewed runner checkout and immutable +image: + +```bash +export OPENCODE_EVAL_RUNNER_BIN=/path/to/opencode-eval-runner/bin/opencode-eval-runner + +bun run eval:live -- \ + --opencode-image 'ghcr.io/bateau84/opencode-eval-runner@sha256:' \ + --model '' \ + +``` + +Existing case selection, runtime projects, agents, permissions, retries, judge +execution, artifacts, and result classification remain Loom-owned behavior. + +## Hook registrations + +A reviewed installed plugin may subscribe to: + +```ts +await ctx.tool.hook("execute.native-observed", event => { + // diagnostic runtime event; not authenticated evidence +}) + +await ctx.tool.hook("execute.observed", event => { + // diagnostic Code Mode inner event; not authenticated evidence +}) +``` + +The callbacks are ordinary in-process plugin hooks. They are not a security +boundary. A callback failure is suppressed by the observation seam; a callback +that never returns can still delay execution and therefore remains unsupported +for a strong noninterference claim. + +## Shared ordering + +Both event families allocate from one process-wide monotonic `sequence`. + +Sequence is assigned at the observation boundary before awaiting the plugin +callback. It expresses observed boundary order, not serial execution. + +For example: + +```text +A.call_end.sequence < B.call_start.sequence +``` + +can establish that B's observed start followed A's observed completion. Merely +sorting starts or terminals cannot establish that relation for overlapping calls. + +## Native/direct events — `opencode-native-observation/v2` + +### Start + +A native start is emitted only after: + +1. normal `tool.execute.before` processing has selected the executed tool/input; +2. the tool input has decoded successfully; +3. the tool is about to execute. + +Shape: + +```json +{ + "schema": "opencode-native-observation/v2", + "sequence": 1, + "actor": { + "agent": "actual-executing-agent", + "session_id": "actual-session", + "message_id": "actual-assistant-message" + }, + "observer_failures": 0, + "kind": "call_start", + "invocation_id": "runtime-observation-uuid", + "call_id": "actual-provider-tool-call-id", + "tool": "actual_resolved_registration", + "mode": "native", + "parent": null, + "input": { + "state": "available", + "value": {} + }, + "boundary": "executable-input" +} +``` + +### Identity + +- `tool` is the resolved runtime registration actually dispatched after normal + request-definition/repair processing. +- `actor.agent`, `session_id`, and `message_id` come from the real + `Tool.Context`. +- `call_id` is the actual provider/runtime ToolCall ID. +- `invocation_id` is a separate observation identity. It is not substituted + into tool input or public Tool.Context. +- `parent: null` means no enclosing Code Mode invocation. It does **not** state + that the Session has no parent Session. + +### Successful terminal + +The terminal is emitted only **after** the normal session writer has: + +1. applied `ToolOutput.truncate`; +2. constructed the canonical `SessionEvent.Tool.Success`; +3. successfully published that durable session event. + +Shape: + +```json +{ + "schema": "opencode-native-observation/v2", + "sequence": 9, + "actor": { "...": "same start binding" }, + "observer_failures": 0, + "kind": "call_end", + "invocation_id": "same-runtime-observation-uuid", + "call_id": "same-tool-call-id", + "boundary": "session-tool-terminal", + "unavailable_fields": 0, + "outcome": "returned", + "result": { + "state": "available", + "value": { + "content": [], + "metadata": {}, + "executed": false + } + } +} +``` + +The `result.value` payload is the canonical session result delivered through +`SessionEvent.Tool.Success`, excluding duplicate identity fields. If output was +truncated, the observed content/metadata are the **truncated session form**, not +the larger pre-truncation Tool result. + +### Failed terminal + +The failed terminal is emitted only after the canonical +`SessionEvent.Tool.Failed` has been published. + +```json +{ + "schema": "opencode-native-observation/v2", + "sequence": 10, + "actor": { "...": "same start binding" }, + "observer_failures": 0, + "kind": "call_end", + "invocation_id": "same-runtime-observation-uuid", + "call_id": "same-tool-call-id", + "boundary": "session-tool-terminal", + "unavailable_fields": 0, + "outcome": "threw", + "error_representation": "session-tool-failed/v1", + "error": { + "state": "available", + "value": { + "error": { + "type": "pinned-runtime-session-error-type", + "message": "canonical session error" + }, + "metadata": {}, + "executed": false + } + } +} +``` + +The representation preserves the actual pinned-runtime `SessionError` result; +it does not normalize error types to a preferred vocabulary. + +A decode failure before executable-input admission produces no native start and +must not manufacture a terminal-only record. + +## Code Mode inner events — `opencode-local-observation/v1` + +The Code Mode seam retains its existing schema. + +### Parent + +One Code Mode engine invocation emits `parent_start` and `parent_end`. +When host observations are enabled, the parent binding uses the actual observed +outer native `execute` invocation ID: + +```json +{ + "parent": { + "invocation_id": "outer-native-observation-id", + "session_id": "actual-session", + "message_id": "actual-assistant-message", + "call_id": "actual-outer-execute-call-id" + } +} +``` + +This is runtime-carried binding, not FIFO, input equality, or a reused parent +call ID. + +### Inner start + +`call_start` contains: + +- runtime-generated inner `invocation_id`; +- `tool`: actual internal registration selected for execution; +- `catalog_path`: exact Code Mode callable path; +- the same actual actor/session/message binding as its execution context; +- the actual decoded executable input after normal before-hook changes; +- `boundary: "executable-input"`. + +### Inner terminal + +`call_end` uses the same inner invocation ID. + +- `outcome: "returned"` with `result` is the final JSON value delivered to + the Code Mode caller after the interpreter conversion boundary. +- `outcome: "threw"` uses + `error_representation: "codemode-catch-name-message/v1"`, matching the + name/message view received by Code Mode catch handling. +- `outcome: "interrupted"` has no invented result/error. + +Script-controlled outer return text never supplies an inner value. + +## Snapshot/value semantics + +Both seams create owned snapshots rather than retaining mutable tool objects. + +Supported exact values are JSON-compatible values composed from: + +- null; +- booleans; +- finite numbers; +- strings; +- dense arrays; +- plain data objects. + +Snapshots reject cycles, accessors, symbols, sparse arrays, non-finite numbers, +non-data properties, `toJSON` behavior, and non-plain objects. Rejected values +become: + +```json +{"state":"omitted","reason":"unsupported_snapshot"} +``` + +No value is silently converted to a different JSON type. + +## Redaction + +**The normal-invoke runtime seams do not implement secret redaction.** + +They emit owned in-memory snapshots to installed plugin callbacks. They do not +persist those snapshots themselves. + +A trusted installed observer can apply reviewed redaction before persistence for +diagnostic use. That does not establish protection from an arbitrary evaluated +plugin sharing the same process: such a plugin can subscribe to the same hooks or +reach same-process collection/signing authority. + +Therefore raw normal-invoke events are not eligible evidence and should not be +persisted as general live-eval artifacts until the redaction/trust boundary is +resolved. + +## Completeness + +The current normal-invoke interface does **not** prove run-wide completeness. + +- Native v2 pairs starts/terminals by invocation ID when both are observed, but + has no trusted run header/footer/drain fence. +- Code Mode v1 `parent_end` supplies per-`execute` admitted/dispatched/terminal + accounting only. +- `observer_failures` reports callback failures visible to the runtime, but + cannot prove that the tail of a process-level stream was not lost. +- There is no protected omission count covering native calls, delegated Sessions, + callback queues, process interruption, or final collector drain. + +Consequently no normal-invoke event stream may claim complete positive evidence. + +## Host projection + +There is currently **no protected normal-invoke projection** for these two hook +families. + +The runner's existing `observed_execution` field belongs to the older +observer/import candidate (`--observer-key-file`) or reports capture unavailable. +It must not be silently reinterpreted as a projection of +`execute.native-observed` / `execute.observed`. + +Until a protected normal-host collector exists: + +- hook events are diagnostic only; +- `evidence_eligible` for this normal-host path is effectively false; +- Loom must not backfill missing fields from stdout, script output, requested + actor names, old metadata, or source text. + +A future protected collector must introduce an explicitly versioned projection +contract and consumer migration rather than relabeling an existing version. + +## Supported / unsupported coverage + +| Capability | Status | +| --- | --- | +| Existing `eval:live -> run-evals.py -> invoke` behavior | supported / regression-tested | +| Native executable input | diagnostic support | +| Native final session success payload | diagnostic support | +| Native final session error payload | diagnostic support | +| Code Mode executable input | diagnostic support | +| Code Mode final caller value/error | diagnostic support | +| Actual direct-session actor/session/message/call identity | diagnostic support | +| Runtime outer-execute -> inner parent binding | diagnostic support | +| Shared native/inner ordering | diagnostic support | +| Real delegated foreground child-session identity/ancestry | exercised provider-free through built-in `subagent` | +| Child Session parentID | verified in the delegated probe by querying the actual child Session; not emitted as a normal observation field | +| Provider-hosted tool results | outside this local-execution seam | +| Run-wide completeness | unsupported | +| General live-eval redaction | unsupported | +| Observer callback hang noninterference | unsupported | +| Protection from arbitrary in-process evaluated plugins | unsupported | +| Normal-host positive evidence projection | unsupported | +| Cosign image/result authentication | manual default-branch rebuild/sign workflow implemented separately; final signing requires protected `release-signing` environment approval | +| Loom checkpoint `f8439e4` composition | Loom session reported provider-free composition PASS; runner fixtures remain separate evidence | + +## Trust statement + +A reviewed runtime plus a reviewed installed observer can be trusted *by policy* +to report these runtime facts. That is a deployment trust decision. + +It is categorically different from proving that arbitrary evaluated plugin code +cannot fabricate, suppress, or interfere with observations while sharing the same +process. PR #41 does not claim the latter. + + +## Delegated-session composition checkpoint + +The normal-invoke workflow includes a provider-free test using OpenCode's real +built-in foreground `subagent` tool. It verifies: + +- the parent `subagent` invocation is observed under the actual parent actor and + Session; +- the child provider request runs as the configured `reviewer` agent; +- the child tool observation carries the child Session ID; +- the child Session ID returned by the real `subagent` result matches the child + observation; +- the actual child Session object's `parentID` matches the parent Session; +- child tool completion precedes the foreground `subagent` terminal; +- a resource-scoped permission denial for `subagent/reviewer` prevents any child + Session request or child-tool observation. + +This closes the focused foreground identity/ancestry/permission proof. Background +delivery, cancellation, and OQ-specific lifecycle behavior remain separate Loom +composition obligations and are not inferred from this test. + +See [Cosign provenance](signing-provenance.md) for the manual approved-rebuild signing path and +[plugin isolation feasibility](plugin-isolation-feasibility.md) for the stronger +same-process trust boundary. diff --git a/docs/loom-protected-channel-handoff.md b/docs/loom-protected-channel-handoff.md new file mode 100644 index 0000000..c7f56c5 --- /dev/null +++ b/docs/loom-protected-channel-handoff.md @@ -0,0 +1,206 @@ +# Loom handoff: isolated capture profile + +> **Superseding compatibility boundary:** Loom's external entrypoint remains +> `bun run eval:live ...`, with existing case behavior and +> `scripts/run-evals.py -> opencode-eval-runner invoke` unchanged. Runner/runtime +> work belongs underneath that contract. The `observe` command and remote-tool +> profile below are supplemental restricted-smoke machinery, **not** the migration +> path for normal Loom agent/session evals. See +> [host-semantics feasibility](loom-host-semantics-feasibility.md). + +## Scope of this change + +The runner owns the isolated launch, private transport, collector adapter, import +validation and finalized host artifact. Its downstream OpenCode patch owns the +existing interpreter's final inner-return observation point. No second interpreter +or synthetic inner-result extractor is introduced. + +The working profile is **`codemode-inner/direct-session/v1`**, exposed by the +runner's **`observe`** command. It is a provider-free, direct-session profile for +Code Mode programs calling isolated remote tools. The current result projection +is **`observed_execution.version = 5`**, with +`collection_profile = "private-supervisor-receipt/v1"`. + +This is not a drop-in protected replacement for Loom's normal `invoke` path. +Arbitrary evaluated plugins cannot share the trusted OpenCode/collector process. +Native final results, delegated sessions and full Loom acceptance remain open. +The scope restriction is the same with observation on and off; it is not a claim +that restricting the profile leaves an unrestricted product environment unchanged. + +## Runner-side boundary + +The launcher snapshots program, tool definitions and redaction policy once. Its +launch record binds their exact hashes, the requested immutable runtime/tool +image digests, Docker's resolved image config digests, the fresh run ID, profile, +and host collector source hashes. Docker's actual started containers are checked +against that launch record. Input originals are never reopened for execution. + +Only the trusted runtime receives private `/input` and `/capture` mounts. The +evaluated tool server gets neither, no host workspace or Docker socket, and no +shared PID namespace. The runtime loads only its image-owned adapter. Code Mode +has no direct network extension in this profile; external calls go through that +adapter. Remote redirects are rejected, including the tool readiness endpoint. +Both containers are non-root with read-only roots, bounded resources, dropped +capabilities, and no retained raw container logs. The provider binds only loopback. + +The adapter records existing runtime events and redacts before writing. After the +runtime writer exits, the supervisor seals the file and returns a receipt +`{sha256, bytes}` over its separate, trusted control output. The host requires that +receipt as well as the file's header/footer, exact run/launch/policy bindings, +source/wire sequencing, real invocation IDs, parent identity and complete counts. +Recomputing the file's own checksum cannot replace the control receipt. + +**No HMAC or signing service is used by this connection.** A checksum alone is not +authentication. The host launcher, selected trusted runtime image, immutable +adapter, and Docker/kernel isolation protect both paths. A compromised host, +collector runtime or container escape is outside this claim. Calling a parser +with invented bytes and an invented receipt does not reproduce the protected +launcher. The final JSON is a host artifact, not a portable signed attestation. + +## Required Loom work + +The next normal-host step is **not** to move Loom behind the remote tool service. +First preserve the existing eval entrypoint and wait for a runner/runtime path +whose protection does not replace the in-process host semantics. Loom may use the +restricted profile below only as a supplemental smoke. + +### 1. Preserve and pin the actual producer and smoke + +Commit or otherwise supply the exact bytes of Loom's existing +`scripts/fixtures/eval-tool-observer.ts`, +`scripts/fixtures/eval-observer-smoke-tools.ts`, and +`scripts/test_eval_observer_image.py`, plus its assertion consumer and any imports. +The previously supplied Architect draft explicitly said those prototype files +were uncommitted. A surrounding Loom HEAD does not pin their contents. + +Keep that original smoke and its past artifacts intact. Add a separate integration +variant for this profile; do not relabel a runner fixture run as the original +Loom smoke, backfill old results, or reconstruct missing records from script output. + +### 2. Supplemental restricted-profile pilot only: move evaluated code out of collector authority + +This section does **not** define the normal `eval:live` migration. For a +fixture/tool-level supplemental pilot only, package the actual evaluated tool implementations +as an isolated tool-server image. It must serve `GET /health` and `POST /call` on +port 8080. The request is: + +```json +{ + "name": "tool_name", + "input": {}, + "context": { + "invocation_id": "runtime-allocated-id", + "ordinal": 0, + "agent": "actual-agent", + "session_id": "actual-session", + "message_id": "actual-message", + "call_id": "enclosing-runtime-call" + } +} +``` + +Returned data uses `{"outcome":"returned","result":}`; a remote failure +uses `{"outcome":"threw","error":"message"}`. The same code path and permissions +must be exercised in the pilot with observation on and off. Keep tool business +logic in its existing owner; do not write separate toy replacements to claim Loom +integration. The runtime/adapter chooses execution identity; identity echoed by +the untrusted server is never admitted as an observation. + +The registered runtime name is `isolated_` and its catalog path is +`isolated.`. Do not rename these records as native `loom_*` invocations. +The error view is explicitly `codemode-catch-name-message/v1`, not preservation of +arbitrary JavaScript exception identity, stack or cause. + +**For Loom's whole in-process plugin, this is architecture work, not script glue.** +Loom operations that require OpenCode plugin/session APIs need an explicit, +restricted broker before they can run in the tool container. This runner does not +provide a generic session API tunnel. Do not mount the collector state, move the +observer into the tool container, expose its callbacks, or grant a signing oracle +to make the plugin work. The whole-plugin/delegation profile remains unsupported +until that separate interface and its identity/permission semantics are reviewed. + +### 3. Supply reviewed redaction policy + +Materialize host-selected `tools.json` and `policy.json`. Policy v1 contains +`secrets` and `allowed_values`. The latter permits retention of exact safe values +for deterministic fixtures; it never substitutes expected values for observations. +A real unexpected return remains omitted/ineligible, not replaced with a sentinel. +Include both positive and denial/error outcomes where safe. + +Unknown free text is not automatically declared safe. The collector scrubs known +literal/encoded secrets and sensitive structured keys before persistence/limits. +Changed fields remain `redacted`; unsupported, unknown and oversized fields cannot +be exact evidence. A general Loom redactor must be explicitly reviewed and placed +on the trusted side, reusing Loom's owned logic where appropriate. Do not forward +raw Loom observer files and assume host-side scrubbing undoes prior disclosure. + +### 4. Call the host command and enforce the new admission rule + +Use tested immutable digests from the completed **Protected runtime channel** +workflow's `protected-image.txt` and `fixture-tools-image.txt` (the latter is a +runner fixture only; substitute Loom's independently pinned actual tool image). +Record the runner/producer/consumer revisions and launch hashes. + +```sh +bin/opencode-eval-runner observe \ + --image "$VERIFIED_PROTECTED_RUNTIME_DIGEST" \ + --tool-image "$PINNED_LOOM_TOOL_IMAGE_DIGEST" \ + --program-file program.js \ + --tools-file tools.json \ + --policy-file policy.json \ + --output results.json +``` + +The consumer must explicitly support projection version 5 and this profile. Require +matching expected run ID, launch ID, image/input/policy hashes, complete coverage, +no issues and eligible matching records before checking an outcome. Distinguish +`returned` from domain success. Interpret JSON-looking result strings only through +an explicit predicate; preserve their recorded type. The shared `runtime_call_id` +is not the unique child ID: use `invocation_id` and the full parent identity. + +An eligible restricted capture still has `full_handoff_eligible: false` and explicit +native/delegated/in-process-plugin exclusions. A case requiring any excluded +surface must be BLOCKED/unsupported, not PASS. No expected call can pass on an empty +record set. Required incomplete capture exits nonzero. Never recover eligibility +from prose, outer script results, old metadata, or model claims. + +### 5. Run the preserved integration composition + +Run the real producer → patched image → runner host CLI/importer → actual Loom +predicates. Test overlapping identical calls with reversed completion, actual +actor/input/parent bindings, discarded returns, caught errors, denial data/null, +replayed/edited/deleted/incomplete capture, and I/O failure with unchanged outcomes. +Compare observation on/off in the same restricted profile. Preserve new and altered +fault-injection captures separately. Run with disposable HOME/XDG/workspace state +and local deterministic provider only, no installation-wide database or paid model. + +Runner tests and independent Loom composition are separate evidence. Independent +code/security review remains outstanding; no merge or full acceptance follows +from CI or the fixture proof alone. + +## Signing remains separate + +The result explicitly reports `image_signatures_verified: false` and +`artifact_signed: false` until signing and identity verification are implemented. +Cosign should authenticate the image from the expected approved build workflow +and the finalized host JSON from the expected results signer. The trusted verifier +policy must name those identities; target JSON does not choose trusted signers. +Keep signing/OIDC authority outside both evaluated code and arbitrary artifact +submission. Signed eval input files must be verified before executing the same +snapshot. These steps complement, not replace, the protected collection path. + + +### Importer review clarifications + +Consumers must not treat diagnostic prefix counts as evidence: require complete, +eligible capture and validated accounting. `coverage.accounting_complete` is false +on invalid/unfinished accounting; `missing_terminals` may be null when unknown. +The host now rejects inherited image volumes before launch and enforces its 16 KiB +sanitized-value limit. These fixes do not add native or delegated-session support, +turn the restricted receipt profile into full Loom proof, or supply signing. + + +Projection version 5 makes the accounting change explicit; consumers written for +version 4 must not silently accept it. The runtime-event and wire schema versions +and the receipt collection profile are unchanged. Historical v4 artifacts retain +their original version and meaning; a new host checkout does not rewrite them. diff --git a/docs/plugin-isolation-feasibility.md b/docs/plugin-isolation-feasibility.md new file mode 100644 index 0000000..6e0039d --- /dev/null +++ b/docs/plugin-isolation-feasibility.md @@ -0,0 +1,183 @@ +# Bounded feasibility proposal: protect observations without replacing Loom host semantics + +## Why this is separate + +Normal `eval:live -> runner invoke` semantic observation is feasible, but the +current Loom plugin and the observation callbacks execute in the same OpenCode +process. + +Same-process mechanisms such as: + +- a hidden file; +- a random file descriptor; +- a localhost socket; +- an environment-held/HMAC key; +- a closure-only signing callback; +- a private JavaScript object; + +do not establish a security boundary from arbitrary code with the same process +authority. Signing such records would authenticate the signer, not prove that +evaluated plugin code could not fabricate what reaches it. + +The restricted remote-tool profile solves this by changing execution topology, +but that cannot replace the host/session behavior Loom's current evals test. + +A stronger solution is therefore broader runtime work and is **not implemented +implicitly in PR #41**. + +## Candidate direction + +The smallest candidate worth researching is a **process-isolated plugin host with +a narrow capability broker**, while OpenCode core/session/tool execution and the +collector remain in the trusted runtime process. + +Conceptually: + +```text +runner invoke + -> trusted OpenCode core/session runtime + -> protected runtime observation collector + <-> narrow plugin capability RPC + -> isolated evaluated Loom plugin process +``` + +This is a feasibility direction, not accepted architecture. + +## Phase 0 result — Loom checkpoint `f8439e4` + +The inventory is bounded enough to justify a **separate prototype**, not an +implicit expansion of PR #41. + +Static inspection of `plugins/loom/index.ts` at the checkpoint found these +direct OpenCode-context capabilities: + +| Capability | Direct uses | Broker class | +| --- | ---: | --- | +| `storage.get` | 167 | typed request/response | +| `storage.set` | 117 | typed request/response | +| `storage.scan` | 10 | typed request/response/stream | +| `location.directory` | 43 | immutable launch data | +| `location.project` | 5 | immutable launch data | +| `session.hook` | 2 | typed callback registration | +| `session.synthetic` | 1 | typed request/response | +| `session.get` | 1 | typed request/response | +| `tool.hook` | 2 | typed callback registration | +| `tool.transform` | 1 | typed registration transform | +| `tool.list` | 1 | typed request/response | +| `agent.transform` | 1 | typed registration transform | +| `agent.list` | 1 | typed request/response | +| `permission.hook` | 1 | typed callback registration | +| `rpc.register` | 1 | plugin-owned RPC registration | + +The plugin also directly uses Node filesystem/path/crypto/async-hooks and +`child_process.execFile`. Those do not require execution inside the trusted +OpenCode process: an isolated Loom process can retain a disposable workspace +mount and its own bounded OS process authority while the collector mount, +signing authority, and OpenCode core remain absent from that process. + +### Phase 0 classification + +- **Pure/brokerable state:** storage get/set/scan, session get/synthetic, + tool/agent list. +- **Typed lifecycle callbacks:** session/tool/agent/permission hooks and + transforms. These require ordered request/response callback RPC, but not a + generic `eval` or arbitrary function execution facility in the trusted + runtime. +- **Launch facts:** location/project identity can be immutable launch data. +- **Plugin-local authority:** workspace filesystem/process operations and Loom's + own RPC endpoint can remain in the isolated process. + +No direct checkpoint usage requires handing the isolated process arbitrary +memory access, arbitrary module execution, or an unrestricted OpenCode API in +the trusted collector process. + +**Phase 0 verdict: GO for a separate bounded prototype.** This is not approval to +build it inside PR #41. The prototype must still falsify the assumption that the +typed callback surface is sufficient for cancellation, grants, child delivery, +OQ continuation, and all current plugin behavior. + +## Phase 0 — inventory method + +Pin Loom checkpoint `f8439e4` and inventory every OpenCode capability used by the +actual plugin/smoke: + +- tool registration and transformations; +- tool before/after hooks; +- permission decisions and grants; +- Session creation, prompt/continuation and lookup; +- child/subagent attachment and completion delivery; +- OQ continuation; +- cancellation/interrupt; +- workflow/gate state and persistence access; +- filesystem/process access; +- events and lifecycle callbacks. + +For each capability classify: + +1. pure request/response and safely serializable; +2. stateful but brokerable with an explicit identity/capability token; +3. callback/stream requiring lifecycle semantics; +4. same-process assumption that cannot be preserved without redesign. + +Stop if required capability surface is effectively an unrestricted OpenCode API. + +## Phase 1 — one semantic slice + +Only if Phase 0 is bounded, prototype one real Loom-owned runtime case using: + +- the unchanged `bun run eval:live -> scripts/run-evals.py -> runner invoke` + entrypoint; +- actual Loom plugin code in the isolated plugin process; +- actual OpenCode permission enforcement in the trusted runtime; +- one native tool execution and one Code Mode inner execution; +- disposable HOME/XDG/workspace/database state; +- no real provider. + +The broker must not accept actor/session/invocation identities from the plugin when +the trusted runtime already owns those identities. + +## Phase 2 — discriminating composition tests + +Before broadening the API, prove that isolation preserves the behaviors remote +tool substitution currently cannot establish: + +- permission denial remains denial; +- grant/elevation scope is unchanged; +- child Session actor/session/message identity is genuine; +- foreground/background subagent completion still follows normal semantics; +- cancellation interrupts the same owned work; +- OQ/continuation delivery is not replaced by successful stubs; +- runtime tool result/error conversion remains identical; +- observation on/off produces the same product outcome; +- plugin attempts to write/submit collector records cannot create eligible + observations. + +Use Loom checkpoint `f8439e4` composition tests as the product-side authority; +runner fixtures are supporting evidence only. + +## Go / no-go checkpoint + +Proceed to a separate runtime project only if the prototype demonstrates: + +- a finite reviewed capability surface; +- no generic arbitrary callback/code execution back into the trusted process; +- genuine runtime-owned identity and permission decisions; +- cancellation/lifecycle parity; +- a collector capability inaccessible to the plugin process; +- acceptable complexity relative to the eval-security goal. + +Otherwise keep protected normal-host evidence **unsupported** and retain the +restricted direct-session profile only as supplemental smoke. + +## Explicit non-goals + +This proposal does not authorize: + +- a general remote OpenCode platform; +- replacing `invoke` with `observe`; +- rewriting Loom's case model; +- permissive permission callbacks; +- fabricated Sessions or continuation responses; +- production deployment; +- a default-image change; +- signing as a substitute for collector isolation. diff --git a/docs/protected-channel-baseline-c162.md b/docs/protected-channel-baseline-c162.md new file mode 100644 index 0000000..c87297d --- /dev/null +++ b/docs/protected-channel-baseline-c162.md @@ -0,0 +1,52 @@ +# Preserved protected-channel baseline (2026-10-03) + +This records the prior version-3 connection proof. It is not a claim that these +older images implement the new mandatory supervisor-receipt profile/version 4. +The later collector image must be obtained from its own completed publication run. + +- Implementation: `c1629415a814476f7b62f159530e7edd29fde738`. +- Documentation checkpoint: `9153cc30c6d80deac15dbe284408d7bffa17ff38`. +- Protected connection run: https://github.com/bateau84/opencode-eval-runner/actions/runs/37112938698 +- Artifact: https://github.com/bateau84/opencode-eval-runner/actions/runs/37112938698/artifacts/11270611290 +- Artifact ZIP SHA-256: `1bff745c9db5f9921f93add9eb4613c1262b38f985420b0fef0b91747250b6ed`. +- Result: all 33 restricted-profile integration checks passed. This used the runner's isolated fixture tools, not Loom's full plugin or actual assertion consumer. +- Protected runtime: `ghcr.io/bateau84/opencode-eval-runner@sha256:658f4a53fba33c74f653abb613a6f38f82ec763891bc7f8c09ce1c7b7b19483d`. +- Fixture tools: `ghcr.io/bateau84/opencode-eval-runner@sha256:1e1c58ac92edace961246ce577bb293c2cbcc57fabdb282aa93b6f4c481ffac0`. + +The historical run compared actual runtime records with an independently recorded +tool-server test oracle, exercised the public CLI, tested target file/process +access and fake outputs, replay/deletion/reordering, capture I/O failure, input +snapshot integrity and redaction. A deletion with recomputed inline checksum was +rejected by source-sequence/accounting checks. The current receipt extension adds +a separate trusted digest/length comparison so result edits with recomputed inline +checksums are rejected too. This does not expand protection to a malicious host +that controls both paths. + +No historical artifact or observation has been rewritten. Independent review, +full Loom composition, native/delegated sessions, production redaction and image/ +artifact signing remained open at this checkpoint. + +## Loom script routing carried forward + +Adapt `scripts/run-evals.py` only for cases explicitly declaring the supported +profile. The `observe` command is a deterministic single-program transport, not +a replacement for normal multi-turn `invoke`. Keep result artifacts on the +trusted host. Pin actual adapter/producer/consumer bytes and preserve +`scripts/test_eval_observer_image.py` and its original fixtures. See the current +[handoff](loom-protected-channel-handoff.md) for version-5 admission and actual +Loom execution-isolation work. + + +## Contract erratum and current compatibility direction + +The tested `c1629415...` implementation used runtime events +`opencode-local-observation/v1`, wire `opencode-protected-observation/v2`, and +projection version **3**. Earlier prose describing that implementation as wire +v1 / projection v2 was stale. This erratum does not rewrite its historical +artifact. + +The normal Loom migration target is now explicitly the existing +`bun run eval:live -> scripts/run-evals.py -> opencode-eval-runner invoke` path. +The restricted `observe` profile remains supplemental and must not replace +normal Loom host/session semantics. See +[host-semantics feasibility](loom-host-semantics-feasibility.md). diff --git a/docs/protected-channel.md b/docs/protected-channel.md new file mode 100644 index 0000000..5ba3621 --- /dev/null +++ b/docs/protected-channel.md @@ -0,0 +1,142 @@ +# Protected runtime channel + +## Supported scope + +**This profile is supplemental. It is not the implementation path underneath +Loom's normal `bun run eval:live` entrypoint.** Normal-host feasibility and the +fixed `opencode-eval-runner invoke` compatibility boundary are documented in +[loom-host-semantics-feasibility.md](loom-host-semantics-feasibility.md). + +`observe` connects the existing patched OpenCode final inner-call observations to +the host importer. Profile: **`codemode-inner/direct-session/v1`**. The projection +is **`observed_execution.version = 5`**, with collection profile +**`private-supervisor-receipt/v1`**. Runtime events remain +`opencode-local-observation/v1`; wire frames use +`opencode-protected-observation/v2`. + +Evaluated tool code runs in a separate container. Only the selected immutable +runtime and its image-owned adapter receive the private input and capture mounts. +The runtime uses fixed configuration, no arbitrary plugin root, and the existing +confined interpreter with its direct network extension disabled. The same scope +restrictions apply with capture on and off. The normal `invoke` command, default +image and user's installed runtime are unchanged. + +**This is not protected execution of Loom's current in-process plugin.** Native +results, delegated sessions and arbitrary in-process untrusted plugins are +unsupported. `full_handoff_eligible` is always false. See +[the Loom handoff](loom-protected-channel-handoff.md) for the required split. + +## Protection and trust assumptions + +Trusted: launcher/importer, selected runtime image, immutable adapter, host artifact +storage, Docker and the host kernel. Untrusted: interpreted program, remote tool +code, tool responses, target stdout and target-writable files. A compromised host, +trusted runtime or container escape is outside this model. + +The target gets neither `/input` nor `/capture`, no host workspace, no Docker +socket and no shared PID namespace. Both containers are non-root, read-only, +resource-limited and capability-restricted, with raw container logging disabled. +The provider listens only on runtime loopback. The bridge accepts only host-listed +tools and rejects response redirects; the readiness check also rejects redirects. + +The launcher reads each input once and hashes the exact snapshot it executes. +The launch ID binds run, profile, images, collector source hashes and input/policy +hashes. The actual started container image identities are checked through Docker. + +The adapter redacts runtime-owned snapshots before writing the capture. After the +writer process group exits, the supervisor seals the transcript and returns a +SHA-256/length receipt over a separate trusted control path. The host requires +that receipt before parsing. It also validates the header/footer, run and launch +IDs, policy, source/wire sequences, parent relationships, each terminal and counts. +An edit followed by recomputing the file's own checksum is still rejected. + +**No HMAC or signing service is used by this collection path.** Neither a checksum +nor a fabricated receipt authenticates an arbitrary file. Origin follows from +the isolated launch and protected control/data paths. The parser alone is not an +attester and there is no unsigned-target-file import command. Final host artifacts +are not portable signed proofs; image and result signing remain separate. + +## Host interface + +```sh +bin/opencode-eval-runner observe \ + --image \ + --tool-image \ + --program-file fixture.js \ + --tools-file tools.json \ + --policy-file policy.json \ + --output result.json +``` + +This command uses a deterministic local provider, not a paid model. `--no-observe` +runs the same restricted execution profile without collection. Both image references +must be immutable. Exact image digests from each completed publication/test workflow +are preserved in its artifact; they are not interchangeable with a mutable tag or +local image ID. + +The target serves `GET /health` and `POST /call` on port 8080. The request contains +`name`, unchanged `input`, and runtime-selected `context` (invocation ID, ordinal, +agent/session/message/enclosing call). Returned JSON and remote failure replies are +defined in [the handoff](loom-protected-channel-handoff.md). Target-supplied identity +or observer-shaped response fields remain ordinary data, never collector claims. + +`tools.json` is the host-selected list of names and input schemas. `policy.json` +contains `version: 1`, configured `secrets`, and exact `allowed_values` for safe +fixture retention. Known literal/encoded secrets and sensitive structured keys are +redacted before persistence and size checks. Unknown values are omitted; changed +fields stay redacted and oversized fields have no preview. Allowed values permit +retention only: they never replace an actual unexpected result. + +No raw CLI stdout/stderr or tool logs become evidence artifacts. The separate +`script_output` field is a sanitized behavior-test diagnostic, not inner evidence. + +## Admission and verification + +Consumers must explicitly accept version 5 and this profile, match the trusted +launch record, require complete eligible capture/records, and inspect actual tool +outcomes. A returned denial is not domain success; a JSON-looking string remains +a string. Any unsupported scope, missing terminal, failed receipt, omitted field, +redaction, interruption, corruption or unknown accounting prevents positive +assertion. An empty capture cannot satisfy an expected call. + +`tests/integration/test_protected_connection.py` runs real Docker, the compiled +runtime, public host CLI and importer. Target attacks and host receive-side fault +injection are separate: the former attempt mount/process access and fake outputs; +the latter alter newly collected streams, including replay, deletion, reordering +and a result edit with recomputed checksum. Original and altered captures are +stored separately. Source/runtime tests, fixture integration, independent review +and actual Loom composition remain distinct requirements. + +[Complete Loom integration instructions and remaining scope](loom-protected-channel-handoff.md). + + +## Review hardening + +Capture accounting is separate from evidence eligibility. `coverage.starts` and +`coverage.terminals` retain counts from the validated prefix on a later semantic +failure; `missing_terminals` is unknown (`null`) until that accounting starts. +`accounting_complete` stays false unless all source accounting is validated. +Unknown omissions remain null, and invalid captures still admit no records. +These counters are diagnostic, never a way to promote a valid-looking prefix. +Parent and child invocation IDs share one capture-wide uniqueness requirement. +The host enforces the 16 KiB sanitized-value bound using compact UTF-8 JSON. + +Image-declared volumes are rejected before any container is launched. Tools that +need temporary state must use the profile's existing disposable writable space, +not an inherited image `VOLUME`. Cleanup removes anonymous volumes associated +with this launch's containers; it does not prune unrelated host volumes. Fixed +launcher-policy failures retain their non-sensitive reason codes. + +The experimental image workflows separate read-only build/test jobs from fresh +publication jobs that have no checkout and never execute their image artifacts. +A further read-only job tests the published digests. Publication is not code +approval, a Cosign signature, or a change to default image pins. Repository owners +and authorized workflow editors remain trusted; this boundary prevents tested +code/process residue from sharing publication credentials, not malicious edits to +the publisher workflow itself. + + +Projection version 5 makes the accounting change explicit; consumers written for +version 4 must not silently accept it. The runtime-event and wire schema versions +and the receipt collection profile are unchanged. Historical v4 artifacts retain +their original version and meaning; a new host checkout does not rewrite them. diff --git a/docs/runner-evidence-safety.md b/docs/runner-evidence-safety.md new file mode 100644 index 0000000..e00a8c5 --- /dev/null +++ b/docs/runner-evidence-safety.md @@ -0,0 +1,281 @@ +# Runner pre-output evidence safety (RSP v1) + +## Scope and compatibility + +This adapter realizes the runner-owned portion of RSP-001–004 under the unchanged +`bun run eval:live -> scripts/run-evals.py -> runner invoke` entrypoint. It does +not change tool implementations, permissions, sessions, runtime call ordering, +OpenCode's binary, or default image pins. It is explicitly opt-in at the host and +ships in a separate immutable image variant. + +Without the safety options, ordinary `invoke` keeps its existing execution +semantics and v1 product fields, but the host always owns the additive +`observed_execution` field: it replaces any container-supplied value and reports +`capture_not_requested` when `--observer-key-file` is absent. Consumers that +validate exact result keys must therefore account for this host-owned projection. + +A safety invocation uses a **new** result schema, never silently reinterprets +older v1 results, and must be consumed by a compatible Loom adapter. + +References: Loom's accepted `eval-evidence-safety-projection.md` at +`1a85b1a9e6707f720b95bd81b1e245ffa73202bf`; branch +`functionality-anchor-requirements-coherance` resolved to +`0756f7519fc7791bc31e2ef2aa41ec53168782a9` during implementation. At these fetched +commits `CredentialInventory.private_policy()` is in `scripts/run-evals.py`. +The named `scripts/eval_evidence_safety.py` is not present in the published tree. +No claim is made to have loaded that missing helper or the user's local checkout. + +## Private policy and request + +Loom owns source/path/role classification. The runner consumes the exact existing +`private_policy()` delivery shape: + +```json +{ + "schema": "loom-eval-credential-inventory/v1", + "policy_version": "source-path-roles/v1", + "complete": true, + "sources": { + "env": "complete", + "auth": "not_selected", + "config": "not_selected", + "models": "not_selected", + "credential_seed": "not_selected", + "config_root": "not_selected" + }, + "values": ["synthetic-credential"] +} +``` + +Exactly these source categories and `complete`, `incomplete`, `not_selected` +states are supported. Missing/unknown versions, malformed/duplicate JSON keys, +non-UTF-8 bytes, unknown source categories, size overflow, or invalid source states +cannot produce a complete policy. The private policy limit remains **128,000 +UTF-8 bytes**; values are bounded to 4,096 distinct-entry candidates. An incomplete +inventory never falls back to using a partial matcher. + +The host snapshots these bytes in memory. It audits the normal command's selected +seed categories against the declaration. Runner-resolved default/env-only seed +selection without an explicit corresponding argument downgrades the inventory: +v1 cannot bind a category name to an independently resolved path. Config-root +credential discovery remains unsupported and downgrades retention; it does not +remove the configuration from execution. Loom must classify the same selected +input snapshots it passes, including workspace configuration when credential +bearing. A complete inventory is a trusted input declaration, not new discovery +or a proof that a tool cannot generate other secrets. + +The host sends `opencode-eval-runner/evidence-safety-request/v1` on **private +supervisor stdin**, not in argv, a mounted policy file, or the child environment. +It contains `run_id` (fresh 64-hex nonce), `policy` (the shape above or null), and +`binding_key` (private per-invocation 32-byte random value encoded as hex). +Supervisor stdin is consumed before execution. Product subprocess stdin is +`DEVNULL`, so the policy is not inherited as tool/model input. Authorized ordinary +credential seeds are separate execution inputs, not evidence copies. + +## Acknowledgement and host admission + +The supervisor returns `evidence_safety_ack`: + +```text +schema = opencode-eval-runner/evidence-safety-ack/v1 +consumer = runner-evidence-safety/v1 +run_id = the expected host nonce +policy_schema = loom-eval-credential-inventory/v1 +policy_version = source-path-roles/v1 +projection_schema = loom-eval-evidence-safety/v1 +stages = [container.before_clip, container.before_output] +policy_valid = boolean +inventory_complete = boolean +module_sha256 = digest of the loaded projection module +image_source_revision = image build's source commit +policy_receipt = per-request keyed receipt over the parsed private policy +``` + +The host validates all fields, types, dispositions, and loss counts before its +first evidence-file write or print. The private binding key never appears in the +result; the keyed receipt permits checking that the same policy bytes were +consumed without exposing a public low-entropy credential hash. This receipt is +**not an observation signature, a signing service, or protection from a malicious +plugin**. It only binds policy acknowledgement within a reviewed installation. + +Before launching product code, the host requires an immutable image reference +and validates the inspected image's adapter version, projection-module hash and +container-entrypoint hash against its checkout. It runs the resolved local image +configuration ID, not a mutable tag. Unsupported images are rejected before they +can run with the new policy; there is no fallback to unprotected legacy execution. +Extra mounts that can replace the interpreter/emitter are unsupported; ordinary +workspace and workspace dependency submounts are retained. + +An accepted result includes host-generated `evidence_safety_validation` with +`acknowledged: true` and stages `host.before_write`, `host.before_print`. It also +includes `evidence_load`: immutable image reference, resolved image configuration +ID, image source revision, actual host executable/adapter/module SHA-256 values, +and host checkout revision/clean status when available. These are **code** hashes, +not hashes of policy values or tool payloads. The host never accepts a container's +claim that host verification occurred. + +A missing, incompatible, replayed, wrong-policy, or malformed acknowledgement +results only in a fixed omission projection and a nonzero host exit. Raw container +stdout/stderr/exception text is never used as a diagnostic fallback. + +## Public projection + +Root result schema: `opencode-eval-runner/safe-result/v1`. +Tool-event projection: `opencode-eval-runner/safe-tool-results/v1`, under +`tool_result_evidence`. Native event values come from the actual existing CLI +JSONL stream; no script-source inference or replacement runtime is introduced. +This is not a new producer of Code Mode inner results and does not promote +normal-invoke diagnostic hook records into trustworthy evidence. + +`evidence_safety` uses Loom's accepted `loom-eval-evidence-safety/v1` shape: +`policy_version`, `inventory_complete`, `coverage_complete`, `fields`, and the +fixed reason-to-counter `loss_counts` map. Each field entry has `event` (null for +transport fields or a safe zero-based event ordinal), `field`, and `state`. +Non-exact entries also have an accepted fixed `reason` and `stage`. + +States are **exact**, **redacted**, and **omitted**. Redacted values are diagnostic +only. Omitted fields have no value slot, preview, original-key list, original-value +hash, prefix, or suffix. Every retained payload/selector has a disposition. Fixed +validated envelope keys, enums, counters and booleans are protocol, not payload. +For example `exit_code: 1` survives a credential equal to `1`; a payload integer +that could disclose that credential is omitted rather than changed to an invalid +JSON token. + +Transport fields include `text`, `tools`, `actions`, `skills_loaded`, dynamic model/ +agent/session/reasoning identities, and raw-stream omissions. Event fields include +`tool`, actual `call_id` (`callID` or pinned V2 `id`, **not** `partID`), `session_id`, +`input`, `output`, and `error`. Status is validated against the fixed runtime +status vocabulary; `completed` is not interpreted as domain success. + +- Structured payloads are copied without mutation. Safe keys stay unchanged; + credential-bearing keys omit the enclosing field instead of colliding under a + common replacement key. Numbers, null and strings keep their type when exact. +- String matching protects real short credentials including `0`, `1`, `text`, + and `low`. There is no word exemption or length floor. Only correctly + source-classified credentials become match material. Generated mask markers + are never scanned again. +- Raw and up-to-three-layer JSON-escaped credential forms are supported. A + declared nested JSON-string adapter may decode and re-encode, retaining the + original bytes when unchanged. Opaque JSON-looking strings with a match are + omitted rather than having their syntax rewritten. Unsupported deeper forms, + cycles, duplicate JSON keys and non-finite values are omitted. +- Identity fields are exact or omitted, never masked into another usable identity. + Any action selector/input loss omits the whole action list. +- Native output needs the pinned runtime's explicit non-truncation metadata; + known upstream truncation is `upstream_clipped`, absent/unknown declaration is + `opaque_payload_unverified`. Host sanitation cannot undo an earlier cut. +- Sanitization occurs **before** runner field-size decisions. An oversized safe + field is omitted with `size_limit`, never stored as clipped JSON. Limits are + 2,000 encoded UTF-8 bytes for input, 6,000 for tool output/error, and 200,000 for + assistant text. Up to 64 outer tool events are retained. +- Raw stdout/stderr, plugin preflight detail, and arbitrary metadata are deliberately + omitted before runner clipping/output. There is no claimed safe adapter for + their arbitrary encodings. Consequently whole-projection `coverage_complete` + remains false; per-field exactness does not imply a complete capture. + +Reasons match the architecture: `credential_match`, `sensitive_key`, +`inventory_incomplete`, `upstream_clipped`, `unsupported_schema`, +`unsupported_representation`, `opaque_payload_unverified`, `size_limit`, `missing`, +`invalid`, `write_failed`. Container losses use stage `runner`; host admission +failures use `transport`. No raw error/credential/path is inserted into reasons. + +## First sinks and failures + +Covered runner-owned boundaries are field rendering/size decisions, raw-stream +prefix creation, container result stdout, host result-file creation/replacement, +and `--print-result`. Capture stays in memory until projection. The host writes +an already-validated projection into a mode-0600 temporary file and replaces the +requested result atomically; no raw intermediate result file is created. Docker +logging is disabled for the safety invocation, avoiding a second daemon log copy. + +Success, nonzero product exit, caught tool failure, timeout with partial output, +invalid JSON, unknown policy, missing acknowledgement and output write failure +have explicit omission behavior. Product return values and exit status remain +separate from evidence eligibility. Argument-parser errors in this mode use a +fixed diagnostic, not reflected arguments. No tool retry is added. + +Not covered or changed: OpenCode's own product Session database/tool-output +storage, arbitrary tool filesystem writes, Loom-owned observer sidecars, general +live-data encoding discovery, Issue 42 plugin isolation, or signing. Those are +not certified by this acknowledgement. A newly discovered runner-owned earlier +sink must be added to this contract before it can be claimed covered. + +## Loom integration and invocation + +Loom should retain `bun run eval:live ...` and add the following internally to its +normal runner invocation after creating the private inventory file: + +```sh +/path/to/runner/bin/opencode-eval-runner invoke \ + --image ghcr.io/bateau84/opencode-eval-runner@sha256: \ + --require-evidence-safety \ + --evidence-policy-file /private/disposable/inventory.json \ + --model --workspace \ + --prompt-file --output +``` + +All existing model/agent/permission/seed options remain normal `invoke` options. +The policy file is consumed by the host and is not mounted for the target. Missing +policy on a compatible image still allows ordinary product execution, but all +matcher-dependent evidence is omitted. Unknown image support fails before launch. + +Loom must explicitly recognize the new result/event schemas, validate the +acknowledgement and loaded revisions, and consume dispositions for every scoring +path. Exact fields alone do not authorize PASS; redacted/omitted required fields +and unknown absence coverage remain indeterminate. Do not fall back to old +`observed_tool_results`, raw stdout, or marker absence. No historical artifact is +rewritten or promoted. + +## Verification boundary + +The new CI workflow builds/publishes a separate safety image from the existing +OpenCode eval.4 digest plus changed runner Python files. Build and verification +have no package/OIDC signing authority; publication is a separate job that does +not execute the workload. The resulting immutable digest must match the tested +source; a repository commit does not change an old image. + +Unit tests invoke the actual container result emitter and host writer, with +synthetic inventories. The image probe exercises the public host CLI, real +OpenCode binary and deterministic loopback provider. It compares product outcomes +with safety off, then checks actual file/print output for short and escaped +credentials, pre-clip redaction, size omission, native errors, and policy failures. +The deliberately unsafe baseline uses disposable synthetic data only and is not +published as admitted evidence. These runner tests do not replace Loom's own +composition tests. Both publication and test results must be checked on the +actual PR source before acceptance. + + +## Disposable OpenCode state profile + +For provider-free composition that must exclude installed auth/database state, +use `--opencode-state-profile disposable`. The complete lifecycle and policy +source-state contract are defined in +[`disposable-opencode-state.md`](disposable-opencode-state.md). + +The disposable profile is opt-in. It does not change ordinary `invoke` defaults. +In RSP mode its `opencode-eval-runner/runtime-state/v1` attestation is carried as +an exact protocol field and must be understood by the Loom consumer before that +composition can be admitted. + + +## Docker and Podman image-ID compatibility + +The host preflight accepts both valid engine renderings of a local image config +ID: + +- Docker: `sha256:<64 lowercase hex>` +- Podman: `<64 lowercase hex>` (some Podman versions may also render the + Docker-style prefix) + +`evidence_load.image_config` is always canonicalized to +`sha256:<64 lowercase hex>`. + +Execution remains content-addressed: after immutable RepoDigest/label/source +validation, the runner replaces the requested image reference with the exact +config ID representation returned by that engine. It never falls back to a +mutable tag. Malformed, uppercase, truncated, overlong, or non-SHA256 IDs are +rejected. + +The evidence-safety workflow includes unit regressions for both representations +and an actual Podman preflight against the published immutable safety image. +No provider inference is used by that preflight. diff --git a/docs/signing-provenance.md b/docs/signing-provenance.md new file mode 100644 index 0000000..3b9252e --- /dev/null +++ b/docs/signing-provenance.md @@ -0,0 +1,161 @@ +# Cosign provenance for approved normal-invoke images and results + +## Security goal + +Cosign authenticates **approved bytes and signer identity**. It does not make an +observation truthful when evaluated code shares the collector's authority. + +The trusted signing path is deliberately separate from pull-request CI: + +```text +reviewed PR commit + -> manual default-branch workflow_dispatch + -> unprivileged rebuild from that exact commit + -> immutable image publication + -> unprivileged provider-free verification + -> protected release-signing environment approval + -> Cosign image + finalized-result signatures +``` + +PR pushes and green PR workflows cannot automatically request a signature. + +## Why PR artifacts are not signed directly + +A PR controls its source files, Containerfile, test scripts, and workflow inputs. +A default-branch signer that automatically consumes successful PR artifacts would +authenticate attacker-controlled bytes after only self-produced checks. + +The signer therefore does **not** use `workflow_run` and does not consume +`runtime-publication-*` or `local-runtime-*` PR artifacts. + +Instead, an authorized operator manually dispatches the trusted default-branch +workflow with: + +- the reviewed 40-hex source commit; +- the PR number that currently has that exact head. + +The trusted workflow verifies the PR head, checks out the selected commit in an +unprivileged build job, rebuilds the runtime from pinned OpenCode source, and +reruns the provider-free runtime, eval-live compatibility, and delegated-session +probes. + +## Privilege separation + +The workflow has four jobs. + +### build + +- runs only when the workflow definition itself is invoked from `refs/heads/main`; +- has read-only repository permission; +- validates the selected commit is the current head of the supplied PR; +- rebuilds OpenCode from pinned upstream source plus the reviewed downstream + patch; +- runs the source tests; +- saves image bytes as an artifact; +- has no package-write or OIDC authority. + +### publish + +- receives only the image archive; +- has package-write permission but no OIDC token; +- does not check out or execute source code; +- publishes a commit/run-scoped image and returns its immutable digest. + +### verify + +- has read-only permission; +- checks out the exact selected commit; +- pulls the immutable digest; +- verifies revision/runtime labels; +- runs the real normal-invoke runtime probe, eval-live compatibility probe, and + delegated-session probe; +- uploads the finalized diagnostic evidence; +- has no package-write or OIDC authority. + +### sign + +- depends on successful build, publish, and verify; +- uses the GitHub environment **`release-signing`**; +- repository configuration must protect that environment with required + reviewers; +- has OIDC and package-write permission; +- checks out no source and executes no selected-commit scripts; +- validates only fixed result shapes from the verified artifacts; +- signs the exact immutable image and a separately finalized result JSON. + +Thus a PR push cannot obtain signing authority merely by making its own tests +green. The final approval occurs after rebuild and verification. + +## Signed result binding + +`eval-result.json` version 2 contains: + +- reviewed source commit; +- PR number; +- trusted signing workflow run ID; +- exact immutable image digest; +- SHA-256 digests of the finalized compatibility, delegated-session, and runtime + probe summaries; +- their pass states; +- explicit `protected_capture_accepted: false`; +- explicit `in_process_plugin_protection: "unsupported"`. + +The signature therefore authenticates the diagnostic result without upgrading it +into a protected-capture claim. + +## Expected signer identity + +When invoked from the protected default-branch workflow, the expected identity is: + +```text +https://github.com/bateau84/opencode-eval-runner/.github/workflows/sign-normal-invoke-evidence.yml@refs/heads/main +``` + +Issuer: + +```text +https://token.actions.githubusercontent.com +``` + +The workflow verifies both the image signature and blob bundle immediately after +creation. + +## Required repository configuration + +Before this signing path is treated as release authority: + +1. merge/install the workflow on the default branch; +2. create the `release-signing` GitHub environment; +3. configure required reviewers for that environment; +4. limit who may deploy to it according to repository policy. + +Without those environment protections, the YAML alone is not a human approval +gate. + +## Current PR limitation + +PR #41 is intentionally unmerged. The trusted workflow therefore cannot yet be +invoked from the default branch, and no valid release-signing environment run can +be produced from this PR alone. + +That limitation is intentional: producing a keyless signature from a PR-controlled +copy of the workflow would defeat the security boundary this design is meant to +provide. + + +## Final signer admission + +Before the protected `release-signing` job creates either signature, it validates +the downloaded summaries against the immutable image selected for that run: + +- eval-live compatibility: kind `eval-live-invoke-compatibility`, version 2, + matching `image`, passed, and diagnostic/non-evidence status; +- delegated-session proof: kind `delegated-session-normal-invoke`, version 1, + matching `image`, and passed; +- runtime seam: kind `normal-invoke-runtime-seam-probe`, version 2, matching + `image`, matching reviewed source revision, seam passed, handoff still + `BLOCKED`, and independent-code-approval still false. + +The signing job authenticates to GHCR with its package-scoped `GITHUB_TOKEN` +before `cosign sign`. GitHub OIDC provides the keyless certificate identity; it +does not replace registry authentication for publishing the OCI signature. diff --git a/evidence-safety/Containerfile b/evidence-safety/Containerfile new file mode 100644 index 0000000..29c1337 --- /dev/null +++ b/evidence-safety/Containerfile @@ -0,0 +1,22 @@ +# Reuse the already-tested eval.5 OpenCode binary. Only runner adapters change. +FROM ghcr.io/bateau84/opencode-eval-runner@sha256:9e5af1397c385eabff32daa8e3b4a295713c38515d231af6c5759ddd320bdb40 +ARG RUNNER_REVISION +ARG PACKAGE_INIT_SHA256 +ARG SAFETY_MODULE_SHA256 +ARG INVOKE_SHA256 +LABEL org.opencontainers.image.revision=$RUNNER_REVISION \ + io.opencode-eval.evidence-safety="runner-evidence-safety/v1" \ + io.opencode-eval.evidence-safety-init=$PACKAGE_INIT_SHA256 \ + io.opencode-eval.evidence-safety-module=$SAFETY_MODULE_SHA256 \ + io.opencode-eval.evidence-safety-invoke=$INVOKE_SHA256 +COPY container/__init__.py /opt/opencode-eval-runner/container/__init__.py +COPY container/evidence_safety.py /opt/opencode-eval-runner/container/evidence_safety.py +COPY container/invoke.py /opt/opencode-eval-runner/container/invoke.py +USER root +RUN echo "$PACKAGE_INIT_SHA256 /opt/opencode-eval-runner/container/__init__.py" | sha256sum -c - \ + && echo "$SAFETY_MODULE_SHA256 /opt/opencode-eval-runner/container/evidence_safety.py" | sha256sum -c - \ + && echo "$INVOKE_SHA256 /opt/opencode-eval-runner/container/invoke.py" | sha256sum -c - \ + && printf '%s\n' "$RUNNER_REVISION" > /opt/opencode-eval-runner/container/evidence-safety-revision.txt +USER 1000:1000 +# Direct image invocation without a policy emits omissions, never legacy raw evidence. +ENV EVAL_EVIDENCE_SAFETY=1 diff --git a/protected-runtime/Containerfile b/protected-runtime/Containerfile new file mode 100644 index 0000000..7bd065f --- /dev/null +++ b/protected-runtime/Containerfile @@ -0,0 +1,13 @@ +# Existing locally patched runtime: no second interpreter or tool-call observer. +FROM ghcr.io/bateau84/opencode-eval-runner@sha256:800480450524bb9dd4f99a55d2c3cd40617b04b06625f74ae2d7e912bf4b0151 AS protected +ARG RUNNER_REVISION +LABEL io.opencode-eval.capture-profile="codemode-inner/direct-session/v1" \ + io.opencode-eval.runner-revision=$RUNNER_REVISION +COPY protected-runtime/invoke.py protected-runtime/bridge.ts /opt/protected/ +ENTRYPOINT ["python3", "/opt/protected/invoke.py"] + +# Adversarial fixture is a DIFFERENT runtime process/container, never an imported +# plugin. Sharing a read-only base image does not share capture mounts or PIDs. +FROM ghcr.io/bateau84/opencode-eval-runner@sha256:800480450524bb9dd4f99a55d2c3cd40617b04b06625f74ae2d7e912bf4b0151 AS fixture-tools +COPY tests/integration/protected_tools.py /opt/tools/server.py +ENTRYPOINT ["python3", "/opt/tools/server.py"] diff --git a/protected-runtime/bridge.ts b/protected-runtime/bridge.ts new file mode 100644 index 0000000..cc3eb2a --- /dev/null +++ b/protected-runtime/bridge.ts @@ -0,0 +1,148 @@ +import { appendFileSync, readFileSync, writeFileSync } from "node:fs" + +// Trusted transport adapter for the existing execute.observed runtime seam. +// No observation is reconstructed here. Never load target plugins in this process. +const request = JSON.parse(readFileSync("/input/request.json", "utf8")) +const policy = JSON.parse(readFileSync("/input/policy.json", "utf8")) +const path = "/capture/events.jsonl" +const token = /^[A-Za-z0-9_.:/@-]{1,256}$/ +const sensitive = /password|passwd|secret|token|authorization|credential|api[-_]?key|private[-_]?key/i +const literals: string[] = [...new Set((policy.secrets ?? []).flatMap((s: string) => + [s, Buffer.from(s).toString("base64"), Buffer.from(s).toString("hex"), encodeURIComponent(s)]))] + .filter(Boolean).sort((a, b) => b.length - a.length) +function canonical(v: any): string { + if (Array.isArray(v)) return "[" + v.map(canonical).join(",") + "]" + if (v !== null && typeof v === "object") return "{" + Object.keys(v).sort().map(k => JSON.stringify(k) + ":" + canonical(v[k])).join(",") + "}" + return JSON.stringify(v) +} +const allowed = new Set((policy.allowed_values ?? []).map(canonical)) +function safeField(field: any): any { + if (field?.state !== "available") return { state: "omitted", reason: "unsupported_snapshot" } + let changed = false + const scrub = (value: any): any => { + if (typeof value === "string") { + let out = value + for (const secret of literals) out = out.split(secret).join("[REDACTED]") + changed ||= out !== value + return out + } + if (Array.isArray(value)) return value.map(scrub) + if (value !== null && typeof value === "object") { + const out = Object.create(null) + for (const [key, item] of Object.entries(value)) { + const safeKey = scrub(key) + if (Object.hasOwn(out, safeKey)) throw new Error("key_collision") + if (sensitive.test(key)) { out[safeKey] = "[REDACTED]"; changed = true } + else out[safeKey] = scrub(item) + } + return out + } + return value + } + try { + const value = scrub(field.value) + // Exact host-approved values, not a claim that regexes discover unknown secrets. + if (!allowed.has(canonical(value))) return { state: "omitted", reason: "policy_omission" } + if (Buffer.byteLength(JSON.stringify(value), "utf8") > 16384) return { state: "truncated", reason: "field_limit" } + return { state: changed ? "redacted" : "available", redaction: "safe", value } + } catch { return { state: "omitted", reason: "policy_omission" } } +} +let seq = 0 +let bytes = 0 +let broken = false +function write(body: any) { + if (broken) return + try { + const line = JSON.stringify(body) + "\n" + bytes += Buffer.byteLength(line) + if (bytes > 8 * 1024 * 1024 - 4096 || seq > 10000) throw new Error("capture_limit") + appendFileSync(path, line, { mode: 0o600 }) + } catch { + broken = true + try { writeFileSync("/capture/fault", "capture_io_or_limit") } catch {} + } +} +function observed(event: any) { + if (!request.observe) return + // All metadata comes from the runtime. Reject unexpected fields rather than + // copying unreviewed free text through the redaction boundary. + const names = new Set(["schema", "sequence", "parent", "actor", "observer_failures", "kind", "boundary", "mode", + "invocation_id", "tool", "catalog_path", "input", "result", "error", "error_representation", "outcome", "dispatched", + "admitted", "terminals", "missing_terminals", "unsupported_dispatches", "unavailable_fields", "scope", "evidence_eligible"]) + const safe: any = {} + try { + for (const [key, value] of Object.entries(event)) { + if (!names.has(key)) throw new Error("unknown_field") + if (["input", "result", "error"].includes(key)) safe[key] = safeField(value) + else if (key === "actor" || key === "parent") { + if (!value || typeof value !== "object") throw new Error("identity") + for (const item of Object.values(value)) { + if (typeof item !== "string" || !token.test(item) || literals.some(s => item.includes(s))) throw new Error("identity") + } + safe[key] = value + } else { + if (typeof value === "string" && (!token.test(value) || literals.some(s => value.includes(s)))) throw new Error("metadata") + if (!["string", "number", "boolean"].includes(typeof value)) throw new Error("metadata") + safe[key] = value + } + } + write({ kind: "observation", run_id: request.run_id, seq: ++seq, observation: safe }) + } catch { + broken = true + try { writeFileSync("/capture/fault", "capture_metadata_unsupported") } catch {} + } +} + +export default { + id: "protectedbridge", + async setup(ctx: any) { + if (request.observe) write({ kind: "capture_start", schema: "opencode-protected-observation/v2", + profile: "codemode-inner/direct-session/v1", run_id: request.run_id, seq: 0, policy_id: request.policy_id, launch_id: request.launch_id }) + await ctx.tool.hook("execute.observed", observed) + await ctx.tool.hook("execute.before", (event: any) => { + if (event.tool !== "execute" && !request.tools.some((t: any) => event.tool === "isolated_" + t.name)) + throw new Error("tool_not_in_protected_profile") + }) + await ctx.tool.transform((editor: any) => { + // Fixed restricted profile, identical with capture ON and OFF. No local + // shell, dynamic plugin, delegated-session or other executable surface. + for (const tool of editor.list()) editor.remove(tool.id) + editor.namespace({ name: "isolated", description: "Tools executed in a separate untrusted container." }) + for (const definition of request.tools) editor.add({ + name: definition.name, description: definition.description ?? definition.name, + input: definition.input, output: {}, options: { namespace: "isolated", codemode: true }, + execute: async (input: unknown, tool: any) => { + if (typeof tool.evaluationInvocationID !== "string" || !token.test(tool.evaluationInvocationID) + || !Number.isSafeInteger(tool.evaluationDispatchOrdinal) || tool.evaluationDispatchOrdinal < 0) + throw new Error("protected_runtime_context_missing") + const response = await fetch(request.tool_url + "/call", { + method: "POST", headers: { "content-type": "application/json" }, + body: JSON.stringify({ name: definition.name, input, context: { + invocation_id: tool.evaluationInvocationID, ordinal: tool.evaluationDispatchOrdinal, + agent: tool.agent, session_id: tool.sessionID, message_id: tool.messageID, call_id: tool.id, + }}), signal: AbortSignal.timeout(10000), redirect: "error", + }).catch(() => { throw new Error("isolated_tool_transport_failed") }) + if (!response.ok) throw new Error("isolated_tool_transport_failed") + if (!response.body) throw new Error("isolated_tool_invalid_response") + const reader = response.body.getReader() + const chunks: Uint8Array[] = [] + let size = 0 + try { + for (;;) { + const next = await reader.read() + if (next.done) break + size += next.value.byteLength + if (size > 1024 * 1024) throw new Error("isolated_tool_response_limit") + chunks.push(next.value) + } + } finally { await reader.cancel().catch(() => {}) } + const text = Buffer.concat(chunks).toString("utf8") + const value = JSON.parse(text) + if (value.outcome === "threw" && typeof value.error === "string") throw new Error(value.error) + if (value.outcome !== "returned" || !Object.hasOwn(value, "result")) throw new Error("isolated_tool_invalid_response") + return { output: value.result, content: "" } + }, + }) + }) + }, +} diff --git a/protected-runtime/invoke.py b/protected-runtime/invoke.py new file mode 100644 index 0000000..7b8475d --- /dev/null +++ b/protected-runtime/invoke.py @@ -0,0 +1,194 @@ +"""Trusted supervisor: fixed config/code, loopback deterministic provider. + +Only this container sees /capture. Tool code runs in another container, never as +an in-process plugin. Raw OpenCode stdout/stderr are held in memory and discarded. +""" +from __future__ import annotations +import base64 +import hashlib +import json +import os +from pathlib import Path +import shutil +import signal +import subprocess +import threading +import time +import urllib.request +import urllib.parse +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +PROFILE = "codemode-inner/direct-session/v1" + + +class NoRedirect(urllib.request.HTTPRedirectHandler): + def redirect_request(self, req, fp, code, msg, headers, newurl): + raise urllib.error.HTTPError(req.full_url, code, "redirect_refused", headers, fp) + + +def health_check(url): + # A target-controlled /health response must not redirect into runtime services. + with urllib.request.build_opener(NoRedirect).open(url, timeout=1) as response: + return response.status == 200 + + +def safe_output(value, policy): + """Script diagnostics follow the same redact-before-retain rule as capture.""" + if not isinstance(value, str): + return {"state": "omitted", "reason": "not_observed"} + secrets = {form for raw in policy["secrets"] for form in + (raw, base64.b64encode(raw.encode()).decode(), raw.encode().hex(), urllib.parse.quote(raw, safe="~()*!.'-"))} + safe = value + for secret in sorted(secrets, key=len, reverse=True): + safe = safe.replace(secret, "[REDACTED]") + if safe not in policy["allowed_values"]: + return {"state": "omitted", "reason": "policy_omission"} + if len(safe.encode()) > 16384: + return {"state": "truncated", "reason": "field_limit"} + return {"state": "available" if safe == value else "redacted", "value": safe} + + +def main(): + request = json.loads(Path("/input/request.json").read_text()) + policy_raw = Path("/input/policy.json").read_bytes() + if hashlib.sha256(policy_raw).hexdigest() != request["policy_id"]: + raise ValueError("policy_mismatch") + launch_raw = Path("/input/launch.json").read_bytes() + if hashlib.sha256(launch_raw).hexdigest() != request["launch_id"]: + raise ValueError("launch_mismatch") + launch = json.loads(launch_raw) + tools_raw = Path("/input/tools.json").read_bytes() + if (launch["run_id"] != request["run_id"] or launch["profile"] != PROFILE + or launch["program_sha256"] != hashlib.sha256(request["program"].encode()).hexdigest() + or launch["tools_sha256"] != hashlib.sha256(tools_raw).hexdigest() + or json.loads(tools_raw) != request["tools"] or launch["policy_sha256"] != request["policy_id"]): + raise ValueError("launch_inputs_mismatch") + version = subprocess.run(["opencode", "--version"], capture_output=True, text=True, check=True).stdout.strip() + if version != "opencode v2.0.18-eval.2": + raise ValueError("protected_runtime_version_required") + for attempt in range(100): + try: + if health_check(request["tool_url"] + "/health"): + break + except OSError: + time.sleep(0.05) + else: + raise ValueError("isolated_tool_unavailable") + stage, script_output = 0, {"state": "omitted", "reason": "not_observed"} + policy = json.loads(policy_raw) + + class Provider(BaseHTTPRequestHandler): + def log_message(self, *args): + pass + + def do_POST(self): + nonlocal stage, script_output + length = int(self.headers.get("Content-Length", "0")) + if not self.path.endswith("/chat/completions") or length > 4 * 1024 * 1024 or stage >= 4: + self.send_error(400) + return + body = json.loads(self.rfile.read(length)) + if stage == 0: + tool = {"index": 0, "id": "protected-execute", "type": "function", + "function": {"name": "execute", "arguments": json.dumps({"code": request["program"]})}} + delta, finish = {"role": "assistant", "tool_calls": [tool]}, "tool_calls" + else: + # A behavior-test diagnostic only, never a source for inner values. + messages = [m for m in body.get("messages", []) if m.get("role") == "tool"] + if messages: + value = messages[-1].get("content") + script_output = safe_output(value, policy) + delta, finish = {"role": "assistant", "content": "capture-complete"}, "stop" + stage += 1 + common = {"id": "protected-fixture-" + str(stage), "created": 1, "model": body["model"]} + if body.get("stream"): + chunks = [ + {**common, "object": "chat.completion.chunk", "choices": [{"index": 0, "delta": delta, "finish_reason": None}]}, + {**common, "object": "chat.completion.chunk", "choices": [{"index": 0, "delta": {}, "finish_reason": finish}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}}, + ] + raw = ("".join("data: " + json.dumps(c) + "\n\n" for c in chunks) + "data: [DONE]\n\n").encode() + mime = "text/event-stream" + else: + raw = json.dumps({**common, "object": "chat.completion", "choices": [{"index": 0, "message": delta, "finish_reason": finish}]}).encode() + mime = "application/json" + self.send_response(200) + self.send_header("Content-Type", mime) + self.send_header("Content-Length", str(len(raw))) + self.end_headers() + self.wfile.write(raw) + + # Never copy ambient host config, credentials, plugin roots or database seeds. + home = Path("/tmp/protected") + for part in ("home", "config", "data", "state", "cache", "run"): + (home / part).mkdir(parents=True, exist_ok=True) + root = Path("/workspace") + plugins = root / ".opencode/plugins" + plugins.mkdir(parents=True) + shutil.copyfile("/opt/protected/bridge.ts", plugins / "bridge.ts") + server = ThreadingHTTPServer(("127.0.0.1", 0), Provider) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + (root / "opencode.json").write_text(json.dumps({ + "$schema": "https://opencode.ai/config.json", "model": "capture/mock", "enabled_providers": ["capture"], + "provider": {"capture": {"npm": "@ai-sdk/openai-compatible", "name": "Local capture fixture", + "options": {"baseURL": f"http://127.0.0.1:{server.server_port}/v1", "apiKey": "fixture"}, + "models": {"mock": {"name": "Mock", "limit": {"context": 1000000, "output": 32768}}}}}, + })) + env = {"PATH": "/usr/local/bin:/usr/bin:/bin", "HOME": str(home / "home"), + "XDG_CONFIG_HOME": str(home / "config"), "XDG_DATA_HOME": str(home / "data"), + "XDG_STATE_HOME": str(home / "state"), "XDG_CACHE_HOME": str(home / "cache"), + "XDG_RUNTIME_DIR": str(home / "run"), "OPENCODE_DISABLE_AUTOUPDATE": "1", + "OPENCODE_EVAL_OBSERVATIONS": "1" if request["observe"] else "0", "OPENCODE_EVAL_PROTECTED_CHANNEL": "1", "OPENCODE_DB": "opencode.db"} + runtime_exit, exited = 124, False + proc = subprocess.Popen(["opencode", "run", "--standalone", "--format", "json", "--auto", + "--title", "protected capture", "--model", "capture/mock", "Run the prescribed capture program."], + cwd=root, env=env, stdout=subprocess.PIPE, stderr=subprocess.PIPE, start_new_session=True) + try: + proc.communicate(timeout=50) + runtime_exit = proc.returncode + # The leader exiting alone is not a drain proof if children still hold state. + try: + os.killpg(proc.pid, 0) + except ProcessLookupError: + exited = True + if not exited: + os.killpg(proc.pid, signal.SIGKILL) + except subprocess.TimeoutExpired: + os.killpg(proc.pid, signal.SIGKILL) + proc.communicate(timeout=5) + finally: + server.shutdown() + server.server_close() + thread.join(timeout=2) + path = Path("/capture/events.jsonl") + if request["observe"] and exited and runtime_exit == 0 and path.is_file() and not Path("/capture/fault").exists(): + raw = path.read_bytes() + if len(raw) < 8 * 1024 * 1024 - 1024 and raw.endswith(b"\n"): + # Seal only after the whole writer process group is gone. The importer + # independently checks runtime sequence, parent accounting and terminals. + lines = raw.splitlines() + footer = {"kind": "capture_end", "run_id": request["run_id"], "seq": len(lines), + "event_count": len(lines) - 1, "sha256": hashlib.sha256(raw).hexdigest(), + "writer_exited": True, "runtime_exit": runtime_exit} + with path.open("ab") as out: + out.write((json.dumps(footer) + "\n").encode()) + # Host and container may have different UIDs. The host's mode-0700 temporary + # parent provides privacy; only this runtime container receives the child mount. + receipt = None + if path.is_file(): + path.chmod(0o644) + # Independent trusted control channel; not a hash accepted from this file. + # The launcher checks this receipt against the received capture, so editing + # records and recomputing the inline footer cannot conceal the alteration. + sealed = path.read_bytes() + if exited and len(sealed) <= 8 * 1024 * 1024: + receipt = {"sha256": hashlib.sha256(sealed).hexdigest(), "bytes": len(sealed)} + print(json.dumps({"profile": PROFILE, "run_id": request["run_id"], "launch_id": request["launch_id"], + "runtime_exit": runtime_exit, "writer_exited": exited, + "script_output": script_output, "receipt": receipt})) + return 0 if exited and runtime_exit == 0 else 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/runner/cli.py b/runner/cli.py index dfe69e6..b7a9a0b 100644 --- a/runner/cli.py +++ b/runner/cli.py @@ -11,6 +11,8 @@ import tempfile from pathlib import Path +from runner.observer import ObserverCapture, unavailable + DEFAULT_IMAGES = { "opencode": "ghcr.io/bateau84/opencode-eval-runner:opencode-edge", "github-copilot-cli": "ghcr.io/bateau84/opencode-eval-runner:copilot-edge", @@ -28,6 +30,19 @@ RUNTIME_UID = 1000 RUNTIME_GID = 1000 +OPENCODE_STATE_PROFILES = ("default", "disposable") +DISPOSABLE_STATE_CONTROL_ENVS = { + "EVAL_OPENCODE_STATE_PROFILE", + "EVAL_OPENCODE_AUTH_SOURCE", + "EVAL_OPENCODE_DATABASE_SOURCE", +} +RUNNER_SEED_ENVS = ( + "OPENCODE_EVAL_RUNNER_AUTH", + "OPENCODE_EVAL_RUNNER_CONFIG", + "OPENCODE_EVAL_RUNNER_MODELS", + "OPENCODE_EVAL_RUNNER_DB", + "OPENCODE_EVAL_RUNNER_CONFIG_ROOT", +) class RunnerError(RuntimeError): @@ -177,6 +192,52 @@ def existing_dir(explicit: str | None, env_name: str) -> Path | None: return path +def explicit_seed(explicit: str | None, label: str) -> Path | None: + if not explicit: + return None + path = Path(explicit).expanduser() + if not path.is_file(): + raise RunnerError(f"{label} file not found: {path}") + return path + + +def explicit_dir(explicit: str | None, label: str) -> Path | None: + if not explicit: + return None + path = Path(explicit).expanduser() + if not path.is_dir(): + raise RunnerError(f"{label} directory not found: {path}") + return path + + +def opencode_state_profile(args: argparse.Namespace, host_env: dict[str, str] | None = None) -> str: + profile = getattr(args, "opencode_state_profile", "default") or "default" + if profile not in OPENCODE_STATE_PROFILES: + raise RunnerError(f"unsupported OpenCode state profile: {profile}") + if profile == "disposable": + if args.transport != "opencode": + raise RunnerError("--opencode-state-profile disposable is only supported by the opencode transport") + if getattr(args, "database", None): + raise RunnerError("--database is incompatible with --opencode-state-profile disposable") + env = dict(os.environ) if host_env is None else host_env + selected = [name for name in RUNNER_SEED_ENVS if env.get(name)] + if selected: + raise RunnerError( + "--opencode-state-profile disposable rejects implicit runner seed overrides: " + + ", ".join(sorted(selected)) + ) + return profile + + +def resolve_database_seed( + args: argparse.Namespace, destination: Path, host_env: dict[str, str] | None = None +) -> Path | None: + if opencode_state_profile(args, host_env) == "disposable": + return None + source = existing_seed(args.database, "OPENCODE_EVAL_RUNNER_DB", default_database_path()) + return sanitize_database_seed(source, destination) if source else None + + def host_environment_for_transport(transport: str) -> dict[str, str]: env = dict(os.environ) if transport != "github-copilot-cli" or any(env.get(name, "").strip() for name in COPILOT_AUTH_ENVS): @@ -270,8 +331,8 @@ def build_container_command( command += ["--security-opt", "label=disable"] elif hasattr(os, "getuid") and os.getuid() != 0: # Rootful Docker preserves numeric ownership on bind mounts. Match the - # non-root host caller so mode-0600 seed files remain readable. When - # invoked by root, do not add --user: keep the image's non-root USER. + # non-root host caller so mode-0600 seed files stay readable. A root + # caller leaves the image's non-root USER intact. command += ["--user", f"{os.getuid()}:{os.getgid()}"] command += [ "--workdir", @@ -282,13 +343,19 @@ def build_container_command( for spec in getattr(args, "mount", []): command += extra_mount_arg(spec) - auth = existing_seed(args.auth, "OPENCODE_EVAL_RUNNER_AUTH", default_auth_path()) - config = existing_seed(args.config, "OPENCODE_EVAL_RUNNER_CONFIG") - models = existing_seed( - args.models_catalog, - "OPENCODE_EVAL_RUNNER_MODELS", - default_models_path(), - ) + state_profile = opencode_state_profile(args, host_env) + if state_profile == "disposable": + auth = explicit_seed(args.auth, "auth") + config = explicit_seed(args.config, "config") + models = explicit_seed(args.models_catalog, "models catalog") + else: + auth = existing_seed(args.auth, "OPENCODE_EVAL_RUNNER_AUTH", default_auth_path()) + config = existing_seed(args.config, "OPENCODE_EVAL_RUNNER_CONFIG") + models = existing_seed( + args.models_catalog, + "OPENCODE_EVAL_RUNNER_MODELS", + default_models_path(), + ) if auth: command += bind_arg(auth, "/seed/auth.json", readonly=True) if config: @@ -297,7 +364,11 @@ def build_container_command( command += bind_arg(models, "/seed/models.json", readonly=True) if database_seed: command += bind_arg(database_seed, "/seed/opencode.db", readonly=True) - config_root = existing_dir(getattr(args, "config_root", None), "OPENCODE_EVAL_RUNNER_CONFIG_ROOT") + config_root = ( + explicit_dir(getattr(args, "config_root", None), "config root") + if state_profile == "disposable" + else existing_dir(getattr(args, "config_root", None), "OPENCODE_EVAL_RUNNER_CONFIG_ROOT") + ) if config_root: command += bind_arg(config_root, "/seed/opencode-config", readonly=True) @@ -311,10 +382,22 @@ def build_container_command( "--env", "EVAL_PROMPT_FILE=/input/prompt.txt", "--env", "EVAL_SYSTEM_FILE=/input/system.txt", "--env", f"EVAL_TIMEOUT_SECONDS={args.timeout_seconds}", + "--env", f"EVAL_OPENCODE_STATE_PROFILE={state_profile}", + "--env", f"EVAL_OPENCODE_AUTH_SOURCE={'explicit' if auth else 'none' if state_profile == 'disposable' else 'implicit-or-none'}", + "--env", f"EVAL_OPENCODE_DATABASE_SOURCE={'runtime-bootstrap' if state_profile == 'disposable' else 'seed-or-runtime'}", ] - env_names = list(dict.fromkeys(DEFAULT_ENV_ALLOWLIST + tuple(args.env))) - if args.transport == "github-copilot-cli": + if state_profile == "disposable": + forbidden = sorted(DISPOSABLE_STATE_CONTROL_ENVS.intersection(args.env)) + if forbidden: + raise RunnerError( + "--opencode-state-profile disposable reserves runner state environment names: " + + ", ".join(forbidden) + ) + env_names = list(dict.fromkeys(tuple(args.env))) + else: + env_names = list(dict.fromkeys(DEFAULT_ENV_ALLOWLIST + tuple(args.env))) + if args.transport == "github-copilot-cli" and state_profile != "disposable": env_names.extend(COPILOT_AUTH_ENVS) pass_env(command, env_names, host_env) @@ -323,6 +406,12 @@ def build_container_command( def invoke(args: argparse.Namespace) -> int: + if getattr(args, 'require_evidence_safety', False) or getattr(args, 'evidence_policy_file', None): + from runner.safe_invoke import invoke as safe_invoke + return safe_invoke(args) + observer_key = getattr(args, "observer_key_file", None) + if observer_key and args.transport != "opencode": + raise RunnerError("--observer-key-file is only supported by the opencode transport") prompt_path = Path(args.prompt_file).resolve() if not prompt_path.is_file(): raise RunnerError(f"prompt file not found: {prompt_path}") @@ -345,18 +434,11 @@ def invoke(args: argparse.Namespace) -> int: else: (input_dir / "system.txt").write_text("", encoding="utf-8") - database_source = existing_seed( - args.database, - "OPENCODE_EVAL_RUNNER_DB", - default_database_path(), - ) - database_seed = ( - sanitize_database_seed(database_source, root / "opencode-credentials.db") - if database_source - else None + host_env = host_environment_for_transport(args.transport) + database_seed = resolve_database_seed( + args, root / "opencode-credentials.db", host_env ) - host_env = host_environment_for_transport(args.transport) command, _ = build_container_command( args, input_dir, @@ -364,30 +446,52 @@ def invoke(args: argparse.Namespace) -> int: host_env=host_env, database_seed=database_seed, ) - proc = subprocess.run( - command, - env=host_env, - text=True, - capture_output=True, - timeout=args.container_timeout, - check=False, - ) + capture = None + if observer_key: + try: + capture = ObserverCapture(Path(observer_key).expanduser(), root, command, host_env) + except (OSError, ValueError) as exc: + raise RunnerError("cannot configure isolated observer capture") from exc try: - result = json.loads(proc.stdout) - except json.JSONDecodeError as exc: - detail = " | ".join(part.strip() for part in (proc.stderr, proc.stdout) if part.strip()) - raise RunnerError( - f"container produced invalid result JSON (exit {proc.returncode}): {exc}" - + (f": {detail[:2000]}" if detail else "") - ) from exc - - if not isinstance(result, dict): - raise RunnerError("container result must be a JSON object") - + proc = subprocess.run( + command, + env=host_env, + text=True, + capture_output=True, + timeout=args.container_timeout, + check=False, + ) + try: + result = json.loads(proc.stdout) + except json.JSONDecodeError as exc: + detail = " | ".join(part.strip() for part in (proc.stderr, proc.stdout) if part.strip()) + raise RunnerError( + f"container produced invalid result JSON (exit {proc.returncode}): {exc}" + + (f": {detail[:2000]}" if detail else "") + ) from exc + + if not isinstance(result, dict): + raise RunnerError("container result must be a JSON object") + returncode = proc.returncode + except (RunnerError, subprocess.TimeoutExpired, OSError): + if capture is None: + raise + # Preserve an explicit non-evidence artifact even on interruption. + # Never echo untrusted stdout/stderr into observer diagnostics. + result = {"error": "observer_transport_failed", "exit_code": 2} + returncode = 2 + + # ALWAYS replace this field: container/script JSON cannot attest itself. + result["observed_execution"] = ( + capture.finish(transport_ok=returncode == 0) if capture + else unavailable("capture_not_requested") + ) + if capture and not result["observed_execution"]["evidence_eligible"] and returncode == 0: + returncode = 4 result_host.write_text(json.dumps(result, indent=2) + "\n", encoding="utf-8") if args.print_result: sys.stdout.write(result_host.read_text(encoding="utf-8")) - return 0 if proc.returncode == 0 else proc.returncode + return returncode def parser() -> argparse.ArgumentParser: @@ -429,10 +533,29 @@ def parser() -> argparse.ArgumentParser: run.add_argument("--prompt-file", required=True) run.add_argument("--system-file") run.add_argument("--output", required=True) + run.add_argument("--require-evidence-safety", action="store_true", + help="Require pre-output projection and matching image acknowledgement; missing policy omits payloads.") + run.add_argument("--evidence-policy-file", + help="Private Loom credential inventory JSON (128000-byte maximum); implies --require-evidence-safety.") + run.add_argument( + "--observer-key-file", + metavar="PATH", + help="Require authenticated execution-observer evidence (OpenCode only). The trusted producer's verification key must remain outside every container mount. Missing or indeterminate capture exits 4; see docs/execution-observer.md.", + ) run.add_argument("--auth") run.add_argument("--config") run.add_argument("--models-catalog") run.add_argument("--database") + run.add_argument( + "--opencode-state-profile", + choices=OPENCODE_STATE_PROFILES, + default="default", + help=( + "OpenCode host-state lifecycle. 'default' preserves existing implicit auth/model/database seeds; " + "'disposable' rejects implicit runner seed overrides, disables ambient provider auth forwarding, " + "forbids database seeds, and lets OpenCode bootstrap a fresh migrated database in disposable XDG state." + ), + ) run.add_argument("--config-root") run.add_argument("--env", action="append", default=[], metavar="NAME") run.add_argument("--timeout-seconds", type=int, default=240) @@ -442,7 +565,29 @@ def parser() -> argparse.ArgumentParser: def main() -> int: - args = parser().parse_args() + # argparse accepts unambiguous long-option prefixes. Protect diagnostics + # for every candidate safety prefix (including ambiguous ones), not only + # the full spelling. Preserve normal option abbreviation behavior. + safety_options = ('--require-evidence-safety', '--evidence-policy-file') + requested_safety = any( + name.startswith('--') and len(name) > 2 and + any(option.startswith(name) for option in safety_options) + for arg in sys.argv[1:] for name in (arg.partition('=')[0],) + ) + if requested_safety: + # argparse can echo invalid arguments (including accidental credential + # values) before invoke has loaded policy. Emit only a fixed diagnostic. + import contextlib + import io + try: + with contextlib.redirect_stderr(io.StringIO()): + args = parser().parse_args() + except SystemExit as exc: + if exc.code: + print('opencode-eval-runner: invalid safety invocation', file=sys.stderr) + return int(exc.code or 0) + else: + args = parser().parse_args() try: if args.command == "invoke": return invoke(args) diff --git a/runner/observer.py b/runner/observer.py new file mode 100644 index 0000000..f072bb0 --- /dev/null +++ b/runner/observer.py @@ -0,0 +1,375 @@ +"""Fail-closed importer for an independently trusted execution-hook observer. + +This module never observes tools or signs records. The producer must implement +and attest the boundary described in docs/execution-observer.md. In particular, +a signing key in the evaluated container is NOT a trusted observer deployment. +""" +from __future__ import annotations + +import base64 +import binascii +import hashlib +import hmac +import json +import os +import re +import secrets +import stat +from pathlib import Path +from typing import Any + +PROTOCOL = b"opencode-eval-observer/v1\0" +MOUNT = "/eval-observer" +MAX_CAPTURE_BYTES = 8 * 1024 * 1024 +MAX_FRAME_BYTES = 256 * 1024 +MAX_FRAMES = 10001 +MAX_FIELD_BYTES = 16384 +IDENTIFIER = re.compile(r"[A-Za-z0-9_.:/@-]{1,256}\Z") +SENSITIVE_KEY = re.compile( + r"password|passwd|secret|token|authorization|credential|api[-_]?key|private[-_]?key", + re.IGNORECASE, +) + + +class InvalidCapture(ValueError): + """A constant, non-sensitive reason code; never includes producer text.""" + + +def _object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + result: dict[str, Any] = {} + for key, value in pairs: + if key in result: + raise InvalidCapture("duplicate_json_key") + result[key] = value + return result + + +def _invalid_number(_: str) -> None: + raise InvalidCapture("nonfinite_number") + + +def _json(raw: bytes) -> Any: + value = json.loads(raw.decode("utf-8"), object_pairs_hook=_object, parse_constant=_invalid_number) + # Also reject overflowed floats and unpaired UTF-16 surrogates. + json.dumps(value, ensure_ascii=False, allow_nan=False).encode("utf-8") + return value + + +def projection(run_id: str | None = None) -> dict[str, Any]: + return { + "kind": "execution-observer-projection", "version": 1, + "run_id": run_id, "status": "unavailable", "evidence_eligible": False, + "records": [], "issues": [], + "coverage": { + "capture_started": False, "capture_ended": False, + "observed_starts": 0, "observed_terminals": 0, + "missing_terminals": 0, "omitted_records": None, + "truncated": False, "unsupported": [], + }, + } + + +def unavailable(reason: str, run_id: str | None = None) -> dict[str, Any]: + result = projection(run_id) + result["issues"].append(reason) + return result + + +def _keys(value: Any, required: set[str], optional: set[str] = frozenset()) -> None: + if not isinstance(value, dict) or not required <= value.keys() or value.keys() - required - optional: + raise InvalidCapture("invalid_record_shape") + + +def _integer(value: Any) -> bool: + return type(value) is int and 0 <= value <= 2**53 - 1 + + +def _identity(value: Any, known_secrets: tuple[str, ...]) -> str: + if not isinstance(value, str) or not IDENTIFIER.fullmatch(value): + raise InvalidCapture("invalid_identity") + if any(secret in value for secret in known_secrets): + raise InvalidCapture("unsafe_identity") + return value + + +def _scrub(value: Any, known_secrets: tuple[str, ...]) -> tuple[Any, bool]: + """Defense in depth AFTER the producer's authenticated safe-redaction claim.""" + changed = False + if isinstance(value, dict): + result = {} + for key, item in value.items(): + safe_key, key_changed = _scrub(key, known_secrets) + if safe_key in result: + # Redaction must not merge two distinct input keys. + raise InvalidCapture("redaction_key_collision") + if SENSITIVE_KEY.search(key): + result[safe_key] = "[REDACTED]" + changed = True + else: + result[safe_key], item_changed = _scrub(item, known_secrets) + changed |= item_changed + changed |= key_changed + return result, changed + if isinstance(value, list): + items = [_scrub(item, known_secrets) for item in value] + return [item for item, _ in items], any(flag for _, flag in items) + if isinstance(value, str): + safe = value + for secret in known_secrets: + safe = safe.replace(secret, "[REDACTED]") + return safe, safe != value + return value, False + + +def _field(value: Any, known_secrets: tuple[str, ...], limit: int) -> dict[str, Any]: + if not isinstance(value, dict): + return {"state": "omitted", "reason": "unsafe_redaction"} + state = value.get("state") + if state not in {"available", "redacted", "omitted", "truncated"}: + raise InvalidCapture("invalid_field_state") + if state in {"omitted", "truncated"}: + # Reasons and previews from the producer might themselves contain secrets. + return {"state": state, "reason": "producer_" + state} + if value.get("redaction") != "safe" or "value" not in value: + return {"state": "omitted", "reason": "unsafe_redaction"} + safe, changed = _scrub(value["value"], known_secrets) + # Redaction happens BEFORE any clipping/size decision. Oversized fields are + # omitted, never returned as a deceptively complete prefix or parsed JSON. + size = len(json.dumps(safe, ensure_ascii=False, allow_nan=False).encode("utf-8")) + if size > limit: + return {"state": "truncated", "reason": "field_limit"} + return {"state": "redacted" if changed or state == "redacted" else "available", "value": safe} + + +def _read_regular(path: Path, limit: int) -> bytes: + fd = os.open(path, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK) + try: + info = os.fstat(fd) + if not stat.S_ISREG(info.st_mode) or info.st_nlink != 1: + raise InvalidCapture("unsafe_capture_file") + if info.st_size > limit: + raise InvalidCapture("capture_limit") + with os.fdopen(fd, "rb", closefd=False) as stream: + raw = stream.read(limit + 1) + if len(raw) > limit: + raise InvalidCapture("capture_limit") + return raw + finally: + os.close(fd) + + +def load_capture( + path: Path, *, key: bytes, run_id: str, known_secrets: tuple[str, ...] = (), + field_limit: int = MAX_FIELD_BYTES, transport_ok: bool = True, +) -> dict[str, Any]: + """Return a sanitized projection. No malformed/partial capture can be eligible. + + Authentication is relative to a trusted producer with an isolated key. It + does not prove that an arbitrary hook implementation sees the final result. + """ + result = projection(run_id) + coverage = result["coverage"] + calls: dict[str, dict[str, Any]] = {} + known_secrets = tuple(sorted({s for s in known_secrets if s}, key=len, reverse=True)) + try: + raw = _read_regular(path, MAX_CAPTURE_BYTES) + if not raw: + return unavailable("empty_capture", run_id) + if not raw.endswith(b"\n"): + coverage["truncated"] = True + raise InvalidCapture("unterminated_capture") + lines = raw.splitlines() + if len(lines) > MAX_FRAMES: + raise InvalidCapture("record_limit") + previous = bytes(32) + ended = False + for sequence, line in enumerate(lines): + if ended: + raise InvalidCapture("records_after_capture_end") + if len(line) > MAX_FRAME_BYTES: + raise InvalidCapture("frame_limit") + envelope = _json(line) + _keys(envelope, {"payload", "mac"}) + if not isinstance(envelope["payload"], str) or not isinstance(envelope["mac"], str): + raise InvalidCapture("invalid_envelope") + payload = base64.b64decode(envelope["payload"], validate=True) + mac = hmac.new(key, PROTOCOL + run_id.encode("ascii") + b"\0" + previous + payload, hashlib.sha256).hexdigest() + if not hmac.compare_digest(mac, envelope["mac"]): + raise InvalidCapture("authentication_failed") + previous = bytes.fromhex(mac) + event = _json(payload) + if not isinstance(event, dict) or event.get("run_id") != run_id: + raise InvalidCapture("wrong_run") + if type(event.get("seq")) is not int or event["seq"] != sequence: + raise InvalidCapture("ambiguous_order") + kind = event.get("kind") + common = {"kind", "run_id", "seq"} + if sequence == 0: + _keys(event, common | {"version", "source", "boundary", "correlation", "ordering"}) + if kind != "capture_start" or type(event["version"]) is not int or event["version"] != 1: + raise InvalidCapture("unsupported_version") + expected = { + "source": "loom-execution-hook", "boundary": "tool-return-to-caller", + "correlation": "execution-invocation-id", "ordering": "monotonic-sequence", + } + if any(event[name] != value for name, value in expected.items()): + raise InvalidCapture("unsupported_capture_boundary") + coverage["capture_started"] = True + elif kind == "call_start": + _keys(event, common | {"invocation_id", "tool", "input", "actor", "parent", "mode"}) + invocation = _identity(event["invocation_id"], known_secrets) + if invocation in calls: + raise InvalidCapture("duplicate_invocation") + _keys(event["actor"], {"agent", "session_id"}) + actor = {name: _identity(value, known_secrets) for name, value in event["actor"].items()} + if event["mode"] not in {"native", "code_mode"}: + raise InvalidCapture("unsupported_call_mode") + parent = event["parent"] + if parent is not None: + _keys(parent, {"session_id", "call_id"}) + parent = {name: _identity(value, known_secrets) for name, value in parent.items()} + elif event["mode"] == "code_mode": + raise InvalidCapture("missing_parent") + calls[invocation] = { + "invocation_id": invocation, "tool": _identity(event["tool"], known_secrets), + "actor": actor, "parent": parent, "mode": event["mode"], + "input": _field(event["input"], known_secrets, field_limit), + "start_sequence": sequence, "terminal_sequence": None, + "outcome": "missing", "evidence_eligible": False, + } + coverage["observed_starts"] += 1 + elif kind == "call_end": + _keys(event, common | {"invocation_id", "outcome"}, {"result", "error"}) + invocation = _identity(event["invocation_id"], known_secrets) + call = calls.get(invocation) + if call is None or call["terminal_sequence"] is not None: + raise InvalidCapture("ambiguous_terminal") + outcome = event["outcome"] + field = {"returned": "result", "threw": "error"}.get(outcome) + if field is None or field not in event or ("error" if field == "result" else "result") in event: + raise InvalidCapture("invalid_terminal") + call.update(outcome=outcome, terminal_sequence=sequence) + call[field] = _field(event[field], known_secrets, field_limit) + coverage["observed_terminals"] += 1 + elif kind == "capture_end": + _keys(event, common | {"calls_started", "calls_ended", "omitted_records", "truncated", "unsupported"}) + for name in ("calls_started", "calls_ended", "omitted_records"): + if not _integer(event[name]): + raise InvalidCapture("invalid_completeness") + if event["calls_started"] != len(calls) or event["calls_ended"] != coverage["observed_terminals"]: + raise InvalidCapture("count_mismatch") + if type(event["truncated"]) is not bool or not isinstance(event["unsupported"], list): + raise InvalidCapture("invalid_completeness") + coverage.update( + capture_ended=True, omitted_records=event["omitted_records"], + truncated=event["truncated"], + unsupported=[_identity(item, known_secrets) for item in event["unsupported"]], + ) + ended = True + else: + raise InvalidCapture("unknown_record_kind") + result["records"] = list(calls.values()) # start order, NEVER completion order + coverage["missing_terminals"] = sum(call["outcome"] == "missing" for call in calls.values()) + if not ended: + result["issues"].append("missing_capture_end") + if coverage["missing_terminals"]: + result["issues"].append("missing_terminals") + if coverage["omitted_records"]: + result["issues"].append("omitted_records") + if coverage["unsupported"]: + result["issues"].append("unsupported_capture") + if coverage["truncated"]: + result["issues"].append("truncated_capture") + for call in calls.values(): + fields = [call[name] for name in ("input", "result", "error") if name in call] + if any(field["state"] != "available" for field in fields): + if "incomplete_fields" not in result["issues"]: + result["issues"].append("incomplete_fields") + if any(field["state"] == "truncated" for field in fields): + coverage["truncated"] = True + if not transport_ok: + result["issues"].append("transport_failed") + result["evidence_eligible"] = not result["issues"] + result["status"] = "complete" if result["evidence_eligible"] else "incomplete" + for call in calls.values(): + call["evidence_eligible"] = result["evidence_eligible"] + except FileNotFoundError: + return unavailable("missing_capture", run_id) + except InvalidCapture as exc: + code = str(exc) + result["status"] = "unsupported" if code.startswith("unsupported_") else "invalid" + result["issues"] = [code] + result["records"] = [] + if code in {"capture_limit", "record_limit", "frame_limit", "unterminated_capture"}: + coverage["truncated"] = True + except (OSError, ValueError, TypeError, UnicodeError, RecursionError, binascii.Error): + result["status"] = "invalid" + result["issues"] = ["malformed_capture"] + result["records"] = [] + coverage["missing_terminals"] = coverage["observed_starts"] - coverage["observed_terminals"] + return result + + +class ObserverCapture: + """Fresh per-invoke transport. Only the public nonce/path reach the target.""" + + def __init__(self, key_path: Path, root: Path, command: list[str], host_env: dict[str, str]): + self.key = _read_regular(key_path, 4096) + if len(self.key) < 32: + raise ValueError("observer key must contain at least 32 bytes") + resolved = key_path.resolve() + if resolved.stat().st_mode & 0o077: + raise ValueError("observer key file must be private (mode 0600)") + key_encodings = { + self.key, + self.key.hex().encode("ascii"), + self.key.hex().upper().encode("ascii"), + base64.b64encode(self.key), + base64.urlsafe_b64encode(self.key), + base64.b64encode(self.key).rstrip(b"="), + base64.urlsafe_b64encode(self.key).rstrip(b"="), + } + for index, argument in enumerate(command[:-1]): + if argument not in {"--volume", "--env"}: + continue + specification = command[index + 1] + if argument == "--volume": + source = Path(specification.rsplit(":", 2)[0]).resolve() + if resolved == source or source in resolved.parents: + raise ValueError("observer key must be outside all container mounts") + if source.is_file() and os.path.samefile(source, resolved): + raise ValueError("observer key must not be mounted") + target = specification.rsplit(":", 2)[1] + if target == "/" or target == MOUNT or target.startswith(MOUNT + "/"): + raise ValueError("observer mount target is reserved") + else: + name, _, inline = specification.partition("=") + value = inline if "=" in specification else host_env.get(name, "") + value_bytes = value.encode("utf-8") + if any(encoded and encoded in value_bytes for encoded in key_encodings): + raise ValueError("observer key must not be forwarded in the environment") + if name.startswith("EVAL_OBSERVER_"): + raise ValueError("observer environment names are reserved") + self.run_id = secrets.token_hex(32) + capture_dir = root / "observer" + capture_dir.mkdir(mode=0o777) + capture_dir.chmod(0o777) # writable by the image's non-root UID under Docker + self.path = capture_dir / "records.jsonl" + command[-1:-1] = [ + "--volume", f"{capture_dir}:{MOUNT}:rw", + "--env", "EVAL_OBSERVER_PROTOCOL=1", + "--env", f"EVAL_OBSERVER_RUN_ID={self.run_id}", + "--env", f"EVAL_OBSERVER_PATH={MOUNT}/records.jsonl", + ] + # No automatic forwarding of credentials/signing material. Redact known + # host secrets as a second layer; the producer must sanitize unknown ones. + self.known_secrets = tuple(value for name, value in host_env.items() if value and SENSITIVE_KEY.search(name)) + try: + self.known_secrets += (self.key.decode("utf-8"),) + except UnicodeError: + pass + self.known_secrets += (self.key.hex(), base64.b64encode(self.key).decode("ascii")) + + def finish(self, *, transport_ok: bool) -> dict[str, Any]: + return load_capture(self.path, key=self.key, run_id=self.run_id, + known_secrets=self.known_secrets, transport_ok=transport_ok) diff --git a/runner/protected.py b/runner/protected.py new file mode 100644 index 0000000..c74573e --- /dev/null +++ b/runner/protected.py @@ -0,0 +1,292 @@ +"""Explicit, restricted runtime-to-host capture profile; never trust target files. + +The host launcher, selected immutable runtime image, kernel/container engine, and +private host capture directory are trusted. Evaluated code runs as Code Mode data +or in the separate tool container. In-process untrusted plugins are unsupported. +No HMAC or target-held signing key is used by this profile. +""" +from __future__ import annotations + +import argparse +import hashlib +import json +import os +from pathlib import Path +import re +import secrets +import stat +import subprocess +import sys +import tempfile +from typing import Any + +PROFILE = "codemode-inner/direct-session/v1" +SCHEMA = "opencode-protected-observation/v2" +RUNTIME_SCHEMA = "opencode-local-observation/v1" +MAX_BYTES = 8 * 1024 * 1024 +MAX_EVENTS = 10000 +MAX_FIELD_BYTES = 16 * 1024 +TOKEN = re.compile(r"[A-Za-z0-9_.:/@-]{1,256}\Z") +IMAGE = re.compile(r"[A-Za-z0-9._:/-]+@sha256:[0-9a-f]{64}\Z") + + +class CaptureError(ValueError): + """Only fixed reason codes; never echo raw capture data.""" + + +def _require(ok: bool, reason: str) -> None: + if not ok: + raise CaptureError(reason) + + +def _pairs(items): + result = {} + for key, value in items: + _require(key not in result, "duplicate_key") + result[key] = value + return result + + +def _bad_number(_): + raise CaptureError("nonfinite_number") + + +def strict_json(raw: bytes | str) -> Any: + value = json.loads(raw, object_pairs_hook=_pairs, parse_constant=_bad_number) + json.dumps(value, ensure_ascii=False, allow_nan=False).encode("utf-8") + return value + + +def _token(value) -> bool: + return isinstance(value, str) and bool(TOKEN.fullmatch(value)) + + +def _int(value) -> bool: + return type(value) is int and 0 <= value <= 2**53 - 1 + + +def _shape(value, fields): + _require(isinstance(value, dict) and set(value) == set(fields), "invalid_shape") + + +def _field(value): + _require(isinstance(value, dict), "invalid_field") + state = value.get("state") + _require(state in ("available", "redacted", "omitted", "truncated"), "invalid_field") + if state in ("available", "redacted"): + _shape(value, ("state", "value", "redaction")) + _require(value["redaction"] == "safe", "unsafe_field") + _require(len(json.dumps(value["value"], ensure_ascii=False, separators=(",", ":")).encode("utf-8")) <= MAX_FIELD_BYTES, "field_limit") + return {"state": state, "value": value["value"]} + _shape(value, ("state", "reason")) + _require(value["reason"] in ("policy_omission", "field_limit", "unsupported_snapshot"), "invalid_reason") + return dict(value) + + +def empty_projection(run_id: str) -> dict: + return { + "kind": "execution-observer-projection", "version": 5, + "profile": PROFILE, "run_id": run_id, "status": "unavailable", + "evidence_eligible": False, "full_handoff_eligible": False, + "records": [], "parents": [], "issues": [], + "coverage": {"scope": PROFILE, "native": "unsupported", "delegated_sessions": "unsupported", + "in_process_untrusted_plugins": "unsupported", "capture_started": False, + "capture_ended": False, "starts": 0, "terminals": 0, + "missing_terminals": None, "omitted_records": None, "truncated": False, + "accounting_complete": False}, + } + + +def import_capture(raw: bytes, *, run_id: str, policy_id: str, launch_id: str, receipt: dict | None, tools: set[str], transport_ok: bool) -> dict: + """Validate bytes read by the protected launcher, NOT arbitrary target bytes. + + A checksum detects stream damage; it is not an origin proof. Origin comes from + the launcher's private mount and isolated runtime. This function alone cannot + attest a file, and is deliberately not exposed as a file-import CLI command. + """ + result = empty_projection(run_id) + result["launch_id"] = launch_id + result["collection_profile"] = "private-supervisor-receipt/v1" + parents, calls = {}, {} + accounting_started = False + try: + _require(len(raw) <= MAX_BYTES, "capture_limit") + _shape(receipt, ("sha256", "bytes")) + _require(type(receipt["bytes"]) is int and receipt["bytes"] == len(raw) + and receipt["sha256"] == hashlib.sha256(raw).hexdigest(), "receipt_mismatch") + _require(bool(raw) and raw.endswith(b"\n"), "missing_or_partial_capture") + lines = raw.splitlines(keepends=True) + _require(len(lines) <= MAX_EVENTS + 2, "frame_limit") + _require(len(lines) >= 2, "frame_count") + _require(all(len(line) <= 256 * 1024 for line in lines), "frame_limit") + frames = [strict_json(line) for line in lines] + for seq, frame in enumerate(frames): + _require(isinstance(frame, dict) and frame.get("run_id") == run_id, "wrong_run") + _require(type(frame.get("seq")) is int and frame["seq"] == seq, "sequence_gap") + header, footer = frames[0], frames[-1] + _shape(header, ("kind", "schema", "profile", "run_id", "seq", "policy_id", "launch_id")) + _require(header["kind"] == "capture_start" and header["schema"] == SCHEMA + and header["profile"] == PROFILE and header["policy_id"] == policy_id, "wrong_profile") + _require(header["launch_id"] == launch_id, "wrong_launch") + result["coverage"]["capture_started"] = True + _shape(footer, ("kind", "run_id", "seq", "event_count", "sha256", "writer_exited", "runtime_exit")) + _require(footer["kind"] == "capture_end" and footer["writer_exited"] is True, "missing_capture_end") + _require(type(footer["runtime_exit"]) is int and footer["runtime_exit"] == 0, "runtime_failed") + _require(type(footer["event_count"]) is int and footer["event_count"] == len(frames) - 2, "count_mismatch") + _require(footer["sha256"] == hashlib.sha256(b"".join(lines[:-1])).hexdigest(), "stream_modified") + result["coverage"]["capture_ended"] = True + session = None + accounting_started = True + for expected_source_seq, frame in enumerate(frames[1:-1], 1): + _shape(frame, ("kind", "run_id", "seq", "observation")) + _require(frame["kind"] == "observation", "unknown_record") + event = frame["observation"] + _require(isinstance(event, dict) and event.get("schema") == RUNTIME_SCHEMA, "unknown_runtime_schema") + _require(type(event.get("sequence")) is int and event["sequence"] == expected_source_seq, "source_sequence_gap") + _require(type(event.get("observer_failures")) is int and event["observer_failures"] == 0, "observer_failed") + actor, parent = event.get("actor"), event.get("parent") + _shape(actor, ("agent", "session_id", "message_id")) + _shape(parent, ("invocation_id", "session_id", "message_id", "call_id")) + _require(all(_token(v) for v in (*actor.values(), *parent.values())), "invalid_identity") + _require(actor["session_id"] == parent["session_id"] and actor["message_id"] == parent["message_id"], "conflicting_parent") + session = session or actor["session_id"] + _require(actor["session_id"] == session, "delegated_session_unsupported") + common = {"schema", "sequence", "parent", "actor", "observer_failures", "kind"} + pid, kind = parent["invocation_id"], event.get("kind") + if kind == "parent_start": + _shape(event, common | {"boundary", "mode"}) + _require(event["boundary"] == "codemode-engine" and event["mode"] == "code_mode", "wrong_boundary") + _require(pid not in parents and pid not in calls, "duplicate_parent") + parents[pid] = {"identity": parent, "actor": actor, "start_sequence": frame["seq"], + "terminal_sequence": None, "starts": 0, "terminals": 0} + continue + _require(pid in parents and parents[pid]["terminal_sequence"] is None, "unknown_or_closed_parent") + bound = parents[pid] + _require(parent == bound["identity"] and actor == bound["actor"], "conflicting_identity") + if kind == "call_start": + _shape(event, common | {"invocation_id", "tool", "catalog_path", "input", "boundary"}) + iid = event["invocation_id"] + _require(_token(iid) and iid not in calls and iid not in parents, "duplicate_or_invalid_invocation") + _require(event["tool"] in tools and event["catalog_path"] == "isolated." + event["tool"][len("isolated_"):], "unapproved_registration") + _require(event["boundary"] == "executable-input", "wrong_boundary") + calls[iid] = {"invocation_id": iid, "tool": event["tool"], "catalog_path": event["catalog_path"], + "actor": actor, "parent": parent, "mode": "code_mode", + "runtime_call_id": parent["call_id"], "input": _field(event["input"]), + "start_sequence": frame["seq"], "terminal_sequence": None, + "outcome": "missing", "evidence_eligible": False} + bound["starts"] += 1 + elif kind == "call_end": + outcome = event.get("outcome") + extra = {"result"} if outcome == "returned" else {"error", "error_representation"} if outcome == "threw" else set() + _shape(event, common | {"invocation_id", "dispatched", "boundary", "outcome"} | extra) + iid = event["invocation_id"] + _require(iid in calls and calls[iid]["terminal_sequence"] is None, "ambiguous_terminal") + call = calls[iid] + _require(call["parent"] == parent and call["actor"] == actor, "conflicting_identity") + _require(event["dispatched"] is True and event["boundary"] == "codemode-json-return", "wrong_boundary") + _require(outcome in ("returned", "threw", "interrupted"), "invalid_outcome") + if outcome == "returned": + call["result"] = _field(event["result"]) + elif outcome == "threw": + _require(event["error_representation"] == "codemode-catch-name-message/v1", "unsupported_error_view") + call["error"] = _field(event["error"]) + call["error_representation"] = event["error_representation"] + else: + result["issues"].append("interrupted_call") + # A malformed result must not make its call look settled. + call.update(outcome=outcome, terminal_sequence=frame["seq"]) + bound["terminals"] += 1 + elif kind == "parent_end": + _shape(event, common | {"admitted", "dispatched", "terminals", "missing_terminals", "unsupported_dispatches", + "unavailable_fields", "scope", "evidence_eligible"}) + _require(event["scope"] == "one-codemode-engine-invocation" and event["evidence_eligible"] is False, "wrong_parent_profile") + for name in ("admitted", "dispatched", "terminals", "missing_terminals", "unsupported_dispatches", "unavailable_fields"): + _require(_int(event[name]), "invalid_accounting") + _require(event["admitted"] == event["dispatched"] == bound["starts"] + and event["terminals"] == bound["terminals"], "parent_count_mismatch") + _require(event["missing_terminals"] == event["unsupported_dispatches"] == 0, "incomplete_parent") + if event["unavailable_fields"]: + result["issues"].append("unsupported_snapshot") + bound["terminal_sequence"] = frame["seq"] + else: + raise CaptureError("unknown_runtime_record") + _require(bool(parents) and bool(calls), "empty_capture") + _require(all(p["terminal_sequence"] is not None for p in parents.values()), "missing_parent_end") + _require(all(c["terminal_sequence"] is not None for c in calls.values()), "missing_terminal") + for call in calls.values(): + for name in ("input", "result", "error"): + if name in call and call[name]["state"] != "available": + result["issues"].append("incomplete_field") + if call[name]["state"] == "truncated": + result["coverage"]["truncated"] = True + if not transport_ok: + result["issues"].append("transport_failed") + result["issues"] = sorted(set(result["issues"])) + result["status"] = "incomplete" if result["issues"] else "complete" + result["evidence_eligible"] = not result["issues"] + result["coverage"].update(starts=len(calls), terminals=sum(c["terminal_sequence"] is not None for c in calls.values()), + missing_terminals=0, omitted_records=0, accounting_complete=True) + for call in calls.values(): + call["evidence_eligible"] = result["evidence_eligible"] + result["records"], result["parents"] = list(calls.values()), list(parents.values()) + except (CaptureError, ValueError, TypeError, KeyError, UnicodeError, RecursionError) as exc: + result["status"] = "invalid" + result["issues"] = [str(exc) if isinstance(exc, CaptureError) else "malformed_capture"] + result["records"], result["parents"] = [], [] + result["evidence_eligible"] = False + if isinstance(exc, CaptureError) and str(exc) in {"capture_limit", "frame_limit", "field_limit"}: + result["coverage"]["truncated"] = True + if accounting_started: + # Verified-prefix counts are diagnostic only. An invalid suffix can hide + # more calls, so completeness and omitted_records remain separately unknown. + terminals = sum(call["terminal_sequence"] is not None for call in calls.values()) + result["coverage"].update(starts=len(calls), terminals=terminals, + missing_terminals=len(calls) - terminals) + return result + + +def _run(args, *, timeout=120, **kwargs): + return subprocess.run(args, check=True, capture_output=True, text=True, timeout=timeout, **kwargs) + + +def read_private(path: Path) -> bytes: + fd = os.open(path, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK) + try: + info = os.fstat(fd) + _require(stat.S_ISREG(info.st_mode) and info.st_nlink == 1 and info.st_size <= MAX_BYTES, "unsafe_capture_file") + with os.fdopen(fd, "rb", closefd=False) as stream: + data = stream.read(MAX_BYTES + 1) + _require(len(data) <= MAX_BYTES, "capture_limit") + return data + finally: + os.close(fd) + + +def invoke(args, *, _test_receive=None, _test_prepare=None, _test_target_probe=None) -> int: + from .protected_launch import invoke as launch + return launch(args, _test_receive=_test_receive, _test_prepare=_test_prepare, _test_target_probe=_test_target_probe) + + +def parser(): + p = argparse.ArgumentParser(description="Protected, provider-free direct-session Code Mode capture; in-process target plugins are unsupported.") + p.add_argument("--image", required=True, help="Explicitly trusted immutable protected runtime image") + p.add_argument("--tool-image", required=True, help="Untrusted isolated JSON tool server image, listening on port 8080") + p.add_argument("--program-file", required=True) + p.add_argument("--tools-file", required=True) + p.add_argument("--policy-file", required=True) + p.add_argument("--output", required=True) + p.add_argument("--timeout", type=int, default=90) + p.add_argument("--no-observe", action="store_true", help="Same restricted profile with capture disabled, for behavior comparison") + return p + + +def main(): + try: + return invoke(parser().parse_args()) + except (CaptureError, OSError, ValueError): + print("protected-capture: invalid configuration", file=sys.stderr) + return 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/runner/protected_launch.py b/runner/protected_launch.py new file mode 100644 index 0000000..72bc546 --- /dev/null +++ b/runner/protected_launch.py @@ -0,0 +1,239 @@ +"""Trusted host launch and private collection; not an unsigned-file verifier.""" +from __future__ import annotations +import argparse +import hashlib +import json +import os +from pathlib import Path +import re +import secrets +import stat +import subprocess +import tempfile + +from .protected import (CaptureError, IMAGE, PROFILE, _require, _shape, empty_projection, + import_capture, read_private, strict_json) + + +def run(command, *, timeout=120): + return subprocess.run(command, check=True, capture_output=True, text=True, timeout=timeout) + + +def bounded_file(path, limit): + # Open each input once, then execute this snapshot, not a reopened original. + fd = os.open(path, os.O_RDONLY | os.O_NONBLOCK) + with os.fdopen(fd, "rb") as source: + _require(stat.S_ISREG(os.fstat(source.fileno()).st_mode), "input_not_regular") + raw = source.read(limit + 1) + _require(len(raw) <= limit, "input_limit") + return raw + + +def image_info(reference): + run(["docker", "pull", reference]) + info = strict_json(run(["docker", "image", "inspect", reference]).stdout)[0] + _require(reference in info.get("RepoDigests", []), "image_digest_not_resolved") + _require(bool(re.fullmatch(r"sha256:[0-9a-f]{64}", info.get("Id", ""))), "invalid_image_id") + # Docker otherwise creates anonymous persistent storage before the later + # mount audit can reject it. This profile permits no image-declared volumes. + _require(not (info.get("Config") or {}).get("Volumes"), "image_declares_volumes") + return info + + +OWNER_LABEL = "io.opencode-eval-runner.capture-owner" + + +def cleanup_resources(attempted, runtime, target, network, owner): + """Reconcile create attempts, including lost/timeout create responses. + + A successful inventory with no matching name establishes absence. An + unreachable engine or malformed inventory is a cleanup failure, never + absence. Inspect the run label and delete by immutable resource ID so a + same-named resource from another run is not removed. + """ + failures = [] + for resource, name in (("runtime", runtime), ("target", target), ("network", network)): + if not attempted.get(resource, False): + continue + kind = "network" if resource == "network" else "container" + command = ["docker", kind, "ls", "--no-trunc", "--filter", "name=" + name, + "--format", "{{json .}}"] + if kind == "container": + command.append("--all") + try: + listed = subprocess.run(command, capture_output=True, text=True, timeout=20, check=False) + _require(listed.returncode == 0, "cleanup_failed") + rows = [strict_json(line) for line in listed.stdout.splitlines() if line.strip()] + name_field = "Name" if kind == "network" else "Names" + _require(all(isinstance(row, dict) and isinstance(row.get(name_field), str) for row in rows), "cleanup_failed") + matches = [row for row in rows if row[name_field].lstrip("/") == name] + if not matches: + continue + _require(len(matches) == 1, "cleanup_failed") + identity = matches[0].get("ID") + _require(isinstance(identity, str) and bool(re.fullmatch(r"[0-9a-f]{64}", identity)), "cleanup_failed") + inspected = subprocess.run(["docker", kind, "inspect", identity], + capture_output=True, text=True, timeout=20, check=False) + _require(inspected.returncode == 0, "cleanup_failed") + snapshots = strict_json(inspected.stdout) + _require(isinstance(snapshots, list) and len(snapshots) == 1, "cleanup_failed") + snapshot = snapshots[0] + labels = snapshot.get("Labels") if kind == "network" else (snapshot.get("Config") or {}).get("Labels") + _require(isinstance(labels, dict) and labels.get(OWNER_LABEL) == owner, "cleanup_failed") + _require(snapshot.get("Id") == identity and snapshot.get("Name", "").lstrip("/") == name, "cleanup_failed") + removal = (["docker", "network", "rm", identity] if kind == "network" else + ["docker", "rm", "--volumes", "-f", identity]) + cleaned = subprocess.run(removal, capture_output=True, timeout=20, check=False) + _require(cleaned.returncode == 0, "cleanup_failed") + except (OSError, subprocess.SubprocessError, ValueError, TypeError, KeyError, AttributeError): + failures.append(resource) + return failures + + +def invoke(args: argparse.Namespace, *, _test_receive=None, _test_prepare=None, _test_target_probe=None) -> int: + for reference in (args.image, args.tool_image): + _require(bool(IMAGE.fullmatch(reference)), "immutable_image_required") + _require(type(args.timeout) is int and 1 <= args.timeout <= 300, "invalid_timeout") + policy_raw = bounded_file(args.policy_file, 1024 * 1024) + policy = strict_json(policy_raw) + _shape(policy, ("version", "secrets", "allowed_values")) + _require(type(policy["version"]) is int and policy["version"] == 1, "invalid_policy") + _require(isinstance(policy["allowed_values"], list) and isinstance(policy["secrets"], list) + and all(isinstance(s, str) and s for s in policy["secrets"]), "invalid_policy") + tools_raw = bounded_file(args.tools_file, 1024 * 1024) + tools = strict_json(tools_raw) + _require(isinstance(tools, list) and 0 < len(tools) <= 64, "invalid_tools") + for tool in tools: + _require(isinstance(tool, dict) and {"name", "input"} <= tool.keys() + and not tool.keys() - {"name", "input", "description"}, "invalid_tools") + _require(isinstance(tool["name"], str) and bool(re.fullmatch(r"[a-z][a-z0-9_]{0,63}", tool["name"])), "invalid_tools") + _require(isinstance(tool["input"], dict) and isinstance(tool.get("description", ""), str), "invalid_tools") + _require(len({tool["name"] for tool in tools}) == len(tools), "duplicate_tool") + program_raw = bounded_file(args.program_file, 128 * 1024) + program = program_raw.decode("utf-8") + run_id = secrets.token_hex(32) + prefix = "capture-" + run_id[:16] + network, target, runtime = prefix + "-net", prefix + "-tools", prefix + "-runtime" + output = Path(os.path.abspath(args.output)) + _require(all(output.resolve() != Path(p).resolve() for p in (args.program_file, args.tools_file, args.policy_file)), "output_overwrites_input") + output.parent.mkdir(parents=True, exist_ok=True) + policy_id = hashlib.sha256(policy_raw).hexdigest() + repo = Path(__file__).resolve().parents[1] + collector_sources = {name: hashlib.sha256((repo / name).read_bytes()).hexdigest() for name in + ("runner/protected.py", "runner/protected_launch.py", "bin/opencode-eval-runner")} + launch = {"collector_sources_sha256": collector_sources, "run_id": run_id, "profile": PROFILE, "runtime_image": args.image, "tool_image": args.tool_image, + "policy_sha256": policy_id, "program_sha256": hashlib.sha256(program_raw).hexdigest(), + "tools_sha256": hashlib.sha256(tools_raw).hexdigest()} + result = {"schema": "opencode-eval-runner/v1", "profile": PROFILE, "run_id": run_id, + "runtime_image": args.image, "tool_image": args.tool_image, "policy_id": policy_id, + "observed_execution": empty_projection(run_id), + "image_signatures_verified": False, "artifact_signed": False} + code = 4 + with tempfile.TemporaryDirectory(prefix="protected-capture-") as tmp: + root = Path(tmp) + inputs, capture = root / "input", root / "capture" + inputs.mkdir(mode=0o755) + capture.mkdir(mode=0o777) + capture.chmod(0o777) # Only its parent is host-private; the target gets no mount. + hardening = ["--read-only", "--cap-drop", "ALL", "--security-opt", "no-new-privileges", + "--pids-limit", "128", "--memory", "1g", "--cpus", "2", "--user", "1000:1000", + "--log-driver", "none"] + attempted = {"network": False, "target": False, "runtime": False} + try: + runtime_info, target_info = image_info(args.image), image_info(args.tool_image) + launch.update(runtime_config_digest=runtime_info["Id"], tool_config_digest=target_info["Id"]) + launch_raw = json.dumps(launch, sort_keys=True, separators=(",", ":")).encode() + launch_id = hashlib.sha256(launch_raw).hexdigest() + request = {"run_id": run_id, "program": program, "tools": tools, "tool_url": "http://" + target + ":8080", + "policy_id": policy_id, "launch_id": launch_id, "observe": not args.no_observe} + (inputs / "request.json").write_text(json.dumps(request)) + (inputs / "policy.json").write_bytes(policy_raw) + (inputs / "tools.json").write_bytes(tools_raw) + (inputs / "launch.json").write_bytes(launch_raw) + for file in inputs.iterdir(): + file.chmod(0o644) + result.update(launch=launch, launch_id=launch_id) + if _test_prepare is not None: + _test_prepare(capture) + attempted["network"] = True + run(["docker", "network", "create", "--internal", "--label", OWNER_LABEL + "=" + run_id, network]) + # Create and start are separate so cleanup ownership is established + # before an image entrypoint/start failure can occur. + attempted["target"] = True + run(["docker", "create", "--name", target, "--network", network, + "--label", OWNER_LABEL + "=" + run_id, *hardening, + "--tmpfs", "/tmp:rw,nosuid,nodev,size=64m", args.tool_image]) + run(["docker", "start", target]) + # No arbitrary mounts, plugin roots, host credentials or command override. + create_runtime = ["docker", "create", "--name", runtime, "--network", network, + "--label", OWNER_LABEL + "=" + run_id, *hardening, + "--tmpfs", "/tmp:rw,exec,nosuid,nodev,size=1g", + "--tmpfs", "/workspace:rw,nosuid,nodev,size=32m,mode=1777", + "--volume", str(inputs) + ":/input:ro", "--volume", str(capture) + ":/capture:rw", + "--entrypoint", "python3", args.image, "/opt/protected/invoke.py"] + attempted["runtime"] = True + run(create_runtime) + completed = subprocess.run( + ["docker", "start", "--attach", runtime], + capture_output=True, text=True, timeout=args.timeout, check=False, + ) + runtime_state = strict_json(run(["docker", "inspect", runtime]).stdout)[0] + # Verify the engine's actual container identities, not claims in JSON. + target_state = strict_json(run(["docker", "inspect", target]).stdout)[0] + _require(runtime_state["Image"] == launch["runtime_config_digest"] + and target_state["Image"] == launch["tool_config_digest"], "launched_image_mismatch") + _require(runtime_state["State"]["Running"] is False, "writer_still_running") + _require(not target_state.get("Mounts") and not target_state["HostConfig"].get("PidMode"), "unexpected_target_access") + result["launched_images_verified"] = True + envelope = strict_json(completed.stdout) + _shape(envelope, ("profile", "run_id", "launch_id", "runtime_exit", "script_output", "writer_exited", "receipt")) + _require(envelope["profile"] == PROFILE and envelope["run_id"] == run_id + and envelope["launch_id"] == launch_id, "invalid_supervisor") + result["collection_receipt"] = envelope["receipt"] + result.update(runtime_exit=envelope["runtime_exit"], script_output=envelope["script_output"]) + transport_ok = completed.returncode == 0 and type(envelope["runtime_exit"]) is int and envelope["runtime_exit"] == 0 and envelope["writer_exited"] is True + if args.no_observe: + result["observed_execution"]["issues"] = ["capture_not_requested"] + code = 0 if transport_ok else 2 + else: + try: + raw = read_private(capture / "events.jsonl") + except FileNotFoundError: + raw = b"" + if _test_receive is not None: + raw = _test_receive(raw, run_id, policy_id) + result["observed_execution"] = import_capture(raw, run_id=run_id, policy_id=policy_id, launch_id=launch_id, + receipt=envelope["receipt"], tools={"isolated_" + t["name"] for t in tools}, transport_ok=transport_ok) + code = 0 if result["observed_execution"]["evidence_eligible"] else 4 + # Test-only oracle access. Never a CLI option or an evidence producer. + if _test_target_probe is not None: + _test_target_probe(target, runtime_state, target_state) + except CaptureError as exc: + result["observed_execution"] = empty_projection(run_id) + result["observed_execution"]["issues"] = [str(exc)] + code = 2 + except (OSError, subprocess.SubprocessError, ValueError, TypeError, KeyError, RecursionError): + result["observed_execution"] = empty_projection(run_id) + result["observed_execution"]["issues"] = ["protected_transport_failed"] + code = 2 + finally: + cleanup_failures = cleanup_resources(attempted, runtime, target, network, run_id) + if cleanup_failures: + result["observed_execution"]["evidence_eligible"] = False + result["observed_execution"]["status"] = "incomplete" + if "cleanup_failed" not in result["observed_execution"]["issues"]: + result["observed_execution"]["issues"].append("cleanup_failed") + result["cleanup_failures"] = cleanup_failures + for record in result["observed_execution"]["records"]: + record["evidence_eligible"] = False + code = 2 + # Atomically replace a host artifact without following an existing symlink. + fd, staged = tempfile.mkstemp(prefix=".capture-", dir=output.parent) + try: + with os.fdopen(fd, "w") as stream: + json.dump(result, stream, indent=2) + stream.write("\n") + os.replace(staged, output) + finally: + Path(staged).unlink(missing_ok=True) + return code diff --git a/runner/safe_invoke.py b/runner/safe_invoke.py new file mode 100644 index 0000000..d28e99c --- /dev/null +++ b/runner/safe_invoke.py @@ -0,0 +1,276 @@ +"""Opt-in normal invoke safety at host sinks, with explicit image handshake. + +This is not an alternate tool/session runner. It uses cli's normal command and +input resolution; only the evidence channel and admission are changed. +""" +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path, PurePosixPath +import re +import secrets +import shutil +import subprocess +import sys +import tempfile + +from container import evidence_safety as s + + +def load_policy(path): + try: + with Path(path).open('rb') as f: + raw = f.read(s.POLICY_LIMIT + 1) + s.require(len(raw) <= s.POLICY_LIMIT) + return s.Policy(s.strict_loads(raw)) + except (ValueError, TypeError, OSError, RecursionError): + return s.Policy() + + +def audit_selected_inputs(policy, command, host_env, explicit_sources=None, state_profile='default'): + """Downgrade omissions/defaults not represented by Loom's selected profile. + + This does not discover new real state, classify arbitrary configs, or infer + credentials from public values. Loom owns source/path classification. + """ + if not policy.complete: + return policy + selected = set() + categories = {'/seed/auth.json': 'auth', '/seed/opencode.json': 'config', + '/seed/models.json': 'models', '/seed/opencode.db': 'credential_seed', + '/seed/opencode-config': 'config_root'} + contradictory = False + for index, arg in enumerate(command[:-1]): + value = command[index + 1] + if arg == '--volume': + target = value.rsplit(':', 2)[1] + if target in categories: + selected.add(categories[target]) + # The legacy/default profile has no path-role contract for an + # ambient plugin config root. The disposable profile does: + # explicit --config-root is allowed when Loom marks config_root + # complete and the explicit-source set agrees. + contradictory |= ( + target == '/seed/opencode-config' and state_profile != 'disposable' + ) + elif arg == '--env': + name, _, inline = value.partition('=') + actual = inline if '=' in value else host_env.get(name, '') + if s.sensitive_key(name) and actual: + selected.add('env') + contradictory |= actual not in policy.values + states = policy.private['sources'] + contradictory |= any(states[category] != 'complete' for category in selected) + if state_profile == 'disposable': + # The disposable profile makes default database/auth/config selection an + # explicit runner guarantee. Policy source states must describe the + # actually selected explicit inputs, not an ambient fallback. + expected = { + 'auth': 'complete' if 'auth' in selected else 'not_selected', + 'config': 'complete' if 'config' in selected else 'not_selected', + 'models': 'complete' if 'models' in selected else 'not_selected', + 'credential_seed': 'not_selected', + 'config_root': 'complete' if 'config_root' in selected else 'not_selected', + } + contradictory |= any(states[name] != value for name, value in expected.items()) + if explicit_sources is not None: + # The v1 policy has no path bindings for runner-resolved defaults. + # Never infer that a category label describes an ambient fallback. + # Environment selection is bound by the actual forwarded name/value + # above, not by a seed path. The explicit-source set covers only + # mount/path-backed seed categories. + contradictory |= bool((selected - {'env'}) - explicit_sources) + if contradictory: + private = dict(policy.private) + private['complete'] = False + private['values'] = [] + return s.Policy(private) + return policy + + +def canonical_image_config_id(value): + """Accept Docker/Podman config-ID renderings and return canonical + engine ref.""" + s.require(type(value) is str) + if re.fullmatch(r'sha256:[0-9a-f]{64}', value): + return value, value + if re.fullmatch(r'[0-9a-f]{64}', value): + return 'sha256:' + value, value + raise s.Invalid('invalid') + + +def resolved_image(command): + reference = command[-1] + s.require(type(reference) is str and re.fullmatch(r'[A-Za-z0-9._:/-]+@sha256:[0-9a-f]{64}', reference) is not None) + engine = command[0] + # No selected credentials, inputs or policy are provided to this preflight. + inspect = subprocess.run([engine, 'image', 'inspect', reference], capture_output=True, timeout=120, check=False) + if inspect.returncode: + pulled = subprocess.run([engine, 'pull', reference], capture_output=True, timeout=120, check=False) + s.require(pulled.returncode == 0) + inspect = subprocess.run([engine, 'image', 'inspect', reference], capture_output=True, timeout=120, check=False) + s.require(inspect.returncode == 0) + info = s.strict_loads(inspect.stdout)[0] + labels = info.get('Config', {}).get('Labels') or {} + s.require(reference in info.get('RepoDigests', [])) + checkout = Path(__file__).resolve().parents[1] + init_sha = hashlib.sha256((checkout / 'container/__init__.py').read_bytes()).hexdigest() + invoke_sha = hashlib.sha256((checkout / 'container/invoke.py').read_bytes()).hexdigest() + s.require(labels.get('io.opencode-eval.evidence-safety') == s.CONSUMER) + s.require(labels.get('io.opencode-eval.evidence-safety-init') == init_sha) + s.require(labels.get('io.opencode-eval.evidence-safety-module') == s.module_sha()) + s.require(labels.get('io.opencode-eval.evidence-safety-invoke') == invoke_sha) + revision = labels.get('org.opencontainers.image.revision') + s.require(type(revision) is str and re.fullmatch('[0-9a-f]{40}', revision) is not None) + canonical_config_id, engine_config_ref = canonical_image_config_id(info.get('Id')) + # Execute the exact inspected local config object. Docker renders this as + # sha256:<64>; Podman may render the same content ID as bare <64>. + # Never fall back to the mutable tag/digest text after inspection. + command[-1] = engine_config_ref + return { + 'image': reference, + 'image_config': canonical_config_id, + 'image_source_revision': revision, + 'image_package_init_sha256': init_sha, + 'image_policy_module_sha256': s.module_sha(), + 'image_invoke_sha256': invoke_sha, + } + + +def execute_container(command, request, timeout, run_id, host_env=None): + """Create before attach so client timeouts still have an owned cleanup name.""" + engine = command[0] + name = 'rsp-' + run_id[:32] + create = [engine, 'create', *command[2:]] + create.remove('--rm') + create[-1:-1] = ['--name', name] + try: + proc = subprocess.run(create, env=host_env, capture_output=True, timeout=120, check=False) + s.require(proc.returncode == 0) + return subprocess.run([engine, 'start', '--attach', '--interactive', name], + input=request, capture_output=True, timeout=timeout, check=False) + finally: + # This randomly named invocation is owned before attach can time out. + # No returned exception/engine output is reflected into evidence. + removed = subprocess.run([engine, 'rm', '--force', '--volumes', name], + capture_output=True, timeout=30, check=False) + s.require(removed.returncode == 0) + + +def write_projection(path, result, print_result): + """Atomic, private first write, never raw stdout or a raw result tempfile.""" + raw = s.encode(result) + s.require(len(raw) <= s.RESULT_LIMIT) + path = Path(path) + path.parent.mkdir(parents=True, exist_ok=True) + fd, temporary = tempfile.mkstemp(prefix='.safe-evidence-', dir=path.parent) + try: + with os.fdopen(fd, 'wb') as stream: + stream.write(raw + b'\n') + stream.flush() + os.fsync(stream.fileno()) + os.replace(temporary, path) + finally: + Path(temporary).unlink(missing_ok=True) + if print_result: + sys.stdout.write(raw.decode('utf-8') + '\n') + + +def invoke(args): + from runner import cli + result = s.fallback('invalid') + code = 4 + try: + # Do not accidentally treat the legacy HMAC path as this acknowledgement. + s.require(not getattr(args, 'observer_key_file', None), 'unsupported_schema') + policy = load_policy(getattr(args, 'evidence_policy_file', None)) + run_id = secrets.token_hex(32) + binding_key = secrets.token_bytes(32) + s.require(type(args.container_timeout) is int and args.container_timeout > 0) + # Preserve ordinary invoke mounts; reject safety negotiation if a mount + # could replace the reviewed emitter/interpreter, rather than silently + # changing that mount or running an unprotected image. + for spec in args.mount: + parts = spec.rsplit(':', 2) + target = parts[-2] if parts[-1] in {'ro', 'rw'} else parts[-1] + target = PurePosixPath(target) + s.require(target.is_absolute() and '..' not in target.parts and str(target).startswith('/workspace/'), 'unsupported_schema') + with tempfile.TemporaryDirectory(prefix='runner-safe-invoke-') as tmp: + root = Path(tmp) + inputs, output = root / 'input', root / 'output' + inputs.mkdir() + output.mkdir() + # Authorized execution inputs are distinct from evidence copies. + shutil.copyfile(args.prompt_file, inputs / 'prompt.txt') + if args.system_file: + shutil.copyfile(args.system_file, inputs / 'system.txt') + else: + (inputs / 'system.txt').write_text('') + host_env = cli.host_environment_for_transport(args.transport) + state_profile = cli.opencode_state_profile(args, host_env) + database = cli.resolve_database_seed(args, root / 'credentials.db', host_env) + command, _ = cli.build_container_command(args, inputs, output, host_env, database) + explicit = {name for name, value in { + 'auth': args.auth, 'config': args.config, 'models': args.models_catalog, + 'credential_seed': args.database, 'config_root': args.config_root}.items() if value} + policy = audit_selected_inputs(policy, command, host_env, explicit, state_profile) + if state_profile == 'disposable' and not policy.complete: + result = s.fallback('inventory_incomplete', 'transport', policy) + write_projection(args.output, result, args.print_result) + return 4 + loaded = resolved_image(command) + # Send only to the supervisor's consumed stdin; do not expose policy + # values in a bind-mounted file, argv, or inherited runtime env. + command[-1:-1] = ['-i', '--log-driver', 'none', '--env', 'EVAL_EVIDENCE_SAFETY=1'] + request = s.encode({'schema': s.REQUEST, 'run_id': run_id, 'policy': policy.private, 'binding_key': binding_key.hex()}) + s.require(len(request) <= s.WIRE_LIMIT) + # Environment-selected normal --env values must reach docker create; + # the private policy still travels only on the attached stdin. + proc = execute_container(command, request, args.container_timeout, run_id, host_env) + try: + result = s.validate_reply(proc.stdout, policy, run_id, loaded['image_source_revision'], binding_key) + except (ValueError, TypeError, KeyError, RecursionError): + result = s.fallback('unsupported_schema') + else: + runtime_state = result.get('runtime_state') + if state_profile == 'disposable': + s.require(type(runtime_state) is dict and + runtime_state.get('profile') == 'disposable' and + runtime_state.get('database_source') == 'runtime-bootstrap' and + runtime_state.get('database_seed_present') is False and + runtime_state.get('database_created') is True) + else: + s.require(runtime_state is None) + loaded['host_executable_sha256'] = hashlib.sha256( + (Path(__file__).resolve().parents[1] / 'bin/opencode-eval-runner').read_bytes()).hexdigest() + loaded['host_adapter_sha256'] = hashlib.sha256(Path(__file__).read_bytes()).hexdigest() + loaded['policy_module_sha256'] = s.module_sha() + # Code provenance is generated by the host, never accepted from + # a container reply or target output. No source policy is copied. + checkout = Path(__file__).resolve().parents[1] + try: + rev = subprocess.run(['git', '-C', str(checkout), 'rev-parse', 'HEAD'], + capture_output=True, timeout=5, check=False).stdout.decode().strip() + dirty = subprocess.run(['git', '-C', str(checkout), 'status', '--porcelain', '--untracked-files=no'], + capture_output=True, timeout=5, check=False) + loaded['host_source_revision'] = rev if re.fullmatch('[0-9a-f]{40}', rev) else None + loaded['host_tracked_tree_clean'] = dirty.returncode == 0 and not dirty.stdout + except (OSError, subprocess.SubprocessError, UnicodeError): + loaded['host_source_revision'] = None + loaded['host_tracked_tree_clean'] = False + result['evidence_load'] = loaded + result['evidence_safety_validation'] = {'schema': 'opencode-eval-runner/evidence-safety-validation/v1', + 'acknowledged': True, 'stages': ['host.before_write', 'host.before_print']} + code = proc.returncode or (0 if policy.complete else 4) + write_projection(args.output, result, args.print_result) + return code + except (Exception,): + # Includes timeouts with partial stdout/stderr and default-resolution + # errors. Never format exception text or a command containing secrets. + result = s.fallback('invalid') + try: + write_projection(args.output, result, args.print_result) + except Exception: + print('runner evidence write failed', file=sys.stderr) + return 2 diff --git a/runtime-patches/Containerfile b/runtime-patches/Containerfile new file mode 100644 index 0000000..8294793 --- /dev/null +++ b/runtime-patches/Containerfile @@ -0,0 +1,14 @@ +# Semantics-preserving downstream runtime used underneath the normal runner invoke path. +# The restricted protected-channel image remains a separate supplemental profile. +FROM ghcr.io/bateau84/opencode-eval-runner@sha256:68ef7322c75aede0e8cc76d0e3531e8b82dd417bbb5e5100264a89eab7fe8627 +ARG RUNNER_REVISION +LABEL org.opencontainers.image.revision=$RUNNER_REVISION \ + io.opencode.runtime.source=cd9a14a6b688d4021bee381dfd39d2cef9c0f862 \ + io.opencode.runtime.version=2.0.18-eval.5 \ + io.opencode.runtime.patch=host-semantics-observation-v3 \ + io.opencode.runtime.entrypoint=normal-invoke \ + io.opencode.runtime.trust=unprotected-in-process-observer +ENV OPENCODE_EVAL_OBSERVATIONS=1 \ + OPENCODE_EVAL_HOST_OBSERVATIONS=1 +COPY --chmod=0755 opencode /usr/local/bin/opencode +RUN opencode --version diff --git a/runtime-patches/README.md b/runtime-patches/README.md new file mode 100644 index 0000000..8404572 --- /dev/null +++ b/runtime-patches/README.md @@ -0,0 +1,132 @@ +# Local OpenCode observation patch + +This is a downstream patch maintained **in this runner repository**, not an +upstream OpenCode PR and not another Code Mode interpreter. The normal release +image and host installation remain unchanged. + +The experimental **normal-invoke** image builds the existing OpenCode CLI from +`cd9a14a6b688d4021bee381dfd39d2cef9c0f862` (v2.0.18). `apply.py` verifies every +changed upstream blob before applying exact substitutions. + +## Compatibility boundary + +The patch exists underneath Loom's current path: + +```text +bun run eval:live ... + -> scripts/run-evals.py + -> opencode-eval-runner invoke + -> patched OpenCode +``` + +No alternate Loom case runner is introduced. The restricted `observe` command and +remote-tool profile are a separate supplemental experiment and are not applied to +this image. + +The image enables: + +```text +OPENCODE_EVAL_OBSERVATIONS=1 +OPENCODE_EVAL_HOST_OBSERVATIONS=1 +``` + +Those flags only expose diagnostic hook events. They do not change tool inputs, +permissions, session selection, provider flow or normal result handling. + +## Runtime seams + +### Native/direct tool boundary + +A plugin may subscribe to: + +```ts +ctx.tool.hook("execute.native-observed", callback) +``` + +The central Tool service emits `opencode-native-observation/v2` starts and final +returns/errors for normal native calls, including the outer `execute` call. + +The start is emitted after input decoding, using the real Tool.Context: +session, agent, assistant message and runtime call ID. The terminal is emitted +after existing `execute.after` hooks and final content normalization. + +### Code Mode inner boundary + +The existing downstream seam remains: + +```ts +ctx.tool.hook("execute.observed", callback) +``` + +It emits `opencode-local-observation/v1` records for actual Code Mode inner calls. +The runtime-generated inner invocation ID survives to the final interpreter +terminal. Returned values are captured after Code Mode conversion; caught throws +use the explicit `codemode-catch-name-message/v1` view. + +Native and inner events share one monotonic runtime sequence. With host-semantics +observation enabled, the inner parent invocation ID is the actual native outer +`execute` observation ID; it is not reconstructed from input equality or FIFO. + +The observation metadata is kept outside tool input. Existing product call IDs +and Tool.Context values are not replaced. + +## Evidence boundary + +These events are **diagnostic, not protected evidence**. + +The normal Loom plugin is loaded inside the same OpenCode process as the runtime +seam. Code in that process shares the authority needed to reach same-process +files, sockets, descriptors and any signing/collector capability exposed there. +A hidden filename, HMAC key, random FD or localhost listener does not establish a +supported isolation boundary against arbitrary in-process plugin code. + +Therefore the normal-invoke image proves semantic feasibility, not target-unforgeable +collection. Do not turn these events into PASS evidence merely because their +schema validates. + +The restricted direct-session profile proves a different property by moving +evaluated tools out of collector authority. It intentionally changes the host +execution shape and cannot replace normal Loom eval semantics. + +## Build and verification + +The `Normal-invoke runtime observation seam` workflow: + +1. applies only `apply.py` — **not** the restricted `apply-protected.py`; +2. runs source tests against the real core/interpreter; +3. builds `OpenCode 2.0.18-eval.5`; +4. publishes a commit/run-scoped immutable image in a separate credentialed job; +5. probes the runtime with a deterministic local provider; +6. runs `run_eval_live_compat_probe.py` through the public + `opencode-eval-runner invoke` CLI. + +The compatibility probe verifies that the existing runner result behavior remains +usable while native and Code Mode observations are emitted underneath it. It uses +disposable HOME/XDG/workspace state and no real-provider credentials. + +The runtime probe additionally covers overlapping identical inner calls, reverse +completion, actual actor/input bindings, parent linkage, final modified returns, +caught throws and observer callback failure. + +## Remaining feasibility boundary + +The central runtime seam is suitable for Loom to test its existing host behavior +without moving its tool implementations behind a new service. + +What is **not** solved by this patch is a protected collector for an arbitrary +in-process Loom plugin. Achieving that requires a real plugin isolation/capability +boundary while retaining the APIs and lifecycle semantics Loom depends on. That is +a broader runtime project and must not be implemented implicitly in PR #41. + +See [Loom host-semantics feasibility](../docs/loom-host-semantics-feasibility.md). + + +## Loom-facing contract + +The exact event fields, enablement, ordering, value/error semantics, redaction +boundary, completeness limits and unsupported coverage are defined in +[docs/loom-normal-invoke-observation.md](../docs/loom-normal-invoke-observation.md). + +A bounded proposal for stronger plugin isolation, without replacing `invoke`, is +documented in +[docs/plugin-isolation-feasibility.md](../docs/plugin-isolation-feasibility.md). diff --git a/runtime-patches/apply-protected.py b/runtime-patches/apply-protected.py new file mode 100644 index 0000000..0a8b02d --- /dev/null +++ b/runtime-patches/apply-protected.py @@ -0,0 +1,34 @@ +#!/usr/bin/env python3 +"""Apply the pinned runtime patch, then restrict its protected-channel profile.""" +from pathlib import Path +import shutil +import sys +from apply import apply + + +def apply_protected(root: Path): + # apply() first verifies the upstream revision and every original blob. + apply(root) + path = root / "packages/core/src/codemode/tool.ts" + text = path.read_text() + replacements = [ + (' const executed = yield* executeTool(name, tool, input, context,', + ' const dispatchedContext = process.env.OPENCODE_EVAL_PROTECTED_CHANNEL === "1" && invocation\n' + ' ? Object.assign({}, context, { evaluationInvocationID: invocation.id, evaluationDispatchOrdinal: index })\n' + ' : context\n' + ' const executed = yield* executeTool(name, tool, input, dispatchedContext,'), + ('extensions: [CodeModeWeb.extension], hooks', + 'extensions: process.env.OPENCODE_EVAL_PROTECTED_CHANNEL === "1" ? [] : [CodeModeWeb.extension], hooks'), + ] + for old, new in replacements: + if text.count(old) != 1: + raise ValueError("protected runtime patch anchor mismatch") + text = text.replace(old, new, 1) + path.write_text(text) + shutil.copyfile(Path(__file__).with_name("protected-profile.test.ts"), + root / "packages/core/test/protected-profile.test.ts") + print("Applied protected profile: no script network extension; runtime-carried transport identity") + + +if __name__ == "__main__": + apply_protected(Path(sys.argv[1]).resolve()) diff --git a/runtime-patches/apply.py b/runtime-patches/apply.py new file mode 100755 index 0000000..fe5c7ba --- /dev/null +++ b/runtime-patches/apply.py @@ -0,0 +1,187 @@ +#!/usr/bin/env python3 +"""Apply the downstream patch to exactly OpenCode v2.0.18 sources.""" +import hashlib +from pathlib import Path +import shutil +import subprocess +import sys + +REVISION = "cd9a14a6b688d4021bee381dfd39d2cef9c0f862" +PINS = { + "packages/codemode/src/tool-runtime.ts": "9c179bbfe070959704bc6e750e6e2d0045da2b6a", + "packages/codemode/src/tool.ts": "1b23d20c36839aded39df3eca4d666aa50122b36", + "packages/codemode/src/codemode.ts": "edae39aeebf996f57baf19115cbc66638548445a", + "packages/codemode/src/interpreter/errors.ts": "587347645f182c4a1742b3d47748c9dfb5510aff", + "packages/core/src/codemode/tool.ts": "74742f64997e085aee5e8ee15dba30ee63bf5a6a", + "packages/core/src/tool.ts": "5e2ca8401aa550b1bd980cd9d7f513a3db9da0bc", + "packages/core/src/tool/runtime.ts": "f21c67a533ed942c06747f51474908bce099fc33", + "packages/core/src/session/runner/publish-llm-event.ts": "03367a8ed1b020f031d9f75eaedcb4a215ba6d18", + "packages/plugin/src/effect/tool.ts": "04494b2630eda63a169a8905815b438fae8358ba", +} + + +def apply(root: Path): + revision = subprocess.check_output(["git", "rev-parse", "HEAD"], cwd=root, text=True).strip() + if revision != REVISION: + raise ValueError("unsupported runtime revision; refusing fuzzy patch") + sources = {} + for path, expected in PINS.items(): + raw = (root / path).read_bytes() + actual = hashlib.sha1(b"blob " + str(len(raw)).encode() + b"\0" + raw).hexdigest() + if actual != expected: + raise ValueError(f"source pin mismatch: {path}") + sources[path] = raw.decode() + + def edit(path, old, new): + if sources[path].count(old) != 1: + raise ValueError(f"patch anchor not unique: {path}: {old[:70]}") + sources[path] = sources[path].replace(old, new, 1) + + p = "packages/codemode/src/tool-runtime.ts" + edit(p, 'export type ToolInvocation = { readonly name: string; readonly input: unknown }', + 'export type ToolInvocation = { readonly id: string; readonly name: string; readonly input: unknown }') + edit(p, ' return yield* hooked(\n { name, input },', + ' const call: ToolInvocation = { id: crypto.randomUUID(), name, input }\n return yield* hooked(\n call,') + edit(p, 'Effect.suspend(() => tool.execute(input))', 'Effect.suspend(() => tool.execute(input, call))') + p = "packages/codemode/src/tool.ts" + edit(p, 'import type { Tools } from "./tools.js"', + 'import type { Tools } from "./tools.js"\nimport type { ToolInvocation } from "./tool-runtime.js"') + edit(p, 'readonly execute: (input: unknown) =>', 'readonly execute: (input: unknown, invocation?: ToolInvocation) =>') + edit(p, 'readonly execute: (input: InputType) =>', 'readonly execute: (input: InputType, invocation?: ToolInvocation) =>') + edit(p, 'execute: (input) => options.execute(input as InputType),', + 'execute: (input, invocation) => options.execute(input as InputType, invocation),') + p = "packages/codemode/src/interpreter/errors.ts" + edit(p, ' const type = thrown instanceof Error && isErrorType(thrown.name) ? thrown.name : "Error"\n return createErrorValue(builtins[type], normalizeError(thrown).message)', + ' const error = callerError(thrown)\n return createErrorValue(builtins[error.name], error.message)') + edit(p, '/** Error.prototype.toString:', + 'export const callerError = (thrown: unknown): { name: ErrorType; message: string } => ({\n name: thrown instanceof Error && isErrorType(thrown.name) ? thrown.name : "Error",\n message: normalizeError(thrown).message,\n})\n\n/** Error.prototype.toString:') + p = "packages/codemode/src/codemode.ts" + edit(p, 'import { Effect, Schema } from "effect"', + 'import { Effect, Schema } from "effect"\nexport { callerError } from "./interpreter/errors.js"') + + p = "packages/core/src/tool/runtime.ts" + edit(p, 'export const execute = (tool: Tool.Info, input: unknown, context: Tool.Context) =>', + 'export const execute = (tool: Tool.Info, input: unknown, context: Tool.Context, observedInput?: (input: unknown) => Effect.Effect) =>') + edit(p, ' const decoded = yield* decodeInput(tool, input)', + ' const decoded = yield* decodeInput(tool, input)\n if (observedInput) yield* observedInput(decoded)') + + p = "packages/core/src/tool.ts" + edit(p, 'import { CodeModeTool } from "./codemode/tool.js"', + 'import { CodeModeTool } from "./codemode/tool.js"\nimport { LocalObservation } from "./codemode/local-observation.js"') + edit(p, ' context: Tool.Context,\n ) {', + ' context: Tool.Context,\n observedInput?: (input: unknown) => Effect.Effect,\n codeModeDispatch = false,\n ) {') + edit(p, 'execute(tool, input, context).pipe(', 'execute(tool, input, context, observedInput).pipe(') + edit(p, ' ) {\n const execution = yield* execute(tool, input, context, observedInput).pipe(', + ' ) {\n const nativeObservation =\n !codeModeDispatch && observedInput === undefined && process.env.OPENCODE_EVAL_HOST_OBSERVATIONS === "1"\n ? LocalObservation.makeNative(context, name, (event) =>\n hooks.trigger("tool", "execute.native-observed", event).pipe(Effect.asVoid),\n )\n : undefined\n const execution = yield* execute(tool, input, context, observedInput ?? nativeObservation?.start).pipe(') + edit(p, 'CodeModeTool.create(codeModeInventory, (name, tool, input, context) =>\n beforeExecute(name, input, context).pipe(\n Effect.flatMap((event) => executeTool(tool, name, event.input, context)),\n ),\n )', + 'CodeModeTool.create(codeModeInventory, (name, tool, input, context, observedInput) =>\n beforeExecute(name, input, context).pipe(\n Effect.flatMap((event) => executeTool(tool, name, event.input, context, observedInput, true)),\n ),\n process.env.OPENCODE_EVAL_OBSERVATIONS === "1"\n ? (event) => hooks.trigger("tool", "execute.observed", event).pipe(Effect.asVoid)\n : undefined,\n )') + p = "packages/core/src/session/runner/publish-llm-event.ts" + edit(p, 'import type { Tool } from "../../tool.js"', + 'import type { Tool } from "../../tool.js"\nimport { LocalObservation } from "../../codemode/local-observation.js"') + edit(p, ''' yield* bus.publish(SessionEvent.Tool.Failed, { + sessionID: input.sessionID, + assistantMessageID, + id, + error: + tool.name === "subagent" && error.type === "aborted" && typeof tool.progress?.sessionID === "string" + ? { ...error, message: `${error.message} (sessionID: ${tool.progress.sessionID})` } + : error, + ...failureSnapshot(tool, metadata), + executed: tool.providerExecuted, + }) + return true''', + ''' const terminal = { + sessionID: input.sessionID, + assistantMessageID, + id, + error: + tool.name === "subagent" && error.type === "aborted" && typeof tool.progress?.sessionID === "string" + ? { ...error, message: `${error.message} (sessionID: ${tool.progress.sessionID})` } + : error, + ...failureSnapshot(tool, metadata), + executed: tool.providerExecuted, + } + yield* bus.publish(SessionEvent.Tool.Failed, terminal) + if (process.env.OPENCODE_EVAL_HOST_OBSERVATIONS === "1") + yield* LocalObservation.finishNative({ + sessionID: input.sessionID, + messageID: assistantMessageID, + callID: id, + outcome: "threw", + value: { + error: terminal.error, + ...(terminal.metadata === undefined ? {} : { metadata: terminal.metadata }), + executed: terminal.executed, + }, + errorRepresentation: "session-tool-failed/v1", + }) + return true''') + edit(p, ''' yield* bus.publish(SessionEvent.Tool.Success, { + sessionID: input.sessionID, + assistantMessageID, + id, + content, + ...(result.metadata === undefined ? {} : { metadata: result.metadata }), + executed: tool.providerExecuted, + }) + })''', + ''' const terminal = { + sessionID: input.sessionID, + assistantMessageID, + id, + content, + ...(result.metadata === undefined ? {} : { metadata: result.metadata }), + executed: tool.providerExecuted, + } + yield* bus.publish(SessionEvent.Tool.Success, terminal) + if (process.env.OPENCODE_EVAL_HOST_OBSERVATIONS === "1") + yield* LocalObservation.finishNative({ + sessionID: input.sessionID, + messageID: assistantMessageID, + callID: id, + outcome: "returned", + value: { + content: terminal.content, + ...(terminal.metadata === undefined ? {} : { metadata: terminal.metadata }), + executed: terminal.executed, + }, + }) + })''') + + p = "packages/plugin/src/effect/tool.ts" + edit(p, 'export interface ToolHooks {', + 'export interface ToolHooks {\n /** Downstream eval-only observations; not an authenticated evidence channel. */\n readonly "execute.observed": Readonly>\n /** Native/tool-service observations for normal invoke feasibility; also unauthenticated. */\n readonly "execute.native-observed": Readonly>') + edit(p, 'export interface ToolFailures extends Record {', + 'export interface ToolFailures extends Record {\n readonly "execute.observed": never\n readonly "execute.native-observed": never') + + p = "packages/core/src/codemode/tool.ts" + edit(p, 'import { CodeModeWeb } from "./web.js"', + 'import { CodeModeWeb } from "./web.js"\nimport { LocalObservation } from "./local-observation.js"') + edit(p, ' executeTool: (name: string, tool: Info, input: unknown, context: Context) => Effect.Effect,\n)', + ' executeTool: (name: string, tool: Info, input: unknown, context: Context, observedInput?: (input: unknown) => Effect.Effect) => Effect.Effect,\n observe?: (event: Readonly>) => Effect.Effect,\n)') + edit(p, ' const callIndex = yield* Ref.make(0)', + ' const observation = observe ? LocalObservation.make(context, observe) : undefined\n if (observation) yield* observation.open()\n const callIndex = yield* Ref.make(0)') + edit(p, ' (name, tool, input) =>\n Effect.gen(function* () {', + ' (name, tool, input, invocation) =>\n Effect.gen(function* () {') + edit(p, ' const executed = yield* executeTool(name, tool, input, context)', + ' const executed = yield* executeTool(name, tool, input, context,\n observation && invocation ? (decoded) => observation.dispatch(invocation, name, decoded) : undefined)') + edit(p, ' progressHooks(record),\n ).execute(code)', + ' observation ? observation.hooks(progressHooks(record)) : progressHooks(record),\n ).execute(code).pipe(Effect.onExit(() => observation ? observation.close() : Effect.void))') + edit(p, ' executeTool: (name: string, tool: Info, input: unknown) => Effect.Effect,', + ' executeTool: (name: string, tool: Info, input: unknown, invocation?: CodeMode.ToolInvocation) => Effect.Effect,') + edit(p, ' execute: (input) => executeTool(name, registration, input),', + ' execute: (input, invocation) => executeTool(name, registration, input, invocation),') + + # Complete all checks in memory before touching the checkout. + for path, text in sources.items(): + (root / path).write_text(text) + here = Path(__file__).resolve().parent + shutil.copyfile(here / "local-observation.ts", root / "packages/core/src/codemode/local-observation.ts") + target = root / "packages/core/test/local-observation.test.ts" + target.parent.mkdir(exist_ok=True) + shutil.copyfile(here / "local-observation.test.ts", target) + print(f"Applied local eval observation patch to {REVISION}; no upstream write") + + +if __name__ == "__main__": + apply(Path(sys.argv[1]).resolve()) diff --git a/runtime-patches/local-observation.test.ts b/runtime-patches/local-observation.test.ts new file mode 100644 index 0000000..10c3071 --- /dev/null +++ b/runtime-patches/local-observation.test.ts @@ -0,0 +1,223 @@ +import { test, expect } from "bun:test" +import { Effect, Schema } from "effect" +import { CodeMode, Tool } from "@opencode/codemode" +import { CodeModeTool } from "../src/codemode/tool.js" +import { LocalObservation } from "../src/codemode/local-observation.js" + +const context = { + sessionID: "actual-session", messageID: "actual-message", id: "shared-parent", agent: "actual-agent", + progress: () => Effect.void, +} as any +const registration = (name: string) => ({ + name, description: "local observer fixture", input: Schema.Struct({ tag: Schema.optionalKey(Schema.String) }), + options: { namespace: "fixture", codemode: true }, execute: () => Effect.succeed({ content: "unused" }), +}) + +test("real core and interpreter: identical overlap, final conversion, caught errors, discarded returns", async () => { + const events: any[] = [] + const inventory = { tools: new Map(["echo", "denied", "throws", "mutate"].map((name) => [`fixture_${name}`, registration(name)])) } + let ordinal = 0 + let release!: () => void + const first = new Promise((resolve) => { release = resolve }) + const execute = (name: string, _tool: unknown, input: unknown, actual: unknown, observed?: (value: unknown) => Effect.Effect) => Effect.gen(function* () { + expect(actual).toBe(context) + const dispatched = { ...(input as object), actual: true } + if (observed) yield* observed(dispatched) + if (name === "fixture_echo") { + const n = ++ordinal + if (n === 1) yield* Effect.promise(() => first) + if (n === 2) setTimeout(release, 10) + return { content: `CALL-${n}` } + } + if (name === "fixture_throws") return yield* Effect.die(new Error("THROWN-FINAL")) + if (name === "fixture_denied") return { content: '{"ok":false}' } + return { content: "FINAL-RETURN" } + }) + const tool = CodeModeTool.create(inventory as any, execute as any, (event) => Effect.sync(() => { events.push(event) })) + const result = await Effect.runPromise(tool.execute({ code: ` + const pair = await Promise.all([tools.fixture.echo({tag:"same"}),tools.fixture.echo({tag:"same"})]); + if (pair[0] !== "CALL-1" || pair[1] !== "CALL-2") throw Error("pair"); + const denied = await tools.fixture.denied({}); + if (denied !== '{"ok":false}') throw Error("denial"); + try { await tools.fixture.throws({}); } catch (error) { + if (error.name !== "Error" || error.message !== "THROWN-FINAL") throw Error("error view"); + } + if (await tools.fixture.mutate({}) !== "FINAL-RETURN") throw Error("final"); + return "DISCARDED"; + ` }, context)) + expect(result.output.output).toBe("DISCARDED") + const starts = events.filter((e) => e.kind === "call_start") + const ends = events.filter((e) => e.kind === "call_end") + expect(starts).toHaveLength(5) + expect(ends).toHaveLength(5) + expect(new Set(starts.map((e) => e.invocation_id)).size).toBe(5) + expect(starts[0].input.value).toEqual({ tag: "same", actual: true }) + expect(starts[1].input.value).toEqual(starts[0].input.value) + expect(ends[0].invocation_id).toBe(starts[1].invocation_id) + expect(ends[1].invocation_id).toBe(starts[0].invocation_id) + expect(ends[0].result.value).toBe("CALL-2") + expect(ends[1].result.value).toBe("CALL-1") + expect(starts[1].sequence).toBeLessThan(ends[0].sequence) + expect(ends[2].result.value).toBe('{"ok":false}') + expect(ends[3].outcome).toBe("threw") + expect(ends[3].error.value).toEqual({ name: "Error", message: "THROWN-FINAL" }) + expect(ends[4].result.value).toBe("FINAL-RETURN") + expect(starts[0].actor).toEqual({ agent: "actual-agent", session_id: "actual-session", message_id: "actual-message" }) + expect(starts[0].parent.call_id).toBe("shared-parent") + expect(events.at(-1)).toMatchObject({ kind: "parent_end", missing_terminals: 0, unsupported_dispatches: 0, evidence_eligible: false }) +}) + +test("native outer start binds the actual Code Mode parent; finalization is session-owned", async () => { + const nativeEvents: any[] = [] + const innerEvents: any[] = [] + const native = LocalObservation.makeNative(context, "execute", (event) => Effect.sync(() => nativeEvents.push(event))) + await Effect.runPromise(native.start({ code: "return await tools.fixture.echo({})" })) + + const inventory = { tools: new Map([["fixture_echo", registration("echo")]]) } + const execute = (_name: unknown, _tool: unknown, input: unknown, actual: unknown, observed?: (value: unknown) => Effect.Effect) => + Effect.gen(function* () { + expect(actual).toBe(context) + if (observed) yield* observed(input) + return { content: "INNER-FINAL" } + }) + const tool = CodeModeTool.create(inventory as any, execute as any, (event) => Effect.sync(() => innerEvents.push(event))) + await Effect.runPromise(tool.execute({ code: "return await tools.fixture.echo({});" }, context)) + await Effect.runPromise(LocalObservation.finishNative({ + sessionID: context.sessionID, + messageID: context.messageID, + callID: context.id, + outcome: "returned", + value: { content: [{ type: "text", text: "SESSION-FINAL" }], metadata: { truncated: false }, executed: false }, + })) + + const nativeStart = nativeEvents.find((event) => event.kind === "call_start") + const nativeEnd = nativeEvents.find((event) => event.kind === "call_end") + const innerStart = innerEvents.find((event) => event.kind === "call_start") + expect(nativeStart).toMatchObject({ + schema: "opencode-native-observation/v2", + invocation_id: native.invocationID, + call_id: "shared-parent", + tool: "execute", + mode: "native", + parent: null, + actor: { agent: "actual-agent", session_id: "actual-session", message_id: "actual-message" }, + }) + expect(nativeEnd).toMatchObject({ + invocation_id: native.invocationID, + boundary: "session-tool-terminal", + outcome: "returned", + }) + expect(nativeEnd.result.value.metadata.truncated).toBe(false) + expect(innerStart.parent.invocation_id).toBe(native.invocationID) + expect(innerStart.parent.call_id).toBe("shared-parent") + expect(nativeStart.sequence).toBeLessThan(innerStart.sequence) + expect(innerEvents.at(-1).sequence).toBeLessThan(nativeEnd.sequence) + expect(context).not.toHaveProperty("evaluationInvocationID") +}) + +test("native observation keeps delegated actor identity from the real tool context", async () => { + const child = { + sessionID: "child-session", messageID: "child-message", id: "child-call", agent: "reviewer", + progress: () => Effect.void, + } as any + const events: any[] = [] + const observation = LocalObservation.makeNative(child, "read", (event) => Effect.sync(() => events.push(event))) + await Effect.runPromise(observation.start({ file: "README.md" })) + await Effect.runPromise(LocalObservation.finishNative({ + sessionID: child.sessionID, + messageID: child.messageID, + callID: child.id, + outcome: "threw", + value: { error: { type: "aborted", message: "cancelled" }, executed: false }, + errorRepresentation: "session-tool-failed/v1", + })) + expect(events[0].actor).toEqual({ agent: "reviewer", session_id: "child-session", message_id: "child-message" }) + expect(events[0]).toMatchObject({ call_id: "child-call", mode: "native", parent: null }) + expect(events[1]).toMatchObject({ + invocation_id: events[0].invocation_id, + outcome: "threw", + error_representation: "session-tool-failed/v1", + }) +}) + +test("native decode failures do not fabricate terminal-only observations", async () => { + const events: any[] = [] + LocalObservation.makeNative(context, "read", (event) => Effect.sync(() => events.push(event))) + await Effect.runPromise(LocalObservation.finishNative({ + sessionID: context.sessionID, + messageID: context.messageID, + callID: context.id, + outcome: "threw", + value: { error: { type: "tool.input", message: "decode failed" } }, + })) + expect(events).toEqual([]) +}) + +test("native terminal omission increments unavailable accounting", async () => { + const omittedContext = { + sessionID: "omit-session", messageID: "omit-message", id: "omit-call", agent: "build", + progress: () => Effect.void, + } as any + const events: any[] = [] + const observation = LocalObservation.makeNative(omittedContext, "read", (event) => Effect.sync(() => events.push(event))) + await Effect.runPromise(observation.start({ path: "x" })) + await Effect.runPromise(LocalObservation.finishNative({ + sessionID: omittedContext.sessionID, + messageID: omittedContext.messageID, + callID: omittedContext.id, + outcome: "returned", + value: undefined, + })) + expect(events.at(-1)).toMatchObject({ + kind: "call_end", + unavailable_fields: 1, + result: { state: "omitted", reason: "unsupported_snapshot" }, + }) +}) + +test("runtime-carried invocation IDs reach actual executable and terminal", async () => { + const executing: string[] = [], after: string[] = [] + const tools = { one: Tool.make({ description: "one", input: Schema.Struct({}), output: Schema.Null, + execute: (_input, call) => Effect.sync(() => { executing.push(call!.id); return null }) }) } + await Effect.runPromise(CodeMode.execute({ tools, code: "await tools.one({}); await tools.one({}); return null;", + hooks: { "tool.after": (call) => Effect.sync(() => { after.push(call.id) }) } })) + expect(executing).toEqual(after) + expect(new Set(executing).size).toBe(2) +}) + +test("observer failure does not replace actual return", async () => { + const inventory = { tools: new Map([["fixture_echo", registration("echo")]]) } + const execute = (_name: unknown, _tool: unknown, input: unknown, _context: unknown, observe?: (input: unknown) => Effect.Effect) => + Effect.andThen(observe?.(input) ?? Effect.void, Effect.succeed({ content: "UNCHANGED" })) + for (const observe of [undefined, () => Effect.die("observer I/O failure")]) { + const tool = CodeModeTool.create(inventory as any, execute as any, observe) + const result = await Effect.runPromise(tool.execute({ code: 'return await tools.fixture.echo({});' }, context)) + expect(result.output.output).toBe("UNCHANGED") + } + + const native = LocalObservation.makeNative(context, "read", () => Effect.die("native-observer-failure")) + await Effect.runPromise(native.start({ path: "x" })) + await Effect.runPromise(LocalObservation.finishNative({ + sessionID: context.sessionID, + messageID: context.messageID, + callID: context.id, + outcome: "returned", + value: { content: [{ type: "text", text: "UNCHANGED" }], executed: false }, + })) +}) + +test("snapshots are owned, immutable, and do not call getters/toJSON", () => { + const source = { nested: { value: "original" } } + const copied: any = LocalObservation.snapshot(source) + source.nested.value = "changed" + expect(copied.value.nested.value).toBe("original") + expect(Object.isFrozen(copied.value.nested)).toBe(true) + let called = false + expect(LocalObservation.snapshot({ get token() { called = true; return "secret" } }).state).toBe("omitted") + expect(LocalObservation.snapshot({ toJSON() { called = true; return "secret" } }).state).toBe("omitted") + expect(called).toBe(false) + expect(LocalObservation.snapshot(undefined).state).toBe("omitted") + expect(LocalObservation.snapshot({ value: Infinity }).state).toBe("omitted") + const cycle: any = {}; cycle.self = cycle + expect(LocalObservation.snapshot(cycle).state).toBe("omitted") +}) diff --git a/runtime-patches/local-observation.ts b/runtime-patches/local-observation.ts new file mode 100644 index 0000000..d3e8f06 --- /dev/null +++ b/runtime-patches/local-observation.ts @@ -0,0 +1,258 @@ +export * as LocalObservation from "./local-observation.js" + +import { CodeMode } from "@opencode/codemode" +import type { Context } from "@opencode/schema/tool" +import { Effect } from "effect" + +// Runtime observation seam only. Nothing here authenticates a collector. +// Sequence is process-wide so native and Code Mode records can be merged +// without inferring ordering from timestamps or array position. +let sequence = 0 +const nativeExecuteParents = new WeakMap() + +type Event = Readonly> +type Field = + | { readonly state: "available"; readonly value: unknown } + | { readonly state: "omitted"; readonly reason: string } + +type NativeState = { + readonly invocationID: string + readonly context: Context + readonly tool: string + readonly actor: Readonly<{ agent: string; session_id: string; message_id: string }> + readonly emit: (event: Event) => Effect.Effect + lost: number + unavailable: number +} + +const nativeStates = new Map() + +const nextSequence = () => ++sequence +const nativeKey = (sessionID: string, messageID: string, callID: string) => + [sessionID, messageID, callID].join("\u0000") + +export function snapshot(value: unknown): Field { + const seen = new Set() + let nodes = 0 + const copy = (value: unknown, depth: number): unknown => { + if (++nodes > 10000 || depth > 32) throw new Error("snapshot_limit") + if (value === null || typeof value === "boolean" || typeof value === "string") return value + if (typeof value === "number" && Number.isFinite(value)) return value + if (!value || typeof value !== "object") throw new Error("non_json_value") + if (seen.has(value)) throw new Error("cyclic_value") + if (!Array.isArray(value) && Object.getPrototypeOf(value) !== Object.prototype && Object.getPrototypeOf(value) !== null) + throw new Error("non_plain_value") + seen.add(value) + const descriptors = Object.getOwnPropertyDescriptors(value) + const result: Record | unknown[] = Array.isArray(value) ? [] : Object.create(null) + for (const key of Reflect.ownKeys(descriptors)) { + if (Array.isArray(value) && key === "length") continue + if (typeof key !== "string") throw new Error("symbol_key") + const property = descriptors[key] + if (!property.enumerable || !("value" in property)) throw new Error("non_data_property") + Object.defineProperty(result, key, { enumerable: true, value: copy(property.value, depth + 1) }) + } + if (Array.isArray(value) && Object.keys(result).length !== value.length) throw new Error("sparse_array") + seen.delete(value) + return Object.freeze(result) + } + try { + return Object.freeze({ state: "available", value: copy(value, 0) }) + } catch { + return Object.freeze({ state: "omitted", reason: "unsupported_snapshot" }) + } +} + +const sendNative = (state: NativeState, body: Event) => + Effect.suspend(() => { + const event = Object.freeze({ + schema: "opencode-native-observation/v2", + sequence: nextSequence(), + actor: state.actor, + observer_failures: state.lost, + ...body, + }) + return Effect.suspend(() => state.emit(event)).pipe( + Effect.catchCause(() => Effect.sync(() => { state.lost++ })), + ) + }) + +const nativeField = (state: NativeState, value: unknown) => { + const result = snapshot(value) + if (result.state !== "available") state.unavailable++ + return result +} + +/** + * Begin observation only after Tool runtime input decoding succeeds. + * + * The terminal is deliberately NOT emitted here: the canonical native result + * crosses ToolOutput.truncate and SessionEvent publication later in the Step + * writer. finishNative() is called at that final session-owned boundary. + */ +export function makeNative(context: Context, tool: string, emit: (event: Event) => Effect.Effect) { + const invocationID = crypto.randomUUID() + const key = nativeKey(context.sessionID, context.messageID, context.id) + const state: NativeState = { + invocationID, + context, + tool, + actor: Object.freeze({ agent: context.agent, session_id: context.sessionID, message_id: context.messageID }), + emit, + lost: 0, + unavailable: 0, + } + let started = false + return { + invocationID, + start: (input: unknown) => + Effect.suspend(() => { + // Duplicate runtime call identity is a diagnostic capture defect, not + // permission to perturb the real tool execution. Leave it unsupported. + if (nativeStates.has(key)) return Effect.void + started = true + nativeStates.set(key, state) + // The same Context object reaches the Code Mode outer tool. + if (tool === "execute") nativeExecuteParents.set(context as object, invocationID) + return sendNative(state, { + kind: "call_start", + invocation_id: invocationID, + call_id: context.id, + tool, + mode: "native", + parent: null, + input: nativeField(state, input), + boundary: "executable-input", + }) + }), + started: () => started, + } +} + +/** + * Emit the native terminal after the normal Session writer has published the + * same canonical result/error. Missing starts remain non-events, never invented + * terminals. + */ +export function finishNative(input: { + readonly sessionID: string + readonly messageID: string + readonly callID: string + readonly outcome: "returned" | "threw" + readonly value: unknown + readonly errorRepresentation?: string +}) { + const key = nativeKey(input.sessionID, input.messageID, input.callID) + const state = nativeStates.get(key) + if (!state) return Effect.void + nativeStates.delete(key) + if (state.tool === "execute" && nativeExecuteParents.get(state.context as object) === state.invocationID) + nativeExecuteParents.delete(state.context as object) + const terminalField = nativeField(state, input.value) + const common = { + kind: "call_end", + invocation_id: state.invocationID, + call_id: input.callID, + boundary: "session-tool-terminal", + unavailable_fields: state.unavailable, + } + return input.outcome === "returned" + ? sendNative(state, { ...common, outcome: "returned", result: terminalField }) + : sendNative(state, { + ...common, + outcome: "threw", + error: terminalField, + error_representation: input.errorRepresentation ?? "session-error/v1", + }) +} + +export function make(context: Context, emit: (event: Event) => Effect.Effect) { + const parent = Object.freeze({ + invocation_id: nativeExecuteParents.get(context as object) ?? crypto.randomUUID(), + session_id: context.sessionID, + message_id: context.messageID, + call_id: context.id, + }) + const actor = Object.freeze({ agent: context.agent, session_id: context.sessionID, message_id: context.messageID }) + const admitted = new Set() + const dispatched = new Set() + const ended = new Set() + let lost = 0 + let unavailable = 0 + const send = (body: Event) => + Effect.suspend(() => { + const event = Object.freeze({ + schema: "opencode-local-observation/v1", + sequence: nextSequence(), + parent, + actor, + observer_failures: lost, + ...body, + }) + return Effect.suspend(() => emit(event)).pipe( + Effect.catchCause(() => Effect.sync(() => { lost++ })), + ) + }) + const field = (value: unknown) => { + const result = snapshot(value) + if (result.state !== "available") unavailable++ + return result + } + return { + open: () => send({ kind: "parent_start", boundary: "codemode-engine", mode: "code_mode" }), + dispatch: (call: CodeMode.ToolInvocation, registration: string, input: unknown) => + Effect.suspend(() => { + dispatched.add(call.id) + return send({ + kind: "call_start", + invocation_id: call.id, + tool: registration, + catalog_path: call.name, + input: field(input), + boundary: "executable-input", + }) + }), + hooks: (original: CodeMode.Hooks): CodeMode.Hooks => ({ + ...original, + "tool.before": (call) => + Effect.suspend(() => { + admitted.add(call.id) + return original["tool.before"]?.(call) ?? Effect.void + }), + "tool.after": (call, result) => + Effect.andThen( + original["tool.after"]?.(call, result) ?? Effect.void, + Effect.suspend(() => { + ended.add(call.id) + const base = { + kind: "call_end", + invocation_id: call.id, + dispatched: dispatched.has(call.id), + boundary: "codemode-json-return", + } + if (result.status === "success") + return send({ ...base, outcome: "returned", result: field(result.value) }) + if (result.status === "interrupted") return send({ ...base, outcome: "interrupted" }) + return send({ + ...base, + outcome: "threw", + error: field(CodeMode.callerError(result.error)), + error_representation: "codemode-catch-name-message/v1", + }) + }), + ), + }), + close: () => + send({ + kind: "parent_end", + admitted: admitted.size, + dispatched: dispatched.size, + terminals: ended.size, + missing_terminals: [...admitted].filter((id) => !ended.has(id)).length, + unsupported_dispatches: [...admitted].filter((id) => !dispatched.has(id)).length, + unavailable_fields: unavailable, + scope: "one-codemode-engine-invocation", + evidence_eligible: false, + }), + } +} diff --git a/runtime-patches/protected-profile.test.ts b/runtime-patches/protected-profile.test.ts new file mode 100644 index 0000000..745d04d --- /dev/null +++ b/runtime-patches/protected-profile.test.ts @@ -0,0 +1,49 @@ +import { test, expect } from "bun:test" +import { Effect, Schema } from "effect" +import { CodeModeTool } from "../src/codemode/tool.js" + +const context = { + sessionID: "actual-session", messageID: "actual-message", id: "parent", agent: "actual-agent", + progress: () => Effect.void, +} as any + +test("protected profile has no script fetch and preserves dispatched invocation IDs with capture on/off", async () => { + const saved = process.env.OPENCODE_EVAL_PROTECTED_CHANNEL + process.env.OPENCODE_EVAL_PROTECTED_CHANNEL = "1" + try { + for (const observing of [false, true]) { + const executions: any[] = [], events: any[] = [] + const tool = { name: "one", description: "fixture", input: Schema.Struct({}), + options: { namespace: "isolated", codemode: true }, execute: () => Effect.succeed({ content: "unused" }) } + const execute = (_name: string, _tool: unknown, input: unknown, actual: any, record?: (input: unknown) => Effect.Effect) => + Effect.gen(function* () { + if (record) yield* record(input) + executions.push(actual) + return { content: "UNCHANGED" } + }) + const runtime = CodeModeTool.create({ tools: new Map([["isolated_one", tool]]) } as any, execute as any, + observing ? event => Effect.sync(() => { events.push(event) }) : undefined) + const result = await Effect.runPromise(runtime.execute({ code: ` + if (typeof fetch !== "undefined") throw Error("script network available"); + const values = await Promise.all([tools.isolated.one({}), tools.isolated.one({})]); + return values; + ` }, context)) + expect(result.output.error).toBeUndefined() + expect(JSON.parse(result.output.output)).toEqual(["UNCHANGED", "UNCHANGED"]) + expect(executions).toHaveLength(2) + expect(new Set(executions.map(c => c.evaluationInvocationID)).size).toBe(2) + expect(executions.map(c => c.evaluationDispatchOrdinal)).toEqual([0, 1]) + for (const actual of executions) expect(actual).toMatchObject(context) + expect(context).not.toHaveProperty("evaluationInvocationID") + if (observing) { + const starts = events.filter(e => e.kind === "call_start") + expect(starts.map(e => e.invocation_id)).toEqual(executions.map(c => c.evaluationInvocationID)) + expect(events.filter(e => e.kind === "call_end").map(e => e.invocation_id).sort()) + .toEqual(executions.map(c => c.evaluationInvocationID).sort()) + } + } + } finally { + if (saved === undefined) delete process.env.OPENCODE_EVAL_PROTECTED_CHANNEL + else process.env.OPENCODE_EVAL_PROTECTED_CHANNEL = saved + } +}) diff --git a/runtime-patches/run_delegated_session_probe.py b/runtime-patches/run_delegated_session_probe.py new file mode 100755 index 0000000..25d8b7c --- /dev/null +++ b/runtime-patches/run_delegated_session_probe.py @@ -0,0 +1,309 @@ +#!/usr/bin/env python3 +"""Provider-free proof of real delegated Session identity/lifecycle under runner invoke.""" +from __future__ import annotations + +import argparse +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +ROOT = Path(__file__).resolve().parents[1] + +PLUGIN = r''' +import { appendFileSync } from "node:fs" + +const path = "/workspace/delegated-observations.jsonl" +function log(event: any) { appendFileSync(path, JSON.stringify(event) + "\n") } +const calls = new Map() + +export default { + id: "delegateprobe", + async setup(ctx: any) { + await ctx.tool.transform((editor: any) => { + editor.namespace({ name: "delegateprobe", description: "Delegated session fixture" }) + editor.add({ + name: "marker", + description: "Return a child-session marker", + input: { type: "object", properties: { marker: { type: "string" } }, required: ["marker"], additionalProperties: false }, + options: { namespace: "delegateprobe", codemode: false }, + execute: async (input: any) => ({ content: "MARKER-" + input.marker }), + }) + }) + await ctx.tool.hook("execute.native-observed", async (event: any) => { + log(event) + if (event.kind === "call_start") { + calls.set(String(event.invocation_id), { + tool: String(event.tool), + parentSession: String(event.actor?.session_id ?? ""), + }) + return + } + if (event.kind !== "call_end") return + const call = calls.get(String(event.invocation_id)) + if (!call) return + calls.delete(String(event.invocation_id)) + const childID = event.outcome === "returned" + ? event.result?.value?.metadata?.sessionID + : undefined + if (call.tool === "subagent" && typeof childID === "string" && childID) { + const child = await ctx.session.get({ sessionID: childID }) + log({ + kind: "session_ancestry", + child_session_id: child.id, + child_parent_id: child.parentID ?? null, + parent_actor_session_id: call.parentSession, + }) + } + }) + await ctx.session.hook("http.request", (event: any) => { + const headers = new Headers(event.request.headers) + headers.set("x-probe-session", String(event.sessionID)) + headers.set("x-probe-agent", String(event.agent)) + event.request = new Request(event.request, { headers }) + }) + }, +} +''' + +GENERAL_ALLOW = """--- +description: Parent fixture +mode: primary +model: fixture/mock +permissions: + - action: "*" + resource: "*" + effect: allow +--- + +Use the requested tools. +""" + +GENERAL_DENY = """--- +description: Parent fixture with reviewer denial +mode: primary +model: fixture/mock +permissions: + - action: "*" + resource: "*" + effect: allow + - action: subagent + resource: reviewer + effect: deny +--- + +Use the requested tools. +""" + +REVIEWER = """--- +description: Child reviewer fixture +mode: subagent +model: fixture/mock +permissions: + - action: "*" + resource: "*" + effect: allow +--- + +Complete the delegated task. +""" + + +def read_jsonl(path: Path): + if not path.is_file(): + return [] + return [json.loads(line) for line in path.read_text().splitlines() if line] + + +def run_once(image: str, output: Path, *, denied: bool): + request_log = [] + child_calls = 0 + parent_calls = 0 + + class Provider(BaseHTTPRequestHandler): + def log_message(self, *_): + pass + + def do_POST(self): + nonlocal child_calls, parent_calls + size = int(self.headers.get("Content-Length", "0")) + body = json.loads(self.rfile.read(size)) + agent = self.headers.get("x-probe-agent", "") + session = self.headers.get("x-probe-session", "") + names = [item["function"]["name"] for item in body.get("tools", [])] + request_log.append({"agent": agent, "session": session, "tools": names}) + tool = None + if agent == "reviewer": + child_calls += 1 + if child_calls == 1: + marker = next((name for name in names if name.endswith("delegateprobe_marker")), None) + if marker is None: + self.send_error(400, "child marker tool missing") + return + tool = (marker, {"marker": "child"}) + else: + delta, finish = {"role": "assistant", "content": "CHILD-DONE"}, "stop" + else: + parent_calls += 1 + if parent_calls == 1: + if "subagent" not in names: + self.send_error(400, "subagent tool missing") + return + tool = ("subagent", { + "agent": "reviewer", + "description": "delegation fixture", + "prompt": "Call delegateprobe.marker with marker child, then answer CHILD-DONE.", + "background": False, + }) + else: + delta, finish = {"role": "assistant", "content": "PARENT-DONE"}, "stop" + if tool is not None: + delta = {"role": "assistant", "tool_calls": [{ + "index": 0, "id": ("child-marker-call" if agent == "reviewer" else "parent-subagent-call"), + "type": "function", + "function": {"name": tool[0], "arguments": json.dumps(tool[1])}, + }]} + finish = "tool_calls" + common = {"id": f"chatcmpl-{len(request_log)}", "created": 1, "model": body["model"]} + chunks = [ + {**common, "object": "chat.completion.chunk", "choices": [{"index": 0, "delta": delta, "finish_reason": None}]}, + {**common, "object": "chat.completion.chunk", "choices": [{"index": 0, "delta": {}, "finish_reason": finish}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}}, + ] + raw = ("".join("data: " + json.dumps(c) + "\n\n" for c in chunks) + "data: [DONE]\n\n").encode() + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.send_header("Content-Length", str(len(raw))) + self.end_headers() + self.wfile.write(raw) + + server = ThreadingHTTPServer(("127.0.0.1", 0), Provider) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + try: + with tempfile.TemporaryDirectory(prefix="delegated-session-probe-") as tmp: + root = Path(tmp) + workspace = root / "workspace" + plugin_dir = workspace / ".opencode/plugins" + agent_dir = workspace / ".opencode/agents" + plugin_dir.mkdir(parents=True) + agent_dir.mkdir(parents=True) + (plugin_dir / "delegateprobe.ts").write_text(PLUGIN) + (agent_dir / "general.md").write_text(GENERAL_DENY if denied else GENERAL_ALLOW) + (agent_dir / "reviewer.md").write_text(REVIEWER) + (workspace / "opencode.json").write_text(json.dumps({ + "$schema": "https://opencode.ai/config.json", + "default_agent": "general", + "model": "fixture/mock", + "enabled_providers": ["fixture"], + "provider": {"fixture": { + "npm": "@ai-sdk/openai-compatible", + "name": "Delegated fixture", + "options": {"baseURL": f"http://127.0.0.1:{server.server_port}/v1", "apiKey": "fixture"}, + "models": {"mock": {"name": "Mock", "limit": {"context": 1000000, "output": 32768}}}, + }}, + })) + prompt = root / "prompt.txt"; prompt.write_text("Delegate the fixture task to reviewer.") + system = root / "system.txt"; system.write_text("") + result_path = root / "result.json" + xdg = root / "xdg" + for name in ("home", "config", "data", "state", "cache"): + (xdg / name).mkdir(parents=True, exist_ok=True) + env = dict(os.environ) + direct_auth = { + "OPENAI_API_KEY", "ANTHROPIC_API_KEY", "OPENROUTER_API_KEY", + "OPENCODE_API_KEY", "COPILOT_GITHUB_TOKEN", "GH_TOKEN", "GITHUB_TOKEN", + } + for name in list(env): + if name in direct_auth or name.startswith("OPENCODE_EVAL_RUNNER_"): + env.pop(name, None) + env.update({ + "HOME": str(xdg/"home"), "XDG_CONFIG_HOME": str(xdg/"config"), + "XDG_DATA_HOME": str(xdg/"data"), "XDG_STATE_HOME": str(xdg/"state"), + "XDG_CACHE_HOME": str(xdg/"cache"), + }) + command = [ + sys.executable, str(ROOT/"bin/opencode-eval-runner"), "invoke", + "--engine","docker","--network","host","--image",image, + "--transport","opencode","--workspace",str(workspace),"--workspace-mode","rw", + "--model","fixture/mock","--agent","general","--prompt-file",str(prompt), + "--system-file",str(system),"--output",str(result_path), + "--timeout-seconds","60","--container-timeout","90", + ] + proc = subprocess.run(command,cwd=ROOT,env=env,capture_output=True,text=True,timeout=110,check=False) + result = json.loads(result_path.read_text()) if result_path.is_file() else {} + events = read_jsonl(workspace/"delegated-observations.jsonl") + starts = [e for e in events if e.get("kind")=="call_start"] + ends = {e.get("invocation_id"):e for e in events if e.get("kind")=="call_end"} + ancestry = [e for e in events if e.get("kind")=="session_ancestry"] + sub_start = next((e for e in starts if e.get("tool")=="subagent"), None) + sub_end = ends.get(sub_start.get("invocation_id")) if sub_start else None + child_start = next((e for e in starts if str(e.get("tool","")).endswith("delegateprobe_marker")), None) + child_end = ends.get(child_start.get("invocation_id")) if child_start else None + parent_session = sub_start.get("actor",{}).get("session_id") if sub_start else None + child_session = None + if sub_end and sub_end.get("outcome")=="returned": + child_session = sub_end.get("result",{}).get("value",{}).get("metadata",{}).get("sessionID") + environment_clean = ( + not any(env.get(key) for key in direct_auth) + and not any(key.startswith("OPENCODE_EVAL_RUNNER_") for key in env) + ) + if denied: + checks = { + "no_ambient_credentials_or_runner_seeds": environment_clean, + "invoke_completed": proc.returncode == 0 and result.get("exit_code") == 0, + "subagent_attempt_observed": sub_start is not None, + "canonical_denial_terminal": sub_end is not None and sub_end.get("outcome")=="threw" + and "denied" in json.dumps(sub_end).lower(), + "no_child_session_request": all(r["agent"] != "reviewer" for r in request_log), + "no_child_tool_observation": child_start is None, + } + else: + checks = { + "no_ambient_credentials_or_runner_seeds": environment_clean, + "invoke_completed": proc.returncode == 0 and result.get("exit_code") == 0 and result.get("text")=="PARENT-DONE", + "parent_subagent_observed": sub_start is not None and sub_end is not None, + "child_session_returned_by_real_subagent": isinstance(child_session,str) and bool(child_session), + "child_provider_used_reviewer_identity": any(r["agent"]=="reviewer" and r["session"]==child_session for r in request_log), + "child_tool_actor_matches_child": child_start is not None + and child_start.get("actor",{}).get("agent")=="reviewer" + and child_start.get("actor",{}).get("session_id")==child_session, + "parent_and_child_sessions_distinct": bool(parent_session) and bool(child_session) and parent_session != child_session, + "actual_session_parent_id_matches_parent": len(ancestry) == 1 + and ancestry[0].get("child_session_id") == child_session + and ancestry[0].get("child_parent_id") == parent_session + and ancestry[0].get("parent_actor_session_id") == parent_session, + "child_completed_before_parent_subagent_terminal": child_end is not None and sub_end is not None + and child_end.get("sequence",0) < sub_end.get("sequence",0), + "foreground_completion_delivered": sub_end is not None and "CHILD-DONE" in json.dumps(sub_end), + } + report = { + "denied": denied, "image": image, "checks": checks, "passed": all(checks.values()), + "requests": request_log, "events": events, + "result": {"exit_code": result.get("exit_code"), "text": result.get("text")}, + } + (output/("denied.json" if denied else "success.json")).write_text(json.dumps(report,indent=2)+"\n") + return report + finally: + server.shutdown(); server.server_close(); thread.join(timeout=2) + + +def main(): + p=argparse.ArgumentParser(); p.add_argument("--image",required=True); p.add_argument("--output",type=Path,required=True) + args=p.parse_args(); args.output.mkdir(parents=True,exist_ok=True) + success=run_once(args.image,args.output,denied=False) + denied=run_once(args.image,args.output,denied=True) + summary={"kind":"delegated-session-normal-invoke","version":1,"image":args.image, + "success":success["checks"],"denied":denied["checks"],"passed":success["passed"] and denied["passed"], + "scope":"real built-in foreground subagent + actual Session parentID + permission enforcement; provider-free", + "remaining":["background delegation and cancellation/OQ lifecycle are not exercised by this focused probe"]} + (args.output/"summary.json").write_text(json.dumps(summary,indent=2)+"\n") + print(json.dumps(summary,indent=2)) + return 0 if summary["passed"] else 1 + +if __name__=="__main__": + raise SystemExit(main()) diff --git a/runtime-patches/run_eval_live_compat_probe.py b/runtime-patches/run_eval_live_compat_probe.py new file mode 100755 index 0000000..f28f82a --- /dev/null +++ b/runtime-patches/run_eval_live_compat_probe.py @@ -0,0 +1,281 @@ +#!/usr/bin/env python3 +"""Provider-free compatibility proof for Loom's existing eval:live -> runner invoke path.""" +from __future__ import annotations + +import argparse +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +ROOT = Path(__file__).resolve().parents[1] +PROGRAM = 'return await tools.captureprobe.echo({tag:"same"});' + +PLUGIN = r''' +import { appendFileSync } from "node:fs" +const path = "/workspace/runtime-observations.jsonl" +function log(channel: string, event: any) { + appendFileSync(path, JSON.stringify({ channel, event }) + "\n") +} +export default { + id: "eval-live-compat", + async setup(ctx: any) { + await ctx.tool.transform((editor: any) => { + editor.namespace({ name: "captureprobe", description: "eval:live compatibility tools" }) + editor.add({ + name: "native", description: "native sentinel", input: { type: "object", properties: {}, additionalProperties: false }, + options: { namespace: "captureprobe", codemode: false }, + execute: async () => ({ content: "NATIVE-RAW\n" + "x".repeat(60000) }), + }) + editor.add({ + name: "nativefail", description: "native failure sentinel", input: { type: "object", properties: {}, additionalProperties: false }, + options: { namespace: "captureprobe", codemode: false }, + execute: async () => { throw new Error("NATIVE-FAIL-RAW") }, + }) + editor.add({ + name: "echo", description: "Code Mode sentinel", + input: { type: "object", properties: { tag: { type: "string" } }, additionalProperties: false }, + options: { namespace: "captureprobe", codemode: true }, + execute: async (input: any) => ({ content: input.tag === "same" ? "INNER-RAW" : "WRONG" }), + }) + }) + await ctx.tool.hook("execute.native-observed", (event: any) => { log("native", event) }) + await ctx.tool.hook("execute.observed", (event: any) => { log("codemode", event) }) + }, +} +''' + + +def read_jsonl(path: Path): + if not path.is_file(): + return [] + return [json.loads(line) for line in path.read_text().splitlines() if line] + + +def run(image: str, output: Path) -> int: + if "@sha256:" not in image: + raise ValueError("compatibility probe requires immutable image digest") + output.mkdir(parents=True, exist_ok=True) + requests = [] + stage = 0 + + class Provider(BaseHTTPRequestHandler): + def log_message(self, *_): + pass + + def do_POST(self): + nonlocal stage + size = int(self.headers.get("Content-Length", "0")) + body = json.loads(self.rfile.read(size)) + requests.append(body) + names = [item["function"]["name"] for item in body.get("tools", [])] + if stage == 0: + native = next((name for name in names if "captureprobe" in name and name.endswith("native")), None) + if native is None: + self.send_error(400, "native fixture tool missing") + return + delta = {"role": "assistant", "tool_calls": [{ + "index": 0, "id": "native-call", "type": "function", + "function": {"name": native, "arguments": "{}"}, + }]} + finish = "tool_calls" + elif stage == 1: + nativefail = next((name for name in names if "captureprobe" in name and name.endswith("nativefail")), None) + if nativefail is None: + self.send_error(400, "native failure fixture tool missing") + return + delta = {"role": "assistant", "tool_calls": [{ + "index": 0, "id": "native-fail-call", "type": "function", + "function": {"name": nativefail, "arguments": "{}"}, + }]} + finish = "tool_calls" + elif stage == 2: + if "execute" not in names: + self.send_error(400, "execute tool missing") + return + delta = {"role": "assistant", "tool_calls": [{ + "index": 0, "id": "execute-call", "type": "function", + "function": {"name": "execute", "arguments": json.dumps({"code": PROGRAM})}, + }]} + finish = "tool_calls" + else: + delta = {"role": "assistant", "content": "EVAL-LIVE-FINAL"} + finish = "stop" + stage += 1 + common = {"id": f"chatcmpl-{stage}", "created": 1, "model": body["model"]} + chunks = [ + {**common, "object": "chat.completion.chunk", "choices": [{"index": 0, "delta": delta, "finish_reason": None}]}, + {**common, "object": "chat.completion.chunk", "choices": [{"index": 0, "delta": {}, "finish_reason": finish}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}}, + ] + raw = ("".join("data: " + json.dumps(chunk) + "\n\n" for chunk in chunks) + "data: [DONE]\n\n").encode() + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.send_header("Content-Length", str(len(raw))) + self.end_headers() + self.wfile.write(raw) + + server = ThreadingHTTPServer(("127.0.0.1", 0), Provider) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + try: + with tempfile.TemporaryDirectory(prefix="eval-live-compat-") as tmp: + root = Path(tmp) + workspace = root / "workspace" + plugin_dir = workspace / ".opencode/plugins" + plugin_dir.mkdir(parents=True) + (plugin_dir / "compat.ts").write_text(PLUGIN) + (workspace / "opencode.json").write_text(json.dumps({ + "$schema": "https://opencode.ai/config.json", + "model": "fixture/mock", + "enabled_providers": ["fixture"], + "provider": {"fixture": { + "npm": "@ai-sdk/openai-compatible", + "name": "Local eval:live fixture", + "options": {"baseURL": f"http://127.0.0.1:{server.server_port}/v1", "apiKey": "fixture"}, + "models": {"mock": {"name": "Mock", "limit": {"context": 1000000, "output": 32768}}}, + }}, + })) + prompt = root / "prompt.txt" + system = root / "system.txt" + result_path = root / "result.json" + prompt.write_text("Run the fixture actions.") + system.write_text("") + empty = root / "empty" + for name in ("home", "config", "data", "state", "cache"): + (empty / name).mkdir(parents=True, exist_ok=True) + env = dict(os.environ) + direct_auth = { + "OPENAI_API_KEY", "ANTHROPIC_API_KEY", "OPENROUTER_API_KEY", + "OPENCODE_API_KEY", "COPILOT_GITHUB_TOKEN", "GH_TOKEN", "GITHUB_TOKEN", + } + for name in list(env): + if name in direct_auth or name.startswith("OPENCODE_EVAL_RUNNER_"): + env.pop(name, None) + env.update({ + "HOME": str(empty / "home"), + "XDG_CONFIG_HOME": str(empty / "config"), + "XDG_DATA_HOME": str(empty / "data"), + "XDG_STATE_HOME": str(empty / "state"), + "XDG_CACHE_HOME": str(empty / "cache"), + }) + command = [ + sys.executable, str(ROOT / "bin/opencode-eval-runner"), "invoke", + "--engine", "docker", "--network", "host", "--image", image, + "--transport", "opencode", "--workspace", str(workspace), "--workspace-mode", "rw", + "--model", "fixture/mock", "--prompt-file", str(prompt), "--system-file", str(system), + "--output", str(result_path), "--timeout-seconds", "45", "--container-timeout", "75", + ] + proc = subprocess.run(command, cwd=ROOT, env=env, capture_output=True, text=True, timeout=100, check=False) + result = json.loads(result_path.read_text()) if result_path.is_file() else {} + observations = read_jsonl(workspace / "runtime-observations.jsonl") + (output / "result.json").write_text(json.dumps(result, indent=2) + "\n") + (output / "observations.json").write_text(json.dumps(observations, indent=2) + "\n") + native = [row["event"] for row in observations if row.get("channel") == "native"] + inner = [row["event"] for row in observations if row.get("channel") == "codemode"] + native_starts = [event for event in native if event.get("kind") == "call_start"] + native_ends = {event.get("invocation_id"): event for event in native if event.get("kind") == "call_end"} + outer = [event for event in native_starts if event.get("tool") == "execute"] + inner_starts = [event for event in inner if event.get("kind") == "call_start"] + inner_ends = [event for event in inner if event.get("kind") == "call_end"] + evidence = result.get("tool_result_evidence", {}).get("events", []) + native_success_starts = [event for event in native_starts if event.get("tool", "").endswith("native")] + native_failure_starts = [event for event in native_starts if event.get("tool", "").endswith("nativefail")] + native_success_end = ( + native_ends.get(native_success_starts[0].get("invocation_id")) + if len(native_success_starts) == 1 else None + ) + native_failure_end = ( + native_ends.get(native_failure_starts[0].get("invocation_id")) + if len(native_failure_starts) == 1 else None + ) + native_success_value = (native_success_end or {}).get("result", {}).get("value", {}) + native_failure_value = (native_failure_end or {}).get("error", {}).get("value", {}) + native_success_content = native_success_value.get("content", []) if isinstance(native_success_value, dict) else [] + checks = { + "existing_invoke_entrypoint": command[1:3] == [str(ROOT / "bin/opencode-eval-runner"), "invoke"], + "transport_success": proc.returncode == 0 and result.get("exit_code") == 0, + "existing_result_text": result.get("text") == "EVAL-LIVE-FINAL", + "native_product_result_preserved": any( + event.get("tool", "").endswith("native") and "NATIVE-RAW" in str(event.get("output", "")) + for event in evidence + ), + "native_failure_product_result_preserved": any( + event.get("tool", "").endswith("nativefail") + and event.get("status") == "error" + and "NATIVE-FAIL-RAW" in str(event.get("error", "")) + for event in evidence + ), + "execute_product_result_preserved": any( + event.get("tool") == "execute" and "INNER-RAW" in str(event.get("output", "")) + for event in evidence + ), + "native_observation_schema_v2": bool(native) and all( + event.get("schema") == "opencode-native-observation/v2" for event in native + ), + "native_observation_present": len(native_success_starts) == 1, + "native_final_is_post_truncation_session_terminal": ( + isinstance(native_success_value, dict) + and native_success_value.get("metadata", {}).get("truncated") is True + and any("[showing " in str(part.get("text", "")) for part in native_success_content if isinstance(part, dict)) + and (native_success_end or {}).get("boundary") == "session-tool-terminal" + ), + "native_final_error_is_session_terminal": ( + (native_failure_end or {}).get("outcome") == "threw" + and (native_failure_end or {}).get("boundary") == "session-tool-terminal" + and (native_failure_end or {}).get("error_representation") == "session-tool-failed/v1" + and isinstance(native_failure_value, dict) + and native_failure_value.get("error", {}).get("type") == "unknown" + and "NATIVE-FAIL-RAW" in str(native_failure_value.get("error", {}).get("message", "")) + ), + "outer_execute_observation_present": len(outer) == 1 and outer[0].get("invocation_id") in native_ends, + "inner_final_observation_present": len(inner_starts) == len(inner_ends) == 1 + and inner_ends[0].get("result", {}).get("value") == "INNER-RAW", + "inner_parent_is_actual_outer_invocation": len(outer) == 1 and len(inner_starts) == 1 + and inner_starts[0].get("parent", {}).get("invocation_id") == outer[0].get("invocation_id") + and inner_starts[0].get("parent", {}).get("call_id") == "execute-call", + "shared_sequence_orders_boundaries": len(outer) == 1 and len(inner_starts) == len(inner_ends) == 1 + and outer[0].get("sequence", 0) < inner_starts[0].get("sequence", 0) + < inner_ends[0].get("sequence", 0) < native_ends[outer[0].get("invocation_id")].get("sequence", 0), + "no_real_provider_credentials": not any(env.get(key) for key in direct_auth), + "no_runner_seed_overrides": not any( + key.startswith("OPENCODE_EVAL_RUNNER_") for key in env + ), + } + summary = { + "kind": "eval-live-invoke-compatibility", + "version": 2, + "image": image, + "runner_entrypoint": "opencode-eval-runner invoke", + "loom_entrypoint_contract": "bun run eval:live -> python3 scripts/run-evals.py -> opencode-eval-runner invoke", + "checks": checks, + "passed": all(checks.values()), + "semantic_scope": { + "normal_invoke": "demonstrated", + "native_tool_final_result_and_error": "demonstrated_after_session_truncation/publication", + "codemode_inner_final_result": "demonstrated", + "delegated_session_identity": "not_exercised", + "in_process_plugin_protection": "unsupported", + }, + "evidence_status": "diagnostic_non_evidence", + "reason": "The evaluated plugin shares the OpenCode process and can therefore share collector authority.", + } + (output / "summary.json").write_text(json.dumps(summary, indent=2) + "\n") + print(json.dumps(summary, indent=2)) + return 0 if summary["passed"] else 1 + finally: + server.shutdown() + server.server_close() + thread.join(timeout=2) + + +if __name__ == "__main__": + parser = argparse.ArgumentParser() + parser.add_argument("--image", required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + raise SystemExit(run(args.image, args.output)) diff --git a/runtime-patches/run_image_probe.py b/runtime-patches/run_image_probe.py new file mode 100755 index 0000000..bc11b6f --- /dev/null +++ b/runtime-patches/run_image_probe.py @@ -0,0 +1,179 @@ +#!/usr/bin/env python3 +"""Test the local runtime seam with the existing deterministic runner fixture. + +Not Loom's unpublished smoke or a protected-producer acceptance test. Every raw +value is fixed fixture data. No provider credentials or host databases are used. +""" +import argparse +import hashlib +import importlib.util +import json +from pathlib import Path +import secrets +import shutil +import subprocess +import sys +import tempfile + +REPO = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(REPO)) +from runner.observer import ObserverCapture + +spec = importlib.util.spec_from_file_location("base_probe", REPO / "tests/integration/run_capture_probe.py") +probe = importlib.util.module_from_spec(spec) +spec.loader.exec_module(probe) +EXTRA_HOOKS = ''' + await ctx.tool.hook("execute.before", (event: any) => { + if (String(event.tool) === "captureprobe_echo") event.input = { tag: "dispatched" } + }) + await ctx.tool.hook("execute.observed", (event: any) => { + if (process.env.PROBE_FAIL_OBSERVER === "1") throw new Error("fixture-observer-io-failure") + log(hookPath, { phase: "runtime-observed", event }) + }) + await ctx.tool.hook("execute.native-observed", (event: any) => { + if (process.env.PROBE_FAIL_OBSERVER === "1") throw new Error("fixture-native-observer-io-failure") + log(hookPath, { phase: "runtime-native-observed", event }) + }) +''' + + +def run(image: str, output: Path): + if "@sha256:" not in image: + raise ValueError("the runtime probe requires an immutable registry digest") + output.mkdir(parents=True, exist_ok=True) + reports = {} + fixture_hash = None + for name, observed, forge, failure in ( + ("off", False, False, False), ("on", True, False, False), + ("forged", True, True, False), ("observer-fails", True, False, True), + ): + with tempfile.TemporaryDirectory(prefix="local-runtime-probe-") as tmp: + root = Path(tmp) + fixture = root / "fixture" + (fixture / "tests/integration").mkdir(parents=True) + shutil.copyfile(REPO / "tests/integration/run_capture_probe.py", fixture / "tests/integration/run_capture_probe.py") + source = (REPO / "tests/integration/capture_probe.ts").read_text() + anchor = ' writeFileSync("/tmp/capture-probe-loaded", "loaded")' + if source.count(anchor) != 1: + raise ValueError("unrecognized preserved runner fixture") + source = source.replace(anchor, EXTRA_HOOKS + anchor) + fixture_hash = hashlib.sha256(source.encode()).hexdigest() + (fixture / "tests/integration/capture_probe.ts").write_text(source) + key = root / "host-only-key" + key.write_bytes(secrets.token_bytes(32)) + key.chmod(0o600) + command = [ + "docker", "run", "--rm", "--read-only", "--network", "none", "--cap-drop", "ALL", + "--security-opt", "no-new-privileges", "--tmpfs", "/tmp:rw,exec,nosuid,nodev,size=1g", + "--tmpfs", "/workspace:rw,nosuid,nodev,size=32m,mode=1777", "--workdir", "/workspace", + "--volume", f"{fixture}:/probe-repo:ro", "--env", "PROBE_HOOKS=1", + "--env", f"PROBE_FORGE={int(forge)}", "--env", f"PROBE_FAIL_OBSERVER={int(failure)}", + "--env", f"OPENCODE_EVAL_OBSERVATIONS={int(observed)}", "--entrypoint", "python3", image, + ] + capture = ObserverCapture(key, root, command, {}) + command.extend(["/probe-repo/tests/integration/run_capture_probe.py", "--inside"]) + try: + proc = subprocess.run(command, text=True, capture_output=True, timeout=90, check=False) + report = json.loads(proc.stdout) + report["docker_exit_code"] = proc.returncode + report["observed_execution"] = capture.finish( + transport_ok=proc.returncode == 0 and report.get("transport", {}).get("exit_code") == 0) + except (subprocess.TimeoutExpired, json.JSONDecodeError): + report = {"probe_error": "container_timeout_or_invalid_json", "observed_execution": capture.finish(transport_ok=False)} + reports[name] = report + (output / f"{name}.json").write_text(json.dumps(report, indent=2) + "\n") + print(name, "scenario_completed=", probe.scenario_completed(report), flush=True) + + def observations(name): + return [item["record"]["event"] for item in reports[name].get("hooks", []) + if item.get("record", {}).get("phase") == "runtime-observed"] + + def native_observations(name): + return [item["record"]["event"] for item in reports[name].get("hooks", []) + if item.get("record", {}).get("phase") == "runtime-native-observed"] + + off_native_starts = [ + event for event in native_observations("off") + if event.get("kind") == "call_start" + ] + checks = { + "all_scripts_completed": all(probe.scenario_completed(r) for r in reports.values()), + "product_outcomes_unchanged": bool(reports["off"].get("oracle")) and all( + probe.oracle_signature(r) == probe.oracle_signature(reports["off"]) for r in reports.values()), + "native_observation_image_owned": bool(native_observations("off")) and bool(native_observations("on")), + "inner_observation_test_toggle": not observations("off") and bool(observations("on")), + "disabled_inner_observer_does_not_create_native_inner_calls": ( + len(off_native_starts) == 2 + and {event.get("tool") for event in off_native_starts} == {"captureprobe_native", "execute"} + and all(event.get("mode") == "native" for event in off_native_starts) + ), + "observer_failure_does_not_change_results": probe.scenario_completed(reports["observer-fails"]), + "normal_invoke_runtime_version": all( + "2.0.18-eval.5" in str(r.get("opencode_version") or "") for r in reports.values() + ), + "forged_sidecar_rejected": reports["forged"]["observed_execution"]["issues"] == ["authentication_failed"], + "no_false_capture_acceptance": all(r["observed_execution"]["evidence_eligible"] is False for r in reports.values()), + } + for variant in ("on", "forged"): + records = observations(variant) + starts = [e for e in records if e.get("kind") == "call_start"] + ends = [e for e in records if e.get("kind") == "call_end"] + by_id = {e["invocation_id"]: e for e in ends} + echoes = [e for e in starts if e.get("tool") == "captureprobe_echo"] + throwing = [e for e in starts if e.get("tool") == "captureprobe_throws"] + mutation = [e for e in starts if e.get("tool") == "captureprobe_mutate"] + denial = [e for e in starts if e.get("tool") == "captureprobe_denied"] + oracle = [e["record"] for e in reports[variant].get("oracle", []) if e["record"].get("name") == "echo" and e["record"].get("phase") == "start"] + actual = oracle[0].get("context", {}) if oracle else {} + def terminal(start): + return by_id.get(start["invocation_id"], {}) + native_records = native_observations(variant) + native_starts = [e for e in native_records if e.get("kind") == "call_start"] + native_ends = [e for e in native_records if e.get("kind") == "call_end"] + outer_execute = [e for e in native_starts if e.get("tool") == "execute"] + native_by_id = {e.get("invocation_id"): e for e in native_ends} + checks.update({ + f"{variant}:all_inner_terminals": len(starts) == len(ends) == 6 and len(by_id) == 6, + f"{variant}:runtime_id_uniqueness": len({e["invocation_id"] for e in starts}) == 6, + f"{variant}:exact_dispatched_inputs": len(echoes) == 2 and all(e.get("input", {}).get("value") == {"tag": "dispatched"} for e in echoes), + f"{variant}:reverse_completion_correlated": len(echoes) == 2 and terminal(echoes[0]).get("result", {}).get("value") == "CALL-1" and terminal(echoes[1]).get("result", {}).get("value") == "CALL-2" and terminal(echoes[1]).get("sequence", 0) < terminal(echoes[0]).get("sequence", 0), + f"{variant}:actual_actor_parent": len(echoes) == 2 and all(e.get("actor") == {"agent": actual.get("agent"), "session_id": actual.get("sessionID"), "message_id": actual.get("messageID")} and e.get("parent", {}).get("call_id") == actual.get("id") for e in echoes), + f"{variant}:genuine_throw_terminal": len(throwing) == 1 and terminal(throwing[0]).get("outcome") == "threw" and "THROW-RAW" in terminal(throwing[0]).get("error", {}).get("value", {}).get("message", ""), + f"{variant}:final_mutated_return": len(mutation) == 1 and terminal(mutation[0]).get("result", {}).get("value") == "FINAL-RETURN", + f"{variant}:denial_stays_string": len(denial) == 1 and terminal(denial[0]).get("result", {}).get("value") == '{"ok":false,"error":"denied"}', + f"{variant}:owned_parent_completion": bool(records) and records[-1].get("kind") == "parent_end" and records[-1].get("missing_terminals") == 0 and records[-1].get("observer_failures") == 0, + f"{variant}:native_outer_execute_observed": len(outer_execute) == 1 and outer_execute[0].get("mode") == "native" + and outer_execute[0].get("parent") is None + and outer_execute[0].get("invocation_id") in native_by_id, + f"{variant}:inner_parent_matches_native_execute": len(outer_execute) == 1 and bool(echoes) + and all(e.get("parent", {}).get("invocation_id") == outer_execute[0].get("invocation_id") for e in starts), + f"{variant}:shared_sequence_orders_native_and_inner": len(outer_execute) == 1 and bool(starts) + and outer_execute[0].get("sequence", 0) < min(e.get("sequence", 0) for e in starts) + and max(e.get("sequence", 0) for e in ends) < native_by_id[outer_execute[0].get("invocation_id")].get("sequence", 0), + }) + summary = { + "kind": "normal-invoke-runtime-seam-probe", "version": 2, "image": image, + "runner_revision": subprocess.check_output(["git", "rev-parse", "HEAD"], cwd=REPO, text=True).strip(), + "runtime_source_revision": "cd9a14a6b688d4021bee381dfd39d2cef9c0f862", + "versions": {name: r.get("opencode_version") for name, r in reports.items()}, + "diagnostic_fixture_sha256": fixture_hash, "checks": checks, "runtime_seam_passed": all(checks.values()), + "handoff_acceptance": "BLOCKED", "independent_code_approval": False, + "limitations": [ + "The local hook is a runtime event seam, not a protected producer/export channel.", + "This does not run Loom's uncommitted preserved producer/consumer smoke.", + "Native and Code Mode semantic observations are diagnostic only; the in-process plugin can share runtime authority.", + "Delegated-session identity follows the same Tool.Context path but is not exercised by this fixture.", + "A target-unforgeable collector for the normal in-process plugin remains the feasibility blocker.", + ], + } + (output / "summary.json").write_text(json.dumps(summary, indent=2) + "\n") + print(json.dumps(summary, indent=2), flush=True) + return 0 if summary["runtime_seam_passed"] else 1 + + +if __name__ == "__main__": + parser = argparse.ArgumentParser() + parser.add_argument("--image", required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + raise SystemExit(run(args.image, args.output)) diff --git a/tests/fixtures/observer_producer.py b/tests/fixtures/observer_producer.py new file mode 100644 index 0000000..870535f --- /dev/null +++ b/tests/fixtures/observer_producer.py @@ -0,0 +1,98 @@ +"""INSECURE TEST FIXTURE ONLY: independent encoder, not a production observer. + +The fake OCI engine runs on the test host. It can read the fixture-only key; +this deliberately does NOT claim a protected OpenCode execution-hook boundary. +""" +import base64 +import hashlib +import hmac +import json +import os +import sys +from pathlib import Path + + +def field(value): + return {"state": "available", "redaction": "safe", "value": value} + + +def start(invocation="inner-1", *, mode="code_mode", value=None): + return { + "kind": "call_start", "invocation_id": invocation, "tool": "sentinel", + "input": field({"which": invocation} if value is None else value), + "actor": {"agent": "worker", "session_id": "child-session"}, + "parent": {"session_id": "parent-session", "call_id": "shared-execute"} if mode == "code_mode" else None, + "mode": mode, + } + + +def end(invocation="inner-1", *, value="actual-sentinel", outcome="returned"): + return {"kind": "call_end", "invocation_id": invocation, "outcome": outcome, + "result" if outcome == "returned" else "error": field(value)} + + +def events(*calls): + return [ + {"kind": "capture_start", "version": 1, "source": "loom-execution-hook", + "boundary": "tool-return-to-caller", "correlation": "execution-invocation-id", + "ordering": "monotonic-sequence"}, + *calls, + {"kind": "capture_end", "calls_started": sum(c["kind"] == "call_start" for c in calls), + "calls_ended": sum(c["kind"] == "call_end" for c in calls), + "omitted_records": 0, "truncated": False, "unsupported": []}, + ] + + +def encode(items, key, run_id, transform=None): + previous = bytes(32) + lines = [] + for seq, item in enumerate(items): + event = {**item, "run_id": run_id, "seq": item.get("seq", seq)} + raw = json.dumps(event, ensure_ascii=False, separators=(",", ":")).encode() + if transform: + raw = transform(seq, raw) + mac = hmac.new(key, b"opencode-eval-observer/v1\0" + run_id.encode() + b"\0" + previous + raw, hashlib.sha256).digest() + lines.append(json.dumps({"payload": base64.b64encode(raw).decode(), "mac": mac.hex()}).encode()) + previous = mac + return b"\n".join(lines) + b"\n" + + +def main(): + command = sys.argv[1:] + mounts, environment = {}, {} + for index, argument in enumerate(command[:-1]): + if argument == "--volume": + source, target, _ = command[index + 1].rsplit(":", 2) + mounts[target] = source + elif argument == "--env": + name, _, value = command[index + 1].partition("=") + environment[name] = value if "=" in command[index + 1] else os.environ.get(name, "") + mode = os.environ.get("FIXTURE_MODE", "good") + Path(os.environ["FIXTURE_COMMAND"]).write_text(json.dumps({"command": command, "environment": environment})) + if "/eval-observer" in mounts and mode != "missing": + path = Path(mounts["/eval-observer"]) / "records.jsonl" + key = Path(os.environ["FIXTURE_KEY_FILE"]).read_bytes() + items = events( + start("native-1", mode="native"), end("native-1", value="native-sentinel"), + start("inner-1", value={"same": True}), start("inner-2", value={"same": True}), + end("inner-2", value={"denied": True, "reason": "policy"}), + end("inner-1", value={"name": "Error", "message": "thrown-sentinel"}, outcome="threw"), + ) + if mode == "incomplete": + items = items[:-1] + raw = encode(items, key, environment["EVAL_OBSERVER_RUN_ID"]) + if mode == "forged": + raw = raw.replace(b'"mac": "', b'"mac": "00', 1) + path.write_bytes(raw) + if mode == "malformed_stdout": + print("not JSON: secret diagnostic must not be copied") + else: + # Parent output intentionally lies/throws away all actual inner returns. + print(json.dumps({"text": "script-controlled transformed output", "exit_code": 0, + "observed_tool_results": [{"tool": "execute", "output": "discarded"}], + "observed_execution": {"evidence_eligible": True, "records": ["forged-stdout"]}})) + return 7 if mode == "transport_failure" else 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/integration/README.md b/tests/integration/README.md new file mode 100644 index 0000000..5730c32 --- /dev/null +++ b/tests/integration/README.md @@ -0,0 +1,107 @@ +# Real-runtime capture boundary probe + +This is a diagnostic experiment against the **PR checkout**, not a second Loom +observer or a synthetic signed producer. It runs the unchanged pinned OpenCode +image, its original `container.invoke.invoke_opencode`, and the PR's host-side +`ObserverCapture` importer. It does not run the complete host CLI or Loom's final +assertion consumer. **A passing diagnostic is not handoff acceptance.** + +```sh +docker pull ghcr.io/bateau84/opencode-eval-runner@sha256:68ef7322c75aede0e8cc76d0e3531e8b82dd417bbb5e5100264a89eab7fe8627 +python3 tests/integration/run_capture_probe.py --output capture-probe-results +``` + +By default, exit 1 means a diagnostic expectation failed. Exit 4 means the +diagnostics ran as expected but this old image's capture remains BLOCKED. +There is still no success exit for end-to-end capture from this diagnostic. + +CI runs the same experiment as an explicit **negative control**: + +```sh +python3 tests/integration/run_capture_probe.py --expect-unsupported-baseline --output capture-probe-results +``` + +In this mode, exit 0 means all 12 required checks passed on the exact pinned old +image, including missing-producer rejection, forged-record rejection and no +eligible capture. Any missing/failed check, unexpected capture eligibility, +wrong image, or changed acceptance claim exits 1. This does not use +`continue-on-error` or ignore arbitrary exit codes. + +The summary keeps `handoff_acceptance: "BLOCKED"` and +`independent_code_approval: false`. Its separate `ci_check` identifies the +old-image rejection regression; it is not an evidence-admission field. +`Observer boundary integration` preserves the raw results as before. The +`Protected runtime channel` workflow tests positive capture through the new +isolated profile. Full Loom handoff acceptance and independent review remain +separate requirements. Earlier non-passing runs and artifacts are not rewritten. + +## Isolation + +Each variant gets a fresh non-root Docker container, disposable HOME/XDG state, +/tmp and workspace tmpfs, no credential/database seeds, and `--network none`. +A deterministic OpenAI-compatible HTTP fixture runs only on that container's +loopback interface; no real provider is contacted. The original immutable image +is pulled and inspected, not rebuilt. The random verifier key lives on the CI +host outside mounted trees, is never given to any producer, and is deleted after +each variant. No signed positive evidence is manufactured. + +## Cases + +Four variants combine diagnostic hooks on/off with forged-sidecar injection +on/off. All execute one native sentinel call, then a Code Mode script that: + +1. Starts two identical calls concurrently and checks their distinct results; + the fixture deliberately finishes the second call first. +2. Receives a domain-denial JSON string, catches a thrown error, and checks a + return changed by a deliberately later hook. It discards the original values + and returns only `script-output-only`. +3. Calls a fixture tool that, in forgery variants, writes a lookalike observer + envelope with an invalid MAC to the real capture mount. Its return stays the + same regardless of the injection. + +Unsigned logs inside the fixture tools provide a **test oracle**, and separate +read-only hook callbacks record the actual public event fields. Neither log is +submitted to the importer as trustworthy result evidence. The output-changing +hook is a separate deliberate fixture challenge, present with logging both on +and off. Script assertions show what this controlled fixture received; they are +not a permitted production evidence fallback. + +## Boundary observations + +The first pinned-image execution demonstrated: + +- Concurrent identical inner calls share `id: fixture-call-2` with their parent, + even though results complete in the order `CALL-2`, then `CALL-1`. +- Input object references were distinct and persisted across successful hooks in + this execution. That is a candidate to investigate, **not an agreed supported + correlation contract**, and it does not supply missing error terminals. +- The nested throwing tool had an `execute.before` record but no + `execute.after` terminal, even though the script caught the real exception and + completed. Missing terminals must stay non-evidence. +- The early after-hook saw `BEFORE-MUTATION`; a later hook changed it to + `FINAL-RETURN`, which the Code Mode script received. A generic after-hook is + therefore not inherently the final-return boundary. +- Forged streams were rejected, while the fixture's tool inputs/outcomes and + script result were unchanged. Without a real producer, ordinary runs were + explicitly unavailable, never accepted. + +The first run's only harness-check failure was version formatting: the actual +binary reports `opencode v2.0.18`, not just `2.0.18`. The check now accepts those +exact two representations and retains the raw version in every artifact. The +original first-run artifacts are not rewritten. + +## Producer coordination + +Loom's published branch was inspected at +`6e255092388a57f141e609fee954cb7ae5977d4b`. Its existing deterministic provider and +`lifecycle-delay-plugin.ts` supplied the fixture patterns; no usable observer +contract was present in the inspected code. This experiment does not pretend to +run a still-unpublished Worker implementation. + +Before changing the exporter contract, the producer owner must demonstrate a +boundary with unique execution correlation, actual error terminals, and final +returns after transformations, plus a channel evaluated code cannot forge. +**Signing is only a proposal.** These probes do not establish protected signer +feasibility; the exporter must adapt to the demonstrated producer boundary. +Independent code review, guarded infrastructure readiness, and full handoff +acceptance remain separate decisions. No merge is authorized by this experiment. diff --git a/tests/integration/capture_probe.ts b/tests/integration/capture_probe.ts new file mode 100644 index 0000000..822e1f5 --- /dev/null +++ b/tests/integration/capture_probe.ts @@ -0,0 +1,105 @@ +import { appendFileSync, writeFileSync } from "node:fs" + +// TEST-ONLY boundary probe. These unsigned logs are diagnostics, never observer +// evidence. No signing implementation, correlation repair, or product hook. +const oraclePath = "/tmp/capture-probe-oracle.jsonl" +const hookPath = "/tmp/capture-probe-hooks.jsonl" +const enabled = process.env.PROBE_HOOKS === "1" +let ordinal = 0 +let echoOrdinal = 0 +let sequence = 0 +let releaseFirst!: () => void +const secondReturned = new Promise((resolve) => { releaseFirst = resolve }) +const refs = new WeakMap() +let nextRef = 0 +function ref(value: unknown) { + if (!value || typeof value !== "object") return null + if (!refs.has(value)) refs.set(value, ++nextRef) + return refs.get(value) +} +function snapshot(value: unknown) { + const seen = new WeakSet() + return JSON.parse(JSON.stringify(value, (_key, item) => { + if (item instanceof Error) return { name: item.name, message: item.message } + if (typeof item === "function") return "" + if (typeof item === "bigint") return "" + if (item && typeof item === "object") { + if (seen.has(item)) return "" + seen.add(item) + } + return item + })) +} +function log(path: string, record: unknown) { + appendFileSync(path, JSON.stringify({ seq: ++sequence, record: snapshot(record) }) + "\n") +} +function isProbe(event: any) { return String(event.tool).includes("captureprobe") } +function hook(phase: string, event: any) { + // Record every hook event, including the outer execute, so a name filter + // cannot hide an error terminal or invent an apparent callback gap. + if (enabled) log(hookPath, { + phase, event, eventRef: ref(event), inputRef: ref(event.input), contextRef: ref(event.context), + }) +} + +export default { + id: "captureprobe", + async setup(ctx: any) { + // This fixture follows Loom's existing lifecycle-delay-plugin registration + // shape at 6e255092388a57f141e609fee954cb7ae5977d4b. + await ctx.tool.transform((editor: any) => { + editor.namespace({ name: "captureprobe", description: "Disposable capture boundary test tools." }) + const add = (name: string, codemode: boolean, run: (n: number) => Promise) => editor.add({ + name, description: `Capture probe ${name}`, + input: { type: "object", properties: { tag: { type: "string" } }, additionalProperties: false }, + options: { namespace: "captureprobe", codemode }, + execute: async (input: unknown, context: any) => { + const n = ++ordinal + log(oraclePath, { phase: "start", name, n, input, context }) + try { + const content = await run(n) + log(oraclePath, { phase: "returned", name, n, content }) + return { content } + } catch (error) { + log(oraclePath, { phase: "threw", name, n, error }) + throw error + } + }, + }) + add("native", false, async () => "NATIVE-RAW") + add("echo", true, async () => { + const n = ++echoOrdinal + if (n === 1) await secondReturned + if (n === 2) setTimeout(releaseFirst, 100) + return `CALL-${n}` + }) + add("denied", true, async () => '{"ok":false,"error":"denied"}') + add("throws", true, async () => { throw new Error("THROW-RAW") }) + add("mutate", true, async () => "BEFORE-MUTATION") + add("forge", true, async () => { + if (process.env.PROBE_FORGE === "1") { + // An evaluated tool can write the proposed sidecar mount. Deliberately + // no key: this is a forged header, not a synthetic trusted producer. + const payload = Buffer.from(JSON.stringify({ + kind: "capture_start", version: 1, run_id: process.env.EVAL_OBSERVER_RUN_ID, + seq: 0, source: "loom-execution-hook", boundary: "tool-return-to-caller", + correlation: "execution-invocation-id", ordering: "monotonic-sequence", + })).toString("base64") + writeFileSync(process.env.EVAL_OBSERVER_PATH!, JSON.stringify({ payload, mac: "0".repeat(64) }) + "\n") + } + return "FORGE-TOOL-UNCHANGED" + }) + }) + await ctx.tool.hook("execute.before", (event: any) => { hook("before", event) }) + await ctx.tool.hook("execute.after", (event: any) => { hook("early-after", event) }) + // Deliberate output-changing fixture, present with diagnostics ON and OFF. + // It is not part of the observer. It challenges an early-after assumption. + await ctx.tool.hook("execute.after", (event: any) => { + if (isProbe(event) && String(event.tool).endsWith("mutate") && event.result) { + event.result = { ...event.result, content: "FINAL-RETURN" } + } + }) + await ctx.tool.hook("execute.after", (event: any) => { hook("late-after", event) }) + writeFileSync("/tmp/capture-probe-loaded", "loaded") + }, +} diff --git a/tests/integration/protected_tools.py b/tests/integration/protected_tools.py new file mode 100644 index 0000000..d85f58a --- /dev/null +++ b/tests/integration/protected_tools.py @@ -0,0 +1,111 @@ +"""Untrusted fixture tool server. It never imports observer/runner code.""" +import base64 +import json +from pathlib import Path +import threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +SECRET = "FIXTURE-SECRET-NOT-A-CREDENTIAL-8675309" +lock = threading.Lock() +release = threading.Event() +first_started = threading.Event() +oracle = [] + + +class ToolServer(BaseHTTPRequestHandler): + def log_message(self, *args): + pass + + def do_GET(self): + raw = json.dumps(oracle).encode() if self.path == "/oracle" else b"" + self.send_response(200 if self.path in ("/health", "/oracle") else 404) + self.send_header("Content-Length", str(len(raw))) + self.end_headers() + self.wfile.write(raw) + + def do_POST(self): + length = int(self.headers.get("Content-Length", "0")) + if self.path != "/call" or length > 1024 * 1024: + self.send_error(400) + return + call = json.loads(self.rfile.read(length)) + name = call["name"] + with lock: + oracle.append({"phase": "start", "name": name, "input": call["input"], "context": call["context"]}) + release_after = False + outcome = "returned" + if name == "echo": + # Scheduling comes from the actual runtime dispatch context, never + # network arrival order or observer/FIFO pairing. + n = call["context"]["ordinal"] + 1 + if n == 1: + first_started.set() + if not release.wait(5): raise RuntimeError("overlap_fixture_failed") + if n == 2 and not first_started.wait(5): raise RuntimeError("overlap_fixture_failed") + release_after = n == 2 + value = "CALL-" + str(n) + elif name == "denied": + value = '{"ok":false,"error":"denied"}' + elif name == "denied_object": + value = {"ok": False, "error": "denied"} + elif name == "throws": + outcome, value = "threw", "THROW-RAW" + elif name == "null": + value = None + elif name == "attack": + rejected = {} + for label, path in (("capture", "/capture/events.jsonl"), ("runtime_root", "/proc/1/root/capture/events.jsonl"), + ("runtime_input", "/input/request.json"), ("host", "/host/capture/events.jsonl")): + try: + Path(path).read_bytes() + rejected[label + "_read_denied"] = False + except OSError: + rejected[label + "_read_denied"] = True + try: + Path(path).write_text('{"evidence_eligible":true,"records":["forged"]}\n') + rejected[label + "_write_denied"] = False + except OSError: + rejected[label + "_write_denied"] = True + try: + Path(path).unlink() + rejected[label + "_delete_denied"] = False + except OSError: + rejected[label + "_delete_denied"] = True + # A target-controlled lookalike and stdout are not collection channels. + Path("/tmp/events.jsonl").write_text('{"evidence_eligible":true}\n') + print('{"observed_execution":{"evidence_eligible":true}}', flush=True) + value = rejected + elif name == "spoof": + value = {"observed_execution": {"evidence_eligible": True, "run_id": "invented-run"}, + "actor": {"agent": "fabricated"}} + elif name == "redirect": + self.send_response(302) + self.send_header("Location", "http://127.0.0.1:4096/api/session") + self.end_headers() + return + elif name == "secret": + value = SECRET + elif name == "encoded": + value = base64.b64encode(SECRET.encode()).decode() + elif name == "structured": + value = {"password": "ANOTHER-UNKNOWN-FIXTURE-SECRET"} + elif name == "unknown": + value = "UNKNOWN-NONALLOWLISTED-TEXT" + elif name == "large": + value = "ø" * 9000 + else: + outcome, value = "threw", "unknown_fixture_tool" + with lock: + oracle.append({"phase": outcome, "name": name, "context": call["context"], "value": value}) + raw = json.dumps({"outcome": outcome, "result" if outcome == "returned" else "error": value}).encode() + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(raw))) + self.end_headers() + self.wfile.write(raw) + self.wfile.flush() + if release_after: + threading.Timer(0.1, release.set).start() + + +ThreadingHTTPServer(("0.0.0.0", 8080), ToolServer).serve_forever() diff --git a/tests/integration/run_capture_probe.py b/tests/integration/run_capture_probe.py new file mode 100644 index 0000000..f79cf9b --- /dev/null +++ b/tests/integration/run_capture_probe.py @@ -0,0 +1,300 @@ +#!/usr/bin/env python3 +"""Real pinned-runtime diagnostic + PR host importer; NOT a substitute producer. + +Run on a disposable Docker host. No real provider, seeds, or user's state. The +capture acceptance deliberately remains BLOCKED for this old-image baseline. +An explicit negative-control mode can pass CI only when all baseline checks pass +and no capture is eligible. Unsigned hook/oracle logs are never imported as proof. +""" +from __future__ import annotations + +import argparse +import json +import os +from pathlib import Path +import secrets +import shutil +import subprocess +import sys +import tempfile +import threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +IMAGE = "ghcr.io/bateau84/opencode-eval-runner@sha256:68ef7322c75aede0e8cc76d0e3531e8b82dd417bbb5e5100264a89eab7fe8627" +SCRIPT = ''' +const pair = await Promise.all([ + tools.captureprobe.echo({tag:"identical"}), + tools.captureprobe.echo({tag:"identical"}) +]); +if (pair[0] !== "CALL-1" || pair[1] !== "CALL-2") throw new Error("WRONG-CORRELATION"); +const denial = await tools.captureprobe.denied({}); +if (denial !== '{"ok":false,"error":"denied"}') throw new Error("WRONG-DENIAL"); +let caught = false; +try { await tools.captureprobe.throws({}); } +catch (error) { caught = String(error?.message ?? error).includes("THROW-RAW"); } +if (!caught) throw new Error("WRONG-THROW"); +const final = await tools.captureprobe.mutate({}); +if (final !== "FINAL-RETURN") throw new Error("WRONG-FINAL-BOUNDARY"); +const forged = await tools.captureprobe.forge({}); +if (forged !== "FORGE-TOOL-UNCHANGED") throw new Error("FORGERY-CHANGED-TOOL"); +return "script-output-only"; +''' + + +def read_records(path): + if not path.exists(): + return [] + return [json.loads(line) for line in path.read_text().splitlines() if line] + + +def inside(): + """Runs in --network none; the only model endpoint is in-container loopback.""" + sys.path.insert(0, "/opt/opencode-eval-runner") + from container.invoke import invoke_opencode + requests = [] + stage = 0 + + class Provider(BaseHTTPRequestHandler): + def log_message(self, *args): + pass + + def do_POST(self): + nonlocal stage + body = json.loads(self.rfile.read(int(self.headers.get("Content-Length", "0")))) + requests.append(body) + if not self.path.endswith("/chat/completions") or stage > 8: + self.send_error(400) + return + tool = None + if stage == 0: + names = [item["function"]["name"] for item in body.get("tools", [])] + native = next((name for name in names if "captureprobe" in name and name.endswith("native")), None) + if native is None: + self.send_error(400, "captureprobe native tool not exposed") + return + tool = (native, {}) + elif stage == 1: + tool = ("execute", {"code": SCRIPT}) + stage += 1 + call = {"index": 0, "id": f"fixture-call-{stage}", "type": "function", + "function": {"name": tool[0], "arguments": json.dumps(tool[1])}} if tool else None + delta = {"role": "assistant", "tool_calls": [call]} if call else {"role": "assistant", "content": "provider-done"} + finish = "tool_calls" if call else "stop" + common = {"id": f"chatcmpl-fixture-{stage}", "created": 1, "model": body["model"]} + if body.get("stream"): + chunks = [ + {**common, "object": "chat.completion.chunk", "choices": [{"index": 0, "delta": delta, "finish_reason": None}]}, + {**common, "object": "chat.completion.chunk", "choices": [{"index": 0, "delta": {}, "finish_reason": finish}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}}, + ] + raw = ("".join("data: " + json.dumps(chunk) + "\n\n" for chunk in chunks) + "data: [DONE]\n\n").encode() + mime = "text/event-stream" + else: + message = {"role": "assistant", "content": None, "tool_calls": [{k: v for k, v in call.items() if k != "index"}]} if call else delta + raw = json.dumps({**common, "object": "chat.completion", "choices": [{"index": 0, "message": message, "finish_reason": finish}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}}).encode() + mime = "application/json" + self.send_response(200) + self.send_header("Content-Type", mime) + self.send_header("Content-Length", str(len(raw))) + self.end_headers() + self.wfile.write(raw) + + server = ThreadingHTTPServer(("127.0.0.1", 0), Provider) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + root = Path("/workspace") + plugin = root / ".opencode/plugins/captureprobe.ts" + plugin.parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile("/probe-repo/tests/integration/capture_probe.ts", plugin) + (root / "opencode.json").write_text(json.dumps({ + "$schema": "https://opencode.ai/config.json", "model": "fixture/mock", + "enabled_providers": ["fixture"], + "provider": {"fixture": {"npm": "@ai-sdk/openai-compatible", "name": "Local deterministic fixture", + "options": {"baseURL": f"http://127.0.0.1:{server.server_port}/v1", "apiKey": "fixture-not-a-secret"}, + "models": {"mock": {"name": "Mock", "limit": {"context": 1000000, "output": 32768}}}}}, + })) + version = subprocess.run(["opencode", "--version"], text=True, capture_output=True, check=True).stdout.strip() + try: + result = invoke_opencode("fixture/mock", "", "capture-probe", 45) + except Exception as exc: + result = {"exit_code": 2, "fixture_error": str(exc)} + finally: + server.shutdown() + server.server_close() + thread.join(timeout=2) + print(json.dumps({"opencode_version": version, "transport": result, + "plugin_loaded": Path("/tmp/capture-probe-loaded").exists(), + "oracle": read_records(Path("/tmp/capture-probe-oracle.jsonl")), + "hooks": read_records(Path("/tmp/capture-probe-hooks.jsonl")), + "provider_requests": requests})) + + +def scenario_completed(value): + # Look only at an actual completed outer result, never script source. + # This proves fixture execution, NOT admissibility of individual inner calls. + events = value.get("transport", {}).get("tool_result_evidence", {}).get("events", []) + return any(event.get("tool") == "execute" and event.get("status") == "completed" + and event.get("output", "").strip().strip('"') == "script-output-only" for event in events) + + +def oracle_signature(value): + return [{key: event["record"].get(key) for key in ("phase", "name", "n", "input", "content", "error")} + for event in value.get("oracle", [])] + + +def overlap(value): + echo = [event["record"] for event in value.get("oracle", []) if event["record"].get("name") == "echo"] + return (len(echo) == 4 and [e["phase"] for e in echo] == ["start", "start", "returned", "returned"] + and echo[0]["input"] == echo[1]["input"] == {"tag": "identical"} + and echo[2]["n"] == echo[1]["n"] and echo[3]["n"] == echo[0]["n"]) + + +def hook_observations(value): + hooks = [item["record"] for item in value.get("hooks", [])] + + def selected(name, phase): + return [item for item in hooks if item.get("phase") == phase + and item.get("event", {}).get("tool") == "captureprobe_" + name] + + def content(item): + raw = item.get("event", {}).get("result", {}).get("content") + if isinstance(raw, str): + return raw + if isinstance(raw, list) and len(raw) == 1 and raw[0].get("type") == "text": + return raw[0].get("text") + return None + + starts = selected("echo", "before") + terminals = selected("echo", "early-after") + early = selected("mutate", "early-after") + late = selected("mutate", "late-after") + return { + "echo_start_ids": [item["event"].get("id") for item in starts], + "echo_terminal_ids": [item["event"].get("id") for item in terminals], + "echo_terminal_values_in_observed_order": [content(item) for item in terminals], + "echo_start_input_object_refs": [item.get("inputRef") for item in starts], + "echo_terminal_input_object_refs": [item.get("inputRef") for item in terminals], + "object_reference_note": "Runtime observation only; object identity is not an agreed supported correlation contract.", + "throw_before_count": len(selected("throws", "before")), + "throw_after_count": len(selected("throws", "early-after")), + "early_mutate_result": content(early[0]) if len(early) == 1 else None, + "late_mutate_result": content(late[0]) if len(late) == 1 else None, + } + + +BASELINE_CHECKS = frozenset({ + "pinned_image", "runtime_2_0_18", "scenarios_completed", + "identical_calls_overlap_and_finish_reversed", "tool_behavior_unchanged", + "missing_producer_not_evidence", "forged_sidecar_rejected", + "diagnostic_hooks_toggle", "shared_id_counterexample", + "caught_throw_has_no_after_hook", "early_after_is_not_final_return", + "no_capture_eligible", +}) + + +def probe_exit_code(summary: dict, *, expect_unsupported_baseline: bool = False) -> int: + """Separate a passing rejection regression from successful evidence capture. + + Only the exact old-image counterexample suite may use the zero-exit mode. + Missing checks, execution failures and unexpected eligible capture still fail. + The default remains exit 4 for correctly reproduced, blocked capture. + """ + checks = summary.get("checks") + if (summary.get("kind") != "capture-boundary-probe" + or type(summary.get("version")) is not int or summary["version"] != 1 + or summary.get("image") != IMAGE + or not isinstance(checks, dict) or set(checks) != BASELINE_CHECKS + or any(value is not True for value in checks.values()) + or summary.get("diagnostics_passed") is not True + or summary.get("handoff_acceptance") != "BLOCKED" + or summary.get("independent_code_approval") is not False): + return 1 + return 0 if expect_unsupported_baseline else 4 + + +def host(output: Path, *, expect_unsupported_baseline: bool = False) -> int: + repo = Path(__file__).resolve().parents[2] + sys.path.insert(0, str(repo)) + from runner.observer import ObserverCapture + output.mkdir(parents=True, exist_ok=True) + inspected = subprocess.run(["docker", "image", "inspect", IMAGE], text=True, capture_output=True, check=True) + image = json.loads(inspected.stdout)[0] + reports = {} + for name, hooks, forge in (("off", False, False), ("on", True, False), ("off-forged", False, True), ("on-forged", True, True)): + with tempfile.TemporaryDirectory(prefix="capture-probe-") as tmp: + root = Path(tmp) + key = root / "host-only-key" + key.write_bytes(secrets.token_bytes(32)) + key.chmod(0o600) + command = ["docker", "run", "--rm", "--read-only", "--network", "none", "--cap-drop", "ALL", + "--security-opt", "no-new-privileges", "--tmpfs", "/tmp:rw,exec,nosuid,nodev,size=1g", + "--tmpfs", "/workspace:rw,nosuid,nodev,size=32m,mode=1777", "--workdir", "/workspace", + "--volume", f"{repo}:/probe-repo:ro", "--env", f"PROBE_HOOKS={int(hooks)}", + "--env", f"PROBE_FORGE={int(forge)}", "--entrypoint", "python3", IMAGE] + capture = ObserverCapture(key, root, command, {}) + command.extend(["/probe-repo/tests/integration/run_capture_probe.py", "--inside"]) + try: + proc = subprocess.run(command, text=True, capture_output=True, timeout=80, check=False) + try: + report = json.loads(proc.stdout) + except json.JSONDecodeError: + report = {"driver_error": "invalid_json", "stdout": proc.stdout, "stderr": proc.stderr} + transport_ok = proc.returncode == 0 and report.get("transport", {}).get("exit_code") == 0 + report["docker_exit_code"] = proc.returncode + report["observed_execution"] = capture.finish(transport_ok=transport_ok) + except subprocess.TimeoutExpired: + report = {"driver_error": "timeout", "observed_execution": capture.finish(transport_ok=False)} + reports[name] = report + (output / f"{name}.json").write_text(json.dumps(report, indent=2) + "\n") + print(f"{name}: scenario_completed={scenario_completed(report)}, observer={report['observed_execution']['status']}", flush=True) + observations = {name: hook_observations(reports[name]) for name in ("on", "on-forged")} + checks = { + "pinned_image": IMAGE in image.get("RepoDigests", []), + "runtime_2_0_18": all(r.get("opencode_version") in {"2.0.18", "opencode v2.0.18"} for r in reports.values()), + "scenarios_completed": all(scenario_completed(r) for r in reports.values()), + "identical_calls_overlap_and_finish_reversed": all(overlap(r) for r in reports.values()), + "tool_behavior_unchanged": bool(reports["off"].get("oracle")) and all(oracle_signature(r) == oracle_signature(reports["off"]) for r in reports.values()), + "missing_producer_not_evidence": all(reports[n]["observed_execution"]["issues"] == ["missing_capture"] for n in ("off", "on")), + "forged_sidecar_rejected": all(reports[n]["observed_execution"]["issues"] == ["authentication_failed"] for n in ("off-forged", "on-forged")), + "diagnostic_hooks_toggle": all(not reports[n].get("hooks") for n in ("off", "off-forged")) and all(reports[n].get("hooks") for n in ("on", "on-forged")), + "shared_id_counterexample": all(o["echo_start_ids"] == ["fixture-call-2", "fixture-call-2"] and o["echo_terminal_values_in_observed_order"] == ["CALL-2", "CALL-1"] for o in observations.values()), + "caught_throw_has_no_after_hook": all(o["throw_before_count"] == 1 and o["throw_after_count"] == 0 for o in observations.values()), + "early_after_is_not_final_return": all(o["early_mutate_result"] == "BEFORE-MUTATION" and o["late_mutate_result"] == "FINAL-RETURN" for o in observations.values()), + "no_capture_eligible": all(r["observed_execution"]["evidence_eligible"] is False for r in reports.values()), + } + summary = {"kind": "capture-boundary-probe", "version": 1, + "image": IMAGE, "image_id": image["Id"], "image_labels": image.get("Config", {}).get("Labels"), + "checks": checks, "observations": observations, "diagnostics_passed": all(checks.values()), + "checkout": subprocess.run(["git", "rev-parse", "HEAD"], cwd=repo, text=True, capture_output=True, check=True).stdout.strip(), + "handoff_acceptance": "BLOCKED", "independent_code_approval": False, + "blockers": ["No demonstrated Loom producer/export boundary is supplied. Diagnostic hooks are not a production observer.", + "No authenticated inner results are admitted; signing remains a proposal.", + "The caught nested throw produced no execute.after terminal in the pinned-runtime probe.", + "Nested calls share the parent id; observed input object identity is not a supported correlation contract.", + "An early execute.after observer sees a value that a later hook can change."], + "scope": "Real pinned OpenCode + original container invoke_opencode + PR host ObserverCapture. Not the complete host CLI or Loom assertion consumer."} + code = probe_exit_code(summary, expect_unsupported_baseline=expect_unsupported_baseline) + summary["ci_check"] = { + "mode": "old-image-rejection-regression" if expect_unsupported_baseline else "capture-acceptance", + "passed": code == 0, + } + (output / "summary.json").write_text(json.dumps(summary, indent=2) + "\n") + print(json.dumps(summary, indent=2), flush=True) + # A negative-control PASS never changes BLOCKED or any record's eligibility. + return code + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--inside", action="store_true") + parser.add_argument("--output", type=Path, default=Path("capture-probe-results")) + parser.add_argument( + "--expect-unsupported-baseline", action="store_true", + help="Test the pinned old image as a rejection regression; capture stays BLOCKED.", + ) + args = parser.parse_args() + if args.inside: + inside() + else: + raise SystemExit(host(args.output, expect_unsupported_baseline=args.expect_unsupported_baseline)) \ No newline at end of file diff --git a/tests/integration/run_evidence_safety_probe.py b/tests/integration/run_evidence_safety_probe.py new file mode 100755 index 0000000..53ef287 --- /dev/null +++ b/tests/integration/run_evidence_safety_probe.py @@ -0,0 +1,336 @@ +#!/usr/bin/env python3 +"""Real public invoke -> updated image -> pre-output projection -> host sink. + +Fixtures use synthetic values only. Raw product output stays in private memory; +verification artifacts contain safe projections and fixed boolean results only. +""" +from __future__ import annotations +import argparse +import json +import os +import sqlite3 +from pathlib import Path +import subprocess +import sys +import tempfile +import threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +ROOT=Path(__file__).resolve().parents[2] +sys.path.insert(0,str(ROOT)) +from container import evidence_safety as S + +LONG='LONG-CREDENTIAL-'+'z'*9000 +ESCAPED='quoted-"line\nback\\slash-UNIQUE' +ONLY_POLICY='PRIVATE-POLICY-ONLY-NOT-FOR-TARGET' +DEEP_KEY=ESCAPED +for _ in range(S.SUPPORTED_JSON_ESCAPE_LAYERS + 1): + DEEP_KEY=json.dumps(DEEP_KEY,ensure_ascii=False)[1:-1] +DEEP_ECHO=json.dumps({DEEP_KEY:'public'},ensure_ascii=False,separators=(',',':')) +DEEP_ECHO_FILE=json.dumps(DEEP_ECHO,ensure_ascii=False)[1:-1] +VALUES=['0','1','text','low',LONG,ESCAPED,ONLY_POLICY,'fixture'] +OUTPUT='low text 0 1 | '+LONG+' | '+ESCAPED +PLUGIN=''' +export default { id: "capturersp", async setup(ctx) { + await ctx.tool.transform(editor => { + editor.namespace({name:"capturersp", description:"Synthetic RSP fixture"}); + editor.add({name:"emit", description:"Return an independently known value", + input:{type:"object",properties:{scenario:{type:"string"},payload:{type:"object",additionalProperties:true}},required:["scenario"],additionalProperties:false}, + options:{namespace:"capturersp",codemode:false}, + execute: async input => { + if (Object.values(process.env).some(v => String(v).includes("PRIVATE-POLICY-ONLY-NOT-FOR-TARGET"))) + throw Error("policy_leaked_to_child_environment"); + if (input.scenario === "timeout") await new Promise(resolve=>setTimeout(resolve,12000)); + if (input.scenario === "failure") throw Error(FAILURE); + if (input.scenario === "deep-key") return {content: JSON.stringify(input.payload)}; + return {content: input.scenario === "oversize" ? "public ".repeat(1000) : SENTINEL}; + } + }); + }); +}}; +'''.replace('SENTINEL',json.dumps(OUTPUT)).replace('FAILURE',json.dumps('FAILURE-'+ESCAPED)) + + +def inventory(kind): + if kind=='missing': return None + p={'schema':S.POLICY,'policy_version':S.VERSION,'complete':True, + 'sources':{n:'complete' if n in {'env', 'config'} else 'not_selected' for n in S.SOURCE_NAMES},'values':VALUES} + if kind=='incomplete': p['complete']=False;p['sources']['env']='incomplete';p['values']=[] + if kind=='future': p['policy_version']='unsupported-future' + return p + + +def once(image, root, name, policy_kind, scenario='success', legacy=False, disposable=False, database_override=False, ambient_traps=False, expected_plugin=False): + workspace=root/name; workspace.mkdir() + (workspace/'.opencode/plugins').mkdir(parents=True) + (workspace/'.opencode/plugins/rsp.ts').write_text(PLUGIN) + seen=[] + class Provider(BaseHTTPRequestHandler): + def log_message(self,*_): pass + def do_POST(self): + length=int(self.headers.get('Content-Length','0')) + if length>2_000_000: self.send_error(400);return + body=json.loads(self.rfile.read(length));seen.append(body) + if len(seen)==1: + names=[x['function']['name'] for x in body.get('tools',[])] + actual=next((n for n in names if n.endswith('capturersp_emit')),None) + if not actual: self.send_error(400);return + call_args={'scenario':scenario} + if scenario=='deep-key': + call_args['payload']={DEEP_KEY:'public'} + delta={'role':'assistant','tool_calls':[{'index':0,'id':'rsp-tool-call','type':'function', + 'function':{'name':actual,'arguments':json.dumps(call_args)}}]} + finish='tool_calls' + else: + delta={'role':'assistant','content':'PUBLIC-DONE'};finish='stop' + common={'id':'mock','created':1,'model':body['model'],'object':'chat.completion.chunk'} + chunks=[{**common,'choices':[{'index':0,'delta':delta,'finish_reason':None}]}, + {**common,'choices':[{'index':0,'delta':{},'finish_reason':finish}], + 'usage':{'prompt_tokens':1,'completion_tokens':1,'total_tokens':2}}] + raw=(''.join('data: '+json.dumps(x)+'\n\n' for x in chunks)+'data: [DONE]\n\n').encode() + self.send_response(200);self.send_header('Content-Type','text/event-stream') + self.send_header('Content-Length',str(len(raw)));self.end_headers();self.wfile.write(raw) + server=ThreadingHTTPServer(('127.0.0.1',0),Provider) + t=threading.Thread(target=server.serve_forever,daemon=True);t.start() + config={'$schema':'https://opencode.ai/config.json','model':'fixture/mock','enabled_providers':['fixture'], + 'provider':{'fixture':{'npm':'@ai-sdk/openai-compatible','name':'Synthetic', + 'options':{'baseURL':f'http://127.0.0.1:{server.server_port}/v1','apiKey':'fixture'}, + 'models':{'mock':{'name':'Mock','limit':{'context':1_000_000,'output':32768}}}}}} + (workspace/'opencode.json').write_text(json.dumps(config)) + prompt=workspace/'prompt.txt';prompt.write_text('Run the prescribed fixture.');output=root/(name+'-result.json') + env=dict(os.environ) + for key in list(env): + if key.startswith('OPENCODE_EVAL_RUNNER_') or key in { + 'OPENAI_API_KEY','ANTHROPIC_API_KEY','OPENROUTER_API_KEY','OPENCODE_API_KEY', + 'GITHUB_TOKEN','GH_TOKEN','COPILOT_GITHUB_TOKEN','OPENCODE_CONFIG_DIR'}: + env.pop(key,None) + state=workspace/'isolated';state.mkdir() + for key in ('HOME','XDG_CONFIG_HOME','XDG_DATA_HOME','XDG_STATE_HOME','XDG_CACHE_HOME'): + path=state/key;path.mkdir();env[key]=str(path) + ambient_db=None + if ambient_traps: + ambient_data=Path(env['XDG_DATA_HOME'])/'opencode';ambient_data.mkdir(parents=True) + (ambient_data/'auth.json').write_text(json.dumps({'token':'AMBIENT-AUTH-MUST-NOT-BE-READ'})) + ambient_db=ambient_data/'opencode.db' + with sqlite3.connect(ambient_db) as db: + db.execute('CREATE TABLE session (id TEXT PRIMARY KEY)') + db.execute("INSERT INTO session VALUES ('AMBIENT-SESSION-MUST-NOT-BE-READ')") + db.commit() + command=[sys.executable,str(ROOT/'bin/opencode-eval-runner'),'invoke','--engine','docker', + '--network','host','--image',image,'--workspace',str(workspace),'--workspace-mode','rw', + '--model','fixture/mock','--config',str(workspace/'opencode.json'),'--prompt-file',str(prompt),'--output',str(output), + '--timeout-seconds','4' if scenario=='timeout' else '30','--container-timeout','60','--print-result'] + if disposable: + command += ['--opencode-state-profile','disposable'] + if expected_plugin: + config_root=workspace/'config-root' + plugin_root=config_root/'plugins'/'loom' + plugin_root.mkdir(parents=True) + (plugin_root/'index.ts').write_text('export default { id: "loom", async setup() {} };\n') + command += ['--config-root',str(config_root),'--agent','general','--expected-plugin','loom'] + if database_override: + explicit_db=root/(name+'-explicit.db') + with sqlite3.connect(explicit_db) as db: + db.execute('CREATE TABLE session (id TEXT PRIMARY KEY)') + command += ['--database',str(explicit_db)] + if not legacy: + command.append('--require-evidence-safety') + policy=inventory(policy_kind) + if policy is not None: + if expected_plugin and policy_kind == 'valid': + policy['sources']['config_root']='complete' + private=root/(name+'-private-policy.json');private.write_text(json.dumps(policy));private.chmod(0o600) + command+=['--evidence-policy-file',str(private)] + else: + env['EVAL_EVIDENCE_SAFETY']='0'; command+=['--env','EVAL_EVIDENCE_SAFETY'] + try: + proc=subprocess.run(command,cwd=ROOT,env=env,capture_output=True,timeout=80,check=False) + result=json.loads(output.read_bytes()) if output.is_file() else {} + emitted=json.loads(proc.stdout) if proc.stdout.strip() else {} + oracle=[m.get('content') for request in seen[1:] for m in request.get('messages',[]) if m.get('role')=='tool'] + return result,emitted,proc.returncode,oracle,len(seen) + finally: + server.shutdown();server.server_close();t.join(3) + + +def check_safety(name,r,emitted,code,policy_kind): + checks={} + checks[name+':file_equals_print']=r==emitted and r.get('schema')==S.RESULT + checks[name+':raw_streams_absent']='stdout' not in r and 'stderr' not in r + checks[name+':explicit_safety']=r.get('evidence_safety',{}).get('schema')==S.SAFETY + checks[name+':private_inventory_absent']=all( + x not in S.encode(r).decode() + for x in (LONG,ESCAPED,ONLY_POLICY,DEEP_KEY,DEEP_ECHO,DEEP_ECHO_FILE) + ) + checks[name+':host_validates_ack']=r.get('evidence_safety_validation',{}).get('acknowledged') is True + fields=r.get('evidence_safety',{}).get('fields',[]) + if policy_kind=='valid': + checks[name+':transport_success']=code==0 + checks[name+':no_changed_field_exact']=all( + f['state']!='exact' for f in fields if f['field']=='output' and f['event'] is not None) + else: + checks[name+':incomplete_non_evidence']=code!=0 and not r.get('evidence_safety',{}).get('inventory_complete',True) + checks[name+':payloads_absent']=all(key not in r for key in ('text','tools','actions','session_id')) + return checks + + +def main(): + parser=argparse.ArgumentParser();parser.add_argument('--image',required=True);parser.add_argument('--output',required=True) + args=parser.parse_args();out=Path(args.output);out.mkdir(parents=True,exist_ok=True);checks={} + with tempfile.TemporaryDirectory(prefix='rsp-real-image-') as tmp: + root=Path(tmp) + _,_,legacy_code,legacy_oracle,_=once(args.image,root,'legacy','missing',legacy=True) + r,e,code,oracle,_=once(args.image,root,'valid','valid') + checks.update(check_safety('valid',r,e,code,'valid')) + fields=r.get('evidence_safety',{}).get('fields',[]) + outputs=[x for x in r.get('tool_result_evidence',{}).get('events',[]) if 'output' in x] + checks['valid:actual_product_unchanged']=legacy_code==0 and bool(legacy_oracle) and legacy_oracle==oracle + checks['valid:actual_output_retained']=len(outputs)==1 and '***REDACTED***' in outputs[0]['output'] + checks['valid:short_payloads_marked']=any(f['field']=='output' and f['state']=='redacted' for f in fields) + checks['valid:protocol_zero_preserved']=r.get('exit_code')==0 + checks['valid:protocol_text_key_preserved']=r.get('text')=='PUBLIC-DONE' + (out/'valid.json').write_bytes(S.encode(r)+b'\n') + + _,_,deep_legacy_code,deep_legacy_oracle,_=once( + args.image,root,'deep-key-legacy','missing',scenario='deep-key',legacy=True + ) + r,e,code,deep_oracle,_=once(args.image,root,'deep-key','valid',scenario='deep-key') + checks.update(check_safety('deep-key',r,e,code,'valid')) + deep_fields=r.get('evidence_safety',{}).get('fields',[]) + deep_events=r.get('tool_result_evidence',{}).get('events',[]) + checks['deep-key:ordinary_execution_unchanged']=( + deep_legacy_code==0 and bool(deep_legacy_oracle) and deep_legacy_oracle==deep_oracle + and any(DEEP_ECHO in str(value) for value in deep_oracle) + ) + checks['deep-key:input_omitted_before_sink']=( + bool(deep_events) + and all('input' not in event for event in deep_events) + and any(f.get('field')=='input' and f.get('event') is not None + and f.get('state')=='omitted' + and f.get('reason')=='unsupported_representation' + for f in deep_fields) + ) + checks['deep-key:output_omitted_before_sink']=( + bool(deep_events) + and all('output' not in event for event in deep_events) + and any(f.get('field')=='output' and f.get('event') is not None + and f.get('state')=='omitted' + and f.get('reason')=='unsupported_representation' + for f in deep_fields) + ) + sink=S.encode(r).decode() + checks['deep-key:encoded_key_absent']=DEEP_KEY not in sink + checks['deep-key:serialized_echo_absent']=DEEP_ECHO not in sink and DEEP_ECHO_FILE not in sink + representation_inventory=inventory('valid') + representation_inventory['values']=[ESCAPED] + representation_policy=S.Policy(representation_inventory) + checks['deep-key:no_recoverable_representation_in_sink']=( + not representation_policy.matches(sink) + and not representation_policy.unsupported_recoverable(sink) + ) + (out/'deep-key.json').write_bytes(S.encode(r)+b'\n') + + for policy_kind in ('missing','incomplete','future'): + for scenario in ('success','failure','timeout'): + name=policy_kind+'-'+scenario + r,e,code,_,_=once(args.image,root,name,policy_kind,scenario) + checks.update(check_safety(name,r,e,code,policy_kind)) + if scenario=='timeout':checks[name+':product_timeout_preserved']=r.get('timed_out') is True and r.get('exit_code')==124 + (out/(name+'.json')).write_bytes(S.encode(r)+b'\n') + for scenario in ('failure','oversize'): + r,e,code,_,_=once(args.image,root,scenario,'valid',scenario) + checks.update(check_safety(scenario,r,e,code,'valid')) + fields=r.get('evidence_safety',{}).get('fields',[]) + if scenario=='failure':checks['actual_error_protected']=any(f['field']=='error' and f['state']=='redacted' for f in fields) + else:checks['safe_then_size_omission']=any(f['field']=='output' and f['state']=='omitted' and f['reason']=='size_limit' for f in fields) + (out/(scenario+'.json')).write_bytes(S.encode(r)+b'\n') + + # First discriminate the disposable lifecycle itself without the RSP + # envelope. All values in this fixture are synthetic, so preserving the + # fixed transport error in the artifact is safe and helps distinguish + # OpenCode bootstrap failures from safety-admission failures. + raw,e_raw,raw_code,_,raw_requests=once( + args.image,root,'disposable-lifecycle','missing',legacy=True, + disposable=True,ambient_traps=True + ) + checks['disposable-lifecycle:bootstrap_succeeds']=raw_code==0 + checks['disposable-lifecycle:runtime_state_attested']=( + raw.get('runtime_state',{}).get('profile')=='disposable' and + raw.get('runtime_state',{}).get('database_source')=='runtime-bootstrap' and + raw.get('runtime_state',{}).get('migration_count')==48) + checks['disposable-lifecycle:provider_after_bootstrap']=raw_requests>0 + (out/'disposable-lifecycle.json').write_text(json.dumps(raw,indent=2)+'\n') + + # DB handoff: real invoke, fresh runtime-owned DB, ambient host auth/DB traps present. + r,e,code,oracle,requests=once( + args.image,root,'disposable-valid','valid',disposable=True,ambient_traps=True + ) + checks.update(check_safety('disposable-valid',r,e,code,'valid')) + state=r.get('runtime_state',{}) + state_disp=next((x for x in r.get('evidence_safety',{}).get('fields',[]) + if x.get('event') is None and x.get('field')=='runtime_state'),{}) + checks['disposable-valid:runtime_bootstrap_attested']=( + state.get('schema')=='opencode-eval-runner/runtime-state/v1' and + state.get('profile')=='disposable' and state.get('database_source')=='runtime-bootstrap' and + state.get('database_created') is True and state.get('database_seed_present') is False and + state.get('auth_source')=='none' and state.get('session_rows_before_inference')==0 and + state.get('credential_rows_before_inference')==0 and state.get('migration_count')==48 and + state.get('first_migration')=='20260127222353_familiar_lady_ursula' and + state.get('last_migration')=='20260923013825_project_time_active' and + state_disp.get('state')=='exact') + checks['disposable-valid:ambient_state_not_read']=( + 'AMBIENT-AUTH-MUST-NOT-BE-READ' not in json.dumps(r) and + 'AMBIENT-SESSION-MUST-NOT-BE-READ' not in json.dumps(r) and requests > 0) + (out/'disposable-valid.json').write_bytes(S.encode(r)+b'\n') + + # Synthetic-only diagnostic of the same expected-plugin lifecycle + # without the RSP envelope. This preserves the fixed fixture error in + # CI evidence so bootstrap/preflight failures can be distinguished + # without exposing real credentials. + raw_plugin,_,raw_plugin_code,_,raw_plugin_requests=once( + args.image,root,'disposable-expected-plugin-lifecycle','missing', + legacy=True,disposable=True,ambient_traps=True,expected_plugin=True + ) + checks['disposable-expected-plugin-lifecycle:success']=raw_plugin_code==0 + checks['disposable-expected-plugin-lifecycle:provider_after_preflight']=raw_plugin_requests>0 + (out/'disposable-expected-plugin-lifecycle.json').write_text(json.dumps(raw_plugin,indent=2)+'\n') + + # Expected-plugin activation uses a Session API, but it must run in + # separate temporary state. The production DB is re-attested after + # preflight and must still report zero pre-inference Sessions. + r,e,code,oracle,requests=once( + args.image,root,'disposable-expected-plugin','valid', + disposable=True,ambient_traps=True,expected_plugin=True + ) + checks.update(check_safety('disposable-expected-plugin',r,e,code,'valid')) + plugin_state=r.get('runtime_state',{}) + checks['disposable-expected-plugin:production_db_still_empty']=( + plugin_state.get('profile')=='disposable' and + plugin_state.get('session_rows_before_inference')==0 and + plugin_state.get('credential_rows_before_inference')==0 and + requests > 0) + checks['disposable-expected-plugin:preflight_active']=( + raw_plugin.get('plugin_preflight',{}).get('expected')=='loom' and + raw_plugin.get('plugin_preflight',{}).get('plugin',{}).get('state',{}).get('status')=='active') + checks['disposable-expected-plugin:safe_result_omits_opaque_preflight']=( + 'plugin_preflight' not in r) + (out/'disposable-expected-plugin.json').write_bytes(S.encode(r)+b'\n') + + # Missing safety policy and incompatible explicit DB selection fail before provider inference. + r,e,code,_,requests=once(args.image,root,'disposable-missing','missing',disposable=True,ambient_traps=True) + checks['disposable-missing:pre_inference_fail']=code!=0 and requests==0 and not r.get('evidence_safety',{}).get('inventory_complete',True) + checks['disposable-missing:no_payload']=all(k not in r for k in ('text','tools','actions','session_id')) + (out/'disposable-missing.json').write_bytes(S.encode(r)+b'\n') + + r,e,code,_,requests=once(args.image,root,'disposable-explicit-db','valid',disposable=True,database_override=True) + checks['disposable-explicit-db:pre_inference_fail']=code!=0 and requests==0 + checks['disposable-explicit-db:no_payload']=all(k not in r for k in ('text','tools','actions','session_id')) + (out/'disposable-explicit-db.json').write_bytes(S.encode(r)+b'\n') + + summary={'schema':'rsp-image-proof/v1','image':args.image,'checks':checks,'passed':all(checks.values()), + 'source_commit':subprocess.check_output(['git','rev-parse','HEAD'],cwd=ROOT,text=True).strip(), + 'real_provider_inference':False,'full_capture_accepted':False,'loom_composition':'not_run'} + (out/'summary.json').write_text(json.dumps(summary,indent=2)+'\n');print(json.dumps(summary,indent=2)) + return 0 if summary['passed'] else 1 + +if __name__=='__main__':raise SystemExit(main()) diff --git a/tests/integration/run_podman_preflight_probe.py b/tests/integration/run_podman_preflight_probe.py new file mode 100755 index 0000000..897eb88 --- /dev/null +++ b/tests/integration/run_podman_preflight_probe.py @@ -0,0 +1,91 @@ +#!/usr/bin/env python3 +"""Exercise the real Podman image-preflight path without provider inference.""" +from __future__ import annotations + +import argparse +import json +from pathlib import Path +import subprocess +import sys + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT)) + +from runner import safe_invoke + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--image", required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + args.output.mkdir(parents=True, exist_ok=True) + + version = subprocess.run( + ["podman", "--version"], capture_output=True, text=True, check=False, timeout=30 + ) + if version.returncode: + raise SystemExit("podman unavailable") + + requested = args.image + command = ["podman", "run", requested] + loaded = safe_invoke.resolved_image(command) + + inspected = subprocess.run( + ["podman", "image", "inspect", requested], + capture_output=True, text=True, check=False, timeout=120, + ) + if inspected.returncode: + raise SystemExit("podman inspect failed after resolved_image") + info = json.loads(inspected.stdout)[0] + raw_id = info.get("Id") + canonical, execution_ref = safe_invoke.canonical_image_config_id(raw_id) + + exact = subprocess.run( + ["podman", "image", "inspect", execution_ref], + capture_output=True, text=True, check=False, timeout=120, + ) + if exact.returncode: + raise SystemExit("podman cannot inspect exact execution config") + exact_info = json.loads(exact.stdout)[0] + exact_canonical, _ = safe_invoke.canonical_image_config_id(exact_info.get("Id")) + + checks = { + "podman_available": version.returncode == 0, + "requested_immutable_digest": "@sha256:" in requested, + "canonical_image_config": loaded.get("image_config") == canonical, + "canonical_has_sha256_prefix": canonical.startswith("sha256:") and len(canonical) == 71, + "execution_ref_is_exact_inspected_id": command[-1] == execution_ref == raw_id, + "execution_ref_is_content_addressed": execution_ref != requested, + "exact_ref_reinspects_same_config": exact_canonical == canonical, + "source_revision_is_pinned": isinstance(loaded.get("image_source_revision"), str) + and len(loaded["image_source_revision"]) == 40, + "initializer_hash_bound": isinstance(loaded.get("image_package_init_sha256"), str) + and len(loaded["image_package_init_sha256"]) == 64, + "module_hash_bound": isinstance(loaded.get("image_policy_module_sha256"), str) + and len(loaded["image_policy_module_sha256"]) == 64, + "invoke_hash_bound": isinstance(loaded.get("image_invoke_sha256"), str) + and len(loaded["image_invoke_sha256"]) == 64, + } + + report = { + "schema": "opencode-eval-runner/podman-preflight-proof/v1", + "podman_version": version.stdout.strip(), + "image": requested, + "raw_image_id": raw_id, + "canonical_image_config": canonical, + "execution_ref": execution_ref, + "loaded": loaded, + "checks": checks, + "passed": all(checks.values()), + "provider_inference": False, + } + (args.output / "podman-preflight.json").write_text( + json.dumps(report, indent=2, sort_keys=True) + "\n" + ) + print(json.dumps(report, indent=2, sort_keys=True)) + return 0 if report["passed"] else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/integration/test_protected_connection.py b/tests/integration/test_protected_connection.py new file mode 100644 index 0000000..0707f0c --- /dev/null +++ b/tests/integration/test_protected_connection.py @@ -0,0 +1,221 @@ +"""Actual Docker/runtime -> private channel -> host importer acceptance probe. + +Fault injection edits NEW bytes received from actual executions on the host. It +never invents producer records or changes historical captures. Target attacks and +host transport faults are reported separately, not conflated as the same actor. +""" +import argparse +import base64 +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT)) +from runner import protected + +SECRET = "FIXTURE-SECRET-NOT-A-CREDENTIAL-8675309" +ATTACK = {f"{path}_{operation}_denied": True for path in ("capture", "runtime_root", "runtime_input", "host") + for operation in ("read", "write", "delete")} +SPOOF = {"observed_execution": {"evidence_eligible": True, "run_id": "invented-run"}, "actor": {"agent": "fabricated"}} +PROGRAM = ''' +if (typeof fetch !== "undefined") throw new Error("script network must be absent"); +const pair = await Promise.all([tools.isolated.echo({tag:"identical"}), tools.isolated.echo({tag:"identical"})]); +if (pair[0] !== "CALL-1" || pair[1] !== "CALL-2") throw new Error("bad pair"); +if (await tools.isolated.denied({}) !== '{"ok":false,"error":"denied"}') throw new Error("bad denial"); +const obj = await tools.isolated.denied_object({}); +if (obj.ok !== false || obj.error !== "denied") throw new Error("bad object"); +if (await tools.isolated.null({}) !== null) throw new Error("bad null"); +let caught = false; +try { await tools.isolated.throws({}); } catch (e) { caught = String(e.message).includes("THROW-RAW"); } +if (!caught) throw new Error("bad throw"); +const attack = await tools.isolated.attack({}); +if (Object.values(attack).some(v => v !== true)) throw new Error("capture reachable by tool"); +await tools.isolated.spoof({}); +return "script-output-only"; +''' +REDACTION_PROGRAM = ''' +await tools.isolated.secret({}); +await tools.isolated.encoded({}); +await tools.isolated.structured({}); +await tools.isolated.unknown({}); +await tools.isolated.large({}); +return "script-output-only"; +''' + + +def main(): + cli = argparse.ArgumentParser() + cli.add_argument("--image", required=True) + cli.add_argument("--tool-image", required=True) + cli.add_argument("--output", required=True) + options = cli.parse_args() + out = Path(options.output).resolve() + out.mkdir(parents=True, exist_ok=True) + checks = {} + originals = {} + with tempfile.TemporaryDirectory(prefix="protected-proof-") as tmp: + root = Path(tmp) + names = ("echo", "denied", "denied_object", "null", "throws", "attack", "spoof", "redirect", "secret", "encoded", "structured", "unknown", "large") + tools = [{"name": name, "input": {"type": "object", "properties": {"tag": {"type": "string"}}, "additionalProperties": False}} for name in names] + policy = {"version": 1, "secrets": [SECRET], "allowed_values": [ + {}, {"tag": "identical"}, "CALL-1", "CALL-2", '{"ok":false,"error":"denied"}', + {"ok": False, "error": "denied"}, None, {"name": "Error", "message": "THROW-RAW"}, ATTACK, SPOOF, + {"name": "Error", "message": "isolated_tool_transport_failed"}, + "script-output-only", "[REDACTED]", {"password": "[REDACTED]"}, "ø" * 9000, + ]} + for name, value in (("tools.json", tools), ("policy.json", policy)): + (root / name).write_text(json.dumps(value)) + (root / "program.js").write_text(PROGRAM) + args = protected.parser().parse_args(["--image", options.image, "--tool-image", options.tool_image, + "--program-file", str(root / "program.js"), "--tools-file", str(root / "tools.json"), + "--policy-file", str(root / "policy.json"), "--output", str(out / "good.json")]) + + def receive(name, transform=lambda raw: raw): + def process(raw, run_id, policy_id): + originals[name] = raw + (out / (name + ".original.jsonl")).write_bytes(raw) + modified = transform(raw) + if modified != raw: + (out / (name + ".received.jsonl")).write_bytes(modified) + return modified + return process + + tool_oracle = [] + def target_probe(target, runtime_state, target_state): + command = ["docker", "exec", "--user", "1000:1000", target, "python3", "-c", + "import urllib.request; print(urllib.request.urlopen('http://127.0.0.1:8080/oracle').read().decode())"] + tool_oracle.extend(json.loads(subprocess.check_output(command, text=True))) + (out / "tool-oracle.json").write_text(json.dumps(tool_oracle, indent=2) + "\n") + checks["separate_mounts_and_process_namespace"] = not target_state["Mounts"] and not target_state["HostConfig"].get("PidMode") and runtime_state["State"]["Running"] is False + checks["target_no_signer_or_credentials"] = not any("TOKEN" in e or "SECRET" in e for e in target_state["Config"].get("Env", [])) + code = protected.invoke(args, _test_receive=receive("good"), _test_target_probe=target_probe) + good = json.loads(Path(args.output).read_text()) + projection = good["observed_execution"] + records = projection["records"] + checks["legitimate_connection_eligible"] = code == 0 and projection["evidence_eligible"] is True + checks["separate_receipt_binds_capture"] = good.get("collection_receipt") == { + "sha256": hashlib.sha256(originals["good"]).hexdigest(), "bytes": len(originals["good"])} + checks["receipt_profile_explicit"] = projection.get("version") == 5 and projection.get("launch_id") == good["launch_id"] and projection["coverage"].get("accounting_complete") is True + echo = [r for r in records if r["tool"] == "isolated_echo"] + checks["distinct_correlated_reverse_completion"] = (len(echo) == 2 and echo[0]["invocation_id"] != echo[1]["invocation_id"] + and echo[0]["input"]["value"] == echo[1]["input"]["value"] == {"tag": "identical"} + and echo[0]["result"]["value"] == "CALL-1" and echo[1]["result"]["value"] == "CALL-2" + and echo[1]["terminal_sequence"] < echo[0]["terminal_sequence"]) + bytool = {r["tool"]: r for r in records} + checks["target_cannot_access_capture"] = bytool.get("isolated_attack", {}).get("result", {}).get("value") == ATTACK + checks["denial_string_preserved"] = bytool.get("isolated_denied", {}).get("result", {}).get("value") == '{"ok":false,"error":"denied"}' + checks["denial_object_preserved"] = bytool.get("isolated_denied_object", {}).get("result", {}).get("value") == {"ok": False, "error": "denied"} + checks["null_present"] = "value" in bytool.get("isolated_null", {}).get("result", {}) and bytool["isolated_null"]["result"]["value"] is None + checks["caught_error_preserved"] = bytool.get("isolated_throws", {}).get("error", {}).get("value") == {"name": "Error", "message": "THROW-RAW"} + checks["identity_preserved"] = bool(records) and all(r["actor"]["message_id"] == r["parent"]["message_id"] + and r["actor"]["session_id"] == r["parent"]["session_id"] and r["runtime_call_id"] == "protected-execute" for r in records) + checks["actual_image_and_input_binding"] = good.get("launched_images_verified") is True and good["launch"]["runtime_image"] == options.image and good["launch"]["program_sha256"] == hashlib.sha256(PROGRAM.encode()).hexdigest() + checks["spoof_is_only_a_returned_value"] = bytool.get("isolated_spoof", {}).get("result", {}).get("value") == SPOOF and all(r["actor"]["agent"] != "fabricated" for r in records) + starts = {e["context"]["invocation_id"]: e for e in tool_oracle if e["phase"] == "start"} + ends = {e["context"]["invocation_id"]: e for e in tool_oracle if e["phase"] in ("returned", "threw")} + checks["records_match_independent_tool_oracle"] = len(starts) == len(ends) == len(records) == 8 and all( + r["invocation_id"] in starts and r["input"]["value"] == starts[r["invocation_id"]]["input"] + and r["actor"] == {k: starts[r["invocation_id"]]["context"][k] for k in ("agent", "session_id", "message_id")} + and r["outcome"] == ends[r["invocation_id"]]["phase"] + and (r.get("result", {}).get("value") == ends[r["invocation_id"]]["value"] if r["outcome"] == "returned" + else r["error"]["value"]["message"] == ends[r["invocation_id"]]["value"]) + for r in records) + checks["scope_exclusions_explicit"] = projection["full_handoff_eligible"] is False and projection["coverage"]["native"] == "unsupported" and projection["coverage"]["delegated_sessions"] == "unsupported" + checks["no_script_output_as_inner_result"] = bool(records) and all(r.get("result", {}).get("value") != "script-output-only" for r in records) + + # Also exercise the user-facing entrypoint, not only a library call. + args.output = str(out / "off.json") + command = [sys.executable, str(ROOT / "bin/opencode-eval-runner"), "observe", "--image", options.image, + "--tool-image", options.tool_image, "--program-file", args.program_file, + "--tools-file", args.tools_file, "--policy-file", args.policy_file, "--output", args.output, "--no-observe"] + off = subprocess.run(command, capture_output=True, text=True, timeout=120) + off_result = json.loads(Path(args.output).read_text()) + checks["observer_off_same_program_result"] = off.returncode == 0 and off_result["script_output"] == good["script_output"] == {"state": "available", "value": "script-output-only"} + checks["observer_off_no_evidence"] = off_result["observed_execution"]["evidence_eligible"] is False + command.remove("--no-observe") + command[command.index("--output") + 1] = str(out / "public-cli.json") + public = subprocess.run(command, capture_output=True, text=True, timeout=120) + public_result = json.loads((out / "public-cli.json").read_text()) + checks["public_cli_legitimate_capture"] = public.returncode == 0 and public_result["observed_execution"]["evidence_eligible"] is True and len(public_result["observed_execution"]["records"]) == 8 + + def replace_result(raw): + return raw.replace(b'CALL-1', b'FORGED', 1) + def remove_middle(raw): + lines = raw.splitlines(keepends=True) + return b"".join(lines[:2] + lines[3:]) + def reorder(raw): + lines = raw.splitlines(keepends=True) + lines[2], lines[3] = lines[3], lines[2] + return b"".join(lines) + def rehash_deleted_terminal(raw): + frames = [json.loads(line) for line in raw.splitlines()] + index = next(i for i, f in enumerate(frames) if f.get("observation", {}).get("kind") == "call_end") + del frames[index] + for i, f in enumerate(frames): f["seq"] = i + prefix = b"".join((json.dumps(f) + "\n").encode() for f in frames[:-1]) + frames[-1]["sha256"] = hashlib.sha256(prefix).hexdigest() + frames[-1]["event_count"] = len(frames) - 2 + return prefix + (json.dumps(frames[-1]) + "\n").encode() + def rehash_forged_result(raw): + lines = replace_result(raw).splitlines(keepends=True) + footer = json.loads(lines[-1]) + footer["sha256"] = hashlib.sha256(b"".join(lines[:-1])).hexdigest() + return b"".join(lines[:-1]) + (json.dumps(footer) + "\n").encode() + transforms = {"resealed_forgery": rehash_forged_result, "rehashed_deletion": rehash_deleted_terminal, "forged": replace_result, "replay": lambda raw: originals["good"], "deleted_record": remove_middle, + "deleted_footer": lambda raw: b"".join(raw.splitlines(keepends=True)[:-1]), + "partial": lambda raw: raw[:-5], "reordered": reorder, + "appended": lambda raw: raw + b'{"kind":"observation"}\n'} + for name, transform in transforms.items(): + args.output = str(out / (name + ".json")) + code = protected.invoke(args, _test_receive=receive(name, transform)) + result = json.loads(Path(args.output).read_text()) + checks[name + "_real_stream_rejected"] = (out / (name + ".received.jsonl")).exists() and code != 0 and result["observed_execution"]["evidence_eligible"] is False and result["observed_execution"]["records"] == [] + print(name, code, result["observed_execution"]["issues"], flush=True) + + args.output = str(out / "io-failure.json") + code = protected.invoke(args, _test_prepare=lambda capture: capture.chmod(0o555)) + io_failed = json.loads(Path(args.output).read_text()) + checks["io_failure_not_evidence"] = code != 0 and io_failed["observed_execution"]["evidence_eligible"] is False + checks["io_failure_did_not_change_script"] = io_failed.get("script_output") == {"state": "available", "value": "script-output-only"} + + # Verify the CLI's positive path too, including execution of the verified snapshot. + args.output = str(out / "snapshot.json") + def change_original(_capture): + (root / "program.js").write_text('throw new Error("mutated source must not execute");') + snapshot_code = protected.invoke(args, _test_prepare=change_original) + snapshot_result = json.loads(Path(args.output).read_text()) + checks["executes_the_same_input_snapshot"] = snapshot_code == 0 and snapshot_result["script_output"] == {"state": "available", "value": "script-output-only"} + (root / "program.js").write_text('try { await tools.isolated.redirect({}); } catch (e) {} return "script-output-only";') + args.output = str(out / "redirect.json") + redirect_code = protected.invoke(args, _test_receive=receive("redirect")) + redirect_result = json.loads(Path(args.output).read_text()) + redirect_records = redirect_result["observed_execution"]["records"] + checks["tool_cannot_redirect_bridge_to_local_services"] = (redirect_code == 0 and len(redirect_records) == 1 + and redirect_records[0].get("error", {}).get("value") == {"name": "Error", "message": "isolated_tool_transport_failed"}) + (root / "program.js").write_text(REDACTION_PROGRAM) + args.output = str(out / "redaction.json") + code = protected.invoke(args, _test_receive=receive("redaction")) + redaction = json.loads(Path(args.output).read_text()) + states = {r["tool"]: r.get("result", {}).get("state") for r in redaction["observed_execution"]["records"]} + checks["redaction_before_persistence"] = SECRET.encode() not in originals.get("redaction", b"") and base64.b64encode(SECRET.encode()) not in originals.get("redaction", b"") and b"ANOTHER-UNKNOWN-FIXTURE-SECRET" not in originals.get("redaction", b"") + checks["redaction_and_omission_non_evidence"] = code == 4 and states == {"isolated_secret": "redacted", "isolated_encoded": "redacted", "isolated_structured": "redacted", "isolated_unknown": "omitted", "isolated_large": "truncated"} + checks["unsafe_unknown_not_persisted"] = b"UNKNOWN-NONALLOWLISTED-TEXT" not in originals.get("redaction", b"") + checks["redaction_did_not_change_script"] = redaction["script_output"] == {"state": "available", "value": "script-output-only"} + summary = {"profile": protected.PROFILE, "runtime_image": options.image, "tool_image": options.tool_image, + "checks": checks, "passed": all(checks.values()), "full_handoff_accepted": False, + "independent_review": "not_performed", "native": "open", "delegated_sessions": "open", + "in_process_untrusted_plugins": "unsupported", + "fault_injection": "host-side receive shim modifies actual new runtime streams; target attacks are separate", + "checkout": subprocess.check_output(["git", "rev-parse", "HEAD"], cwd=ROOT, text=True).strip()} + (out / "summary.json").write_text(json.dumps(summary, indent=2) + "\n") + print(json.dumps(summary, indent=2), flush=True) + return 0 if summary["passed"] else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_capture_probe.py b/tests/test_capture_probe.py new file mode 100644 index 0000000..4aaa3a6 --- /dev/null +++ b/tests/test_capture_probe.py @@ -0,0 +1,112 @@ +"""CI exit-policy tests; synthetic summaries are not runtime capture evidence.""" +import copy +import importlib.util +from pathlib import Path +import unittest + +SPEC = importlib.util.spec_from_file_location( + "capture_probe_exit_policy", + Path(__file__).parent / "integration" / "run_capture_probe.py", +) +probe = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(probe) + +# The expected old-image diagnostic shape is independent of the policy's set. +BASELINE = { + "kind": "capture-boundary-probe", + "version": 1, + "image": "ghcr.io/bateau84/opencode-eval-runner@sha256:68ef7322c75aede0e8cc76d0e3531e8b82dd417bbb5e5100264a89eab7fe8627", + "diagnostics_passed": True, + "handoff_acceptance": "BLOCKED", + "independent_code_approval": False, + "checks": { + "pinned_image": True, + "runtime_2_0_18": True, + "scenarios_completed": True, + "identical_calls_overlap_and_finish_reversed": True, + "tool_behavior_unchanged": True, + "missing_producer_not_evidence": True, + "forged_sidecar_rejected": True, + "diagnostic_hooks_toggle": True, + "shared_id_counterexample": True, + "caught_throw_has_no_after_hook": True, + "early_after_is_not_final_return": True, + "no_capture_eligible": True, + }, +} + + +class CaptureProbeExitTests(unittest.TestCase): + def check_modes(self, summary, expected): + for negative in (False, True): + with self.subTest(negative_control=negative): + self.assertEqual( + probe.probe_exit_code(summary, expect_unsupported_baseline=negative), expected, + ) + + def test_default_still_rejects_blocked_capture(self): + self.assertEqual(probe.probe_exit_code(BASELINE), 4) + + def test_explicit_negative_control_passes_without_acceptance(self): + before = copy.deepcopy(BASELINE) + self.assertEqual(probe.probe_exit_code(BASELINE, expect_unsupported_baseline=True), 0) + self.assertEqual(BASELINE, before) + self.assertEqual(BASELINE["handoff_acceptance"], "BLOCKED") + self.assertIs(BASELINE["independent_code_approval"], False) + + def test_every_diagnostic_failure_still_fails_both_modes(self): + for name in BASELINE["checks"]: + with self.subTest(check=name): + summary = copy.deepcopy(BASELINE) + summary["checks"][name] = False + self.check_modes(summary, 1) + + def test_missing_check_is_not_a_smaller_passing_suite(self): + for name in BASELINE["checks"]: + with self.subTest(check=name): + summary = copy.deepcopy(BASELINE) + del summary["checks"][name] + self.check_modes(summary, 1) + + def test_empty_extra_and_malformed_checks_fail(self): + for checks in ({}, None, [], True, {**BASELINE["checks"], "unexpected": True}): + with self.subTest(checks=checks): + self.check_modes({**BASELINE, "checks": checks}, 1) + + def test_truthy_nonboolean_diagnostics_are_not_success(self): + for value in (1, "true", [True], None): + with self.subTest(value=value): + summary = copy.deepcopy(BASELINE) + summary["checks"]["no_capture_eligible"] = value + self.check_modes(summary, 1) + self.check_modes({**BASELINE, "diagnostics_passed": value}, 1) + + def test_new_or_mutable_image_cannot_use_old_image_control(self): + for image in ("ghcr.io/bateau84/opencode-eval-runner:opencode-edge", + "ghcr.io/bateau84/opencode-eval-runner@sha256:" + "a" * 64, None): + with self.subTest(image=image): + self.check_modes({**BASELINE, "image": image}, 1) + + def test_unexpected_acceptance_or_approval_fails(self): + for key, value in (("handoff_acceptance", "PASS"), ("handoff_acceptance", None), + ("independent_code_approval", True), ("independent_code_approval", 0), + ("diagnostics_passed", False)): + with self.subTest(key=key, value=value): + self.check_modes({**BASELINE, key: value}, 1) + + def test_wrong_summary_kind_or_version_fails(self): + for key, value in (("kind", "different-probe"), ("version", 2), ("version", True), + ("version", "1")): + with self.subTest(key=key, value=value): + self.check_modes({**BASELINE, key: value}, 1) + + def test_missing_required_summary_fields_fail(self): + for key in BASELINE: + with self.subTest(key=key): + summary = copy.deepcopy(BASELINE) + del summary[key] + self.check_modes(summary, 1) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_cli.py b/tests/test_cli.py index 2c779e8..fea9414 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -21,6 +21,8 @@ sanitize_database_seed, resolve_engine, host_environment_for_transport, + opencode_state_profile, + resolve_database_seed, ) @@ -86,6 +88,72 @@ def test_sanitized_database_keeps_credentials_and_clears_runtime_data(self): self.assertEqual(sessions, 0) self.assertEqual(migrations, 1) + def test_disposable_profile_ignores_ambient_default_auth_models_and_provider_tokens(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + workspace, input_dir, output_dir = root/'workspace', root/'input', root/'output' + for path in (workspace, input_dir, output_dir): path.mkdir() + data = root/'data'/'opencode'; data.mkdir(parents=True) + cache = root/'cache'/'opencode'; cache.mkdir(parents=True) + (data/'auth.json').write_text('{"token":"AMBIENT"}') + (cache/'models.json').write_text('{}') + config = root/'synthetic.json'; config.write_text('{}') + args = argparse.Namespace( + engine='podman', image='test-image', workspace=str(workspace), workspace_mode='ro', + output=str(root/'result.json'), transport='opencode', model='fixture/mock', agent='general', + skill=None, expected_plugin=None, reasoning=None, timeout_seconds=30, env=[], auth=None, + config=str(config), models_catalog=None, database=None, config_root=None, mount=[], + network=None, opencode_state_profile='disposable', + ) + env={'XDG_DATA_HOME':str(root/'data'),'XDG_CACHE_HOME':str(root/'cache'),'OPENAI_API_KEY':'AMBIENT-TOKEN'} + with patch('runner.cli.shutil.which', return_value='/usr/bin/podman'): + command, _ = build_container_command(args,input_dir,output_dir,host_env=env,database_seed=None) + rendered=' '.join(command) + self.assertIn(str(config.resolve())+':/seed/opencode.json:ro', rendered) + self.assertNotIn(str((data/'auth.json').resolve()), rendered) + self.assertNotIn(str((cache/'models.json').resolve()), rendered) + self.assertNotIn('OPENAI_API_KEY', rendered) + self.assertIn('EVAL_OPENCODE_STATE_PROFILE=disposable', rendered) + self.assertIn('EVAL_OPENCODE_DATABASE_SOURCE=runtime-bootstrap', rendered) + self.assertIn('EVAL_OPENCODE_AUTH_SOURCE=none', rendered) + + def test_disposable_profile_rejects_database_and_ambient_seed_overrides(self): + args=argparse.Namespace(transport='opencode', database='/tmp/seed.db', opencode_state_profile='disposable') + with self.assertRaisesRegex(RunnerError, '--database is incompatible'): + opencode_state_profile(args,{}) + args.database=None + with self.assertRaisesRegex(RunnerError, 'rejects implicit runner seed overrides'): + opencode_state_profile(args,{'OPENCODE_EVAL_RUNNER_DB':'/private.db'}) + + def test_disposable_profile_rejects_runner_owned_state_env_overrides(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + workspace, input_dir, output_dir = root/'workspace', root/'input', root/'output' + for path in (workspace, input_dir, output_dir): + path.mkdir() + config = root/'synthetic.json' + config.write_text('{}') + for name in ( + 'EVAL_OPENCODE_STATE_PROFILE', + 'EVAL_OPENCODE_AUTH_SOURCE', + 'EVAL_OPENCODE_DATABASE_SOURCE', + ): + args = argparse.Namespace( + engine='podman', image='test-image', workspace=str(workspace), workspace_mode='ro', + output=str(root/'result.json'), transport='opencode', model='fixture/mock', agent='general', + skill=None, expected_plugin=None, reasoning=None, timeout_seconds=30, env=[name], auth=None, + config=str(config), models_catalog=None, database=None, config_root=None, mount=[], + network=None, opencode_state_profile='disposable', + ) + host_env={name:'default' if name == 'EVAL_OPENCODE_STATE_PROFILE' else 'forged'} + with self.assertRaisesRegex(RunnerError, 'reserves runner state environment names'): + build_container_command(args,input_dir,output_dir,host_env=host_env,database_seed=None) + + def test_disposable_database_resolution_never_reads_host_default(self): + args=argparse.Namespace(transport='opencode', database=None, opencode_state_profile='disposable') + with tempfile.TemporaryDirectory() as tmp, patch('runner.cli.default_database_path', side_effect=AssertionError('host default read')): + self.assertIsNone(resolve_database_seed(args, Path(tmp)/'seed.db', {})) + def test_extra_mount_defaults_to_read_only_and_accepts_rw(self): with tempfile.TemporaryDirectory() as tmp: source = Path(tmp) / "node_modules" diff --git a/tests/test_evidence_safety.py b/tests/test_evidence_safety.py new file mode 100644 index 0000000..cf05e9b --- /dev/null +++ b/tests/test_evidence_safety.py @@ -0,0 +1,634 @@ +"""Provider-free RSP tests: actual container emitter + host first-write path.""" +from __future__ import annotations + +import copy +import importlib.util +import io +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest +from unittest.mock import patch + +from container import evidence_safety as S +from runner import cli, safe_invoke + +ROOT = Path(__file__).resolve().parents[1] +REV = 'a' * 40 +IMAGE = 'ghcr.io/test/runner@sha256:' + 'b' * 64 + + +def inventory(values=(), complete=True): + # Exact private_policy() fields inspected in Loom 1a85b1a/run-evals.py. + return {'schema': S.POLICY, 'policy_version': S.VERSION, 'complete': complete, + 'sources': {name: 'complete' if name == 'env' else 'not_selected' for name in S.SOURCE_NAMES}, + 'values': list(values) if complete else []} + + +def tool(output=None, args=None, *, status='completed', toolname='demo'): + state = {'status': status, 'input': {} if args is None else args, 'metadata': {'truncated': False}} + state['error' if status == 'error' else 'output'] = output + return {'type': 'tool_use', 'timestamp': 1, 'sessionID': 'session', + 'part': {'type': 'tool', 'tool': toolname, 'callID': 'call', 'state': state}} + + +def raw_result(events=None, text='follow workflow context text low', code=0): + events = [tool('safe')] if events is None else events + return {'schema': 'opencode-eval-runner/v1', 'transport': 'opencode', 'reasoning_source': 'explicit', + 'reasoning': 'low', 'model': 'fixture/mock', 'agent': 'worker', 'skill': None, + 'exit_code': code, 'session_id': 'session', 'text': text, 'tools': ['demo'], + 'actions': [{'tool': 'demo', 'args': {'choice': 'text'}}], 'skills_loaded': [], + 'stdout': '\n'.join(json.dumps(e) for e in events), 'stderr': 'opaque', + 'tool_result_evidence': {'schema': 'runner-unclipped-events/internal-v1', 'events': events}} + + +def disposition(result, field, event=None): + return next(f for f in result['evidence_safety']['fields'] if f['field'] == field and f['event'] == event) + + +def invoke_module(): + spec = importlib.util.spec_from_file_location('rsp_test_container', ROOT / 'container/invoke.py') + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +class PolicyTests(unittest.TestCase): + def test_exact_loom_wire_shape_and_short_credentials(self): + p = S.Policy(inventory(['0', '1', 'text', 'low'])) + self.assertTrue(p.valid and p.complete) + self.assertEqual(set(p.values), {'0', '1', 'text', 'low'}) + + def test_disposable_runtime_state_has_strict_exact_shape(self): + state={ + 'schema':S.RUNTIME_STATE,'profile':'disposable','database_source':'runtime-bootstrap', + 'database_created':True,'database_seed_present':False,'auth_source':'none', + 'session_rows_before_inference':0,'credential_rows_before_inference':0,'migration_count':48, + 'first_migration':'20260127222353_familiar_lady_ursula', + 'last_migration':'20260923013825_project_time_active', + } + policy=S.Policy(inventory(['0','1','text','low'])) + r=S.project_result({**raw_result(),'runtime_state':state},policy) + self.assertEqual(r['runtime_state'],state) + self.assertEqual(disposition(r,'runtime_state')['state'],'exact') + run_id='a'*64; revision='b'*40; binding=b'c'*32 + r['evidence_safety_ack']=S.receipt(policy,run_id,revision,binding) + self.assertEqual( + S.validate_reply(S.encode(r),policy,run_id,revision,binding)['runtime_state'], + state, + ) + for bad in ( + {**state,'database_seed_present':True}, + {**state,'migration_count':47}, + {**state,'first_migration':'wrong'}, + {**state,'last_migration':'wrong'}, + ): + r=S.project_result({**raw_result(),'runtime_state':bad},S.Policy(inventory())) + self.assertNotIn('runtime_state',r) + self.assertEqual(disposition(r,'runtime_state')['state'],'omitted') + + def test_missing_incompatible_and_incomplete_policy_never_complete(self): + for data in (None, {}, {'values': ['secret']}, {**inventory(), 'complete': 1}, + {**inventory(), 'policy_version': 'future'}, {**inventory(), 'schema': 'future'}, + {**inventory(), 'sources': {}}, {**inventory(), 'extra': 'secret'}, + {**inventory(), 'sources': {'env': 'incomplete'}}, inventory(complete=False)): + self.assertFalse(S.Policy(data).complete) + + def test_policy_limit_and_utf8_strictness(self): + self.assertFalse(S.Policy(inventory(['x' * S.POLICY_LIMIT])).valid) + self.assertFalse(S.Policy(inventory(['\ud800'])).valid) + + def test_no_marker_rescan(self): + p = S.Policy(inventory(['secret', 'REDACTED'])) + value, changed = p.payload('secret REDACTED') + self.assertTrue(changed) + self.assertEqual(value, '***REDACTED*** ***REDACTED***') + + def test_existing_clip_is_omission_even_without_credentials(self): + p = S.Projection(S.Policy(inventory())) + self.assertIs(p.field('text', 'otherwise safe', clipped=True), S.MISSING) + self.assertEqual(p.fields[0]['reason'], 'upstream_clipped') + + +class ProjectionTests(unittest.TestCase): + def test_protocol_constants_survive_short_secrets_without_payload_waiver(self): + r = S.project_result(raw_result([tool({'choice': 'text', 'priority': 'low', 'yes': True}, {'choice': 'text'})], code=1), + S.Policy(inventory(['0', '1', 'text', 'low']))) + self.assertEqual(r['exit_code'], 1) + self.assertEqual(r['schema'], S.RESULT) + self.assertIn('text', r) + self.assertEqual(disposition(r, 'text')['state'], 'redacted') + e = r['tool_result_evidence']['events'][0] + self.assertEqual(e['sequence'], 1) + self.assertEqual(e['status'], 'completed') + self.assertEqual(e['output']['choice'], '***REDACTED***') + self.assertEqual(e['output']['priority'], '***REDACTED***') + self.assertIs(e['output']['yes'], True) + self.assertEqual(disposition(r, 'output', 0)['state'], 'redacted') + self.assertNotIn('reasoning', r) # identity low is omitted, never renamed. + self.assertNotIn('actions', r) # lost selector fidelity is not absence. + self.assertEqual(disposition(r, 'actions')['state'], 'omitted') + self.assertEqual(S.strict_loads(S.encode(r)), r) + + def test_public_only_inventory_retains_exact_bytes_and_type(self): + raw = raw_result([tool(None)], text='workflow follow context text low') + r = S.project_result(raw, S.Policy(inventory(['unrelated-real-token']))) + self.assertEqual(r['text'], raw['text']) + self.assertEqual(disposition(r, 'text')['state'], 'exact') + self.assertIsNone(r['tool_result_evidence']['events'][0]['output']) + self.assertEqual(disposition(r, 'output', 0)['state'], 'exact') + + def test_payload_numbers_omitted_not_coerced_to_markers(self): + for value in (0, 1): + r = S.project_result(raw_result([tool(value)]), S.Policy(inventory(['0', '1']))) + self.assertNotIn('output', r['tool_result_evidence']['events'][0]) + self.assertEqual(disposition(r, 'output', 0)['reason'], 'credential_match') + + def test_payload_key_collision_omits_enclosing_field(self): + for value in ({'text': 'public', 'low': 'public'}, {'apiKey': 'secret'}, {'token': 'secret'}): + r = S.project_result(raw_result([tool(value)]), S.Policy(inventory(['text', 'low']))) + self.assertNotIn('output', r['tool_result_evidence']['events'][0]) + self.assertEqual(disposition(r, 'output', 0)['reason'], 'sensitive_key') + + def test_status_lookalike_is_protocol_but_identity_is_not(self): + r = S.project_result(raw_result([tool('completed', toolname='completed')]), S.Policy(inventory(['completed']))) + e = r['tool_result_evidence']['events'][0] + self.assertEqual(e['status'], 'completed') + self.assertNotIn('tool', e) + self.assertEqual(disposition(r, 'tool', 0)['state'], 'omitted') + + def test_redaction_before_size_decision(self): + secret = 'z' * 20_000 + r = S.project_result(raw_result([tool(secret)]), S.Policy(inventory([secret]))) + self.assertEqual(r['tool_result_evidence']['events'][0]['output'], '***REDACTED***') + self.assertEqual(disposition(r, 'output', 0)['state'], 'redacted') + self.assertNotIn(secret, S.encode(r).decode()) + + def test_oversize_is_omitted_without_prefix_hash_or_suffix(self): + r = S.project_result(raw_result([tool('safe-' * 2000)]), S.Policy(inventory())) + self.assertNotIn('output', r['tool_result_evidence']['events'][0]) + d = disposition(r, 'output', 0) + self.assertEqual(d, {'event': 0, 'field': 'output', 'state': 'omitted', 'reason': 'size_limit', 'stage': 'runner'}) + + def test_escaped_values_raw_through_three_layers(self): + secret = 'tok-"line\n\\ending' + val = secret + for _ in range(4): + r = S.project_result(raw_result([tool(val)]), S.Policy(inventory([secret]))) + self.assertEqual(disposition(r, 'output', 0)['state'], 'redacted') + val = json.dumps(val)[1:-1] + + def test_fourth_json_escape_layer_omits_values_and_mapping_keys(self): + secret = 'tok-"line\n\\ending' + deep = secret + for _ in range(4): + deep = json.dumps(deep)[1:-1] + for payload in (deep, {deep: 'public'}, {'nested': {deep: 'public'}}): + with self.subTest(kind=type(payload).__name__): + r = S.project_result(raw_result([tool(payload)]), S.Policy(inventory([secret]))) + self.assertNotIn('output', r['tool_result_evidence']['events'][0]) + self.assertEqual(disposition(r, 'output', 0)['reason'], 'unsupported_representation') + self.assertNotIn(deep, S.encode(r).decode()) + + def test_representation_changing_echoes_omit_instead_of_chasing_depth(self): + secret = 'tok-"line\n\\ending' + deep = secret + for _ in range(S.SUPPORTED_JSON_ESCAPE_LAYERS + 1): + deep = json.dumps(deep, ensure_ascii=False)[1:-1] + + payload = {deep: 'public'} + echo = json.dumps(payload, ensure_ascii=False, separators=(',', ':')) + for layer in range(4): + with self.subTest(layer=layer): + r = S.project_result( + raw_result([tool(echo, args=payload)]), + S.Policy(inventory([secret])), + ) + event = r['tool_result_evidence']['events'][0] + self.assertNotIn('input', event) + self.assertNotIn('output', event) + self.assertEqual(disposition(r, 'input', 0)['reason'], 'unsupported_representation') + self.assertEqual(disposition(r, 'output', 0)['reason'], 'unsupported_representation') + blob = S.encode(r).decode() + self.assertNotIn(deep, blob) + self.assertNotIn(echo, blob) + self.assertFalse(S.Policy(inventory([secret])).unsupported_recoverable(blob)) + echo = json.dumps(echo, ensure_ascii=False) + + def test_deep_recoverable_identity_is_omitted(self): + secret = 'identity-"line\n\\ending' + deep = secret + for _ in range(S.SUPPORTED_JSON_ESCAPE_LAYERS + 2): + deep = json.dumps(deep, ensure_ascii=False)[1:-1] + p = S.Projection(S.Policy(inventory([secret]))) + self.assertIs(p.field('session_id', deep, role='identity'), S.MISSING) + self.assertEqual(p.fields[0]['reason'], 'unsupported_representation') + def test_bad_representations_and_unknown_schema_omit(self): + cycle = {}; cycle['self'] = cycle + for value in (cycle, float('nan'), {'a': float('inf')}, b'no'): + p = S.Projection(S.Policy(inventory())) + self.assertIs(p.field('text', value), S.MISSING) + self.assertEqual(p.fields[0]['reason'], 'unsupported_representation') + r = S.project_result({**raw_result(), 'unknown': 'private'}, S.Policy(inventory())) + self.assertNotIn('private', S.encode(r).decode()) + self.assertEqual(disposition(r, 'transport')['reason'], 'unsupported_schema') + + def test_incomplete_inventory_no_payload_slots(self): + r = S.project_result(raw_result(), S.Policy(inventory(complete=False))) + for name in ('text', 'actions', 'tools', 'model', 'reasoning', 'session_id'): + self.assertNotIn(name, r) + self.assertEqual(disposition(r, name)['state'], 'omitted') + self.assertFalse(r['evidence_safety']['inventory_complete']) + + def test_sources_and_values_never_exported_and_input_unchanged(self): + policy = inventory(['secret']) + raw = raw_result([tool('secret')]); original = copy.deepcopy(raw) + r = S.project_result(raw, S.Policy(policy)) + self.assertEqual(raw, original) + self.assertNotIn('sources', r) + self.assertNotIn('values', r) + self.assertFalse(r['evidence_safety']['coverage_complete']) + + +class ContainerBoundaryTests(unittest.TestCase): + def run_container(self, policy, events, *, timeout=False, failure=False): + m = invoke_module(); m.RSP_ACTIVE = True; m.RSP_POLICY = m.rsp_module().Policy(policy); m.RSP_RUN_ID = 'c' * 64 + stdout = '\n'.join(json.dumps(e) for e in events) + def execute(command, cwd, env, seconds): + if timeout: + raise subprocess.TimeoutExpired(command, seconds, stdout.encode(), b'PRIVATE-ERROR') + return subprocess.CompletedProcess(command, 1 if failure else 0, stdout, 'PRIVATE-ERROR') + with patch.object(m, 'prepare_opencode_env', return_value={}), patch.object(m, 'plugin_diagnostic', return_value={}), \ + patch.object(m, 'verify_expected_plugin', return_value={}), patch.object(m, 'run', side_effect=execute): + original = m.invoke_opencode('fixture/mock', 'worker', 'prompt', 5) + emitted = io.StringIO() + with patch.object(m.sys, 'stdout', emitted): + m.emit_result(original) + return original, json.loads(emitted.getvalue()) + + def test_success_and_failure_cross_real_preclip_emitter(self): + secret = 'long-' * 2000 + for failure in (False, True): + before, after = self.run_container(inventory([secret]), [tool(secret)], failure=failure) + self.assertEqual(before['tool_result_evidence']['events'][0]['part']['state']['output'], secret) + self.assertFalse(before['stdout_truncated']) + self.assertEqual(after['tool_result_evidence']['events'][0]['output'], '***REDACTED***') + self.assertNotIn('stdout', after) + self.assertNotIn('stderr', after) + self.assertEqual(after['exit_code'], int(failure)) + + def test_timeout_same_sink_keeps_real_error_outcome(self): + before, after = self.run_container(inventory(['low']), [tool('low', status='error')], timeout=True) + self.assertTrue(after['timed_out']) + self.assertEqual(after['exit_code'], 124) + self.assertEqual(after['tool_result_evidence']['events'][0]['error'], '***REDACTED***') + self.assertNotIn('PRIVATE-ERROR', json.dumps(after)) + + def test_absent_policy_omits_in_all_outcomes(self): + for kw in ({}, {'failure': True}, {'timeout': True}): + _, after = self.run_container(None, [tool('private')], **kw) + self.assertNotIn('private', json.dumps(after)) + self.assertFalse(after['evidence_safety_ack']['inventory_complete']) + + def test_raw_large_stream_is_not_clipped_before_projection(self): + m = invoke_module(); m.RSP_ACTIVE = True + text = 'x' * (m.STDOUT_CAPTURE_LIMIT + 1) + self.assertEqual(m.evidence_slice(text, 5), '') + m.RSP_ACTIVE = False + self.assertEqual(m.evidence_slice(text, 5), 'xxxxx') + + +class HostBoundaryTests(unittest.TestCase): + def reply(self, policy=None, nonce='c' * 64): + policy = S.Policy(inventory(['secret'])) if policy is None else policy + result = S.project_result(raw_result([tool('secret')]), policy) + result['evidence_safety_ack'] = S.receipt(policy, nonce, REV) + return policy, result + + def test_real_reply_accepted_missing_or_forged_ack_rejected(self): + p, r = self.reply() + self.assertEqual(S.validate_reply(S.encode(r), p, 'c' * 64, REV), r) + for name, value in (('run_id', 'd' * 64), ('module_sha256', 'f' * 64), ('stages', []), + ('policy_version', 'future'), ('inventory_complete', False)): + bad = copy.deepcopy(r); bad['evidence_safety_ack'][name] = value + with self.assertRaises(ValueError): + S.validate_reply(S.encode(bad), p, 'c' * 64, REV) + bad = dict(r); del bad['evidence_safety_ack'] + with self.assertRaises(ValueError): + S.validate_reply(S.encode(bad), p, 'c' * 64, REV) + + def test_incomplete_policy_keeps_only_fixed_event_status_exact(self): + policy = S.Policy(inventory(complete=False)) + run_id = 'c' * 64 + revision = REV + binding = b'd' * 32 + result = S.project_result(raw_result([tool('private-output')]), policy) + result['evidence_safety_ack'] = S.receipt(policy, run_id, revision, binding) + event_fields = [ + item for item in result['evidence_safety']['fields'] + if item.get('event') == 0 + ] + status = next(item for item in event_fields if item['field'] == 'status') + self.assertEqual(status['state'], 'exact') + self.assertTrue(all( + item['state'] == 'omitted' + for item in event_fields if item['field'] != 'status' + )) + admitted = S.validate_reply(S.encode(result), policy, run_id, revision, binding) + self.assertEqual(admitted['tool_result_evidence']['events'][0]['status'], 'completed') + self.assertNotIn('private-output', json.dumps(admitted)) + + def test_fallback_has_total_top_level_dispositions_and_preserves_outcome_metadata(self): + policy = S.Policy(inventory(complete=False)) + run_id = 'f' * 64 + revision = REV + binding = b'e' * 32 + result = S.fallback('size_limit', 'runner', policy) + result['exit_code'] = 124 + result['timed_out'] = True + result['evidence_safety_ack'] = S.receipt(policy, run_id, revision, binding) + top = { + item['field']: item + for item in result['evidence_safety']['fields'] + if item.get('event') is None + } + self.assertEqual(set(top), S.TOP_LEVEL_REQUIRED) + self.assertTrue(all(item['state'] == 'omitted' for item in top.values())) + admitted = S.validate_reply(S.encode(result), policy, run_id, revision, binding) + self.assertEqual(admitted['exit_code'], 124) + self.assertTrue(admitted['timed_out']) + + def test_every_required_top_level_field_has_disposition(self): + policy = S.Policy(inventory(['unrelated-secret'])) + run_id = 'c' * 64 + revision = REV + binding = b'd' * 32 + result = S.project_result(raw_result(), policy) + result['evidence_safety_ack'] = S.receipt(policy, run_id, revision, binding) + top = { + item['field'] for item in result['evidence_safety']['fields'] + if item.get('event') is None + } + self.assertTrue(S.TOP_LEVEL_REQUIRED <= top) + self.assertEqual(S.validate_reply(S.encode(result), policy, run_id, revision, binding), result) + + malformed = { + 'schema': S.RESULT, + 'evidence_safety_ack': S.receipt(policy, run_id, revision, binding), + 'evidence_safety': { + 'schema': S.SAFETY, + 'policy_version': S.VERSION, + 'inventory_complete': True, + 'coverage_complete': True, + 'fields': [], + 'loss_counts': {reason: 0 for reason in S.REASONS}, + }, + } + with self.assertRaises(ValueError): + S.validate_reply(S.encode(malformed), policy, run_id, revision, binding) + + def test_omitted_event_count_is_bound_to_event_dispositions(self): + policy = S.Policy(inventory(['unrelated-secret'])) + run_id = 'c' * 64 + revision = REV + binding = b'd' * 32 + malformed_event = { + 'type': 'tool_use', + 'timestamp': 1, + 'sessionID': 'session', + 'part': {'type': 'tool', 'tool': 'broken', 'state': []}, + } + result = S.project_result(raw_result([tool('safe-output'), malformed_event]), policy) + result['evidence_safety_ack'] = S.receipt(policy, run_id, revision, binding) + evidence = result['tool_result_evidence'] + self.assertEqual(evidence['observed_events'], 2) + self.assertEqual(evidence['omitted_events'], 1) + self.assertEqual( + [item for item in result['evidence_safety']['fields'] + if item.get('field') == 'event'], + [{'event': 1, 'field': 'event', 'state': 'omitted', + 'reason': 'unsupported_schema', 'stage': 'runner'}], + ) + self.assertEqual(S.validate_reply(S.encode(result), policy, run_id, revision, binding), result) + + no_reason = copy.deepcopy(result) + no_reason['evidence_safety']['fields'] = [ + item for item in no_reason['evidence_safety']['fields'] + if item.get('field') != 'event' + ] + no_reason['evidence_safety']['loss_counts']['unsupported_schema'] -= 1 + no_reason['evidence_safety']['coverage_complete'] = True + with self.assertRaises(ValueError): + S.validate_reply(S.encode(no_reason), policy, run_id, revision, binding) + + retained_omitted = copy.deepcopy(result) + event_item = next( + item for item in retained_omitted['evidence_safety']['fields'] + if item.get('field') == 'event' + ) + event_item['event'] = 0 + with self.assertRaises(ValueError): + S.validate_reply(S.encode(retained_omitted), policy, run_id, revision, binding) + + out_of_range = copy.deepcopy(result) + event_item = next( + item for item in out_of_range['evidence_safety']['fields'] + if item.get('field') == 'event' + ) + event_item['event'] = 9 + with self.assertRaises(ValueError): + S.validate_reply(S.encode(out_of_range), policy, run_id, revision, binding) + + def test_retained_event_requires_complete_field_dispositions(self): + policy = S.Policy(inventory(['unrelated-secret'])) + run_id = 'c' * 64 + revision = REV + binding = b'd' * 32 + result = S.project_result(raw_result([tool('safe-output')]), policy) + result['evidence_safety_ack'] = S.receipt(policy, run_id, revision, binding) + event_fields = { + item['field'] for item in result['evidence_safety']['fields'] + if item.get('event') == 0 + } + self.assertTrue({'tool','call_id','session_id','input','status','output'} <= event_fields) + self.assertEqual(S.validate_reply(S.encode(result), policy, run_id, revision, binding), result) + + for field in ('tool','call_id','session_id','input','status'): + bad = copy.deepcopy(result) + bad['evidence_safety']['fields'] = [ + item for item in bad['evidence_safety']['fields'] + if not (item.get('event') == 0 and item.get('field') == field) + ] + with self.assertRaises(ValueError): + S.validate_reply(S.encode(bad), policy, run_id, revision, binding) + + bad = copy.deepcopy(result) + bad['evidence_safety']['fields'] = [ + item for item in bad['evidence_safety']['fields'] + if not (item.get('event') == 0 and item.get('field') in {'output','error'}) + ] + with self.assertRaises(ValueError): + S.validate_reply(S.encode(bad), policy, run_id, revision, binding) + + def test_host_rejects_forged_exact_representation_changing_echo(self): + secret = 'host-"line\n\\ending' + deep = secret + for _ in range(S.SUPPORTED_JSON_ESCAPE_LAYERS + 1): + deep = json.dumps(deep, ensure_ascii=False)[1:-1] + echo = json.dumps({deep: 'public'}, ensure_ascii=False, separators=(',', ':')) + + policy = S.Policy(inventory([secret])) + run_id = 'c' * 64 + binding = b'd' * 32 + result = S.project_result(raw_result([tool('public-output')]), policy) + result['tool_result_evidence']['events'][0]['output'] = echo + result['evidence_safety_ack'] = S.receipt(policy, run_id, REV, binding) + + with self.assertRaises(ValueError): + S.validate_reply(S.encode(result), policy, run_id, REV, binding) + def test_omitted_value_smuggling_and_duplicate_disposition_rejected(self): + p, r = self.reply() + bad = copy.deepcopy(r); bad['stdout'] = 'secret' + with self.assertRaises(ValueError): S.validate_reply(S.encode(bad), p, 'c' * 64, REV) + bad = copy.deepcopy(r); bad['evidence_safety']['fields'].append(bad['evidence_safety']['fields'][0]) + with self.assertRaises(ValueError): S.validate_reply(S.encode(bad), p, 'c' * 64, REV) + + def test_atomic_file_and_print_have_only_projected_data(self): + _, result = self.reply() + with tempfile.TemporaryDirectory() as tmp, patch('sys.stdout', new_callable=io.StringIO) as output: + path = Path(tmp) / 'out.json' + safe_invoke.write_projection(path, result, True) + self.assertEqual(json.loads(path.read_text()), result) + self.assertEqual(json.loads(output.getvalue()), result) + self.assertNotIn('secret', path.read_text()) + self.assertEqual(path.stat().st_mode & 0o777, 0o600) + self.assertEqual(list(Path(tmp).glob('.safe*')), []) + + def test_resolved_image_binds_all_executable_container_sources(self): + root = Path(safe_invoke.__file__).resolve().parents[1] + init_sha = __import__('hashlib').sha256((root/'container/__init__.py').read_bytes()).hexdigest() + invoke_sha = __import__('hashlib').sha256((root/'container/invoke.py').read_bytes()).hexdigest() + canonical = 'sha256:' + 'c' * 64 + labels = { + 'io.opencode-eval.evidence-safety': S.CONSUMER, + 'io.opencode-eval.evidence-safety-init': init_sha, + 'io.opencode-eval.evidence-safety-module': S.module_sha(), + 'io.opencode-eval.evidence-safety-invoke': invoke_sha, + 'org.opencontainers.image.revision': REV, + } + for engine, raw_id in (('docker', canonical), ('podman', 'c' * 64)): + with self.subTest(engine=engine): + info = [{'RepoDigests':[IMAGE], 'Id':raw_id, 'Config':{'Labels':dict(labels)}}] + command = [engine, 'run', IMAGE] + with patch.object( + safe_invoke.subprocess, 'run', + return_value=subprocess.CompletedProcess([], 0, json.dumps(info).encode(), b''), + ): + loaded = safe_invoke.resolved_image(command) + self.assertEqual(loaded['image_config'], canonical) + self.assertEqual(command[-1], raw_id) + self.assertEqual(loaded['image_package_init_sha256'], init_sha) + self.assertEqual(loaded['image_policy_module_sha256'], S.module_sha()) + self.assertEqual(loaded['image_invoke_sha256'], invoke_sha) + + for bad in ('', 'c'*63, 'c'*65, 'SHA256:'+'c'*64, 'sha256:'+'C'*64, 'md5:'+'c'*64): + with self.subTest(bad=bad[:20]): + info = [{'RepoDigests':[IMAGE], 'Id':bad, 'Config':{'Labels':dict(labels)}}] + with patch.object( + safe_invoke.subprocess, 'run', + return_value=subprocess.CompletedProcess([], 0, json.dumps(info).encode(), b''), + ), self.assertRaises(ValueError): + safe_invoke.resolved_image(['podman', 'run', IMAGE]) + + bad_labels = dict(labels) + bad_labels['io.opencode-eval.evidence-safety-init'] = '0' * 64 + info = [{'RepoDigests':[IMAGE], 'Id':canonical, 'Config':{'Labels':bad_labels}}] + with patch.object( + safe_invoke.subprocess, 'run', + return_value=subprocess.CompletedProcess([], 0, json.dumps(info).encode(), b''), + ), self.assertRaises(ValueError): + safe_invoke.resolved_image(['docker', 'run', IMAGE]) + + + def test_unacknowledged_or_timeout_transport_never_writes_raw_details(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp); prompt = root/'prompt'; prompt.write_text('execute') + policy_path = root/'private.json'; policy_path.write_text(json.dumps(inventory(['secret']))) + args = cli.parser().parse_args(['invoke', '--model', 'fixture/mock', '--prompt-file', str(prompt), + '--output', str(root/'result.json'), '--evidence-policy-file', str(policy_path), '--workspace', tmp, + '--image', IMAGE, '--engine', 'docker', '--print-result']) + for failure in (subprocess.CompletedProcess([], 0, b'{"text":"secret"}', b'secret'), + subprocess.TimeoutExpired('raw-secret-command', 1, b'secret', b'secret')): + with patch.object(cli, 'existing_seed', return_value=None), \ + patch.object(cli, 'resolve_engine', return_value='docker'), \ + patch.object(safe_invoke, 'resolved_image', return_value={'image_source_revision': REV}), \ + patch.object(safe_invoke.subprocess, 'run', side_effect=failure if isinstance(failure, Exception) else None, + return_value=failure), patch('sys.stdout', new_callable=io.StringIO) as output: + code = cli.invoke(args) + self.assertNotEqual(code, 0) + self.assertNotIn('secret', (root/'result.json').read_text()) + self.assertNotIn('secret', output.getvalue()) + + def test_disposable_profile_policy_must_match_explicit_seed_selection(self): + p=S.Policy(inventory()) + command=['docker','--volume','/synthetic/config:/seed/opencode.json:ro',IMAGE] + q=safe_invoke.audit_selected_inputs(p,command,{}, {'config'}, 'disposable') + self.assertFalse(q.complete) # policy claimed config not_selected + data=inventory(); data['sources']['config']='complete' + q=safe_invoke.audit_selected_inputs(S.Policy(data),command,{}, {'config'}, 'disposable') + self.assertTrue(q.complete) + self.assertEqual(q.private['sources']['credential_seed'],'not_selected') + + def test_disposable_profile_accepts_explicit_config_root_when_inventory_matches(self): + data = inventory() + data['sources']['config_root'] = 'complete' + command = [ + 'docker', '--volume', '/synthetic/root:/seed/opencode-config:ro', IMAGE + ] + q = safe_invoke.audit_selected_inputs( + S.Policy(data), command, {}, {'config_root'}, 'disposable' + ) + self.assertTrue(q.complete) + + legacy = safe_invoke.audit_selected_inputs( + S.Policy(data), command, {}, {'config_root'}, 'default' + ) + self.assertFalse(legacy.complete) + + def test_forwarded_sensitive_env_must_be_declared_selected_and_present(self): + command = ['docker', '--env', 'OPENAI_API_KEY', IMAGE] + host_env = {'OPENAI_API_KEY': 'synthetic-forwarded-secret'} + + not_selected = inventory(['synthetic-forwarded-secret']) + not_selected['sources']['env'] = 'not_selected' + q = safe_invoke.audit_selected_inputs( + S.Policy(not_selected), command, host_env, set(), 'disposable' + ) + self.assertFalse(q.complete) + + complete = inventory(['synthetic-forwarded-secret']) + complete['sources']['env'] = 'complete' + q = safe_invoke.audit_selected_inputs( + S.Policy(complete), command, host_env, set(), 'disposable' + ) + self.assertTrue(q.complete) + + missing_value = inventory(['different-secret']) + missing_value['sources']['env'] = 'complete' + q = safe_invoke.audit_selected_inputs( + S.Policy(missing_value), command, host_env, set(), 'disposable' + ) + self.assertFalse(q.complete) + + def test_selected_default_source_not_selected_downgrades_policy(self): + p = S.Policy(inventory()) + q = safe_invoke.audit_selected_inputs(p, ['docker', '--volume', '/private/auth:/seed/auth.json:ro', IMAGE], {}) + self.assertFalse(q.complete) + self.assertTrue(p.complete) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_evidence_safety_failures.py b/tests/test_evidence_safety_failures.py new file mode 100644 index 0000000..adc3198 --- /dev/null +++ b/tests/test_evidence_safety_failures.py @@ -0,0 +1,137 @@ +"""Negative controls for policy acknowledgement and real first output boundaries.""" +import copy +import io +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest +from unittest.mock import patch + +from container import evidence_safety as S +from runner import cli, safe_invoke +from test_evidence_safety import inventory, raw_result, tool, disposition, invoke_module, REV + +class SafetyFailureTests(unittest.TestCase): + def test_bound_receipt_rejects_swapped_complete_policy(self): + a, b = S.Policy(inventory(['AAA'])), S.Policy(inventory(['BBB'])) + key = b'private-test-binding-key' + r = S.project_result(raw_result(), a) + r['evidence_safety_ack'] = S.receipt(a, 'a'*64, REV, key) + with self.assertRaises(ValueError): + S.validate_reply(S.encode(r), b, 'a'*64, REV, key) + self.assertNotIn(key.decode(), S.encode(r).decode()) + self.assertNotIn('AAA', S.encode(r).decode()) + + def test_full_schema_dispositions_required_and_exact_secret_rejected(self): + p=S.Policy(inventory(['UNIQUE-SECRET'])) + r=S.project_result(raw_result([tool('plain')]),p) + r['evidence_safety_ack']=S.receipt(p,'c'*64,REV) + for alter in ('exact-secret','delete','omit-value','event-metadata'): + bad=copy.deepcopy(r) + if alter=='exact-secret': bad['text']='UNIQUE-SECRET' + elif alter=='delete': bad['evidence_safety']['fields']=[x for x in bad['evidence_safety']['fields'] if x['field']!='text'] + elif alter=='omit-value': bad['stdout']='UNIQUE-SECRET' + else: bad['tool_result_evidence']['events'][0]['metadata']='UNIQUE-SECRET' + with self.subTest(alter=alter), self.assertRaises(ValueError): + S.validate_reply(S.encode(bad),p,'c'*64,REV) + + def test_real_clipping_declaration_drops_output_before_any_replacement(self): + ev=tool('VISIBLE-CREDENTIAL-PREFIX') + ev['part']['state']['metadata']={'metadata':{'truncated':True}} + r=S.project_result(raw_result([ev]),S.Policy(inventory())) + self.assertNotIn('output', r['tool_result_evidence']['events'][0]) + self.assertEqual(disposition(r,'output',0)['reason'],'upstream_clipped') + self.assertNotIn('VISIBLE-CREDENTIAL-PREFIX',S.encode(r).decode()) + + def test_native_call_id_uses_id_not_part_id(self): + ev=tool('plain') + ev['part']['id']=ev['part'].pop('callID') + ev['part']['partID']='part-not-invocation' + r=S.project_result(raw_result([ev]),S.Policy(inventory())) + self.assertEqual(r['tool_result_evidence']['events'][0]['call_id'],'call') + + def test_opaque_json_string_omitted_but_declared_json_string_keeps_structure(self): + p=S.Projection(S.Policy(inventory(['0','1','text']))) + raw=' {"label": "text", "ok": true} ' + self.assertIs(p.field('output',raw),S.MISSING) + safe=p.field('nested_json',raw,role='json_string') + self.assertEqual(json.loads(safe),{'label':'***REDACTED***','ok':True}) + self.assertEqual(p.fields[-1]['state'],'redacted') + p=S.Projection(S.Policy(inventory())) + self.assertEqual(p.field('nested_json',raw,role='json_string'),raw) + self.assertEqual(p.fields[-1]['state'],'exact') + for raw in ('{"a":1,"a":2}', '{"a":NaN}', '{bad', '[[['): + self.assertIs(p.field('nested_json',raw,role='json_string'),S.MISSING) + + def test_deeper_escape_not_a_false_exact(self): + secret='quoted-"\n\\secret' + value=secret + for _ in range(4): value=json.dumps(value)[1:-1] + p=S.Projection(S.Policy(inventory([secret]))) + self.assertIs(p.field('output',value),S.MISSING) + + def test_request_rejects_bad_versions_encoding_and_duplicates(self): + request={'schema':S.REQUEST,'run_id':'a'*64,'policy':inventory(['0','1']),'binding_key':'b'*64} + p,nonce,key=S.read_request(io.BytesIO(S.encode(request))) + self.assertTrue(p.complete);self.assertEqual(nonce,'a'*64);self.assertEqual(key,bytes.fromhex('b'*64)) + for raw in (b'', S.encode(request).replace(S.REQUEST.encode(),b'future'), + json.dumps(request).encode('utf-16'),b'{"a":1,"a":2}',b'x'*(S.WIRE_LIMIT+2)): + self.assertFalse(S.read_request(io.BytesIO(raw))[0].complete) + + def test_actual_attach_timeout_still_removes_owned_container(self): + calls=[] + def engine(command, **kwargs): + calls.append(command) + if command[1] == 'start': + raise subprocess.TimeoutExpired(command, 1, b'private partial output', b'private error') + return subprocess.CompletedProcess(command, 0, b'', b'') + with patch.object(safe_invoke.subprocess,'run',side_effect=engine): + with self.assertRaises(subprocess.TimeoutExpired): + safe_invoke.execute_container(['docker','run','--rm','-i','immutable-image'], b'private-policy', 1, 'a'*64) + self.assertEqual([c[1] for c in calls], ['create','start','rm']) + self.assertIn('--volumes', calls[-1]) + self.assertNotIn('private-policy', repr(calls)) + + def test_invalid_cli_diagnostics_do_not_echo_arguments(self): + command=[sys.executable, str(Path(__file__).resolve().parents[1]/'bin/opencode-eval-runner'), + 'invoke','--require-evidence-safety','--model','NOT-A-CREDENTIAL', '--unknown=PRIVATE'] + proc=subprocess.run(command,capture_output=True,text=True) + self.assertNotEqual(proc.returncode,0) + self.assertNotIn('PRIVATE',proc.stdout+proc.stderr) + self.assertNotIn('NOT-A-CREDENTIAL',proc.stdout+proc.stderr) + + def test_cli_abbreviated_safety_flags_do_not_echo_failed_arguments(self): + for flag in ('--evidence-policy-fil', '--evidence-p', '--require-evidence-safet', '--require-e', '--e'): + for assigned in (False, True): + tail = ([flag + '=PRIVATE-POLICY'] if assigned else [flag, 'PRIVATE-POLICY']) + command = [sys.executable, str(Path(__file__).resolve().parents[1]/'bin/opencode-eval-runner'), + 'invoke', '--model', 'PRIVATE-MODEL', *tail, '--unknown=PRIVATE-VALUE'] + proc = subprocess.run(command, capture_output=True, text=True) + with self.subTest(flag=flag, assigned=assigned): + self.assertNotEqual(proc.returncode, 0) + self.assertNotIn('PRIVATE', proc.stdout + proc.stderr) + self.assertIn('invalid safety invocation', proc.stderr) + + def test_missing_policy_emit_success_failure_timeout_without_echo(self): + # Real emitter and main path, not a substitute result serializer. + for kw in ({'exit_code':0},{'exit_code':1},{'exit_code':124,'timed_out':True}): + m=invoke_module() + with tempfile.TemporaryDirectory() as tmp: + prompt=Path(tmp)/'prompt';prompt.write_text('script') + env={'EVAL_EVIDENCE_SAFETY':'1','EVAL_MODEL':'fixture/mock', 'EVAL_PROMPT_FILE':str(prompt)} + class In: + buffer=io.BytesIO(b'') + out=io.StringIO() + with patch.dict(os.environ,env,clear=True),patch.object(m.sys,'stdin',In()),patch.object(m.sys,'stdout',out),\ + patch.object(m,'invoke_opencode',return_value={**raw_result([tool('PRIVATE')]),**kw}): + result=m.main() + parsed=json.loads(out.getvalue()) + self.assertEqual(result,0) # Transport envelope emitted; product exit is in the result. + self.assertNotIn('PRIVATE',out.getvalue()) + self.assertEqual(parsed['exit_code'],kw['exit_code']) + self.assertFalse(parsed['evidence_safety_ack']['policy_valid']) + +if __name__=='__main__': unittest.main() diff --git a/tests/test_invoke.py b/tests/test_invoke.py index b1495e7..1ecfbf7 100644 --- a/tests/test_invoke.py +++ b/tests/test_invoke.py @@ -3,6 +3,7 @@ from pathlib import Path import json import os +import sqlite3 import subprocess import tempfile import unittest @@ -17,6 +18,7 @@ invoke_copilot, invoke_opencode, resolve_opencode_reasoning, + disposable_runtime_state, verify_expected_plugin, ) @@ -139,6 +141,68 @@ def test_session_export_preserves_tool_inputs_as_actions(self): ], ) + def test_disposable_runtime_state_attests_runtime_owned_bootstrap(self): + with tempfile.TemporaryDirectory() as tmp: + root=Path(tmp); data=root/'data'/'opencode'; data.mkdir(parents=True); (root/'config').mkdir() + env={ + 'EVAL_OPENCODE_STATE_PROFILE':'disposable', + 'EVAL_OPENCODE_DATABASE_SOURCE':'runtime-bootstrap', + 'EVAL_OPENCODE_AUTH_SOURCE':'none', + 'XDG_DATA_HOME':str(root/'data'), + } + migrations=['20260127222353_familiar_lady_ursula']+[f'202602{i:08d}_fixture' for i in range(1,47)]+['20260923013825_project_time_active'] + def bootstrap(command,cwd,actual_env,timeout): + self.assertEqual(command[:3], ['opencode','session','list']) + db=data/'opencode.db' + with sqlite3.connect(db) as conn: + conn.execute('CREATE TABLE session_v2 (id TEXT)') + conn.execute('CREATE TABLE credential (id TEXT)') + conn.execute('CREATE TABLE migration (id TEXT PRIMARY KEY, time_completed INTEGER NOT NULL)') + conn.executemany('INSERT INTO migration VALUES (?,1)',[(m,) for m in migrations]) + return subprocess.CompletedProcess(command,0,'[]','') + with patch('container.invoke.run', side_effect=bootstrap): + state=disposable_runtime_state(env,30) + self.assertEqual(state['migration_count'],48) + self.assertEqual(state['database_source'],'runtime-bootstrap') + self.assertFalse(state['database_seed_present']) + self.assertEqual(state['session_rows_before_inference'],0) + self.assertEqual(state['credential_rows_before_inference'],0) + + def test_disposable_invoke_rechecks_production_state_after_plugin_preflight(self): + first = { + "schema": "opencode-eval-runner/runtime-state/v1", + "profile": "disposable", + "database_source": "runtime-bootstrap", + "database_created": True, + "database_seed_present": False, + "auth_source": "none", + "session_rows_before_inference": 0, + "credential_rows_before_inference": 0, + "migration_count": 48, + "first_migration": "20260127222353_familiar_lady_ursula", + "last_migration": "20260923013825_project_time_active", + } + + class Result: + returncode = 0 + stdout = "" + stderr = "" + + with patch("container.invoke.prepare_opencode_env", return_value={"EVAL_OPENCODE_STATE_PROFILE": "disposable"}), patch( + "container.invoke.disposable_runtime_state", return_value=first + ) as bootstrap, patch( + "container.invoke.verify_expected_plugin", return_value={"expected":"loom"} + ) as preflight, patch( + "container.invoke.attest_disposable_runtime_state", return_value=first + ) as attest, patch( + "container.invoke.run", return_value=Result() + ): + invoke_opencode("openai/gpt-5.5", "general", "prompt", 30) + + bootstrap.assert_called_once() + preflight.assert_called_once() + attest.assert_called_once() + def test_v2_invocation_does_not_use_models_refresh_preflight(self): class Result: returncode = 0 @@ -541,6 +605,57 @@ def fake_request(base_url, path, *, method="GET", payload=None, timeout=5.0, aut ], ) + def test_disposable_expected_plugin_preflight_uses_separate_state(self): + server = object() + seen = {} + + def fake_start(env, timeout): + seen.update(env) + return server, "http://127.0.0.1:1234", "Basic test-auth" + + def fake_request(base_url, path, *, method="GET", payload=None, timeout=5.0, authorization=None): + if path == "/api/session": + return {"data": {"id": "ses_test"}} + if path == "/api/session/ses_test/prompt": + return {"data": {"id": "msg_test"}} + return {"data": [{"id": "loom", "state": {"status": "active"}}]} + + with tempfile.TemporaryDirectory() as tmp, patch( + "container.invoke._start_preflight_server", side_effect=fake_start + ), patch( + "container.invoke._standalone_json_request", side_effect=fake_request + ), patch( + "container.invoke._stop_preflight_server", return_value="" + ): + root = Path(tmp) + config = root / "config" / "opencode" + plugins = config / "plugins" + plugins.mkdir(parents=True) + (plugins / "loom.ts").write_text("export default {}\n", encoding="utf-8") + production_data = root / "data" + production_cache = root / "cache" + production_state = root / "state" + production_home = root / "home" + for path in (production_data, production_cache, production_state, production_home): + path.mkdir(parents=True) + env = { + "OPENCODE_CONFIG_DIR": str(config), + "EVAL_OPENCODE_STATE_PROFILE": "disposable", + "XDG_DATA_HOME": str(production_data), + "XDG_CACHE_HOME": str(production_cache), + "XDG_STATE_HOME": str(production_state), + "XDG_CONFIG_HOME": str(root / "config"), + "HOME": str(production_home), + } + verify_expected_plugin(env, "general", "openai/gpt-5.5", "loom", 30) + isolated_data = Path(seen["XDG_DATA_HOME"]) + isolated_root = isolated_data.parent + self.assertNotEqual(isolated_data, production_data) + self.assertNotEqual(Path(seen["XDG_STATE_HOME"]), production_state) + self.assertNotEqual(Path(seen["XDG_CACHE_HOME"]), production_cache) + self.assertEqual(seen["OPENCODE_CONFIG_DIR"], str(config)) + self.assertFalse(isolated_root.exists()) + def test_expected_plugin_preflight_uses_bounded_server_timeout(self): seen = [] diff --git a/tests/test_observer.py b/tests/test_observer.py new file mode 100644 index 0000000..9c68d5a --- /dev/null +++ b/tests/test_observer.py @@ -0,0 +1,389 @@ +import base64 +import copy +import importlib.util +import json +import os +from pathlib import Path +import subprocess +import tempfile +import unittest +from unittest.mock import patch + +from runner import cli +from runner.observer import ObserverCapture, load_capture, MAX_CAPTURE_BYTES + +ROOT = Path(__file__).resolve().parents[1] +FIXTURE = ROOT / "tests" / "fixtures" / "observer_producer.py" +spec = importlib.util.spec_from_file_location("observer_test_producer", FIXTURE) +producer = importlib.util.module_from_spec(spec) +spec.loader.exec_module(producer) +field, start, end, events, encode = (getattr(producer, name) for name in ("field", "start", "end", "events", "encode")) +KEY = b"fixture-only-key-not-a-real-secret-0123456789" +RUN = "a" * 64 + + +class CaptureTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.path = Path(self.temp.name) / "records.jsonl" + + def load(self, items=None, **kwargs): + self.path.write_bytes(encode(items or events(start(), end()), KEY, RUN)) + return load_capture(self.path, key=KEY, run_id=RUN, **kwargs) + + def assert_rejected(self, result, issue=None): + self.assertIs(result["evidence_eligible"], False) + self.assertTrue(all(record["evidence_eligible"] is False for record in result["records"])) + if issue: + self.assertIn(issue, result["issues"]) + + def test_actual_results_and_exact_identity(self): + result = self.load(events(start(mode="native", value={"n": 42}), end(value="raw-sentinel"))) + self.assertTrue(result["evidence_eligible"]) + self.assertEqual(result["records"], [{ + "invocation_id": "inner-1", "tool": "sentinel", + "actor": {"agent": "worker", "session_id": "child-session"}, + "parent": None, "mode": "native", "input": {"state": "available", "value": {"n": 42}}, + "start_sequence": 1, "terminal_sequence": 2, "outcome": "returned", + "evidence_eligible": True, "result": {"state": "available", "value": "raw-sentinel"}, + }]) + + def test_same_input_shared_parent_reverse_completion_order(self): + result = self.load(events(start("a", value={}), start("b", value={}), + end("b", value="second"), end("a", value="first"))) + self.assertTrue(result["evidence_eligible"]) + self.assertEqual([r["result"]["value"] for r in result["records"]], ["first", "second"]) + self.assertEqual([r["terminal_sequence"] for r in result["records"]], [4, 3]) + self.assertEqual(result["records"][0]["parent"], result["records"][1]["parent"]) + + def test_domain_denial_is_returned_not_thrown(self): + denial = '{"ok":false,"error":"denied"}' + result = self.load(events(start("a"), end("a", value=denial), start("b"), + end("b", value={"name": "Error", "message": "boom"}, outcome="threw"))) + self.assertTrue(result["evidence_eligible"]) + self.assertEqual(result["records"][0]["outcome"], "returned") + self.assertEqual(result["records"][0]["result"]["value"], denial) + self.assertEqual(result["records"][1]["outcome"], "threw") + self.assertNotIn("result", result["records"][1]) + + def test_explicit_null_return_is_preserved(self): + self.assertIsNone(self.load(events(start(), end(value=None)))["records"][0]["result"]["value"]) + + def test_missing_capture_and_empty_capture(self): + self.assert_rejected(load_capture(self.path, key=KEY, run_id=RUN), "missing_capture") + self.path.touch() + self.assert_rejected(load_capture(self.path, key=KEY, run_id=RUN), "empty_capture") + + def test_missing_footer(self): + result = self.load(events(start(), end())[:-1]) + self.assert_rejected(result, "missing_capture_end") + self.assertEqual(result["records"][0]["result"]["value"], "actual-sentinel") + self.assertIsNone(result["coverage"]["omitted_records"]) + + def test_missing_terminal_with_authenticated_footer(self): + result = self.load(events(start())) + self.assert_rejected(result, "missing_terminals") + self.assertEqual(result["coverage"]["missing_terminals"], 1) + self.assertEqual(result["records"][0]["outcome"], "missing") + + def test_completeness_and_supported_boundary_flags(self): + for name, value, issue in (("omitted_records", 1, "omitted_records"), + ("truncated", True, "truncated_capture"), + ("unsupported", ["inner_correlation"], "unsupported_capture")): + with self.subTest(name=name): + items = events(start(), end()) + items[-1][name] = value + self.assert_rejected(self.load(items), issue) + items = events(start(), end()) + items[0]["boundary"] = "before_other_plugins_transform_result" + self.assert_rejected(self.load(items), "unsupported_capture_boundary") + + def test_parent_call_id_cannot_be_reused_as_unique_invocation(self): + self.assert_rejected(self.load(events(start(), start(), end())), "duplicate_invocation") + + def test_orphan_and_duplicate_terminals(self): + for items in (events(end()), events(start(), end(), end())): + self.assert_rejected(self.load(items), "ambiguous_terminal") + + def test_code_mode_requires_parent(self): + record = start() + record["parent"] = None + self.assert_rejected(self.load(events(record, end())), "missing_parent") + + def test_malformed_identity_never_falls_back_to_requested_actor(self): + for actor in ({"agent": "worker"}, {"agent": None, "session_id": "s"}): + record = start() + record["actor"] = actor + self.assert_rejected(self.load(events(record, end()))) + + def test_wrong_key_forged_payload_and_replay(self): + raw = encode(events(start(), end()), KEY, RUN) + self.path.write_bytes(raw) + self.assert_rejected(load_capture(self.path, key=b"wrong" * 8, run_id=RUN), "authentication_failed") + self.assert_rejected(load_capture(self.path, key=KEY, run_id="b" * 64), "authentication_failed") + lines = raw.splitlines() + frame = json.loads(lines[2]) + payload = json.loads(base64.b64decode(frame["payload"])) + payload["result"]["value"] = "FORGED PASS" + frame["payload"] = base64.b64encode(json.dumps(payload).encode()).decode() + lines[2] = json.dumps(frame).encode() + self.path.write_bytes(b"\n".join(lines) + b"\n") + self.assert_rejected(load_capture(self.path, key=KEY, run_id=RUN), "authentication_failed") + + def test_removed_reordered_appended_records_and_partial_last_line(self): + raw = encode(events(start(), end()), KEY, RUN) + lines = raw.splitlines(keepends=True) + for bad in (b"".join([lines[0], lines[2], lines[3]]), + b"".join([lines[0], lines[2], lines[1], lines[3]]), + raw + lines[-1], raw[:-1]): + with self.subTest(raw=bad[:16]): + self.path.write_bytes(bad) + self.assert_rejected(load_capture(self.path, key=KEY, run_id=RUN)) + + def test_sequence_gaps_bool_and_count_mismatch(self): + for seq in (8, True, "1"): + items = events(start(), end()) + items[1]["seq"] = seq + self.assert_rejected(self.load(items), "ambiguous_order") + items = events(start(), end()) + items[-1]["calls_started"] = 9 + self.assert_rejected(self.load(items), "count_mismatch") + + def test_duplicate_json_keys_nonfinite_and_deep_json(self): + transforms = ( + lambda seq, raw: raw[:-1] + b',"seq":0}' if seq == 0 else raw, + lambda seq, raw: raw.replace(b'"version":1', b'"version":NaN') if seq == 0 else raw, + lambda seq, raw: b'[' * 2000 + b'0' + b']' * 2000 if seq == 0 else raw, + ) + for transform in transforms: + self.path.write_bytes(encode(events(start(), end()), KEY, RUN, transform)) + self.assert_rejected(load_capture(self.path, key=KEY, run_id=RUN)) + + def test_malformed_records_and_unsupported_versions(self): + for raw in (b'{bad}\n', b'{"payload": "!", "mac": "x"}\n', b'[]\n', b'\xff\n'): + self.path.write_bytes(raw) + self.assert_rejected(load_capture(self.path, key=KEY, run_id=RUN)) + items = events(start(), end()) + items[0]["version"] = 2 + self.assert_rejected(self.load(items), "unsupported_version") + + def test_redact_before_clipping(self): + secret = "secret-material-" * 100 + result = self.load(events(start(), end(value=secret)), known_secrets=(secret,), field_limit=64) + self.assert_rejected(result, "incomplete_fields") + self.assertEqual(result["records"][0]["result"], {"state": "redacted", "value": "[REDACTED]"}) + self.assertNotIn("secret-material", json.dumps(result)) + self.assertFalse(result["coverage"]["truncated"]) + + def test_unknown_redaction_is_omitted_without_preview(self): + record = end(value="private-raw-value") + del record["result"]["redaction"] + result = self.load(events(start(), record)) + self.assert_rejected(result, "incomplete_fields") + self.assertNotIn("private-raw-value", json.dumps(result)) + self.assertEqual(result["records"][0]["result"]["state"], "omitted") + + def test_sensitive_nested_keys_and_embedded_known_secrets(self): + result = self.load(events(start(value={"nested": {"password": "unlisted-secret"}}), + end(value='{"text":"known-secret"}')), known_secrets=("known-secret",)) + self.assert_rejected(result, "incomplete_fields") + self.assertNotIn("unlisted-secret", json.dumps(result)) + self.assertNotIn("known-secret", json.dumps(result)) + + def test_omissions_truncation_and_unsafe_identity_never_pass(self): + result = self.load(events(start(), end(value="a" * 2000)), field_limit=64) + self.assert_rejected(result, "incomplete_fields") + self.assertTrue(result["coverage"]["truncated"]) + self.assertNotIn("value", result["records"][0]["result"]) + record = end() + record["result"] = {"state": "omitted", "reason": "SECRET", "value": "SECRET"} + result = self.load(events(start(), record)) + self.assert_rejected(result) + self.assertNotIn("SECRET", json.dumps(result)) + self.assert_rejected(self.load(known_secrets=("worker",)), "unsafe_identity") + + def test_filesystem_symlink_fifo_hardlink_and_limits(self): + target = self.path.with_name("other") + target.write_bytes(encode(events(start(), end()), KEY, RUN)) + self.path.symlink_to(target) + self.assert_rejected(load_capture(self.path, key=KEY, run_id=RUN)) + self.path.unlink() + os.mkfifo(self.path) + self.assert_rejected(load_capture(self.path, key=KEY, run_id=RUN), "unsafe_capture_file") + self.path.unlink() + os.link(target, self.path) + self.assert_rejected(load_capture(self.path, key=KEY, run_id=RUN), "unsafe_capture_file") + self.path.unlink() + with self.path.open("wb") as stream: + stream.truncate(MAX_CAPTURE_BYTES + 1) + result = load_capture(self.path, key=KEY, run_id=RUN) + self.assert_rejected(result, "capture_limit") + self.assertTrue(result["coverage"]["truncated"]) + + def test_import_is_read_only_and_input_objects_unchanged(self): + items = events(start(), end()) + original = copy.deepcopy(items) + raw = encode(items, KEY, RUN) + self.path.write_bytes(raw) + self.assertTrue(load_capture(self.path, key=KEY, run_id=RUN)["evidence_eligible"]) + self.assertEqual(self.path.read_bytes(), raw) + self.assertEqual(items, original) + + def test_failed_transport_disqualifies_even_complete_capture(self): + self.assert_rejected(self.load(transport_ok=False), "transport_failed") + + +class TransportTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.workspace = self.root / "workspace" + self.workspace.mkdir() + self.key = self.root / "signing-key" + self.key.write_bytes(KEY) + self.key.chmod(0o600) + self.prompt = self.root / "prompt.txt" + self.prompt.write_text("fixture prompt; no provider inference") + self.output = self.root / "result.json" + self.command_log = self.root / "command.json" + fake = self.root / "docker" + fake.write_text(f'#!/bin/sh\nexec "{os.sys.executable}" "{FIXTURE}" "$@"\n') + fake.chmod(0o755) + env = {"PATH": str(self.root) + os.pathsep + os.defpath, + "HOME": str(self.root / "home"), "XDG_DATA_HOME": str(self.root / "data"), + "XDG_STATE_HOME": str(self.root / "state"), "XDG_CACHE_HOME": str(self.root / "cache"), + "XDG_CONFIG_HOME": str(self.root / "config"), + "FIXTURE_KEY_FILE": str(self.key), "FIXTURE_COMMAND": str(self.command_log)} + self.environment = patch.dict(os.environ, env, clear=True) + self.environment.start() + self.addCleanup(self.environment.stop) + self.args = cli.parser().parse_args([ + "invoke", "--engine", "docker", "--model", "fixture/model", "--workspace", str(self.workspace), + "--prompt-file", str(self.prompt), "--output", str(self.output), "--observer-key-file", str(self.key), + "--image", "fixture@sha256:" + "6" * 64, + ]) + + def run_mode(self, mode): + with patch.dict(os.environ, {"FIXTURE_MODE": mode}): + code = cli.invoke(self.args) + return code, json.loads(self.output.read_text()) + + def test_real_subprocess_export_is_independent_of_outer_output(self): + code, result = self.run_mode("good") + self.assertEqual(code, 0) + self.assertEqual(result["text"], "script-controlled transformed output") + self.assertEqual(result["observed_tool_results"], [{"tool": "execute", "output": "discarded"}]) + projection = result["observed_execution"] + self.assertTrue(projection["evidence_eligible"]) + self.assertEqual([r["invocation_id"] for r in projection["records"]], ["native-1", "inner-1", "inner-2"]) + self.assertEqual(projection["records"][0]["result"]["value"], "native-sentinel") + self.assertEqual(projection["records"][1]["error"]["value"]["message"], "thrown-sentinel") + self.assertEqual(projection["records"][2]["result"]["value"], {"denied": True, "reason": "policy"}) + log = json.loads(self.command_log.read_text()) + self.assertNotIn("FIXTURE_KEY_FILE", log["environment"]) + self.assertNotIn(str(self.key), json.dumps(log)) + self.assertNotIn(KEY.decode(), json.dumps(log)) + self.assertEqual(log["command"][-1], "fixture@sha256:" + "6" * 64) + self.assertFalse(any("opencode.db" in arg for arg in log["command"])) + self.assertEqual(self.prompt.read_text(), "fixture prompt; no provider inference") + self.assertEqual(list(self.workspace.iterdir()), []) + + def test_missing_malformed_and_forged_sidecar_cannot_exit_success(self): + for mode in ("missing", "incomplete", "forged"): + with self.subTest(mode=mode): + code, result = self.run_mode(mode) + self.assertEqual(code, 4) + self.assertFalse(result["observed_execution"]["evidence_eligible"]) + self.assertNotIn("forged-stdout", json.dumps(result)) + + def test_original_failure_code_is_preserved(self): + code, result = self.run_mode("transport_failure") + self.assertEqual(code, 7) + self.assertFalse(result["observed_execution"]["evidence_eligible"]) + + def test_bad_stdout_still_writes_non_evidence_without_raw_diagnostics(self): + code, result = self.run_mode("malformed_stdout") + self.assertEqual(code, 2) + self.assertFalse(result["observed_execution"]["evidence_eligible"]) + self.assertNotIn("secret diagnostic", json.dumps(result)) + + def test_timeout_writes_explicit_non_evidence(self): + with patch.object(cli.subprocess, "run", side_effect=subprocess.TimeoutExpired("docker", 1)): + self.assertEqual(cli.invoke(self.args), 2) + self.assertFalse(json.loads(self.output.read_text())["observed_execution"]["evidence_eligible"]) + + def test_opt_out_cannot_launder_observer_claims_from_stdout(self): + self.args.observer_key_file = None + code, result = self.run_mode("good") + self.assertEqual(code, 0) + self.assertEqual(result["observed_execution"]["issues"], ["capture_not_requested"]) + self.assertNotIn("forged-stdout", json.dumps(result)) + self.assertNotIn("/eval-observer", json.dumps(json.loads(self.command_log.read_text()))) + + def test_unique_nonce_and_private_capture_per_invocation(self): + _, one = self.run_mode("good") + _, two = self.run_mode("good") + self.assertNotEqual(one["observed_execution"]["run_id"], two["observed_execution"]["run_id"]) + + def test_key_cannot_be_mounted_via_workspace_or_explicit_bind(self): + for mode in ("workspace", "explicit"): + args = copy.deepcopy(self.args) + if mode == "workspace": + args.workspace = str(self.root) + else: + args.mount = [f"{self.key}:/visible-key:ro"] + with self.assertRaises(cli.RunnerError): + cli.invoke(args) + self.assertFalse(self.command_log.exists()) + + def test_key_and_reserved_variables_cannot_be_forwarded(self): + for name, value in (("KEY", KEY.decode()), ("EVAL_OBSERVER_RUN_ID", "forged")): + with patch.dict(os.environ, {name: value}): + args = copy.deepcopy(self.args) + args.env.append(name) + with self.assertRaises(cli.RunnerError): + cli.invoke(args) + self.assertFalse(self.command_log.exists()) + + def test_embedded_key_encodings_are_rejected(self): + encodings = [ + KEY.decode(), + KEY.hex(), + KEY.hex().upper(), + base64.b64encode(KEY).decode(), + base64.urlsafe_b64encode(KEY).decode(), + base64.b64encode(KEY).decode().rstrip("="), + base64.urlsafe_b64encode(KEY).decode().rstrip("="), + ] + for value in encodings: + with self.subTest(value=value[:24]): + with patch.dict(os.environ, {"KEY": "prefix-" + value + "-suffix"}): + args = copy.deepcopy(self.args) + args.env.append("KEY") + with self.assertRaises(cli.RunnerError): + cli.invoke(args) + self.assertFalse(self.command_log.exists()) + + def test_encoded_key_and_nonprivate_key_are_rejected(self): + for value in (KEY.hex(), base64.b64encode(KEY).decode()): + with patch.dict(os.environ, {"KEY": value}): + args = copy.deepcopy(self.args) + args.env.append("KEY") + with self.assertRaises(cli.RunnerError): + cli.invoke(args) + self.key.chmod(0o644) + with self.assertRaises(cli.RunnerError): + cli.invoke(self.args) + self.assertFalse(self.command_log.exists()) + + def test_observer_only_opencode(self): + self.args.transport = "github-copilot-cli" + with self.assertRaises(cli.RunnerError): + cli.invoke(self.args) + self.assertFalse(self.command_log.exists()) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_protected.py b/tests/test_protected.py new file mode 100644 index 0000000..f5f6f27 --- /dev/null +++ b/tests/test_protected.py @@ -0,0 +1,209 @@ +import importlib.util +from pathlib import Path +import tempfile +import os +import base64 +from runner.protected_launch import bounded_file +"""Parser unit cases. These fixtures are not runtime integration evidence.""" +import copy +import hashlib +import json +import unittest + +from runner.protected import import_capture, PROFILE, SCHEMA, RUNTIME_SCHEMA + +RUN = "a" * 64 +POLICY = "b" * 64 +LAUNCH = "d" * 64 +ACTOR = {"agent": "build", "session_id": "s", "message_id": "m"} +PARENT = {"invocation_id": "p", "session_id": "s", "message_id": "m", "call_id": "c"} + + +def stream(events=None): + events = events or [ + {"kind": "parent_start", "boundary": "codemode-engine", "mode": "code_mode"}, + {"kind": "call_start", "invocation_id": "i", "tool": "isolated_echo", "catalog_path": "isolated.echo", + "input": {"state": "available", "redaction": "safe", "value": {}}, "boundary": "executable-input"}, + {"kind": "call_end", "invocation_id": "i", "dispatched": True, "boundary": "codemode-json-return", "outcome": "returned", + "result": {"state": "available", "redaction": "safe", "value": "actual"}}, + {"kind": "parent_end", "admitted": 1, "dispatched": 1, "terminals": 1, "missing_terminals": 0, + "unsupported_dispatches": 0, "unavailable_fields": 0, "scope": "one-codemode-engine-invocation", "evidence_eligible": False}, + ] + frames = [{"kind": "capture_start", "schema": SCHEMA, "profile": PROFILE, "run_id": RUN, "seq": 0, "policy_id": POLICY, "launch_id": LAUNCH}] + for n, event in enumerate(events, 1): + frames.append({"kind": "observation", "run_id": RUN, "seq": n, "observation": { + "schema": RUNTIME_SCHEMA, "sequence": n, "parent": PARENT, "actor": ACTOR, "observer_failures": 0, **event}}) + raw = b"".join((json.dumps(f) + "\n").encode() for f in frames) + footer = {"kind": "capture_end", "run_id": RUN, "seq": len(frames), "event_count": len(events), + "sha256": hashlib.sha256(raw).hexdigest(), "writer_exited": True, "runtime_exit": 0} + return raw + (json.dumps(footer) + "\n").encode() + + +def load(raw, **kwargs): + return import_capture(raw, run_id=RUN, policy_id=POLICY, launch_id=LAUNCH, receipt=kwargs.get("receipt", {"sha256": hashlib.sha256(raw).hexdigest(), "bytes": len(raw)}), tools={"isolated_echo"}, transport_ok=kwargs.get("transport_ok", True)) + + +class ProtectedParserTests(unittest.TestCase): + def test_receipt_rejects_rewritten_resealed_realistic_stream(self): + raw = stream() + receipt = {"sha256": hashlib.sha256(raw).hexdigest(), "bytes": len(raw)} + events = [json.loads(line)["observation"] for line in raw.splitlines()[1:-1]] + events[2]["result"]["value"] = "forged" + rewritten = stream(events) # attacker also recomputes the inline footer + result = load(rewritten, receipt=receipt) + self.assertEqual(result["issues"], ["receipt_mismatch"]) + self.assertFalse(result["evidence_eligible"]) + self.assertEqual(result["records"], []) + + def test_no_separate_receipt_is_not_protected_evidence(self): + self.assertFalse(load(stream(), receipt=None)["evidence_eligible"]) + + def test_complete_restricted_profile_preserves_all_bindings(self): + value = load(stream()) + self.assertTrue(value["evidence_eligible"]) + self.assertFalse(value["full_handoff_eligible"]) + call = value["records"][0] + self.assertEqual(call["actor"], ACTOR) + self.assertEqual(call["parent"], PARENT) + self.assertEqual(call["runtime_call_id"], "c") + self.assertEqual(call["result"]["value"], "actual") + + def test_wrong_launch_binding_rejected(self): + value = import_capture(stream(), run_id=RUN, policy_id=POLICY, launch_id="e" * 64, + receipt={"sha256": hashlib.sha256(stream()).hexdigest(), "bytes": len(stream())}, tools={"isolated_echo"}, transport_ok=True) + self.assertFalse(value["evidence_eligible"]) + self.assertEqual(value["issues"], ["wrong_launch"]) + + def test_changes_and_replay_cannot_pass(self): + for raw in (stream().replace(b"actual", b"FORGED"), stream().replace(RUN.encode(), ("c" * 64).encode()), + stream().splitlines(keepends=True)[0], stream()[:-10], stream() + b"{}\n"): + with self.subTest(raw=raw[:10]): + self.assertFalse(load(raw)["evidence_eligible"]) + self.assertEqual(load(raw)["records"], []) + + def test_bad_json_empty_duplicate_and_nonfinite(self): + for raw in (b"", b'{}\n', b'[]\n[]\n', b'{"a":1,"a":2}\n{}\n', b'{"x":NaN}\n{}\n'): + self.assertFalse(load(raw)["evidence_eligible"]) + + def test_transport_failure_does_not_promote_valid_capture(self): + value = load(stream(), transport_ok=False) + self.assertFalse(value["evidence_eligible"]) + self.assertTrue(all(r["evidence_eligible"] is False for r in value["records"])) + + def test_unknown_record_and_closed_parent_fail(self): + frames = [json.loads(s) for s in stream().splitlines()] + events = [f["observation"] for f in frames[1:-1]] + events[2]["kind"] = "something_else" + self.assertFalse(load(stream(events))["evidence_eligible"]) + + def test_omitted_redacted_and_truncated_are_non_evidence(self): + for field in ({"state": "omitted", "reason": "policy_omission"}, + {"state": "truncated", "reason": "field_limit"}, + {"state": "redacted", "redaction": "safe", "value": "[REDACTED]"}): + frames = [json.loads(s) for s in stream().splitlines()] + events = [f["observation"] for f in frames[1:-1]] + events[2]["result"] = field + value = load(stream(events)) + self.assertFalse(value["evidence_eligible"]) + self.assertEqual(value["status"], "incomplete") + + def test_deleted_middle_and_forged_completeness_rejected(self): + frames = stream().splitlines(keepends=True) + self.assertFalse(load(b"".join(frames[:2] + frames[3:]))["evidence_eligible"]) + events = [f["observation"] for f in map(json.loads, frames[1:-1])] + events[-1]["terminals"] = 10 + self.assertFalse(load(stream(events))["evidence_eligible"]) + + def test_unknown_error_representation_and_conflicting_actor_rejected(self): + frames = [json.loads(s) for s in stream().splitlines()] + events = [f["observation"] for f in frames[1:-1]] + events[2]["actor"]["agent"] = "other" + self.assertFalse(load(stream(events))["evidence_eligible"]) + + +class ProtectedLaunchTests(unittest.TestCase): + def test_bounded_snapshot_is_independent_of_later_input_change(self): + import tempfile + from pathlib import Path + from runner.protected_launch import bounded_file + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "program" + path.write_bytes(b"original") + snapshot = bounded_file(path, 8) + path.write_bytes(b"changed source") + self.assertEqual(snapshot, b"original") + with self.assertRaisesRegex(ValueError, "input_limit"): + bounded_file(path, 8) + + def test_new_wire_and_projection_are_explicit_not_v1_aliases(self): + result = load(stream()) + self.assertEqual(SCHEMA, "opencode-protected-observation/v2") + self.assertEqual(result["version"], 5) + self.assertEqual(result["profile"], PROFILE) + + +class ProtectedSupervisorTests(unittest.TestCase): + @classmethod + def setUpClass(cls): + path = Path(__file__).resolve().parents[1] / "protected-runtime/invoke.py" + spec = importlib.util.spec_from_file_location("protected_supervisor", path) + cls.supervisor = importlib.util.module_from_spec(spec) + spec.loader.exec_module(cls.supervisor) + + def test_readiness_redirect_does_not_reach_second_endpoint(self): + import threading + import urllib.error + from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + hits = [] + class Handler(BaseHTTPRequestHandler): + def log_message(self, *args): + pass + def do_GET(self): + hits.append(self.path) + if self.path == "/health": + self.send_response(302) + self.send_header("Location", "/private") + else: + self.send_response(200) + self.end_headers() + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + thread = threading.Thread(target=server.serve_forever) + thread.start() + try: + with self.assertRaises(urllib.error.HTTPError): + self.supervisor.health_check(f"http://127.0.0.1:{server.server_port}/health") + self.assertEqual(hits, ["/health"]) + finally: + server.shutdown() + server.server_close() + thread.join() + + def test_script_diagnostics_redact_encoded_secrets_before_retaining(self): + secret = "fixture-private" + variants = (secret, base64.b64encode(secret.encode()).decode(), secret.encode().hex()) + policy = {"secrets": [secret], "allowed_values": [*variants, "[REDACTED]"]} + for value in variants: + self.assertEqual(self.supervisor.safe_output(value, policy), + {"state": "redacted", "value": "[REDACTED]"}) + self.assertEqual(self.supervisor.safe_output("unknown", policy)["state"], "omitted") + + def test_script_diagnostic_limits_follow_redaction(self): + secret = "x" * 20000 + policy = {"secrets": [secret], "allowed_values": ["[REDACTED]", "ø" * 9000]} + self.assertEqual(self.supervisor.safe_output(secret, policy)["state"], "redacted") + self.assertEqual(self.supervisor.safe_output("ø" * 9000, policy)["state"], "truncated") + + def test_input_read_is_bounded_and_rejects_fifo_without_blocking(self): + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "input" + path.write_bytes(b"1234") + with self.assertRaises(ValueError): + bounded_file(path, 2) + path.unlink() + os.mkfifo(path) + with self.assertRaises(ValueError): + bounded_file(path, 2) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_review_accounting.py b/tests/test_review_accounting.py new file mode 100644 index 0000000..c36931d --- /dev/null +++ b/tests/test_review_accounting.py @@ -0,0 +1,69 @@ +"""Regression cases for capture identity/accounting, not origin attestation.""" +import json +import unittest + +from test_protected import load, stream + + +def events(): + return [json.loads(line)["observation"] for line in stream().splitlines()[1:-1]] + + +class CaptureIdentityTests(unittest.TestCase): + def test_child_cannot_reuse_its_parent_invocation_id(self): + records = events() + for record in records: + if "invocation_id" in record: + record["invocation_id"] = "p" + result = load(stream(records)) + self.assertFalse(result["evidence_eligible"]) + self.assertEqual(result["records"], []) + self.assertIn("duplicate_or_invalid_invocation", result["issues"]) + + def test_later_parent_cannot_reuse_a_completed_child_id(self): + records = events() + next_parent = dict(records[0], parent=dict(records[0]["parent"], invocation_id="i"), sequence=5) + records.append(next_parent) + result = load(stream(records)) + self.assertFalse(result["evidence_eligible"]) + self.assertEqual(result["issues"], ["duplicate_parent"]) + + +class CaptureAccountingTests(unittest.TestCase): + def test_changed_accounting_contract_has_a_new_projection_version(self): + result = load(stream()) + self.assertEqual(result["version"], 5) + self.assertTrue(result["coverage"]["accounting_complete"]) + + def test_missing_terminal_retains_diagnostic_prefix_counts(self): + records = events() + del records[2] + records[-1].update(terminals=0, missing_terminals=1) + for index, record in enumerate(records, 1): + record["sequence"] = index + result = load(stream(records)) + self.assertFalse(result["evidence_eligible"]) + self.assertEqual(result["records"], []) + self.assertEqual(result["issues"], ["incomplete_parent"]) + self.assertEqual(result["coverage"]["starts"], 1) + self.assertEqual(result["coverage"]["terminals"], 0) + self.assertEqual(result["coverage"]["missing_terminals"], 1) + self.assertIsNone(result["coverage"]["omitted_records"]) + + def test_unverified_capture_does_not_claim_zero_missing_terminals(self): + result = load(stream(), receipt=None) + self.assertFalse(result["evidence_eligible"]) + self.assertIsNone(result["coverage"]["missing_terminals"]) + + def test_invalid_terminal_field_is_not_counted_as_a_terminal(self): + records = events() + records[2]["result"] = {"state": "available", "redaction": "unsafe", "value": "x"} + result = load(stream(records)) + self.assertFalse(result["evidence_eligible"]) + self.assertEqual(result["coverage"]["starts"], 1) + self.assertEqual(result["coverage"]["terminals"], 0) + self.assertEqual(result["coverage"]["missing_terminals"], 1) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_review_image_policy.py b/tests/test_review_image_policy.py new file mode 100644 index 0000000..4964ef8 --- /dev/null +++ b/tests/test_review_image_policy.py @@ -0,0 +1,161 @@ +"""QA of launch policy at the container-engine I/O boundary.""" +import json +from pathlib import Path +import subprocess +import tempfile +import unittest +from unittest.mock import patch + +from runner import protected_launch +from runner.protected import CaptureError, parser + +REFERENCE = "ghcr.io/example/tool@sha256:" + "1" * 64 + + +def inspected(volumes): + return {"Id": "sha256:" + "2" * 64, "RepoDigests": [REFERENCE], "Config": {"Volumes": volumes}} + + +class ImagePolicyTests(unittest.TestCase): + def test_image_declared_volumes_are_rejected_before_launch(self): + replies = [subprocess.CompletedProcess([], 0, "", ""), + subprocess.CompletedProcess([], 0, json.dumps([inspected({"/state": {}})]), "")] + with patch.object(protected_launch, "run", side_effect=replies): + with self.assertRaisesRegex(CaptureError, "image_declares_volumes"): + protected_launch.image_info(REFERENCE) + + def test_volume_free_image_remains_usable(self): + for volumes in (None, {}): + replies = [subprocess.CompletedProcess([], 0, "", ""), + subprocess.CompletedProcess([], 0, json.dumps([inspected(volumes)]), "")] + with patch.object(protected_launch, "run", side_effect=replies): + self.assertEqual(protected_launch.image_info(REFERENCE), inspected(volumes)) + + @staticmethod + def engine(*, removed, failures=(), foreign=False, missing=False, unavailable=False): + """Independent daemon inventory, not the launcher's ownership flags.""" + identities = {"runtime": "a" * 64, "target": "b" * 64, "network": "c" * 64} + def call(command, **kwargs): + if command[2:3] == ["ls"]: + if unavailable: + return subprocess.CompletedProcess(command, 1, "", "offline") + name = next(arg[5:] for arg in command if arg.startswith("name=")) + name_field = "Name" if command[1] == "network" else "Names" + output = "" if missing else json.dumps({name_field: name, "ID": identities[name]}) + "\n" + return subprocess.CompletedProcess(command, 0, output, "") + if command[2:3] == ["inspect"]: + identity = command[-1] + name = next(n for n, i in identities.items() if i == identity) + labels = {protected_launch.OWNER_LABEL: "someone-else" if foreign else "owner"} + payload = {"Id": identity, "Name": name} + payload["Labels" if name == "network" else "Config"] = labels if name == "network" else {"Labels": labels} + return subprocess.CompletedProcess(command, 0, json.dumps([payload]), "") + removed.append(command) + name = next(n for n, i in identities.items() if i == command[-1]) + return subprocess.CompletedProcess(command, int(name in failures), b"", b"") + return call + + def test_cleanup_only_attempts_resources_with_verified_ownership(self): + removed = [] + with patch.object(protected_launch.subprocess, "run", side_effect=self.engine(removed=removed)): + failures = protected_launch.cleanup_resources( + {"network": True, "target": False, "runtime": True}, "runtime", "target", "network", "owner") + self.assertEqual(failures, []) + self.assertEqual(removed, [["docker", "rm", "--volumes", "-f", "a" * 64], + ["docker", "network", "rm", "c" * 64]]) + + def test_cleanup_failure_is_reported_even_for_negative_runs(self): + removed = [] + with patch.object(protected_launch.subprocess, "run", side_effect=self.engine(removed=removed, failures=("runtime", "network"))): + failures = protected_launch.cleanup_resources( + {"network": True, "target": True, "runtime": True}, "runtime", "target", "network", "owner") + self.assertEqual(failures, ["runtime", "network"]) + self.assertEqual(len(removed), 3) + + def test_cleanup_does_not_delete_a_same_named_foreign_resource(self): + removed = [] + with patch.object(protected_launch.subprocess, "run", side_effect=self.engine(removed=removed, foreign=True)): + failures = protected_launch.cleanup_resources({"runtime": True}, "runtime", "target", "network", "owner") + self.assertEqual(failures, ["runtime"]) + self.assertEqual(removed, []) + + def test_cleanup_distinguishes_absence_from_unavailable_engine(self): + for missing, unavailable in ((True, False), (False, True)): + removed = [] + with patch.object(protected_launch.subprocess, "run", side_effect=self.engine( + removed=removed, missing=missing, unavailable=unavailable)): + failures = protected_launch.cleanup_resources({"runtime": True}, "runtime", "target", "network", "owner") + self.assertEqual(failures, ["runtime"] if unavailable else []) + self.assertEqual(removed, []) + + def test_create_response_loss_is_reconciled_for_network_and_both_containers(self): + for stage in ("network", "target", "runtime"): + with self.subTest(stage=stage), tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + (root / "policy").write_text(json.dumps({"version": 1, "secrets": [], "allowed_values": []})) + (root / "tools").write_text(json.dumps([{"name": "echo", "input": {}}])) + (root / "program").write_text('return "unchanged";') + args = parser().parse_args(["--image", REFERENCE, "--tool-image", REFERENCE, + "--policy-file", str(root / "policy"), "--tools-file", str(root / "tools"), + "--program-file", str(root / "program"), "--output", str(root / "result")]) + live = {} + created = [] + removed = [] + def daemon(command, **kwargs): + if command[1:3] == ["network", "create"] or command[1] == "create": + is_network = command[1] == "network" + name = command[-1] if is_network else command[command.index("--name") + 1] + kind = "network" if is_network else "target" if name.endswith("-tools") else "runtime" + owner = command[command.index("--label") + 1].partition("=")[2] + identity = {"network": "a", "target": "b", "runtime": "c"}[kind] * 64 + live[identity] = {"Id": identity, "Name": name, + "Labels" if is_network else "Config": {protected_launch.OWNER_LABEL: owner} + if is_network else {"Labels": {protected_launch.OWNER_LABEL: owner}}} + created.append(kind) + if kind == stage: + # Daemon has created it; CLI never receives its response. + raise subprocess.TimeoutExpired(command, 1) + return subprocess.CompletedProcess(command, 0, identity, "") + if command[2:3] == ["ls"]: + name = next(arg[5:] for arg in command if arg.startswith("name=")) + name_field = "Name" if command[1] == "network" else "Names" + rows = [{name_field: r["Name"], "ID": i} for i, r in live.items() if r["Name"] == name] + return subprocess.CompletedProcess(command, 0, "\n".join(json.dumps(r) for r in rows), "") + if command[2:3] == ["inspect"]: + return subprocess.CompletedProcess(command, 0, json.dumps([live[command[-1]]]), "") + if command[1] == "rm" or command[1:3] == ["network", "rm"]: + removed.append(command[-1]); del live[command[-1]] + return subprocess.CompletedProcess(command, 0, "", "") + with patch.object(protected_launch, "image_info", return_value=inspected(None)), \ + patch.object(protected_launch.subprocess, "run", side_effect=daemon): + self.assertEqual(protected_launch.invoke(args), 2) + self.assertEqual(created[-1], stage) + self.assertEqual(len(removed), len(created)) + self.assertEqual(live, {}) + result = json.loads((root / "result").read_text()) + self.assertEqual(result["observed_execution"]["issues"], ["protected_transport_failed"]) + self.assertFalse(result["observed_execution"]["evidence_eligible"]) + + def test_rejected_image_writes_specific_non_evidence_and_never_runs(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + (root / "policy").write_text(json.dumps({"version": 1, "secrets": [], "allowed_values": []})) + (root / "tools").write_text(json.dumps([{"name": "echo", "input": {}}])) + (root / "program").write_text('return "unchanged";') + args = parser().parse_args(["--image", REFERENCE, "--tool-image", REFERENCE, + "--policy-file", str(root / "policy"), "--tools-file", str(root / "tools"), + "--program-file", str(root / "program"), "--output", str(root / "result")]) + with patch.object(protected_launch, "image_info", side_effect=CaptureError("image_declares_volumes")), \ + patch.object(protected_launch.subprocess, "run", return_value=subprocess.CompletedProcess([], 0)) as engine: + self.assertEqual(protected_launch.invoke(args), 2) + result = json.loads((root / "result").read_text()) + self.assertFalse(result["observed_execution"]["evidence_eligible"]) + self.assertEqual(result["observed_execution"]["issues"], ["image_declares_volumes"]) + self.assertTrue(all(call.args[0][1] != "run" for call in engine.call_args_list)) + removals = [call.args[0] for call in engine.call_args_list + if len(call.args[0]) > 1 and call.args[0][1] in {"rm", "network"}] + self.assertEqual(removals, []) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_review_limits.py b/tests/test_review_limits.py new file mode 100644 index 0000000..6d20e4c --- /dev/null +++ b/tests/test_review_limits.py @@ -0,0 +1,38 @@ +"""QA boundary cases: the declared UTF-8 value limit must match admission.""" +import unittest + +from test_protected import load, stream +from test_review_accounting import events + + +class FieldLimitTests(unittest.TestCase): + def test_over_16k_utf8_value_cannot_be_admitted_as_complete(self): + records = events() + records[2]["result"]["value"] = "ø" * 8192 # 16,386 JSON UTF-8 bytes, including quotes. + result = load(stream(records)) + self.assertFalse(result["evidence_eligible"]) + self.assertEqual(result["issues"], ["field_limit"]) + self.assertTrue(result["coverage"]["truncated"]) + self.assertEqual(result["records"], []) + + def test_exact_utf8_limit_is_available(self): + records = events() + records[2]["result"]["value"] = "ø" * 8191 # Exactly 16,384 JSON bytes. + result = load(stream(records)) + self.assertTrue(result["evidence_eligible"]) + self.assertEqual(result["records"][0]["result"]["value"], "ø" * 8191) + + def test_size_is_wire_json_not_pretty_printing(self): + records = events() + records[2]["result"]["value"] = [0] * 8000 # Compact 16,001; spaced 24,000. + self.assertTrue(load(stream(records))["evidence_eligible"]) + + def test_too_few_frames_does_not_claim_a_size_truncation(self): + header = stream().splitlines(keepends=True)[0] + result = load(header) + self.assertFalse(result["evidence_eligible"]) + self.assertFalse(result["coverage"]["truncated"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_workflow_boundaries.py b/tests/test_workflow_boundaries.py new file mode 100644 index 0000000..e217c54 --- /dev/null +++ b/tests/test_workflow_boundaries.py @@ -0,0 +1,191 @@ +"""Review checks for publication/signing workflow trust boundaries. + +These inspect declared job boundaries; actual Actions runs must also exercise +image publication and digest-based verification. No YAML parser dependency. +""" +from pathlib import Path +import re +import unittest + +ROOT = Path(__file__).resolve().parents[1] + + +JOB_ID = r"[A-Za-z_][A-Za-z0-9_-]*" + + +def parse_workflow_jobs(text): + prefix, body = text.split("\njobs:\n", 1) + matches = list(re.finditer(r"^ (\S[^\n:]*):\s*$", body, re.MULTILINE)) + jobs = {} + parsed = [] + for match in matches: + raw = match.group(1).strip() + if len(raw) >= 2 and raw[0] == raw[-1] and raw[0] in {'"', "'"}: + key = raw[1:-1] + else: + key = raw + if re.fullmatch(JOB_ID, key) is None or key in jobs: + raise ValueError(f"unsupported or duplicate workflow job key: {raw}") + parsed.append((key, match)) + jobs[key] = None + if not parsed: + raise ValueError("workflow has no parseable jobs") + for i, (key, match) in enumerate(parsed): + jobs[key] = body[match.end(): parsed[i + 1][1].start() if i + 1 < len(parsed) else len(body)] + return prefix, jobs + + +def workflow_jobs(name): + return parse_workflow_jobs((ROOT / ".github/workflows" / name).read_text()) + + +class PublicationBoundaryTests(unittest.TestCase): + def test_job_parser_covers_quoted_digits_and_underscores_fail_closed(self): + prefix, jobs = parse_workflow_jobs( + "name: fixture\non:\n pull_request:\njobs:\n" + " build:\n runs-on: ubuntu-latest\n" + " steps:\n - uses: example/action@deadbeef\n with:\n ref: main\n" + " \"publish_2\":\n runs-on: ubuntu-latest\n" + " '_verify9':\n runs-on: ubuntu-latest\n" + ) + self.assertIn("pull_request", prefix) + self.assertEqual(set(jobs), {"build", "publish_2", "_verify9"}) + with self.assertRaises(ValueError): + parse_workflow_jobs( + "name: fixture\njobs:\n" + " build:\n runs-on: ubuntu-latest\n" + " \"bad job\":\n runs-on: ubuntu-latest\n" + ) + + def test_normal_invoke_workflow_tracks_public_runner_implementation(self): + text = (ROOT / ".github/workflows/local-runtime.yml").read_text() + for path in ( + "runtime-patches/**", + "runner/cli.py", + "runner/observer.py", + "container/**", + "bin/opencode-eval-runner", + "tests/integration/run_capture_probe.py", + "tests/integration/capture_probe.ts", + ): + self.assertIn("- " + path, text) + + def test_only_fresh_publisher_has_package_write_authority(self): + for name in ("protected-channel.yml", "local-runtime.yml", "evidence-safety.yml"): + with self.subTest(workflow=name): + prefix, jobs = workflow_jobs(name) + self.assertNotIn("packages: write", prefix) + self.assertEqual(set(jobs), {"build", "publish", "verify"}) + for job in ("build", "verify"): + self.assertNotIn("packages: write", jobs[job]) + self.assertNotIn("secrets.", jobs[job]) + self.assertNotIn("github.token", jobs[job]) + self.assertNotIn("id-token: write", jobs[job]) + publisher = jobs["publish"] + self.assertIn("needs: build", publisher) + self.assertIn("packages: write", publisher) + self.assertIn("docker load", publisher) + self.assertNotIn("actions/checkout", publisher) + for prohibited in ("docker run", "docker build", "bun ", "python", "./bin/", "./scripts/", "source ", "eval "): + self.assertNotIn(prohibited, publisher) + self.assertIn("needs: [build, publish]", jobs["verify"]) + self.assertIn("needs.publish.outputs", jobs["verify"]) + self.assertIn("github.event.pull_request.head.repo.full_name == github.repository", jobs["build"]) + + def test_evidence_safety_image_copies_and_hashes_only_reviewed_container_sources(self): + containerfile = (ROOT / "evidence-safety/Containerfile").read_text() + workflow = (ROOT / ".github/workflows/evidence-safety.yml").read_text() + self.assertNotIn("COPY container /opt/opencode-eval-runner/container", containerfile) + for path in ("__init__.py", "evidence_safety.py", "invoke.py"): + self.assertIn( + f"COPY container/{path} /opt/opencode-eval-runner/container/{path}", + containerfile, + ) + self.assertIn("container/" + path, workflow) + self.assertIn("ARG PACKAGE_INIT_SHA256", containerfile) + self.assertIn("io.opencode-eval.evidence-safety-init=$PACKAGE_INIT_SHA256", containerfile) + self.assertIn("$PACKAGE_INIT_SHA256 /opt/opencode-eval-runner/container/__init__.py", containerfile) + self.assertIn("--build-arg PACKAGE_INIT_SHA256=", workflow) + + def test_external_actions_are_pinned_and_errors_not_ignored(self): + for name in ( + "protected-channel.yml", + "local-runtime.yml", + "evidence-safety.yml", + "observer-integration.yml", + "sign-normal-invoke-evidence.yml", + ): + text = (ROOT / ".github/workflows" / name).read_text() + for action in re.findall(r"uses: (\S+)", text): + self.assertRegex(action, r"^[A-Za-z0-9_/-]+@[0-9a-f]{40}$") + self.assertNotIn("continue-on-error", text) + self.assertNotIn("pull_request_target", text) + + def test_signer_is_manual_default_branch_rebuild_with_approval_gate(self): + text = (ROOT / ".github/workflows/sign-normal-invoke-evidence.yml").read_text() + prefix, jobs = workflow_jobs("sign-normal-invoke-evidence.yml") + self.assertIn("workflow_dispatch:", prefix) + self.assertNotIn("workflow_run:", prefix) + self.assertNotRegex(prefix, r"(?m)^ pull_request:\s*$") + self.assertNotIn("pull_request_target", text) + self.assertEqual(set(jobs), {"build", "publish", "verify", "sign"}) + + build = jobs["build"] + self.assertIn("github.ref == 'refs/heads/main'", build) + self.assertIn("gh api", build) + self.assertIn(".head.sha", build) + self.assertIn("runtime-patches/apply.py", build) + self.assertIn("docker build", build) + self.assertNotIn("packages: write", build) + self.assertNotIn("id-token: write", build) + + publisher = jobs["publish"] + self.assertIn("packages: write", publisher) + self.assertNotIn("id-token: write", publisher) + self.assertNotIn("actions/checkout", publisher) + self.assertNotIn("docker run", publisher) + self.assertIn("docker load", publisher) + + verifier = jobs["verify"] + self.assertNotIn("packages: write", verifier) + self.assertNotIn("id-token: write", verifier) + self.assertIn("run_image_probe.py", verifier) + self.assertIn("run_eval_live_compat_probe.py", verifier) + self.assertIn("run_delegated_session_probe.py", verifier) + + signer = jobs["sign"] + self.assertIn("environment: release-signing", signer) + self.assertIn("packages: write", signer) + self.assertIn("id-token: write", signer) + self.assertNotIn("actions/checkout", signer) + self.assertNotIn("docker run", signer) + self.assertNotIn("docker build", signer) + self.assertIn("cosign sign --yes", signer) + self.assertIn("cosign sign-blob --yes", signer) + self.assertIn("--certificate-identity", signer) + self.assertIn("protected_capture_accepted", signer) + self.assertIn('"unsupported"', signer) + self.assertIn("docker login ghcr.io", signer) + self.assertIn('eval-live-invoke-compatibility', signer) + self.assertIn('delegated-session-normal-invoke', signer) + self.assertIn('normal-invoke-runtime-seam-probe', signer) + self.assertIn('.image == $image', signer) + self.assertIn('.runner_revision == $source', signer) + + def test_signer_expressions_are_not_escaped_literals(self): + text = (ROOT / ".github/workflows/sign-normal-invoke-evidence.yml").read_text() + self.assertNotIn("\\${{", text) + self.assertNotRegex(text, r"\\\$\{[A-Z_]") + + def test_signer_never_auto_signs_pr_artifacts(self): + text = (ROOT / ".github/workflows/sign-normal-invoke-evidence.yml").read_text() + self.assertNotIn("github.event.workflow_run", text) + self.assertNotIn("runtime-publication-", text) + self.assertNotIn("local-runtime-", text) + self.assertIn("approved-build-", text) + self.assertIn("approved-evidence-", text) + self.assertIn("Reviewed 40-hex source commit", text) + + +if __name__ == "__main__": + unittest.main()