diff --git a/.env.example b/.env.example deleted file mode 100644 index 977ffee..0000000 --- a/.env.example +++ /dev/null @@ -1,15 +0,0 @@ -DATABASE_URL="postgresql://openerrata:openerrata_dev@localhost:5433/openerrata" -OPENAI_API_KEY="sk-..." -OPENAI_MODEL_ID="gpt-5.4-thinking" -OPENAI_MAX_RESPONSE_TOOL_ROUNDS=150 -WORKER_CONCURRENCY=250 -HMAC_SECRET="dev-hmac-secret-change-in-production" -SELECTOR_BUDGET=100 -IP_RANGE_CREDIT_CAP=10 -BLOB_STORAGE_ENDPOINT="http://localhost:9000" -BLOB_STORAGE_BUCKET="openerrata-images" -BLOB_STORAGE_ACCESS_KEY_ID="openerrata_minio" -BLOB_STORAGE_SECRET_ACCESS_KEY="openerrata_minio_secret" -BLOB_STORAGE_PUBLIC_URL_PREFIX="http://localhost:9000/openerrata-images" -DATABASE_ENCRYPTION_KEY="dev-database-encryption-key-change-in-production" -DATABASE_ENCRYPTION_KEY_ID="primary" diff --git a/.github/actions/setup-typescript-workspace/action.yml b/.github/actions/setup-typescript-workspace/action.yml index de0dbde..3ae054e 100644 --- a/.github/actions/setup-typescript-workspace/action.yml +++ b/.github/actions/setup-typescript-workspace/action.yml @@ -3,11 +3,9 @@ description: Install pnpm and Node.js dependencies for the TypeScript workspace inputs: node-version: + # Keep in step with the node:-slim base of api/ and frontend/ Dockerfiles. description: Node.js version - default: "22.13.1" - pnpm-version: - description: pnpm version - default: "10.24.0" + default: "22.23.3" generate-prisma: description: Whether to run pnpm db:generate default: "true" @@ -22,12 +20,13 @@ runs: using: composite steps: - name: Install pnpm - uses: pnpm/action-setup@v4 + uses: pnpm/action-setup@b906affcce14559ad1aafd4ab0e942779e9f58b1 # v4.3.0 with: - version: ${{ inputs.pnpm-version }} + # The version comes from packageManager, which Docker builds also use. + package_json_file: src/typescript/package.json - name: Setup Node.js - uses: actions/setup-node@v4 + uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 with: node-version: ${{ inputs.node-version }} cache: pnpm diff --git a/.github/workflows/deploy.yml b/.github/workflows/deploy.yml index 0e8a99c..9e0f61d 100644 --- a/.github/workflows/deploy.yml +++ b/.github/workflows/deploy.yml @@ -16,9 +16,12 @@ on: - staging - main +# One run per stack at a time, push and manual dispatch alike. A newer run +# waits rather than cancelling one that may be mid-`pulumi up`; GitHub keeps +# only the newest pending run per group. concurrency: - group: deploy-${{ github.event_name == 'push' && github.ref_name || format('dispatch-{0}', github.run_id) }} - cancel-in-progress: true + group: deploy-${{ github.event_name == 'workflow_dispatch' && inputs.environment || github.ref_name }} + cancel-in-progress: false permissions: contents: read @@ -83,7 +86,7 @@ jobs: - name: Checkout if: steps.decide.outputs.should_deploy != 'true' - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 with: fetch-depth: 0 @@ -143,7 +146,7 @@ jobs: - name: Detect deployable path changes since last deploy id: filter if: steps.decide.outputs.should_deploy != 'true' - uses: dorny/paths-filter@v3 + uses: dorny/paths-filter@0e4a8c6effa4802afeda77dc8d303f8176d7dfad # v3.0.4 with: base: ${{ steps.last_deploy.outputs.sha }} filters: | @@ -184,10 +187,13 @@ jobs: timeout-minutes: 10 steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup Helm - uses: azure/setup-helm@v4 + uses: azure/setup-helm@1a275c3b69536ee54be43f2070a358922e12c8d4 # v4.3.1 + with: + # Helm 3, the engine Pulumi's helm.v3.Chart renders the chart with. + version: v3.22.0 - name: Lint chart shell: bash @@ -208,7 +214,7 @@ jobs: DATABASE_URL: postgresql://postgres:postgres@localhost:5432/postgres steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup TypeScript workspace uses: ./.github/actions/setup-typescript-workspace @@ -228,7 +234,7 @@ jobs: DATABASE_URL: postgresql://postgres:postgres@localhost:5432/postgres steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup TypeScript workspace uses: ./.github/actions/setup-typescript-workspace @@ -260,7 +266,7 @@ jobs: --health-retries 5 steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup TypeScript workspace uses: ./.github/actions/setup-typescript-workspace @@ -280,7 +286,7 @@ jobs: DATABASE_URL: postgresql://postgres:postgres@localhost:5432/postgres steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup TypeScript workspace uses: ./.github/actions/setup-typescript-workspace @@ -293,7 +299,7 @@ jobs: run: tar -C extension -czf "$RUNNER_TEMP/extension-dist.tgz" dist - name: Upload extension dist artifact - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 with: name: extension-dist-${{ github.sha }} path: ${{ runner.temp }}/extension-dist.tgz @@ -304,7 +310,7 @@ jobs: run: tar -C frontend -czf "$RUNNER_TEMP/frontend-build.tgz" build - name: Upload frontend build artifact - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 with: name: frontend-build-${{ github.sha }} path: ${{ runner.temp }}/frontend-build.tgz @@ -320,7 +326,7 @@ jobs: working-directory: src/typescript steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup TypeScript workspace uses: ./.github/actions/setup-typescript-workspace @@ -329,7 +335,7 @@ jobs: install-playwright-chromium: "true" - name: Download extension dist artifact - uses: actions/download-artifact@v4 + uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0 with: name: extension-dist-${{ github.sha }} path: ${{ runner.temp }} @@ -351,7 +357,7 @@ jobs: working-directory: src/typescript steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup TypeScript workspace uses: ./.github/actions/setup-typescript-workspace @@ -360,7 +366,7 @@ jobs: install-playwright-chromium: "true" - name: Download frontend build artifact - uses: actions/download-artifact@v4 + uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0 with: name: frontend-build-${{ github.sha }} path: ${{ runner.temp }} @@ -394,17 +400,17 @@ jobs: frontend_image_digest: ${{ steps.build-frontend.outputs.digest }} steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Login to GHCR - uses: docker/login-action@v3 + uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0 with: registry: ghcr.io username: ${{ github.actor }} password: ${{ secrets.GITHUB_TOKEN }} - name: Set up Docker Buildx - uses: docker/setup-buildx-action@v3 + uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0 - name: Compute image metadata id: meta @@ -424,7 +430,7 @@ jobs: - name: Build and push API image id: build - uses: docker/build-push-action@v6 + uses: docker/build-push-action@10e90e3645eae34f1e60eeb005ba3a3d33f178e8 # v6.19.2 with: context: src/typescript file: src/typescript/api/Dockerfile @@ -438,7 +444,7 @@ jobs: - name: Build and push frontend image id: build-frontend - uses: docker/build-push-action@v6 + uses: docker/build-push-action@10e90e3645eae34f1e60eeb005ba3a3d33f178e8 # v6.19.2 with: context: src/typescript file: src/typescript/frontend/Dockerfile @@ -450,6 +456,10 @@ jobs: provenance: true sbom: true + # Holds no long-lived cloud credentials. GitHub's OIDC token (scoped to the + # main/staging environment) joins the tailnet, which routes to the private + # EKS endpoint, and assumes the IAM role from src/aws/ci-iam/setup.sh. EKS + # maps that role to the Kubernetes group bound by src/kubernetes/ci-rbac/. deploy: needs: [resolve-target, decide-deploy, helm-lint, typecheck, lint, test, build, e2e, frontend-e2e, build-image] if: needs.decide-deploy.outputs.should_deploy == 'true' @@ -464,37 +474,119 @@ jobs: working-directory: src/typescript env: DATABASE_URL: postgresql://postgres:postgres@localhost:5432/postgres + DEPLOY_AWS_REGION: us-west-2 + DEPLOY_EKS_CLUSTER: cunningham steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - - name: Setup TypeScript workspace - uses: ./.github/actions/setup-typescript-workspace + - name: Check deploy prerequisites + shell: bash + env: + AWS_DEPLOY_ROLE_ARN: ${{ vars.AWS_DEPLOY_ROLE_ARN }} + TS_OAUTH_CLIENT_ID: ${{ secrets.TS_OAUTH_CLIENT_ID }} + TS_AUDIENCE: ${{ secrets.TS_AUDIENCE }} + run: | + set -euo pipefail + missing=0 + if [ -z "${AWS_DEPLOY_ROLE_ARN:-}" ]; then + echo "::error::Variable AWS_DEPLOY_ROLE_ARN is not set. Set it to the role ARN printed by src/aws/ci-iam/setup.sh." + missing=1 + elif ! [[ "$AWS_DEPLOY_ROLE_ARN" =~ ^arn:aws:iam::[0-9]{12}:role/.+$ ]]; then + echo "::error::Variable AWS_DEPLOY_ROLE_ARN is not an IAM role ARN: $AWS_DEPLOY_ROLE_ARN" + missing=1 + fi + for ts_secret in TS_OAUTH_CLIENT_ID TS_AUDIENCE; do + if [ -z "${!ts_secret:-}" ]; then + echo "::error::Secret $ts_secret is not set. It comes from a Tailscale workload-identity trust credential for this repository with tag:ci." + missing=1 + fi + done + exit "$missing" - - name: Install Pulumi CLI - uses: pulumi/setup-pulumi@v2 + - name: Join tailnet + uses: tailscale/github-action@306e68a486fd2350f2bfc3b19fcd143891a4a2d8 # v4.1.2 + with: + oauth-client-id: ${{ secrets.TS_OAUTH_CLIENT_ID }} + audience: ${{ secrets.TS_AUDIENCE }} + tags: tag:ci + + - name: Accept tailnet routes to the cluster VPC + shell: bash + run: sudo tailscale set --accept-routes + + - name: Assume AWS deploy role + uses: aws-actions/configure-aws-credentials@7474bc4690e29a8392af63c5b98e7449536d5c3a # v4.3.1 + with: + role-to-assume: ${{ vars.AWS_DEPLOY_ROLE_ARN }} + aws-region: ${{ env.DEPLOY_AWS_REGION }} + + - name: Verify deploy role was assumed + shell: bash + env: + AWS_DEPLOY_ROLE_ARN: ${{ vars.AWS_DEPLOY_ROLE_ARN }} + run: | + set -euo pipefail + account_id="$(cut -d: -f5 <<<"$AWS_DEPLOY_ROLE_ARN")" + role_name="${AWS_DEPLOY_ROLE_ARN##*/}" + caller_arn="$(aws sts get-caller-identity --query Arn --output text)" + case "$caller_arn" in + "arn:aws:sts::${account_id}:assumed-role/${role_name}/"*) + echo "Deploying as $caller_arn" + ;; + *) + echo "::error::AWS calls run as $caller_arn, not as the deploy role $AWS_DEPLOY_ROLE_ARN." + exit 1 + ;; + esac - name: Configure kubeconfig + shell: bash + run: | + set -euo pipefail + kubeconfig_path="$RUNNER_TEMP/kubeconfig" + aws eks update-kubeconfig \ + --name "$DEPLOY_EKS_CLUSTER" \ + --region "$DEPLOY_AWS_REGION" \ + --kubeconfig "$kubeconfig_path" \ + --alias "$DEPLOY_EKS_CLUSTER" + echo "KUBECONFIG=$kubeconfig_path" >> "$GITHUB_ENV" + + - name: Verify cluster access shell: bash env: - KUBE_CONFIG_DATA: ${{ secrets.KUBE_CONFIG_DATA }} + TARGET_NAMESPACE: openerrata-${{ needs.resolve-target.outputs.deploy_environment }} run: | set -euo pipefail - if [ -z "${KUBE_CONFIG_DATA:-}" ]; then - echo "KUBE_CONFIG_DATA secret is required" - exit 1 + if kubectl_output="$(kubectl get namespace "$TARGET_NAMESPACE" -o name 2>&1)"; then + echo "Cluster access OK: $kubectl_output" + exit 0 fi + echo "$kubectl_output" + case "$kubectl_output" in + *Unauthorized*|*"provide credentials"*) + echo "::error::EKS cluster $DEPLOY_EKS_CLUSTER rejected the deploy role. Run src/aws/ci-iam/setup.sh, which creates the role's EKS access entry (Kubernetes group openerrata-ci)." + ;; + *[Ff]orbidden*) + echo "::error::The deploy role authenticated but may not read namespace $TARGET_NAMESPACE. Run src/kubernetes/ci-rbac/setup.sh, which binds the openerrata-ci group's RBAC." + ;; + *NotFound*) + echo "::error::Namespace $TARGET_NAMESPACE does not exist. Run src/kubernetes/ci-rbac/setup.sh, which creates it." + ;; + *"getting credentials"*) + echo "::error::kubectl could not get an EKS token from 'aws eks get-token' with the deploy role's credentials." + ;; + *) + echo "::error::Could not reach the private API endpoint of EKS cluster $DEPLOY_EKS_CLUSTER. Check that the tailnet joined with tag:ci and routes the cluster VPC to this runner." + ;; + esac + exit 1 - mkdir -p "$HOME/.kube" - TMP_CONFIG="$(mktemp)" - if printf '%s' "$KUBE_CONFIG_DATA" | base64 --decode > "$TMP_CONFIG" 2>/dev/null \ - && grep -q '^apiVersion:' "$TMP_CONFIG"; then - mv "$TMP_CONFIG" "$HOME/.kube/config" - else - rm -f "$TMP_CONFIG" - printf '%s\n' "$KUBE_CONFIG_DATA" > "$HOME/.kube/config" - fi - chmod 600 "$HOME/.kube/config" + - name: Setup TypeScript workspace + uses: ./.github/actions/setup-typescript-workspace + + - name: Install Pulumi CLI + uses: pulumi/setup-pulumi@b374ceb6168550de27c6eba92e01c1a774040e11 # v2.0.0 - name: Configure Pulumi stack values working-directory: src/typescript/pulumi @@ -503,7 +595,6 @@ jobs: PULUMI_ACCESS_TOKEN: ${{ secrets.PULUMI_ACCESS_TOKEN }} STACK: ${{ needs.resolve-target.outputs.deploy_environment }} PULUMI_BLOB_STORAGE_BUCKET: ${{ secrets.PULUMI_BLOB_STORAGE_BUCKET }} - PULUMI_BLOB_STORAGE_PUBLIC_URL_PREFIX: ${{ secrets.PULUMI_BLOB_STORAGE_PUBLIC_URL_PREFIX }} PULUMI_BLOB_STORAGE_ENDPOINT: ${{ secrets.PULUMI_BLOB_STORAGE_ENDPOINT }} PULUMI_DATABASE_URL: ${{ secrets.PULUMI_DATABASE_URL }} OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} @@ -516,11 +607,6 @@ jobs: PULUMI_CLOUDFLARE_PROXIED: ${{ vars.PULUMI_CLOUDFLARE_PROXIED }} PULUMI_CLOUDFLARE_RECORD_TARGET: ${{ vars.PULUMI_CLOUDFLARE_RECORD_TARGET }} CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }} - AWS_ACCESS_KEY_ID: ${{ secrets.AWS_ACCESS_KEY_ID }} - AWS_SECRET_ACCESS_KEY: ${{ secrets.AWS_SECRET_ACCESS_KEY }} - AWS_SESSION_TOKEN: ${{ secrets.AWS_SESSION_TOKEN }} - AWS_REGION: ${{ secrets.AWS_REGION }} - AWS_DEFAULT_REGION: ${{ secrets.AWS_REGION }} CI_IMAGE_REPOSITORY: ${{ needs.build-image.outputs.image_repository }} CI_IMAGE_TAG: ${{ needs.build-image.outputs.image_tag }} CI_IMAGE_DIGEST: ${{ needs.build-image.outputs.image_digest }} @@ -528,9 +614,17 @@ jobs: CI_FRONTEND_IMAGE_TAG: ${{ needs.build-image.outputs.frontend_image_tag }} CI_FRONTEND_IMAGE_DIGEST: ${{ needs.build-image.outputs.frontend_image_digest }} PULUMI_FRONTEND_HOSTNAME: ${{ vars.PULUMI_FRONTEND_HOSTNAME }} + PULUMI_MANAGED_DATABASE_PUBLICLY_ACCESSIBLE: ${{ vars.PULUMI_MANAGED_DATABASE_PUBLICLY_ACCESSIBLE }} + PULUMI_MANAGED_DATABASE_INGRESS_CIDRS: ${{ vars.PULUMI_MANAGED_DATABASE_INGRESS_CIDRS }} + PULUMI_MANAGED_DATABASE_ENGINE_VERSION: ${{ vars.PULUMI_MANAGED_DATABASE_ENGINE_VERSION }} + PULUMI_WORKER_EGRESS_POLICY_ENABLED: ${{ vars.PULUMI_WORKER_EGRESS_POLICY_ENABLED }} + PULUMI_WORKER_EGRESS_ALLOWED_CIDRS: ${{ vars.PULUMI_WORKER_EGRESS_ALLOWED_CIDRS }} + PULUMI_SELECTOR_DAILY_BUDGET: ${{ vars.PULUMI_SELECTOR_DAILY_BUDGET }} run: | set -euo pipefail + # Stack config is rebuilt from scratch on every run: this step is the + # complete configuration of the hosted stacks. pulumi stack select "$STACK" --create if [ -z "${CI_IMAGE_REPOSITORY:-}" ] || [ -z "${CI_IMAGE_TAG:-}" ] || [ -z "${CI_IMAGE_DIGEST:-}" ]; then @@ -589,6 +683,15 @@ jobs: pulumi config rm ingressClassName --stack "$STACK" >/dev/null 2>&1 || true fi + # selectorBudget (per selector run) was replaced by selectorDailyBudget + # (per UTC day); the Pulumi program refuses the old key. + pulumi config rm selectorBudget --stack "$STACK" >/dev/null 2>&1 || true + if [ -n "${PULUMI_SELECTOR_DAILY_BUDGET:-}" ]; then + pulumi config set selectorDailyBudget "$PULUMI_SELECTOR_DAILY_BUDGET" --stack "$STACK" + else + pulumi config rm selectorDailyBudget --stack "$STACK" >/dev/null 2>&1 || true + fi + dns_provider="${PULUMI_DNS_PROVIDER:-none}" case "$dns_provider" in none|"") @@ -622,11 +725,9 @@ jobs: ;; esac - needs_aws_resources=false - manual_blob_vars=( PULUMI_BLOB_STORAGE_BUCKET - PULUMI_BLOB_STORAGE_PUBLIC_URL_PREFIX + PULUMI_BLOB_STORAGE_ACCESS_KEY_ID PULUMI_BLOB_STORAGE_SECRET_ACCESS_KEY ) configured_manual_blob_values=0 @@ -636,17 +737,16 @@ jobs: fi done - if [ "$configured_manual_blob_values" -ne 0 ] && [ "$configured_manual_blob_values" -ne 3 ]; then + if [ "$configured_manual_blob_values" -ne 0 ] && [ "$configured_manual_blob_values" -ne "${#manual_blob_vars[@]}" ]; then echo "Manual blob storage config is partial." - echo "Set all or none of: PULUMI_BLOB_STORAGE_BUCKET, PULUMI_BLOB_STORAGE_PUBLIC_URL_PREFIX, PULUMI_BLOB_STORAGE_SECRET_ACCESS_KEY" + echo "Set all or none of: ${manual_blob_vars[*]}" exit 1 fi - if [ "$configured_manual_blob_values" -eq 3 ]; then + if [ "$configured_manual_blob_values" -eq "${#manual_blob_vars[@]}" ]; then echo "Using manually provided blob storage configuration." pulumi config set blobStorageBucket "$PULUMI_BLOB_STORAGE_BUCKET" --stack "$STACK" - pulumi config set blobStoragePublicUrlPrefix "$PULUMI_BLOB_STORAGE_PUBLIC_URL_PREFIX" --stack "$STACK" - pulumi config set blobStorageAccessKeyId "${PULUMI_BLOB_STORAGE_ACCESS_KEY_ID:-openerrata}" --stack "$STACK" + pulumi config set blobStorageAccessKeyId "$PULUMI_BLOB_STORAGE_ACCESS_KEY_ID" --stack "$STACK" pulumi config set --secret blobStorageSecretAccessKey "$PULUMI_BLOB_STORAGE_SECRET_ACCESS_KEY" --stack "$STACK" if [ -n "${PULUMI_BLOB_STORAGE_ENDPOINT:-}" ]; then pulumi config set blobStorageEndpoint "$PULUMI_BLOB_STORAGE_ENDPOINT" --stack "$STACK" @@ -655,9 +755,7 @@ jobs: fi else echo "No manual blob storage credentials supplied; using Pulumi-managed AWS S3 blob storage." - needs_aws_resources=true pulumi config rm blobStorageBucket --stack "$STACK" >/dev/null 2>&1 || true - pulumi config rm blobStoragePublicUrlPrefix --stack "$STACK" >/dev/null 2>&1 || true pulumi config rm blobStorageEndpoint --stack "$STACK" >/dev/null 2>&1 || true pulumi config rm blobStorageAccessKeyId --stack "$STACK" >/dev/null 2>&1 || true pulumi config rm blobStorageSecretAccessKey --stack "$STACK" >/dev/null 2>&1 || true @@ -668,52 +766,34 @@ jobs: pulumi config set --secret databaseUrl "$PULUMI_DATABASE_URL" --stack "$STACK" else echo "No database URL supplied; using Pulumi-managed AWS RDS database." - needs_aws_resources=true pulumi config rm databaseUrl --stack "$STACK" >/dev/null 2>&1 || true + # No defaults: network exposure and engine version of a live database + # must be stated, not inherited. The hosted stacks use false, + # 10.0.0.0/16 (the cluster VPC) and 17; see README "Hosted deploy setup". + for managed_db_var in PULUMI_MANAGED_DATABASE_PUBLICLY_ACCESSIBLE PULUMI_MANAGED_DATABASE_INGRESS_CIDRS PULUMI_MANAGED_DATABASE_ENGINE_VERSION; do + if [ -z "${!managed_db_var:-}" ]; then + echo "$managed_db_var (repository/environment variable) is required for the Pulumi-managed RDS database." + exit 1 + fi + done + pulumi config set managedDatabasePubliclyAccessible "$PULUMI_MANAGED_DATABASE_PUBLICLY_ACCESSIBLE" --stack "$STACK" + pulumi config set managedDatabaseIngressCidrs "$PULUMI_MANAGED_DATABASE_INGRESS_CIDRS" --stack "$STACK" + pulumi config set managedDatabaseEngineVersion "$PULUMI_MANAGED_DATABASE_ENGINE_VERSION" --stack "$STACK" fi - if [ "$needs_aws_resources" = "true" ]; then - if [ -z "${AWS_ACCESS_KEY_ID:-}" ] || [ -z "${AWS_SECRET_ACCESS_KEY:-}" ]; then - echo "AWS_ACCESS_KEY_ID and AWS_SECRET_ACCESS_KEY are required when Pulumi-managed AWS resources are enabled." - exit 1 - fi - pulumi config set aws:region "${AWS_REGION:-${AWS_DEFAULT_REGION:-us-east-1}}" --stack "$STACK" + pulumi config set workerEgressPolicyEnabled "${PULUMI_WORKER_EGRESS_POLICY_ENABLED:-false}" --stack "$STACK" + if [ -n "${PULUMI_WORKER_EGRESS_ALLOWED_CIDRS:-}" ]; then + pulumi config set workerEgressAllowedCidrs "$PULUMI_WORKER_EGRESS_ALLOWED_CIDRS" --stack "$STACK" fi + # Pulumi-managed AWS resources live in the cluster's region, and the + # deploy role's credentials (assumed above) are scoped to it. + pulumi config set aws:region "$DEPLOY_AWS_REGION" --stack "$STACK" + if [ -n "${OPENAI_API_KEY:-}" ]; then pulumi config set --secret openaiApiKey "$OPENAI_API_KEY" --stack "$STACK" fi - - name: Cancel stale Pulumi lock - working-directory: src/typescript/pulumi - shell: bash - env: - PULUMI_ACCESS_TOKEN: ${{ secrets.PULUMI_ACCESS_TOKEN }} - STACK: ${{ needs.resolve-target.outputs.deploy_environment }} - run: | - set +e - output="$(pulumi cancel --stack "$STACK" --yes 2>&1)" - rc=$? - set -e - echo "$output" - - if [ $rc -eq 0 ]; then - exit 0 - fi - - if echo "$output" | grep -qi "no update in progress"; then - echo "No active lock; continuing." - exit 0 - fi - - if echo "$output" | grep -qi "has never been updated"; then - echo "Stack has never been updated; no lock to cancel." - exit 0 - fi - - echo "pulumi cancel failed unexpectedly" - exit $rc - - name: Delete stale selector/migrate jobs shell: bash env: @@ -734,12 +814,6 @@ jobs: env: PULUMI_ACCESS_TOKEN: ${{ secrets.PULUMI_ACCESS_TOKEN }} STACK: ${{ needs.resolve-target.outputs.deploy_environment }} - PULUMI_K8S_DELETE_UNREACHABLE: "true" - AWS_ACCESS_KEY_ID: ${{ secrets.AWS_ACCESS_KEY_ID }} - AWS_SECRET_ACCESS_KEY: ${{ secrets.AWS_SECRET_ACCESS_KEY }} - AWS_SESSION_TOKEN: ${{ secrets.AWS_SESSION_TOKEN }} - AWS_REGION: ${{ secrets.AWS_REGION }} - AWS_DEFAULT_REGION: ${{ secrets.AWS_REGION }} CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }} run: | set -euo pipefail @@ -781,16 +855,13 @@ jobs: kubectl -n "$TARGET_NAMESPACE" get events --sort-by=.lastTimestamp | tail -n 200 || true echo "::endgroup::" - echo "::group::Cancel in-flight Pulumi update" - set +e - cancel_output="$(pulumi cancel --stack "$STACK" --yes 2>&1)" - cancel_rc=$? - set -e - echo "$cancel_output" - if [ "$cancel_rc" -ne 0 ] && ! echo "$cancel_output" | grep -qi "no update in progress"; then - echo "::warning::pulumi cancel returned exit code ${cancel_rc}" + if [ "$pulumi_rc" -eq 124 ] || [ "$pulumi_rc" -eq 137 ]; then + # The concurrency group guarantees the only update in progress on + # this stack is the one this job just timed out; release its lock. + echo "::group::Cancel this run's timed-out Pulumi update" + pulumi cancel --stack "$STACK" --yes || echo "::warning::pulumi cancel failed; the stack may stay locked" + echo "::endgroup::" fi - echo "::endgroup::" exit "$pulumi_rc" @@ -804,14 +875,14 @@ jobs: packages: write steps: - name: Login to GHCR - uses: docker/login-action@v3 + uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0 with: registry: ghcr.io username: ${{ github.actor }} password: ${{ secrets.GITHUB_TOKEN }} - name: Set up Docker Buildx - uses: docker/setup-buildx-action@v3 + uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0 - name: Promote deployed API image to latest tag shell: bash @@ -839,7 +910,7 @@ jobs: working-directory: src/typescript steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup TypeScript workspace uses: ./.github/actions/setup-typescript-workspace diff --git a/.github/workflows/pr-checks.yml b/.github/workflows/pr-checks.yml index 3462a37..3a331a8 100644 --- a/.github/workflows/pr-checks.yml +++ b/.github/workflows/pr-checks.yml @@ -22,11 +22,11 @@ jobs: has_relevant_changes: ${{ steps.changes.outputs.relevant }} steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Detect relevant file changes id: changes - uses: dorny/paths-filter@v3 + uses: dorny/paths-filter@0e4a8c6effa4802afeda77dc8d303f8176d7dfad # v3.0.4 with: filters: | relevant: @@ -61,10 +61,13 @@ jobs: timeout-minutes: 10 steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup Helm - uses: azure/setup-helm@v4 + uses: azure/setup-helm@1a275c3b69536ee54be43f2070a358922e12c8d4 # v4.3.1 + with: + # Helm 3, the engine Pulumi's helm.v3.Chart renders the chart with. + version: v3.22.0 - name: Lint chart shell: bash @@ -87,7 +90,7 @@ jobs: DATABASE_URL: postgresql://postgres:postgres@localhost:5432/postgres steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup TypeScript workspace uses: ./.github/actions/setup-typescript-workspace @@ -109,7 +112,7 @@ jobs: DATABASE_URL: postgresql://postgres:postgres@localhost:5432/postgres steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup TypeScript workspace uses: ./.github/actions/setup-typescript-workspace @@ -143,7 +146,7 @@ jobs: --health-retries 5 steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup TypeScript workspace uses: ./.github/actions/setup-typescript-workspace @@ -165,7 +168,7 @@ jobs: DATABASE_URL: postgresql://postgres:postgres@localhost:5432/postgres steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup TypeScript workspace uses: ./.github/actions/setup-typescript-workspace @@ -178,7 +181,7 @@ jobs: run: tar -C extension -czf "$RUNNER_TEMP/extension-dist.tgz" dist - name: Upload extension dist artifact - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 with: name: extension-dist-${{ github.sha }} path: ${{ runner.temp }}/extension-dist.tgz @@ -189,7 +192,7 @@ jobs: run: tar -C frontend -czf "$RUNNER_TEMP/frontend-build.tgz" build - name: Upload frontend build artifact - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 with: name: frontend-build-${{ github.sha }} path: ${{ runner.temp }}/frontend-build.tgz @@ -239,7 +242,7 @@ jobs: working-directory: src/typescript steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup TypeScript workspace uses: ./.github/actions/setup-typescript-workspace @@ -248,7 +251,7 @@ jobs: install-playwright-chromium: "true" - name: Download extension dist artifact - uses: actions/download-artifact@v4 + uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0 with: name: extension-dist-${{ github.sha }} path: ${{ runner.temp }} @@ -272,7 +275,7 @@ jobs: working-directory: src/typescript steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup TypeScript workspace uses: ./.github/actions/setup-typescript-workspace @@ -281,7 +284,7 @@ jobs: install-playwright-chromium: "true" - name: Download frontend build artifact - uses: actions/download-artifact@v4 + uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0 with: name: frontend-build-${{ github.sha }} path: ${{ runner.temp }} diff --git a/.github/workflows/release-extension.yml b/.github/workflows/release-extension.yml index a58c23c..4aa4a3f 100644 --- a/.github/workflows/release-extension.yml +++ b/.github/workflows/release-extension.yml @@ -38,7 +38,7 @@ jobs: --health-retries 5 steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0 - name: Setup TypeScript workspace uses: ./.github/actions/setup-typescript-workspace @@ -93,7 +93,7 @@ jobs: run: npx -y web-ext@8.9.0 lint --source-dir extension/dist/firefox - name: Upload packaged extension artifacts - uses: actions/upload-artifact@v4 + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 with: name: openerrata-extension-packages-${{ steps.version.outputs.version }} path: | @@ -110,13 +110,13 @@ jobs: contents: write steps: - name: Download packaged extension artifacts - uses: actions/download-artifact@v4 + uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0 with: name: openerrata-extension-packages-${{ needs.build-and-test.outputs.version }} path: . - name: Create GitHub Release - uses: softprops/action-gh-release@v2 + uses: softprops/action-gh-release@3bb12739c298aeb8a4eeaf626c5b8d85266b0e65 # v2.6.2 with: files: | openerrata-extension-chrome-${{ needs.build-and-test.outputs.version }}.zip @@ -131,25 +131,48 @@ jobs: timeout-minutes: 20 environment: extension-store-publish steps: + # Missing credentials fail the release instead of silently skipping a store. + - name: Check store credentials + shell: bash + env: + CHROME_EXTENSION_ID: ${{ secrets.CHROME_EXTENSION_ID }} + CHROME_CLIENT_ID: ${{ secrets.CHROME_CLIENT_ID }} + CHROME_CLIENT_SECRET: ${{ secrets.CHROME_CLIENT_SECRET }} + CHROME_REFRESH_TOKEN: ${{ secrets.CHROME_REFRESH_TOKEN }} + FIREFOX_JWT_ISSUER: ${{ secrets.FIREFOX_JWT_ISSUER }} + FIREFOX_JWT_SECRET: ${{ secrets.FIREFOX_JWT_SECRET }} + PUBLISH_TO_FIREFOX_AMO: ${{ vars.PUBLISH_TO_FIREFOX_AMO }} + run: | + set -euo pipefail + required=(CHROME_EXTENSION_ID CHROME_CLIENT_ID CHROME_CLIENT_SECRET CHROME_REFRESH_TOKEN) + if [ "${PUBLISH_TO_FIREFOX_AMO:-}" = "true" ]; then + required+=(FIREFOX_JWT_ISSUER FIREFOX_JWT_SECRET) + else + echo "::notice::Not publishing to Firefox Add-ons (environment variable PUBLISH_TO_FIREFOX_AMO is not true)." + fi + missing=() + for name in "${required[@]}"; do + if [ -z "${!name:-}" ]; then missing+=("$name"); fi + done + if [ "${#missing[@]}" -gt 0 ]; then + echo "::error::Missing extension-store-publish secrets: ${missing[*]}" + exit 1 + fi + - name: Download packaged extension artifacts - uses: actions/download-artifact@v4 + uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0 with: name: openerrata-extension-packages-${{ needs.build-and-test.outputs.version }} path: . - name: Publish to Chrome Web Store - if: >- - env.CHROME_EXTENSION_ID != '' && - env.CHROME_CLIENT_ID != '' && - env.CHROME_CLIENT_SECRET != '' && - env.CHROME_REFRESH_TOKEN != '' env: CHROME_EXTENSION_ID: ${{ secrets.CHROME_EXTENSION_ID }} CHROME_CLIENT_ID: ${{ secrets.CHROME_CLIENT_ID }} CHROME_CLIENT_SECRET: ${{ secrets.CHROME_CLIENT_SECRET }} CHROME_REFRESH_TOKEN: ${{ secrets.CHROME_REFRESH_TOKEN }} run: | - npx -y chrome-webstore-upload-cli@3 upload \ + npx -y chrome-webstore-upload-cli@3.5.0 upload \ --source "openerrata-extension-chrome-${{ needs.build-and-test.outputs.version }}.zip" \ --extension-id "$CHROME_EXTENSION_ID" \ --client-id "$CHROME_CLIENT_ID" \ @@ -157,10 +180,10 @@ jobs: --refresh-token "$CHROME_REFRESH_TOKEN" \ --auto-publish + # The AMO listing is not live yet; set PUBLISH_TO_FIREFOX_AMO=true on the + # extension-store-publish environment once it is. - name: Publish to Firefox Add-ons - if: >- - env.FIREFOX_JWT_ISSUER != '' && - env.FIREFOX_JWT_SECRET != '' + if: vars.PUBLISH_TO_FIREFOX_AMO == 'true' env: FIREFOX_JWT_ISSUER: ${{ secrets.FIREFOX_JWT_ISSUER }} FIREFOX_JWT_SECRET: ${{ secrets.FIREFOX_JWT_SECRET }} diff --git a/AGENTS.md b/AGENTS.md index c7418a9..b49d2a9 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -10,17 +10,22 @@ behavior. ``` src/ +├── aws/ci-iam/ # Bootstrap of the IAM role hosted deploys assume (GitHub OIDC) ├── helm/openerrata/ # Helm chart (single deployment artifact for on-prem + hosted) +├── kubernetes/ci-rbac/ # RBAC for the hosted deploy role's Kubernetes group └── typescript/ ├── shared/ # @openerrata/shared — types, Zod schemas, normalization ├── api/ # @openerrata/api — SvelteKit + tRPC backend, Prisma, job queue ├── extension/ # @openerrata/extension — Chrome MV3 browser extension + ├── frontend/ # @openerrata/frontend — public SvelteKit website (reads the public GraphQL API) └── pulumi/ # @openerrata/pulumi — deploys the Helm chart for hosted env ``` -The monorepo uses pnpm workspaces. Dependencies flow: `shared` → `api` and -`shared` → `extension`. The extension imports the API's `AppRouter` type (type-only, -no runtime code) for tRPC client typing. +The monorepo uses pnpm workspaces. Dependencies flow: `shared` → `api`, +`shared` → `extension` and `shared` → `frontend`. The extension does not import +API code: its tRPC calls are typed by the shared `ExtensionApiProcedureContract` +(which the API asserts its routes against) and validated with the shared output +schemas. ## Key Files @@ -48,7 +53,7 @@ these ways. Always take the time to introduce new features and fix bugs in the codebase in the most complete, ideal, and maintainable way, even if it means modifying more than your user initially expects. Feel free to perform refactors of bad code -not originally mentioned in your prompt. +not originally mentioned in your prompt. **The underlying driving spec behind the software you're tasked with writing should be clearly understandable just by reading your code.** This is the most @@ -94,6 +99,7 @@ var asyncLoadedDataError: str | undefined = undefined ``` Better: + ``` type StreetLightStatus = enum { RED @@ -125,22 +131,23 @@ When key assumptions that your code relies upon to work appear to be broken, and you cannot use the type system or good architectural design to remove the possibility of errors entirely (which is always the first preference), fail early and visibly, rather than attempting to patch things up. In particular: -* Lean towards propagating errors up to callers, instead of silently "warning" + +- Lean towards propagating errors up to callers, instead of silently "warning" about them inside of try/catch blocks. -* Push error handling to type systems where possible by specifying stricter +- Push error handling to type systems where possible by specifying stricter input & output types instead of attempting to throw errors or create fallbacks -* If you are fairly certain data should always exist, assume it does, rather +- If you are fairly certain data should always exist, assume it does, rather than producing code with unnecessary guardrails or existence checks (esp. if such checks might mislead other programmers) -* Avoid the use of hasattr/getattr (or non-python equivalents) when accessing +- Avoid the use of hasattr/getattr (or non-python equivalents) when accessing attributes and fields that should always exist. -* Never produce knowingly incorrect 'defaults' as a result of errors or missing +- Never produce knowingly incorrect 'defaults' as a result of errors or missing data, either for users, or downstream callers. Do not assume your user has context about system architecture, files, line numbers, or programming concepts. Re-explain these details as necessary if they -have not already come up. This is also helpful for verifying that *you* +have not already come up. This is also helpful for verifying that _you_ understand what's going on. ## Testing Philosophy @@ -148,15 +155,17 @@ understand what's going on. The purpose of tests is to automate the collection of evidence that the explicit + implicit spec is being adhered to. This can be accomplished at multiple levels: + - Unit tests of individual or groups of functions that are required to operate correctly in order for the program to work well, like an `assert text == - decompress(compress(text))` test of a compression lib +decompress(compress(text))` test of a compression lib - Mocks of components that have individual properties that we want to verify, like a guard that panics() on invalid SQL statements sent to the DB - Integreation tests of direct components of the SPEC.md - End to end tests of browser functionality on previously downloaded pages Our primary test suite should be: + - Fast, easy to run - Produce few false negatives as possible (i.e., fail the code when it's behaving poorly) - Produce as few false positives as possible (i.e., fail the code when it complies with the spec). This can happen because either: @@ -174,8 +183,10 @@ test coverage of the code high. ## Dev Environment ```bash -# Start local Postgres (port 5433) + MinIO S3-compatible storage (9000 API / 9001 console) +# Start local Postgres (port 5433) + S3-compatible blob storage (port 7070) docker compose up -d +# API/worker/selector configuration; dotenv reads it from the api package directory +cp src/typescript/api/.env.example src/typescript/api/.env ``` For more information about how to run and manage the typescript extensions, see ./src/typescript/AGENTS.md diff --git a/PRIVACY.md b/PRIVACY.md index 9e976c0..3214713 100644 --- a/PRIVACY.md +++ b/PRIVACY.md @@ -1,16 +1,17 @@ # OpenErrata Privacy Policy -**Effective date:** February 22, 2026 +**Effective date:** October 2, 2026 OpenErrata is a browser extension that investigates web content for factual -accuracy using large language models. This policy describes what data the -extension collects, how it is used, and how it is stored. +accuracy using large language models, plus a public website that shows the +results. This policy describes what data the extension and website collect, how +it is used, and how it is stored. ## What Data We Collect ### Post content you visit -When you visit a supported page (LessWrong, X/Twitter, or Substack), the +When you visit a supported page (LessWrong, X/Twitter, Substack, or Wikipedia), the extension extracts the post's text, images, and public metadata (title, author name/handle, publication date, tags, engagement counts) and sends it to the OpenErrata API server for investigation. Only content on supported platforms is @@ -72,7 +73,14 @@ abuse. investigation. OpenAI's data usage policies apply to that processing. See [OpenAI's privacy policy](https://openai.com/privacy). - **Platform APIs** — The server may fetch canonical post content from - LessWrong's public GraphQL API to verify content authenticity. + LessWrong's public GraphQL API and Wikipedia's public API to verify content + authenticity. +- **Cloudflare** — The hosted API and website are served through Cloudflare, + which handles the network connection (including your IP address) to deliver + requests. See [Cloudflare's privacy policy](https://www.cloudflare.com/privacypolicy/). +- **Google Fonts** — The website loads its typeface from Google Fonts, so your + browser requests it from Google's servers when you open the site. See + [Google's privacy policy](https://policies.google.com/privacy). No data is shared with advertising networks, data brokers, analytics providers, or any other third parties. @@ -95,6 +103,12 @@ or any other third parties. server and deleted after use. - The extension requests only the permissions necessary for its operation. +## Website + +The OpenErrata website does not use cookies, accounts, analytics, or tracking. +Search terms you enter are sent to the OpenErrata API to find matching +investigations. + ## Public Investigations OpenErrata is designed for transparency. All completed investigations — diff --git a/README.md b/README.md index 221e574..86dd147 100644 --- a/README.md +++ b/README.md @@ -20,9 +20,10 @@ inspectable by users. 2. **Posts get investigated.** You or someone else clicks "Investigate Now", or has the "auto-investigate posts" setting checked, or the service pre-selects it using a very-likely-to-be-read heuristic. -3. **The LLM investigates.** The full post text (plus images) are sent to - GPT-5.4-thinking, which uses native web search and browsing tools to verify claims. - Only demonstrably incorrect claims are flagged — disputed, ambiguous, or +3. **The LLM investigates.** The full post text (plus images) is sent to + OpenAI's `gpt-6.1-sol`, which uses native web search and browsing tools to + verify claims. Each flagged claim is then re-checked in a separate validation + pass. Only demonstrably incorrect claims are flagged — disputed, ambiguous, or unverifiable claims are left alone. 4. **Incorrect claims are highlighted.** For all extension users, every incorrect sentence gets a red underline in the post. Hover for a summary; @@ -47,9 +48,8 @@ inspectable by users. Errata are first pulled from the public instance before being regenerated. If you do not have an OpenAI API key, you can receive the errata generated by -other users, but they won't be generated on demand. Right now the public -instance is configured to investigate the most popular posts every hour (though -this may change as we hit spend thresholds). +other users, but they won't be generated on demand. The public instance also +investigates the most-viewed posts on its own, up to a daily budget. To point the extension at a self-hosted instance, open the extension options page and change the API URL. @@ -63,12 +63,14 @@ page and change the API URL. | Substack | URL match (`*.substack.com/p/*`) + DOM fingerprint for custom domains | | Wikipedia | URL match (`*.wikipedia.org/wiki/*`, `*.wikipedia.org/w/index.php*`) | -## Public API +## Public API and site All complete investigations are publicly accessible via GraphQL at `POST -/graphql`. No authentication required. Responses include trust signals (content -provenance, corroboration count, server verification timestamps) so consumers -can apply their own trust policy. +/graphql` on the API. No authentication required. Responses include trust +signals (content provenance, corroboration count, server verification +timestamps) so consumers can apply their own trust policy. The public website +(`src/typescript/frontend`) is a client of this API: it lists corrections and +shows each investigation's claims, reasoning and sources. ## Design Principles @@ -82,7 +84,8 @@ can apply their own trust policy. model investigates. - **Single-pass investigation.** The entire post is investigated in one agentic call, giving the model full context for understanding caveats, - qualifications, and claim relationships. + qualifications, and claim relationships; each resulting claim is then + validated on its own. ## Documentation @@ -92,15 +95,19 @@ can apply their own trust policy. ``` src/ +├── aws/ci-iam/ # Bootstrap of the IAM role hosted deploys assume (GitHub OIDC) ├── helm/openerrata/ # Helm chart (single artifact for on-prem + hosted) +├── kubernetes/ci-rbac/ # RBAC for the hosted deploy role's Kubernetes group └── typescript/ ├── shared/ # @openerrata/shared — types, Zod schemas, normalization ├── api/ # @openerrata/api — SvelteKit + tRPC backend, Prisma, job queue ├── extension/ # @openerrata/extension — WebExtension (Chrome + Firefox) + ├── frontend/ # @openerrata/frontend — public SvelteKit website └── pulumi/ # @openerrata/pulumi — deploys the Helm chart for hosted env ``` -The monorepo uses **pnpm workspaces**. Dependencies flow: `shared` -> `api` and `shared` -> `extension`. +The monorepo uses **pnpm workspaces**. Dependencies flow from `shared` to +`api`, `extension` and `frontend`. ## Tech Stack @@ -111,6 +118,7 @@ The monorepo uses **pnpm workspaces**. Dependencies flow: `shared` -> `api` and | Cross-browser | webextension-polyfill | | Type safety | TypeScript + Zod | | API | SvelteKit + tRPC (internal) + GraphQL (public) | +| Public website | SvelteKit (server-rendered, reads the public GraphQL API) | | Database | Postgres + Prisma | | Job queue | Postgres-backed (graphile-worker) | | LLM | OpenAI Responses API with native tool use | @@ -122,18 +130,22 @@ The monorepo uses **pnpm workspaces**. Dependencies flow: `shared` -> `api` and - [Node.js](https://nodejs.org/) (LTS) - [pnpm](https://pnpm.io/) -- [Docker](https://www.docker.com/) (for local Postgres + MinIO) +- [Docker](https://www.docker.com/) (for local Postgres + S3-compatible blob storage) ### Local Development ```bash -# Start Postgres (port 5433) and MinIO S3-compatible storage (ports 9000/9001) +# Start Postgres (port 5433) and S3-compatible blob storage (port 7070) docker compose up -d # Install dependencies cd src/typescript pnpm install +# Configure the API, worker and selector. dotenv reads api/.env; the example +# matches docker-compose. Set OPENAI_API_KEY in it to run the worker. +cp api/.env.example api/.env + # Apply database migrations pnpm db:migrate @@ -145,6 +157,9 @@ pnpm dev:ext # Start the job queue worker (separate terminal) pnpm worker + +# Start the public website against the local API (separate terminal) +API_BASE_URL=http://localhost:5173 pnpm dev:frontend ``` ### Common Commands @@ -177,16 +192,49 @@ building to control the generated `browser_specific_settings.gecko.id`. ## Deployment -The Helm chart at `src/helm/openerrata/` is the single deployment artifact for both on-prem and hosted environments. It does not bundle a database — it takes a `DATABASE_URL` as config. +The Helm chart at `src/helm/openerrata/` is the single deployment artifact for both on-prem and hosted environments. It does not bundle a database or blob storage — it takes a `DATABASE_URL` and an S3-compatible bucket as config. `values.yaml` documents every value; these are required: **On-prem:** ```bash helm install openerrata ./src/helm/openerrata \ + --set image.tag=main-latest \ + --set config.blobStorageProvider=aws \ + --set config.blobStorageRegion=us-east-1 \ + --set config.blobStorageBucket=my-openerrata-images \ --set secrets.databaseUrl="postgresql://..." \ - --set secrets.openaiApiKey="sk-..." + --set secrets.openaiApiKey="sk-..." \ + --set secrets.databaseEncryptionKey="$(openssl rand -hex 32)" \ + --set secrets.blobStorageAccessKeyId="..." \ + --set secrets.blobStorageSecretAccessKey="..." ``` -**Hosted:** The official hosted deployment uses Pulumi (`src/typescript/pulumi/`) to deploy the same Helm chart with hosted-specific overrides (Supabase connection, domain, TLS, autoscaling). This guarantees identical workload definitions between on-prem and hosted — no deployment drift. +Images are published as `ghcr.io/zeropathai/openerrata-api` (and `-frontend`) with `main-` and `main-latest` tags. Database migrations run as a Helm hook, after the first install and before each upgrade; workload pods wait for them, so don't pass `--wait` to the first `helm install`. The public website is off by default (`frontend.enabled`). + +**Hosted:** The official hosted deployment uses Pulumi (`src/typescript/pulumi/`) to deploy the same Helm chart, plus the hostnames, Cloudflare DNS, and (optionally) a managed RDS database and S3 bucket. `.github/workflows/deploy.yml` configures the `staging` and `main` stacks on every deploy from repository and environment secrets and variables. This guarantees identical workload definitions between on-prem and hosted — no deployment drift. + +The deploy job holds no long-lived cloud credentials. It requests a GitHub OIDC token, whose subject names its `main` or `staging` deployment environment, and uses it twice: + +- to join the tailnet as `tag:ci`, because the EKS cluster `cunningham` (`us-west-2`) has a private API endpoint; +- to assume the IAM role `openerrata-github-actions-deploy`, whose policy covers only the AWS resources Pulumi manages. The role's EKS access entry maps it to the Kubernetes group `openerrata-ci`, which `src/kubernetes/ci-rbac/rbac.yaml` allows into the `openerrata-main` and `openerrata-staging` namespaces only. + +The managed RDS databases accept connections only from the cluster VPC (`10.0.0.0/16`, over VPC peering) and are not publicly accessible; reach them from inside the cluster. The managed blob buckets block all public access. + +### Hosted deploy setup + +Run these in order. Steps 1–2 need admin access and are idempotent; rerun them after changing the IAM policy or RBAC. + +1. **AWS role.** With an admin AWS session for the account hosting `cunningham`, run `src/aws/ci-iam/setup.sh`. It creates or updates the role, its policy and its EKS access entry, prints the role ARN to stdout, and prints (without running) the commands that delete the retired `openerrata-ci` IAM user. +2. **Kubernetes RBAC.** With a cluster-admin kubeconfig for `cunningham` (on the tailnet), run `src/kubernetes/ci-rbac/setup.sh`. It creates the two namespaces, binds the RBAC to group `openerrata-ci`, and deletes the retired `openerrata-ci` ServiceAccount and its token Secret. +3. **Tailscale.** Create a workload-identity (OIDC) trust credential with issuer GitHub Actions, tag `tag:ci`, and a subject matching the deploy job's tokens (`repo:ZeroPathAI/OpenErrata:environment:main` and `repo:ZeroPathAI/OpenErrata:environment:staging`). Store its client ID and audience as the repository secrets `TS_OAUTH_CLIENT_ID` and `TS_AUDIENCE`. +4. **Repository variables.** Set + - `AWS_DEPLOY_ROLE_ARN` to the ARN from step 1; + - `PULUMI_MANAGED_DATABASE_PUBLICLY_ACCESSIBLE=false`; + - `PULUMI_MANAGED_DATABASE_INGRESS_CIDRS=10.0.0.0/16`; + - `PULUMI_MANAGED_DATABASE_ENGINE_VERSION=17`. + + The deploy job fails early if the role variable or either Tailscale secret is missing, and the Pulumi step fails if a managed-database variable is. +5. **Deploy** `staging` (a push to the `staging` branch, or a manual run of the Deploy workflow), check it, then `main`. +6. **Retire the static credentials.** Delete the repository secrets `AWS_ACCESS_KEY_ID`, `AWS_SECRET_ACCESS_KEY`, `AWS_REGION` and `KUBE_CONFIG_DATA`, which no workflow reads any more, and run the cleanup commands step 1 printed. ## License diff --git a/SPEC.md b/SPEC.md index 8cfb727..f974c50 100644 --- a/SPEC.md +++ b/SPEC.md @@ -157,8 +157,9 @@ The following are explicitly out of scope: | **Background Worker** | Service worker. Routes messages between content scripts, popup, and the API. Manages local cache. Can auto-trigger `investigateNow` when user key mode is enabled. | | **Popup** | UI for extension state: toggle, summary of current page, settings. | | **OpenErrata API** | Records post views, serves cached investigations, runs selection, and exposes both internal RPC endpoints and a public GraphQL API. All investigations execute asynchronously through the queue (including user-supplied key requests). | -| **Blob Storage** | Stores downloaded investigation-time images (hash-deduplicated) and serves public URLs used in multimodal model input. | -| **Investigation Selector** | Cron job that periodically selects uninvestigated posts with the highest capped unique-view score and enqueues them. Pluggable selection algorithm — v1 uses capped unique-view score; future versions can factor in recency, engagement, author, etc. | +| **Blob Storage** | Stores downloaded investigation-time images (hash-deduplicated). The model receives the image bytes inline, not blob URLs. | +| **Investigation Selector** | Cron job that periodically admits uninvestigated posts with the highest capped unique-view score, up to a per-UTC-day budget, and enqueues them. Pluggable selection algorithm — v1 uses capped unique-view score; future versions can factor in recency, engagement, author, etc. | +| **Public Website** | Server-rendered SvelteKit site listing corrections and showing each investigation's claims, reasoning, and sources. A read-only client of the public GraphQL API (§3.4); it has no database access. | ## 2.4 LLM Investigation Approach @@ -251,9 +252,14 @@ The investigator prompt uses a single content section: Markdown is produced from version-scoped HTML (`HtmlBlob` referenced by `*VersionMeta` rows), then snapshotted into immutable `InvestigationInput` -(`markdown`, `markdownSource`, `markdownRendererVersion`) on first execution. -Retries reuse that snapshot verbatim, so investigation input is stable across -attempts even if markdown conversion logic changes later. +(`markdown`, `markdownSource`, `markdownRendererVersion`) when the investigation +is created, together with the source URL behind each `[IMAGE:N]` placeholder +(resolved to an absolute URL against the post URL; images without a fetchable +source get no placeholder) and the prompt context (post URL, author, publication +time, video flag). The investigator matches placeholders to downloaded images by +source URL, never by position. Every attempt reuses that snapshot verbatim, so +investigation input is stable across attempts even if markdown conversion logic or +the live post metadata change later. Claim `text` and `context` still must anchor against normalized post text in the extension, so markdown formatting characters are stripped from model outputs @@ -288,7 +294,11 @@ LessWrong/SubStack posts. LLMs are bad at counting characters, so we don't ask for offsets. The model returns the **exact claim text** plus **surrounding context** (~10 words before and after). The extension matches claims -to DOM positions using: +against the same normalized content text it reported to the API (one text index over the live +content root, block boundaries included as word breaks — §3.8 "Content normalization"), so claim +and context strings line up with the page exactly as they did for the investigator. Matches map +back to text-node pieces, which are highlighted individually; removing highlights restores the +page's own text nodes (it never merges text nodes the page owns). Matching uses: 1. **Exact substring match** — search for the claim text in the post content. Works for unique sentences. @@ -353,8 +363,11 @@ ISSUES FOUND: CLEAN: NOT YET INVESTIGATED: Settings are split into: - **Basic:** OpenAI API key + auto-investigate toggle. -- **Advanced:** API server URL, attestation/HMAC secret override, instance API - key. If unset, defaults to hosted instance. +- **Advanced:** API server URL and instance API + key. If unset, defaults to hosted instance. A stored value that is set but + unusable (e.g. an API URL that fails validation) is an error shown in the + options page and popup, and the extension makes no API calls until it is + fixed — it never falls back to the hosted instance. ### Annotation Styling @@ -396,20 +409,24 @@ User visits post → Background worker calls API: recordViewAndGetStatus({ postVersionId }) → API looks up PostVersion by primary key - NOT FOUND → return { investigationState: "NOT_INVESTIGATED", claims: null, - priorInvestigationResult: null } - (version was never registered; nothing to do) + NOT FOUND → reject request (unknown post version) → API increments raw viewCount, updates uniqueViewScore, records corroboration credit - → API checks whether this content version has a completed investigation: - HIT → return { investigationState: "INVESTIGATED", provenance, claims: [...] } - (already investigated for this content version; client is done) - MISS + latest complete SERVER_VERIFIED exists for this post → - return { investigationState: "NOT_INVESTIGATED", claims: null, - priorInvestigationResult: { oldClaims: claims, sourceInvestigationId } } - (reuse prior verified claims as interim context; no run is queued) - MISS + only CLIENT_FALLBACK/none latest exists → - return { investigationState: "NOT_INVESTIGATED", claims: null } - (no completed investigation yet for this content version) + → API checks the investigation of this content version (at most one exists, §3.5): + COMPLETE → return { investigationState: "INVESTIGATED", investigationId, provenance, claims } + (already investigated for this content version; client is done) + PENDING / PROCESSING → + return { investigationState: "INVESTIGATING", investigationId, status, provenance, + pendingClaims, confirmedClaims, priorInvestigationResult } + (queued by another viewer, the selector, or an earlier visit; the client polls + getInvestigation({ investigationId }) until it settles, exactly as for its own; + priorInvestigationResult carries claims forward as below) + FAILED → return { investigationState: "FAILED", investigationId, provenance } + (auto-investigate does not start new work for a failed version) + NONE → + return { investigationState: "NOT_INVESTIGATED", + priorInvestigationResult: { oldClaims, sourceInvestigationId } | null } + (claims carried forward from another version, §2.8 "Interim carry-forward"; + null when none applies; no run is queued) → Client renders current state; recordViewAndGetStatus alone does not enqueue a new investigation → Investigation begins only via investigateNow(...) or selector queueing ``` @@ -426,19 +443,26 @@ User clicks "Investigate Now" (or auto-investigate triggers) → Background worker calls API: registerObservedVersion(...) (if not already done) → Background worker calls API: investigateNow({ postVersionId }), optionally including a user OpenAI key via x-openai-api-key header + → The request's funding: its user OpenAI key if it sent one, otherwise the instance + API key it authenticated with; neither → reject (UNAUTHORIZED) → API looks up PostVersion by primary key NOT FOUND → reject request (unknown post version) → API checks for an existing Investigation row for that content version - no row → create PENDING row, start background run, + no row → reject if over the word limit; if funding is a user key, verify it with + OpenAI first (rejected or unverifiable → reject the request, nothing is + created); create the PENDING row with update lineage (§2.4.3) and the + request's funding (origin INSTANCE_REQUEST or USER_KEY_REQUEST, user key + attached in the same transaction), enqueue, return { investigationId, status: PENDING } → If a row already exists, API returns immediately with status-based behavior: COMPLETE → return { investigationId, status: COMPLETE, claims } - FAILED → return { investigationId, status: FAILED } (unless explicit retry action is requested) + FAILED → return { investigationId, status: FAILED } (FAILED is terminal, §3.7) PROCESSING → return { investigationId, status: PROCESSING }; no second run is started - PENDING → if request includes a user OpenAI key and no user-key source is attached yet, - attach one (first key wins) - once PROCESSING starts, user-key source is immutable for that run - ensure a background run exists + (an expired lease is recovered first, §3.7) + PENDING → funded (someone is paying): ensure a queue job exists; the request's + user key is never attached — it never takes over a run someone else funds + unfunded (its user key was dropped, §3.7): fund it with this request + (verifying a user key first), enqueue return { investigationId, status: PENDING } → Extension polls getInvestigation({ investigationId }) until COMPLETE/FAILED → Worker claims queued job, sets PROCESSING, runs investigation, then writes COMPLETE/FAILED @@ -447,19 +471,21 @@ User clicks "Investigate Now" (or auto-investigate triggers) **Auto-investigate (extension-side):** ``` -After recordViewAndGetStatus returns { investigationState: "NOT_INVESTIGATED", claims: null, +After recordViewAndGetStatus returns { investigationState: "NOT_INVESTIGATED", priorInvestigationResult: ... | null } → If user OpenAI key exists and auto-investigate is enabled: background worker calls investigateNow({ postVersionId }) and then polls for completion → Result is cached locally when returned ``` -**Background selection (server-managed budget):** +**Background selection (server-managed daily budget):** ``` Cron job runs every N minutes - → SELECT uninvestigated posts ORDER BY unique_view_score DESC LIMIT :budget - → INSERT Investigation(status=PENDING) for each + → Recover expired leases; re-enqueue funded PENDING investigations that are due + → Admit uninvestigated latest versions (and unfunded investigations) ORDER BY + unique_view_score DESC, while today's SELECTOR admissions < SELECTOR_DAILY_BUDGET + → INSERT Investigation(status=PENDING, origin=SELECTOR) for each new one → Job queue workers pick up and investigate ``` @@ -494,16 +520,26 @@ completed investigation matching the current version. `registerObservedVersion`; the API computes the version key internally. Subsequent calls (`recordViewAndGetStatus`, `investigateNow`) reference the resolved version by `postVersionId`. - **Hit**: Return claims. The view still increments the counter. -- **Miss**: Return `{ investigationState: "NOT_INVESTIGATED", claims: null }` (optionally with - `priorInvestigationResult` when a prior complete `SERVER_VERIFIED` investigation exists). - The view is recorded; the selector may pick this post up later. +- **Miss**: Return `{ investigationState: "NOT_INVESTIGATED", priorInvestigationResult }`, with + claims carried forward from another version when any apply (below). The view is recorded; + the selector may pick this post up later. - **Strict version rule**: Never return or render an investigation for a different content version - of the same post. No fallback to older versions. -- **Interim update rule**: For a version miss with no complete result on the requested - version, the API may return a prior complete `SERVER_VERIFIED` investigation as - interim via `priorInvestigationResult = { oldClaims, sourceInvestigationId }`. - This is a temporary UI state only, does not count as a final cache hit, and does - not by itself queue a new investigation run. + of the same post as that version's result. The only claims of another version a viewer sees + are the interim ones below. +- **Interim carry-forward**: While the requested version has no finished investigation (not + investigated yet, or investigating), `priorInvestigationResult` (`{ oldClaims, + sourceInvestigationId }`) comes from the post's latest complete investigation (by + `checkedAt`) on another version — never the requested version itself — of either + provenance, and holds only the claims whose exact `text` still occurs in the requested + version's content text (after the same normalization claim texts were taken from, §3.8). + If there is no such investigation or none of its claims survives, it is null. + Rationale: a correction is only shown while the exact text it corrects is still on the + page. An edit (or a content-normalization change) then never leaves a correction pointing + at text that is gone, and a correction of text that is still there stays visible until the + new version's own investigation replaces it, however the earlier version's content was + obtained. This is a temporary UI state only: it is not a cache hit, does not by itself + queue an investigation, and does not change which investigation an update run uses as its + parent (still the latest complete `SERVER_VERIFIED` one, §2.4.3). - **Version key semantics**: The version key (`versionHash`) is derived from both normalized text and image occurrences: `sha256(contentHash + "\n" + occurrencesHash)`. For LessWrong and Wikipedia, normalized text is derived server-side from canonical sources; for X/Substack it is @@ -529,7 +565,18 @@ So server-side verification is preferred but best-effort. - Identity binding rule: server-canonical responses define authoritative platform identity. For server-verifiable platforms (for example, Wikipedia `pageId` + `revisionId`), if client-submitted identity disagrees with the platform response, the API records an - integrity anomaly and corrects stored identity to the server value. + integrity anomaly and corrects stored identity to the server value. The post URL and + author are identity too: when the server fetch succeeds (LessWrong: post URL from + `_id` + `slug`, author and title from the post's user; Wikipedia: article URL from the + parse API title), they come from the platform response and `Post.identityVerifiedAt` + latches. Once latched, client-submitted URL/author never overwrite them. +- Client-submitted post URLs are validated per platform before they are stored: `https` + only, on the platform's host (`lesswrong.com`, `x.com`/`twitter.com`, + `.wikipedia.org`; Substack may use custom domains, so only a `/p/` path is + required), naming the same post as `externalId` where the URL encodes it. +- The canonical fetch runs inside `registerObservedVersion`, so it is bounded: one + 10-second deadline across all transient retries and a 10 MB response cap; exceeding + either falls back to `CLIENT_FALLBACK`. - Primary path: server verifies platform content and derives the canonical content version. - Mismatch policy: if identity-bound verification succeeds but conflicts with submitted content, record an integrity anomaly and continue with the server-derived canonical content @@ -568,14 +615,18 @@ consumers can apply their own trust policy. (e.g., on a subsequent `registerObservedVersion` where the server retries), matching `PostVersion` rows latch `serverVerifiedAt` from null to a timestamp. Existing investigation provenance snapshots are immutable and are not rewritten. -- **Interim display policy:** Interim reuse is only enabled when the prior completed investigation has - `provenance = "SERVER_VERIFIED"` on `InvestigationInput`. `CLIENT_FALLBACK` investigations are never reused as - interim results on a new version. +- **Interim display policy:** interim claims may come from a `CLIENT_FALLBACK` investigation as + well as a `SERVER_VERIFIED` one (§2.8 "Interim carry-forward"): what makes a carried-forward + claim safe to show is that the exact text it corrects is on the viewer's page, not how the + earlier version's content was obtained. ## 2.10 Investigation Prioritization The investigation selector is a cron job that runs every N minutes, selecting uninvestigated posts -ordered by capped unique-view score. Budget is configurable (e.g. 100 investigations/day). +ordered by capped unique-view score. The budget (`SELECTOR_DAILY_BUDGET`, default 100) caps the +investigations the selector admits per UTC day, however often it runs. Investigations admitted by +investigateNow (instance-key or user-key funded) do not count against it; re-enqueueing already +admitted work does not either. Scoring rules (v1): @@ -585,6 +636,10 @@ Scoring rules (v1): post per 24h) - IP-range credit cap for this post today has not been exceeded - The IP-range credit cap is configurable. +- The client IP behind both rules is the viewer's, not the proxy's: behind the deployment's + ingress the API reads it from the trusted proxy header (`ADDRESS_HEADER` / `XFF_DEPTH`), and + refuses to start in production without that configuration. A client address that is not an IP + address is a configuration error, not a bucket. This naturally handles edits: if a post was investigated but then edited, the content hash no longer matches any existing investigation, so it re-enters the selection pool. @@ -646,13 +701,25 @@ For every completed investigation, persist audit artifacts: - Normalized input text and `contentHash` - Content provenance and any server-fetch failure reason - Prompt reference (`promptId` → `Prompt.version`, `Prompt.hash`, `Prompt.text`) -- Provider/model metadata (enum values for provider/model, plus provider-reported model version) -- Normalized per-attempt request/response records: - requested tools, output items, output text parts + citations, reasoning summaries, - tool calls (with raw provider payloads), token usage, and provider/parsing errors +- Provider/model metadata: the provider enum, the provider model id the stage-1 + fact-check ran on (recorded at completion, never predicted at queue time), and + the provider-reported model version of the final stage-1 response +- Normalized per-attempt records, with one request record per provider request + (each stage-1 fact-check round and each stage-2 claim validation): the exact + model, instructions, input (image parts recorded by content hash), previous + response id, reasoning options, `include` values and tool definitions sent; and + its response, if any: output items with their text parts + citations, reasoning + summaries and tool calls (raw provider payloads, including web-search sources), + and token usage. Attempt-level provider/parsing errors. +- Attempts recorded before per-request auditing (pre-2026-10) are kept as a single + `LEGACY_COMBINED` request whose fields merge all of that attempt's requests - Input lineage for update runs: `oldClaims`, current article text, and computed content diff included in request/input records so we can reconstruct why only parts changed -- Source snapshots or immutable excerpts used for claims, with hash and retrieval timestamp +- The immutable excerpt (`snippet`) for each claim source; what the model actually retrieved + (web search results, `fetch_url` bodies with retrieval timestamps) is in the attempt's raw + tool-call payloads +- Per-attempt records are insert-only and keyed by a never-reset attempt number, so retries + never overwrite earlier attempts Stored artifacts are the canonical audit record. Re-running the same investigation later may produce different outputs because external web content @@ -669,9 +736,9 @@ out of scope for v1, but the following baseline measures are required: 1. **Extension attestation signal (not authentication).** The extension includes an attestation signal generated from a bundled default secret (with an optional local override in extension settings). Because extensions are inspectable, this is - treated only as a low-confidence abuse signal for filtering/telemetry, not a security boundary. - Missing/invalid attestation is treated as "no signal" rather than an auth failure. Authorization - and trust decisions must not rely on this signal alone. + at most a low-confidence abuse signal for filtering/telemetry, never a security boundary. + The v1 server does not verify or consume it; authorization and trust decisions must not rely + on it. 2. **Server-side content verification.** The server always attempts to fetch canonical content from the platform (see §2.9). Client-submitted text is only used as fallback when the server fetch fails, limiting the attacker's ability to inject fabricated content into investigations. @@ -682,8 +749,14 @@ out of scope for v1, but the following baseline measures are required: won't match a server-verified investigation shown to real users. 5. **User-key handling.** User-supplied OpenAI keys may be persisted locally in the extension, but plaintext keys must never be persisted server-side in application data or durable logs. -6. **SSRF-safe image fetch.** Investigation-time image downloading must block private/internal - network targets and enforce count/size limits before upload to blob storage. +6. **SSRF-safe fetching.** Investigation-time image downloads and the model's `fetch_url` tool + fetch URLs chosen by untrusted input, so the check lives on the connection: every connection + resolves through a DNS lookup that rejects the answer unless all addresses are public unicast + (no private, loopback, link-local, CGNAT, multicast, reserved, IPv4-mapped, NAT64, 6to4 or + Teredo ranges) and pins the connection to the addresses it checked, so DNS rebinding has no + window. IP-literal hosts, embedded credentials and non-HTTP(S) schemes are rejected before + dispatch; redirects are followed manually through the same check; bodies are read up to a + byte cap. Images are downloaded only after the run's funding key is resolved. 7. **Transport limits.** The following limits are enforced on API inputs: - `MAX_OBSERVED_CONTENT_TEXT_CHARS` / `MAX_OBSERVED_CONTENT_TEXT_UTF8_BYTES`: 500,000 - `MAX_IMAGES_PER_INVESTIGATION`: 10 @@ -706,11 +779,12 @@ measures (proof-of-work, behavioral analysis) are planned for future versions. | Cross-browser | **webextension-polyfill** | Normalizes Chrome/Firefox API differences behind a single Promise-based API | | Type safety | **TypeScript + Zod** | Runtime validation at API boundary | | API framework | **SvelteKit + tRPC + GraphQL** | tRPC for extension/internal consumers; GraphQL for public third-party API | -| Database | **Supabase (hosted Postgres) + Prisma** | Stores investigations, view counts, user accounts | -| Job queue | **Postgres-backed** (graphile-worker or `FOR UPDATE SKIP LOCKED`) | No Redis dependency; runs against the same Supabase database | -| LLM | **OpenAI Responses API with tools** | v1 provider. Anthropic support planned via `Investigator` interface | +| Database | **Postgres + Prisma** | Stores posts, content versions, investigations, view credits, audit records | +| Job queue | **Postgres-backed** (graphile-worker) | No Redis dependency; runs against the same database | +| Public website | **SvelteKit (server-rendered)** | Reads only the public GraphQL API; parses responses with the shared public schemas | +| LLM | **OpenAI Responses API with tools** (`gpt-6.1-sol`) | v1 provider and model, fixed in code (`openai-request-config.ts`). Anthropic support planned via `Investigator` interface | | Auth | **Anonymous + required instance OpenAI key + optional request-scoped user OpenAI key** | Instance-managed investigations are always available; users may still override with their own key for on-demand runs | -| Deployment | **Helm chart** (on-prem), **Pulumi** (official hosted, deploys the same chart) | Single artifact for both on-prem and hosted; no deployment drift | +| Deployment | **Helm chart** (on-prem), **Pulumi** (official hosted, deploys the same chart) | Single artifact for both on-prem and hosted; no deployment drift | ## 3.2 Data Model @@ -724,9 +798,10 @@ model Post { id String @id @default(cuid()) platform Platform externalId String // Platform's native ID - url String + url String // https on the platform host; server-verified when identityVerifiedAt is set authorId String? author Author? @relation(fields: [authorId], references: [id]) + identityVerifiedAt DateTime? // Latch: url/author came from a server fetch; client data never overwrites them viewCount Int @default(0) // Raw views (analytics) uniqueViewScore Int @default(0) // Capped selector score lastViewedAt DateTime? @@ -784,6 +859,26 @@ model Author { } ``` +### Instance API keys + +Operators issue API keys that authorize `investigateNow` on the instance's own +OpenAI key. Only a SHA-256 hash of each key is stored; keys are managed with +`pnpm --filter @openerrata/api instance-api-key` (list / activate / revoke). + +```prisma +model InstanceApiKey { + id String @id @default(cuid()) + name String + keyHash String @unique // SHA-256 of the trimmed key + revokedAt DateTime? // Revoked keys are kept for audit and never authorize + createdAt DateTime @default(now()) + updatedAt DateTime @updatedAt + + @@index([name]) + @@index([revokedAt]) +} +``` + ### Platform metadata Platform metadata is version-scoped only. Each `PostVersion` can have one @@ -819,7 +914,6 @@ model HtmlBlob { lesswrongServerVersionMetas LesswrongVersionMeta[] @relation("LesswrongServerHtml") lesswrongClientVersionMetas LesswrongVersionMeta[] @relation("LesswrongClientHtml") - substackServerVersionMetas SubstackVersionMeta[] @relation("SubstackServerHtml") substackClientVersionMetas SubstackVersionMeta[] @relation("SubstackClientHtml") wikipediaServerVersionMetas WikipediaVersionMeta[] @relation("WikipediaServerHtml") wikipediaClientVersionMetas WikipediaVersionMeta[] @relation("WikipediaClientHtml") @@ -914,9 +1008,7 @@ model SubstackVersionMeta { slug String title String subtitle String? - serverHtmlBlobId String? - serverHtmlBlob HtmlBlob? @relation("SubstackServerHtml", fields: [serverHtmlBlobId], references: [id], onDelete: Restrict) - clientHtmlBlobId String? + clientHtmlBlobId String? // No Substack server fetch exists, so only client HTML is stored clientHtmlBlob HtmlBlob? @relation("SubstackClientHtml", fields: [clientHtmlBlobId], references: [id], onDelete: Restrict) imageUrls String[] authorName String @@ -977,17 +1069,22 @@ model Investigation { promptId String prompt Prompt @relation(fields: [promptId], references: [id]) provider InvestigationProvider - model InvestigationModel - modelVersion String? // Provider-reported model revision/version when available + // INV-INV-MODEL-AT-COMPLETION: set iff status = COMPLETE (CHECK + // "Investigation_model_consistency_check"); the provider model id the stage-1 + // fact-check requests were sent to, e.g. "gpt-6.1-sol". + model String? + modelVersion String? // Provider-reported revision of the final stage-1 response checkedAt DateTime? // Null until completion queuedAt DateTime @default(now()) - // Monotonically increasing attempt counter. Incremented atomically when a - // worker claims the lease. Gives each retry a distinct attemptNumber for - // the InvestigationAttempt audit trail. + origin InvestigationOrigin // Who admitted (and pays for) the investigation, §3.7 + admittedAt DateTime // When the current origin admitted it; drives the daily selector budget + // Monotonically increasing attempt counter, never reset. Incremented + // atomically when a worker claims the lease, so every attempt has a distinct + // attemptNumber in the InvestigationAttempt audit trail. attemptCount Int @default(0) - // INV-LEASE: The InvestigationLease row exists iff the investigation is - // PROCESSING and has an active lease holder. Structurally prevents - // leaseOwner/leaseExpiresAt without PROCESSING, and vice versa. + retryAfter DateTime? // Backoff gate for the selector after a transient failure + // INV-LEASE: an InvestigationLease row exists iff status = PROCESSING. + // Enforced at commit by deferred constraint triggers. lease InvestigationLease? openAiKeySource InvestigationOpenAiKeySource? attempts InvestigationAttempt[] @@ -1002,24 +1099,30 @@ model Investigation { @@index([status]) } +// Everything the investigator is given that could change after queue time, +// captured when the investigation is created; every attempt reuses it. model InvestigationInput { - investigationId String @id - investigation Investigation? @relation("InvestigationInputOwner") + investigationId String @id + investigation Investigation? @relation("InvestigationInputOwner") // Immutable after insert; enforced by trigger "reject_investigation_input_updates_trigger". - provenance ContentProvenance - contentHash String - markdownSource MarkdownSource - markdown String? // null iff markdownSource = NONE - markdownRendererVersion String? // null iff markdownSource = NONE - createdAt DateTime @default(now()) + provenance ContentProvenance + contentHash String + markdownSource MarkdownSource + markdown String? // null iff markdownSource = NONE + markdownRendererVersion String? // null iff markdownSource = NONE + imagePlaceholderSourceUrls String[] // Source URL behind each [IMAGE:N], indexed by N; empty when NONE + postUrl String // Prompt context, frozen at queue time + authorName String? + postPublishedAt DateTime? + hasVideo Boolean + createdAt DateTime @default(now()) } -// INV-LEASE: The existence of an InvestigationLease row means "this -// investigation is PROCESSING and has an active lease holder". All fields +// INV-LEASE: an InvestigationLease row exists iff its investigation is +// PROCESSING (checked at commit by deferred constraint triggers). All fields // are NOT NULL — structurally prevents partial lease states. The row is -// deleted on every terminal transition (COMPLETE, FAILED) and on lease -// release (transient failure → PENDING), so progressClaims is automatically -// cleaned up without needing sentinel values. +// deleted on every transition out of PROCESSING (COMPLETE, FAILED, release to +// PENDING, expired-lease recovery), so progressClaims is cleaned up with it. model InvestigationLease { investigationId String @id investigation Investigation @relation(fields: [investigationId], references: [id], onDelete: Cascade) @@ -1072,63 +1175,94 @@ model InvestigationImage { @@index([imageBlobId]) } +// Insert-only: each attemptNumber is claimed once per investigation and its +// audit is written once, at the attempt's terminal transition. model InvestigationAttempt { - id String @id @default(cuid()) - investigationId String - investigation Investigation @relation(fields: [investigationId], references: [id], onDelete: Cascade) - attemptNumber Int - outcome InvestigationAttemptOutcome - requestModel String // Provider request model id (e.g. gpt-5-*) - requestInstructions String // Exact instructions/system prompt sent - requestInput String // Exact user input sent - requestReasoningEffort String? - requestReasoningSummary String? - responseId String? // Provider response id - responseStatus String? - responseModelVersion String? - responseOutputText String? // Raw structured output text returned - startedAt DateTime - completedAt DateTime? - requestedTools InvestigationAttemptRequestedTool[] - outputItems InvestigationAttemptOutputItem[] - toolCalls InvestigationAttemptToolCall[] - usage InvestigationAttemptUsage? - error InvestigationAttemptError? - createdAt DateTime @default(now()) - updatedAt DateTime @updatedAt + id String @id @default(cuid()) + investigationId String + investigation Investigation @relation(fields: [investigationId], references: [id], onDelete: Cascade) + attemptNumber Int + outcome InvestigationAttemptOutcome + startedAt DateTime + completedAt DateTime? + requests InvestigationAttemptRequest[] + error InvestigationAttemptError? + createdAt DateTime @default(now()) + updatedAt DateTime @updatedAt @@unique([investigationId, attemptNumber]) @@index([investigationId, startedAt]) } +// One provider request made during an attempt, exactly as sent. +model InvestigationAttemptRequest { + id String @id @default(cuid()) + attemptId String + attempt InvestigationAttempt @relation(fields: [attemptId], references: [id], onDelete: Cascade) + kind InvestigationAttemptRequestKind + // INV-ATTEMPT-REQUEST-SUBJECT: factCheckRound is set iff kind = FACT_CHECK_ROUND; + // claimIndex is set iff kind = CLAIM_VALIDATION (CHECK "InvestigationAttemptRequest_subject_check"). + factCheckRound Int? + claimIndex Int? + model String // Provider request model id (e.g. gpt-6.1-sol) + instructions String // Exact instructions/system prompt sent + input Json // Exact `input` sent; image parts carry imageContentHash (ImageBlob.contentHash) instead of the data URI + previousResponseId String? // Set on fact-check rounds after the first + reasoningEffort String? + reasoningSummary String? + include String[] + requestedTools InvestigationAttemptRequestedTool[] + response InvestigationAttemptResponse? // Absent when the request failed without a response + createdAt DateTime @default(now()) + + @@unique([attemptId, factCheckRound]) + @@unique([attemptId, claimIndex]) + @@index([attemptId]) +} + model InvestigationAttemptRequestedTool { - id String @id @default(cuid()) - attemptId String - attempt InvestigationAttempt @relation(fields: [attemptId], references: [id], onDelete: Cascade) + id String @id @default(cuid()) + requestId String + request InvestigationAttemptRequest @relation(fields: [requestId], references: [id], onDelete: Cascade) requestOrder Int toolType String rawDefinition Json // Full provider tool-definition payload for this request position - createdAt DateTime @default(now()) + createdAt DateTime @default(now()) - @@unique([attemptId, requestOrder]) - @@index([attemptId]) + @@unique([requestId, requestOrder]) + @@index([requestId]) +} + +model InvestigationAttemptResponse { + id String @id @default(cuid()) + requestId String @unique + request InvestigationAttemptRequest @relation(fields: [requestId], references: [id], onDelete: Cascade) + providerResponseId String + status String? // Provider-reported status; null if omitted (the attempt then fails) + modelVersion String // Provider-reported model revision + receivedAt DateTime? // Null only on LEGACY_COMBINED requests + outputItems InvestigationAttemptOutputItem[] + usage InvestigationAttemptUsage? + createdAt DateTime @default(now()) } model InvestigationAttemptOutputItem { id String @id @default(cuid()) - attemptId String - attempt InvestigationAttempt @relation(fields: [attemptId], references: [id], onDelete: Cascade) + responseId String + response InvestigationAttemptResponse @relation(fields: [responseId], references: [id], onDelete: Cascade) outputIndex Int providerItemId String? itemType String // Provider-defined output item type itemStatus String? + // By itemType: textParts for "message", reasoningSummaries for "reasoning", + // toolCall for every other (tool call) item. textParts InvestigationAttemptOutputTextPart[] reasoningSummaries InvestigationAttemptReasoningSummary[] toolCall InvestigationAttemptToolCall? createdAt DateTime @default(now()) - @@unique([attemptId, outputIndex]) - @@index([attemptId]) + @@unique([responseId, outputIndex]) + @@index([responseId]) } model InvestigationAttemptOutputTextPart { @@ -1175,36 +1309,23 @@ model InvestigationAttemptReasoningSummary { } model InvestigationAttemptToolCall { - id String @id @default(cuid()) - attemptId String - attempt InvestigationAttempt @relation(fields: [attemptId], references: [id], onDelete: Cascade) - outputItemId String - outputItem InvestigationAttemptOutputItem @relation(fields: [outputItemId], references: [id], onDelete: Cascade) - outputIndex Int - providerToolCallId String? - toolType String - status String? - rawPayload Json // Full provider output item payload for this call - capturedAt DateTime - providerStartedAt DateTime? - providerCompletedAt DateTime? - createdAt DateTime @default(now()) - - @@unique([attemptId, outputIndex]) - @@unique([outputItemId]) - @@index([attemptId]) + id String @id @default(cuid()) + outputItemId String @unique + outputItem InvestigationAttemptOutputItem @relation(fields: [outputItemId], references: [id], onDelete: Cascade) + rawPayload Json // Full provider output item payload for this call + createdAt DateTime @default(now()) } model InvestigationAttemptUsage { - id String @id @default(cuid()) - attemptId String @unique - attempt InvestigationAttempt @relation(fields: [attemptId], references: [id], onDelete: Cascade) + id String @id @default(cuid()) + responseId String @unique + response InvestigationAttemptResponse @relation(fields: [responseId], references: [id], onDelete: Cascade) inputTokens Int outputTokens Int totalTokens Int cachedInputTokens Int? reasoningOutputTokens Int? - createdAt DateTime @default(now()) + createdAt DateTime @default(now()) } model InvestigationAttemptError { @@ -1249,10 +1370,7 @@ model Source { claim Claim @relation(fields: [claimId], references: [id], onDelete: Cascade) url String title String - snippet String - snapshotText String? // Immutable excerpt/body used during the run (if retained) - snapshotHash String? // Hash of snapshotText or archived source bytes - retrievedAt DateTime + snippet String // Immutable excerpt the claim relies on; retrieval data is in the attempt's tool calls @@index([claimId]) } @@ -1281,18 +1399,23 @@ enum InvestigationProvider { ANTHROPIC } -enum InvestigationModel { - OPENAI_GPT_5 - OPENAI_GPT_5_MINI - ANTHROPIC_CLAUDE_SONNET - ANTHROPIC_CLAUDE_OPUS -} - enum InvestigationAttemptOutcome { SUCCEEDED FAILED } +enum InvestigationAttemptRequestKind { + FACT_CHECK_ROUND // One round of the stage-1 fact-check tool loop + CLAIM_VALIDATION // One stage-2 per-claim validation call + LEGACY_COMBINED // Pre-2026-10 merged audit of a whole attempt; never written by current code +} + +enum InvestigationOrigin { + SELECTOR // Background selection; server key; counts against SELECTOR_DAILY_BUDGET + INSTANCE_REQUEST // investigateNow from an instance-API-key client; server key + USER_KEY_REQUEST // investigateNow funded by the requester's verified OpenAI key +} + // No Verdict enum. Every Claim in the database is an incorrect claim. // The model only reports claims it has clear evidence are wrong. // Correct, ambiguous, and unverifiable claims are not stored. @@ -1335,48 +1458,54 @@ interface InvestigationResult { // canonical content via server-side verification (best effort) or client fallback. // Returns a postVersionId that subsequent calls use as a cheap PK reference. postRouter.registerObservedVersion - Input: { platform, externalId, url, observedImageUrls?, observedImageOccurrences?, + Input: { platform, externalId, url, + observedImageOccurrences?, // every observed image in page order; the distinct + // image URLs are derived from it (absent = no images) observedContentText?, // required for X/Substack/Wikipedia; omitted for LessWrong metadata: { title?, authorName?, ... } } Output: { platform, externalId, versionHash, postVersionId, provenance: ContentProvenance } -// Record a view and return cached investigation status. Increments raw viewCount -// and updates uniqueViewScore. Uses postVersionId from registerObservedVersion -// for a direct primary-key lookup (no content re-derivation). +// Record a view and return the status of this version's investigation, if any. +// Increments raw viewCount and updates uniqueViewScore. Uses postVersionId from +// registerObservedVersion for a direct primary-key lookup (no content re-derivation); +// rejects unknown post versions. Every variant about an existing investigation +// carries its id, so the client can poll an investigation it did not start. postRouter.recordViewAndGetStatus Input: { postVersionId } Output: | { investigationState: "NOT_INVESTIGATED", priorInvestigationResult: { oldClaims: Claim[], sourceInvestigationId: string } | null } - | { investigationState: "INVESTIGATING", status: "PENDING" | "PROCESSING", + | { investigationState: "INVESTIGATING", investigationId, status: "PENDING" | "PROCESSING", provenance: ContentProvenance, pendingClaims: ClaimPayload[], confirmedClaims: ClaimPayload[], priorInvestigationResult: { oldClaims: Claim[], sourceInvestigationId: string } | null } - | { investigationState: "INVESTIGATED", provenance: ContentProvenance, claims: Claim[] } + | { investigationState: "FAILED", investigationId, provenance: ContentProvenance } + | { investigationState: "INVESTIGATED", investigationId, provenance: ContentProvenance, + claims: Claim[] } // Fetch results for a specific investigation (used for polling) postRouter.getInvestigation Input: { investigationId } Output: - | { investigationState: "NOT_INVESTIGATED", - priorInvestigationResult: { oldClaims: Claim[], sourceInvestigationId: string } | null, - checkedAt?: DateTime } + | { investigationState: "NOT_INVESTIGATED", // unknown investigationId + priorInvestigationResult: null } | { investigationState: "INVESTIGATING", status: "PENDING" | "PROCESSING", provenance: ContentProvenance, pendingClaims: ClaimPayload[], confirmedClaims: ClaimPayload[], - priorInvestigationResult: { oldClaims: Claim[], sourceInvestigationId: string } | null, - checkedAt?: DateTime } - | { investigationState: "FAILED", provenance: ContentProvenance, - checkedAt?: DateTime } + priorInvestigationResult: { oldClaims: Claim[], sourceInvestigationId: string } | null } + | { investigationState: "FAILED", provenance: ContentProvenance } | { investigationState: "INVESTIGATED", provenance: ContentProvenance, claims: Claim[], checkedAt: DateTime } // Request immediate investigation. Uses postVersionId from registerObservedVersion. // Authorization: instance API key OR request-scoped user OpenAI key (`x-openai-api-key`). +// A user key is verified with OpenAI before it funds anything (rejected → UNAUTHORIZED, +// restricted → FORBIDDEN, unverifiable → BAD_GATEWAY) and only funds an investigation this +// request admits (§2.6). // Rejects posts exceeding the word count limit (same 10,000-word cap as the selector). // Idempotent: if an investigation already exists for this content version, returns its // current status (which may be COMPLETE or FAILED, not just PENDING). -// All paths are async queue-backed; user-key requests attach an encrypted short-lived lease. +// All paths are async queue-backed; user-key requests attach an encrypted short-lived key source. postRouter.investigateNow Input: { postVersionId } Output: @@ -1464,6 +1593,11 @@ type PublicInvestigationResult { claims: [PublicClaim!]! } +type ClaimSummary { + id: ID! + summary: String! +} + type PostInvestigationSummary { id: ID! contentHash: String! @@ -1471,6 +1605,7 @@ type PostInvestigationSummary { corroborationCount: Int! checkedAt: DateTime! claimCount: Int! + claimSummaries: [ClaimSummary!]! } type PostInvestigationsResult { @@ -1488,8 +1623,11 @@ type SearchInvestigationSummary { origin: InvestigationOrigin! corroborationCount: Int! claimCount: Int! + claimSummaries: [ClaimSummary!]! } +# SERVER_VERIFIED implies serverVerifiedAt is set; a CLIENT_FALLBACK +# investigation's serverVerifiedAt is set if the content was verified later. type InvestigationOrigin { provenance: ContentProvenance! serverVerifiedAt: DateTime @@ -1497,20 +1635,25 @@ type InvestigationOrigin { type SearchInvestigationsResult { investigations: [SearchInvestigationSummary!]! + hasMore: Boolean! } type PublicMetrics { totalInvestigatedPosts: Int! investigatedPostsWithFlags: Int! - factCheckIncidence: Float! + factCheckIncidence: Float # null when no investigated posts match the filters } type Query { publicInvestigation(investigationId: ID!): PublicInvestigationResult - postInvestigations(platform: Platform!, externalId: String!): PostInvestigationsResult! + postInvestigations( + platform: Platform! + externalId: String! + ): PostInvestigationsResult! searchInvestigations( query: String platform: Platform + minClaimCount: Int limit: Int = 20 offset: Int = 0 ): SearchInvestigationsResult! @@ -1531,7 +1674,12 @@ type Query { when no post exists; otherwise includes all complete investigations for that post. - `searchInvestigations(...)` returns all complete investigations matching the filters. - `publicMetrics(...)` counts all complete investigations matching the filters. -- `searchInvestigations.limit` must be in `[1, 100]`; `offset >= 0`. +- `searchInvestigations.limit` must be in `[1, 100]`; `offset >= 0`; `minClaimCount` (optional, + `>= 0`) keeps only investigations with at least that many claims; `hasMore` is true when + results exist beyond `offset + limit`. +- Post URLs and source URLs are always `http(s)` URLs. +- The shared `public*OutputSchema`s in `shared/src/schemas/public-api.ts` are the + machine-readable form of these response shapes. - Public responses include trust signals (`provenance`, `corroborationCount`, `serverVerifiedAt`) so clients can apply their own thresholds. @@ -1541,7 +1689,11 @@ In v1, public metrics focus on incidence rather than truth-rate leaderboards: ### Public Surface -- External public integrations use GraphQL (`/graphql`). +- External public integrations use GraphQL (`/graphql`); it is the only public read surface + (there is no public tRPC router). +- The public website is one of those integrations: it sends only GraphQL queries and parses + each response with the shared public schemas, so a response outside this contract renders + an error page (HTTP 502) rather than partial or unchecked data. - Extension/internal traffic uses `postRouter.*` tRPC procedures. ## 3.5 Cache & Idempotency Implementation @@ -1560,17 +1712,17 @@ WHERE "postVersionId" = $1 AND "status" = 'COMPLETE' LIMIT 1; ``` -Interim update candidate query (latest complete server-verified investigation for a post): +Interim carry-forward source query (latest complete investigation of the post on another +version, any provenance; its claims are then filtered to those still on the page, §2.8): ```sql SELECT i.* FROM "Investigation" i JOIN "PostVersion" pv ON pv."id" = i."postVersionId" -JOIN "InvestigationInput" ii ON ii."investigationId" = i."id" WHERE pv."postId" = $1 + AND pv."id" <> $2 -- the requested version AND i."status" = 'COMPLETE' - AND ii."provenance" = 'SERVER_VERIFIED' -ORDER BY i."checkedAt" DESC +ORDER BY i."checkedAt" DESC, i."id" DESC LIMIT 1; ``` @@ -1583,51 +1735,62 @@ uniqueness on `postVersionId` in `Investigation` to prevent duplicates under con ## 3.6 Investigation Selector Queries -The selector picks the most recently seen `PostVersion` per post, joins to -`ContentBlob` for word-count filtering, and left-joins `Investigation` + -`InvestigationRun` to find versions that are either uninvestigated or stuck in -a recoverable pending/processing state. +Each selector run (`runSelector({ dailyBudget })`) does three things, isolating +failures per item (a failing item is reported and the run continues; the +entrypoint exits non-zero if any item failed): + +1. **Recover expired leases** — every PROCESSING investigation whose lease has + expired goes through the single recovery path (§3.7). +2. **Re-enqueue funded work** — every funded PENDING investigation whose + `retryAfter` has passed is enqueued again (the per-investigation jobKey + collapses duplicates), so a lost queue job never strands one. Not budgeted. +3. **Admit new work** — while today's SELECTOR admissions (counted by + `origin = SELECTOR AND admittedAt >= start of the UTC day`) are below + `SELECTOR_DAILY_BUDGET`, admit candidates in score order. Each admission + takes a transaction-scoped advisory lock and re-counts, so concurrent runs + cannot overspend. New investigations get update lineage (§2.4.3) and an input + snapshot exactly as investigateNow does; unfunded investigations are re-funded + as SELECTOR. ```sql --- Select candidate post versions for investigation, ordered by unique-view score. +-- Admission candidates: latest version per post with no investigation, or an +-- unfunded one (user key dropped), ordered by unique-view score. WITH latest_versions AS ( SELECT DISTINCT ON (pv."postId") pv."id" AS "postVersionId", pv."postId", - pv."contentBlobId", - pv."lastSeenAt" + pv."contentBlobId" FROM "PostVersion" pv ORDER BY pv."postId", pv."lastSeenAt" DESC, pv."id" DESC ) SELECT lv."postVersionId", - i."id" AS "investigationId", - i."status" AS "investigationStatus" + i."id" AS "unfundedInvestigationId" FROM latest_versions lv JOIN "Post" p ON p."id" = lv."postId" JOIN "ContentBlob" cb ON cb."id" = lv."contentBlobId" LEFT JOIN "Investigation" i ON i."postVersionId" = lv."postVersionId" -LEFT JOIN "InvestigationLease" il ON il."investigationId" = i."id" WHERE cb."wordCount" <= 10000 AND ( - i."id" IS NULL -- no investigation yet - OR i."status" = 'PENDING' -- pending, ready for enqueueing - OR (i."status" = 'PROCESSING' -- stuck processing (lease expired or missing) - AND (il."investigationId" IS NULL OR il."leaseExpiresAt" <= NOW())) + i."id" IS NULL + OR (i."status" = 'PENDING' AND i."origin" = 'USER_KEY_REQUEST' + AND NOT EXISTS (SELECT 1 FROM "InvestigationOpenAiKeySource" ks + WHERE ks."investigationId" = i."id")) ) -ORDER BY p."uniqueViewScore" DESC -LIMIT :budget; +ORDER BY p."uniqueViewScore" DESC, lv."postVersionId" +LIMIT :remainingDailyBudget; ``` -Each candidate is then passed to `ensureInvestigationQueued({ postVersionId, promptId })` -which handles idempotent creation of the `Investigation` row and job enqueueing. - ## 3.7 Job Queue Postgres-backed (graphile-worker). No Redis. Used by selector work and all `investigateNow` requests. -User-key requests attach an encrypted short-lived key source on the Investigation -for worker-side credential handoff. + +Every investigation is admitted with an origin that fixes who pays for its runs: +`SELECTOR` (server key, daily budget), `INSTANCE_REQUEST` (server key, trusted +instance-API-key client) or `USER_KEY_REQUEST` (the requester's verified OpenAI +key, stored as an encrypted, 30-minute InvestigationOpenAiKeySource). A +`USER_KEY_REQUEST` run never falls back to the server key. Each graphile-worker job is enqueued with `maxAttempts: 1` and a per-investigation `jobKey` (`investigate:${investigationId}`). Retry control is managed by the @@ -1635,36 +1798,66 @@ application, not graphile-worker: transient failures reclaim the investigation t PENDING and explicitly re-enqueue with a backoff delay. ``` -Investigation selected (by selector or any investigateNow request) +Investigation admitted (by selector or any investigateNow request) → Upsert investigation for postVersionId (idempotent: one investigation per content version) → If already exists: reuse existing investigation row and do not enqueue duplicate work - → Worker picks up job → claim lease (PENDING → PROCESSING, atomically increment attemptCount) - → Worker calls Investigator.investigate() - → On success: delete lease, UPDATE status = COMPLETE - → On failure: classify and retry or fail permanently + → Worker picks up job → claim lease (funded PENDING → PROCESSING, atomically increment attemptCount) + → Worker resolves the funding key (before downloading any image), then calls Investigator.investigate() + → Heartbeat renews the lease every 15s; if a renewal finds the lease gone (or renewals fail + past its expiry) the run is aborted and writes nothing further + → On success: delete lease, UPDATE status = COMPLETE, record the model that ran + → On failure: classify and retry, drop the user key, or fail permanently + +Lease invariant: + - An InvestigationLease row exists iff status = PROCESSING, enforced by deferred + constraint triggers at commit. + - One recovery path for expired leases (used by the worker, the selector and + investigateNow): delete the lease; PENDING again, or FAILED if the lost attempt + was attempt MAX_INVESTIGATION_ATTEMPTS or later. Retry model: - - Investigation.attemptCount tracks retries (incremented at lease claim). - - MAX_INVESTIGATION_ATTEMPTS = 4. When exhausted, the investigation is marked FAILED. + - Investigation.attemptCount numbers attempts (incremented at lease claim, never reset), + so every InvestigationAttempt audit row keeps its own attemptNumber. + - MAX_INVESTIGATION_ATTEMPTS = 4: a transient failure on attempt 4 or later marks FAILED. - Transient retries use exponential backoff: delay = 10s × 2^(attempt - 1). Failure classes: TRANSIENT (reclaim to PENDING, re-enqueue with backoff, up to MAX_INVESTIGATION_ATTEMPTS): - - Provider 5xx errors, rate limits (429), network timeouts + - Provider 5xx errors, rate limits (429) on the server key, network timeouts and connection errors + USER KEY UNUSABLE (drop the key; the investigation becomes unfunded PENDING, not re-enqueued): + - User-key source missing, expired or undecryptable when the worker starts + - OpenAI refuses the user key: 401, 403, 404 (no model access), 429 (rate/quota) + The failed attempt is still recorded. Unfunded investigations wait until the selector + (within its budget) or a new investigateNow funds them, so a bad key can never make a + post permanently uninvestigable. NON_RETRYABLE (mark FAILED immediately): - - Structured output fails Zod validation (likely prompt/schema issue, not transient) + - Unusable model output: an unparseable validation verdict, or the stage-1 + tool loop exceeding its round limit - Provider content-policy refusal - - Authentication/authorization errors (401, 403) - PARTIAL (mark FAILED, log partial output for debugging): - - Provider returns truncated or incomplete tool-call trace - -If a user-key source is missing/expired when the worker starts, the investigation -fails and requires an explicit user re-request. Key sources are consumed (deleted) -on every terminal transition (COMPLETE, FAILED). - -`FAILED` is terminal for a given `postVersionId` in v1. Re-running that exact content -version requires an explicit operator/admin action (e.g., reset status or delete/recreate row), -not automatic selector retries. + - Provider request rejections on the server key: malformed request (400), + authentication/authorization (401, 403), unknown model/resource (404), + unprocessable (422) + - Investigator input violating its contract + PARTIAL (mark FAILED, keep the partial output in the attempt audit): + - A provider response that ends with any status other than "completed" + (e.g. "incomplete": truncated output or tool-call trace) + +Claim submissions that fail the shared claim schema (e.g. a non-http(s) source +URL) are not failures: the tool call's output tells the model the claim was not +recorded, and the model may resubmit within the same run. + +Before taking jobs, the worker sends a probe with the investigation request's +exact shape (model, tools, include, reasoning options; tool use disabled, output +capped) and refuses to start if OpenAI rejects it; user-key validation in +settings runs the same probe. + +Key sources are consumed (deleted) on every terminal transition (COMPLETE, FAILED) +and when the key is dropped. + +`FAILED` is terminal for a given `postVersionId` in v1: investigateNow returns it +unchanged and the selector never re-admits it. Re-running that exact content +version requires an explicit operator/admin action (e.g. delete/recreate the row), +not automatic retries. ``` ## 3.8 Platform Adapter Interface @@ -1672,11 +1865,13 @@ not automatic selector retries. Each content script implements: ```typescript +// hydrating / ambiguous_dom / missing_identity are transient (the page may +// still be rendering); unsupported is final for the current DOM. type AdapterNotReadyReason = | "hydrating" | "ambiguous_dom" - | "unsupported" - | "missing_identity"; + | "missing_identity" + | "unsupported"; type AdapterExtractionResult = | { kind: "ready"; content: PlatformContent } @@ -1684,16 +1879,19 @@ type AdapterExtractionResult = interface PlatformAdapter { platformKey: Platform; - contentRootSelector: string; - matches(url: string): boolean; - detectFromDom?(document: Document): boolean; + matches(url: string): boolean; // URL-first selection + detectFromDom?(document: Document): boolean; // custom-domain fallback + pageLocator(url: string): PageLocator | null; // what the URL says about the post (below) detectPrivateOrGated?(document: Document): boolean; extract(document: Document): AdapterExtractionResult; getContentRoot(document: Document): Element | null; + // Subtrees under the root that are not post content on this platform; used + // alike for text extraction, claim matching and HTML snapshots. + contentExclusionFilter(root: Element): (element: Element) => boolean; } interface ImageOccurrence { - originalIndex: number; // 0-based ordinal position of the image in the page + originalIndex: number; // 0-based ordinal position of the image in the page normalizedTextOffset: number; // Character offset in the normalized content text sourceUrl: string; captionText?: string; @@ -1701,34 +1899,68 @@ interface ImageOccurrence { interface PlatformContent { platform: Platform; - externalId: string; + externalId: string; // per-platform format, validated: LessWrong post ID, + // numeric tweet ID, numeric Substack post ID, + // Wikipedia `{language}:{pageId}` url: string; - contentText: string; // Client-observed normalized plain text; must be non-empty - mediaState: "text_only" | "has_images" | "has_video"; // Precedence: "has_video" if any video/iframe is detected; otherwise "has_images" when imageUrls is non-empty; otherwise "text_only". - imageUrls: string[]; - imageOccurrences?: ImageOccurrence[]; // Positional image data; sent to API as observedImageOccurrences + contentText: string; // Client-observed normalized plain text + hasVideo: boolean; // any video/audio/video-iframe embed + imageOccurrences: ImageOccurrence[]; // every image, in page order; the only image data + // (sent to the API as observedImageOccurrences) metadata: Record; } ``` +Media state is derived, not stored: `has_video` if `hasVideo`; otherwise +`has_images` when there are image occurrences; otherwise `text_only`. + +A `PageLocator` is what a URL alone says about the post a page shows: +LessWrong post ID, tweet ID, Substack origin + slug, or Wikipedia language + +page ID and/or title. It identifies pages whose external ID is not (yet) +known — skipped pages, and the popup's check that a cached status still +describes the tab's page — and is parsed in one shared module. + Adapter selection is URL-first (`matches(url)`), then optional DOM-fingerprint fallback (`detectFromDom(document)`) for custom-domain platform pages. +When an adapter stays `not_ready` with a transient reason for 5 seconds on the +same URL (e.g. a tweet whose identity cannot be proven), the page is reported +as skipped with `unsupported_content`; it keeps being re-checked on DOM changes. + ### Content normalization (shared package) Both client (extension) and server (API) must produce identical normalized text from -the same HTML. Two shared components ensure this: - -**Block separator injection:** `CONTENT_BLOCK_SEPARATOR_TAGS` defines block-level HTML -elements whose boundaries are treated as word separators during extraction. Both the -extension (DOM TreeWalker) and API (parse5 traversal) inject a space character at the -entry of these elements, ensuring compact HTML without whitespace text nodes still -normalizes identically. +the same HTML. The extension reads content text from the live content root +through one text index — the same index the claim mapper (§2.4.1) and the +content-change check use — skipping `NON_CONTENT_TAGS` and the adapter's +exclusions; word separators exist only in the text (they have no DOM +position). Two shared components ensure parity with the server: + +**Word separator injection:** `WORD_SEPARATOR_TAGS` defines the HTML elements whose +boundaries are treated as word separators during extraction: block-level elements, and the +line-breaking void elements `br` and `hr` (`hard.
This` reads `hard. This`, not +`hard.This`). Both the extension (DOM TreeWalker) and API (parse5 traversal) inject a space +character at the entry and exit of these elements, ensuring compact HTML without whitespace +text nodes still normalizes identically. ```typescript -const CONTENT_BLOCK_SEPARATOR_TAGS = new Set([ - "p", "li", "h1", "h2", "h3", "h4", "h5", "h6", - "figcaption", "blockquote", "tr", "td", "th", "div", +const WORD_SEPARATOR_TAGS = new Set([ + "p", + "li", + "h1", + "h2", + "h3", + "h4", + "h5", + "h6", + "figcaption", + "blockquote", + "tr", + "td", + "th", + "div", + "br", + "hr", ]); const NON_CONTENT_TAGS = new Set(["script", "style", "noscript"]); @@ -1738,10 +1970,10 @@ const NON_CONTENT_TAGS = new Set(["script", "style", "noscript"]); ```typescript const TYPOGRAPHIC_REPLACEMENTS: [RegExp, string][] = [ - [/[\u201C\u201D]/g, '"'], // Left/right double quotes → " - [/[\u2018\u2019]/g, "'"], // Left/right single quotes → ' - [/[\u2010-\u2015]/g, "-"], // Hyphens + en/em dashes → - - [/\u2026/g, "..."], // Horizontal ellipsis → ... + [/[\u201C\u201D]/g, '"'], // Left/right double quotes → " + [/[\u2018\u2019]/g, "'"], // Left/right single quotes → ' + [/[\u2010-\u2015]/g, "-"], // Hyphens + en/em dashes → - + [/\u2026/g, "..."], // Horizontal ellipsis → ... ]; function normalizeContent(raw: string): string { @@ -1750,12 +1982,16 @@ function normalizeContent(raw: string): string { text = text.replace(pattern, replacement); } return text - .replace(/[\u200B-\u200D\uFEFF]/g, "") // Remove zero-width characters - .replace(/\s+/g, " ") // Collapse whitespace + .replace(/[\u200B-\u200D\uFEFF]/g, "") // Remove zero-width characters + .replace(/\s+/g, " ") // Collapse whitespace .trim(); } ``` +Page HTML sent to the API (LessWrong/Substack/Wikipedia `metadata.htmlContent`) is +serialized from a copy of the content root with the extension's own highlight +marks unwrapped and excluded subtrees removed; marks never leave the page. + `contentText` must be non-empty. Posts that normalize to empty text are currently treated as unsupported and are skipped by the extension (`reason: "no_text"`). This includes textless/image-only posts for now. @@ -1766,19 +2002,61 @@ views). In that case, the extension must skip sending content to the API and emi **All skip reasons:** -| Reason | Condition | -| ---------------------- | ------------------------------------------------- | -| `has_video` | Any video/iframe embed detected | -| `word_count` | Normalized text exceeds `WORD_COUNT_LIMIT` (10000)| -| `no_text` | Content normalizes to empty string | -| `private_or_gated` | Private/protected/subscriber-only content | -| `unsupported_content` | Content type not supported by the adapter | - -### Extension message protocol versioning - -All extension messages include a `v` field set to `EXTENSION_MESSAGE_PROTOCOL_VERSION` -(currently `1`). This enables the API to reject or handle messages from outdated -extension versions. +| Reason | Condition | +| --------------------- | -------------------------------------------------------------------------------------------------- | +| `has_video` | Any video/iframe embed detected | +| `word_count` | Normalized text exceeds `WORD_COUNT_LIMIT` (10000) | +| `no_text` | Content normalizes to empty string | +| `private_or_gated` | Private/protected/subscriber-only content | +| `unsupported_content` | Content not extractable (non-article page, or post still unextractable after the 5 s grace period) | + +Skipped pages are identified by platform, page URL and reason only: a skip can +be decided before the post's external ID is known. + +### 3.8.1 Extension Message Protocol + +Messages between the extension's contexts are `{ type, payload }` and every +reply is an envelope `{ ok: true, value } | { ok: false, error, errorCode? }`; +listeners always reply with a promise, so error replies are delivered. +Each direction has one request map (`BACKGROUND_REQUESTS`, +`CONTENT_REQUESTS` in the shared package) from type to payload and response +schemas; senders and handler tables are typed from it and receivers validate +against it. + +| Direction | Type | Purpose | +| -------------------------- | ------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- | +| content → background | `PAGE_CONTENT` | Register observed content, record the view; replies with the page session's status | +| content → background | `PAGE_SKIPPED` | Report a skipped page | +| content → background | `PAGE_RESET` | The page session ended; its status is discarded | +| content → background | `INVESTIGATE_NOW` | Investigate the session's post; replies with its status | +| popup → background | `GET_TAB_STATUS` | A tab's cached status, or the upgrade-required notice | +| background/popup → content | `PING` | Liveness probe; no side effects | +| popup → content | `GET_VISIBILITY` | Highlight visibility; no side effects | +| popup → content | `SHOW_ANNOTATIONS` / `HIDE_ANNOTATIONS` / `REQUEST_INVESTIGATE` / `FOCUS_CLAIM` | User actions | +| background → content | `LOCATION_CHANGED` | Relay of `webNavigation.onHistoryStateUpdated`: content scripts run in an isolated world and cannot observe the page's `pushState` | +| background → content | `STATUS_CHANGED` | The background cached a new status for the tab | + +**Page sessions.** A content script starts a page session (with a random +UUID `tabSessionId`) for each distinct observed page state: a tracked post +version or a skip. The background treats the first message from a new session +as the tab's current one and retires the previous; replies for retired +sessions are dropped. A tab's status lives in `storage.session` (it survives +service-worker restarts, never browser restarts) and is cleared when the tab +commits a new document. `INVESTIGATING` statuses — whether this tab started +the investigation or the API reported one already running — are polled until +they settle; transient poll failures back off and give up after 5 attempts +(`API_ERROR`). + +**Injection.** Each page gets one content-script controller: injection is +probed with `PING` first, and a repeated injection finds the live controller +and does not boot a second one. A controller orphaned by an extension +reload/update shuts itself down and removes its highlights. + +The protocol is not versioned: all contexts ship in one bundle, and a content +script orphaned by an extension update can no longer reach the new background. +API compatibility is versioned separately over HTTP +(`x-openerrata-extension-version`, minimum supported version → upgrade +required). Future work: design a dedicated UI/UX flow for fact-checking image-only posts without relying on text-span highlighting. @@ -1791,11 +2069,17 @@ DOM manipulation is reliable. **Extraction:** 1. Wait for `document_idle`. -2. Locate post body: `document.querySelector('.PostsPage-postContent')`. -3. Extract post ID from URL: `/posts/{postId}/{slug}`. -4. Normalize `textContent`. -5. Extract image URLs (``), filter malformed/data URLs, and compute `mediaState`. -6. Send `{ platform: "LESSWRONG", externalId, url, metadata.htmlContent, observedImageUrls? }` to background worker. +2. Extract post ID from URL: `/posts/{postId}/{slug}`. +3. Locate the post body: the `.PostsPage-postContent` whose `#postBody` JSON-LD names that + post ID (never extract before that identity appears, so SPA switches cannot hash another + post's DOM), then its canonical `#postContent`. +4. Read content text and image occurrences through the shared text index, excluding the + client-rendered linkpost callout (`.LinkPostMessage-root`), which GraphQL `contents.html` + lacks. +5. Tags are the tag chips (`.FooterTag-root` links to `/w/`, formerly `/tag/`), + once each. +6. Send `{ platform: "LESSWRONG", externalId, url, metadata.htmlContent, observedImageOccurrences }` + to the background worker. **Media behavior:** Posts with images and no video are investigated. Posts detected as private/gated are skipped (`reason: "private_or_gated"`). Among public posts, any `has_video` post @@ -1809,9 +2093,12 @@ re-apply annotations. Store annotations in extension state, not DOM. X uses a React SPA with aggressive DOM recycling. 1. `MutationObserver` to detect tweet content in viewport. -2. For individual tweet pages (`/status/{id}`), extract main tweet text. -3. Target `[data-testid="tweetText"]`. Acknowledge this selector is fragile and may need - maintenance. +2. For individual tweet pages (`/status/{id}`), extract main tweet text: the `
` that + links to the status permalink — when several do (replies, quotes), the one whose author + (first profile link) is the URL's handle. +3. Logged in, tweet text is `[data-testid="tweetText"]`; the logged-out frontend has no test + ids and renders it as the article's own first `div[dir="auto"]`. Acknowledge these + selectors are fragile and may need maintenance. **Media behavior:** Extract image URLs separately from video detection. Investigate image-only tweets. Skip private/protected tweets (`reason: "private_or_gated"`). Among accessible tweets, skip @@ -1825,14 +2112,23 @@ any `has_video` tweet (video present, even when extracted images and/or tweet te `chrome.scripting`. 3. `externalId` is the numeric Substack post ID parsed from social image metadata (`post_preview/{numericId}/twitter.jpg` pattern). -4. Content root selector: `.body.markup`. +4. Content root selector: `.body.markup`. Editor components in the body that are not the + post's own text are excluded from content text, image occurrences and transported HTML, + by their `data-component-name`: calls to action (`SubscribeWidget`, + `ButtonCreateButton` share/comment/subscribe buttons) and embeds of content from + elsewhere — tweets (`Twitter2ToDOM`: another author's words, like the quote tweets v1 + does not analyze, §2.1; the avatar and tweet media go with it) and other posts + (`DigestPostEmbed`, `EmbeddedPostToDOM`). 5. Subscriber-only/paywalled views are skipped (`reason: "private_or_gated"`) and are not sent - to `registerObservedVersion`/`recordViewAndGetStatus`/`investigateNow`. + to `registerObservedVersion`/`recordViewAndGetStatus`/`investigateNow`. A post whose JSON-LD + declares `isAccessibleForFree: false` is subscriber-only whatever paywall copy it shows — + including for a paid subscriber who can read it in full; paywall wording and markers are + checked as well. Because custom-domain Substack publishers can use arbitrary hostnames, the extension cannot -pre-enumerate all required origins in the manifest. v1 therefore keeps broad host permissions -and applies strict runtime checks before injection (path must be `/p/*` and Substack fingerprint -must be present). This is an intentional tradeoff for custom-domain support. +pre-enumerate all required origins in the manifest. v1 therefore keeps broad HTTPS host +permissions and applies strict runtime checks before injection (path must be `/p/*` and Substack +fingerprint must be present). This is an intentional tradeoff for custom-domain support. ## 3.12 Wikipedia Content Script @@ -1845,44 +2141,75 @@ JavaScript globals (`mw.config`). and the hostname language code. 3. The adapter reads MediaWiki config values (`wgArticleId`, `wgRevisionId`, `wgNamespaceNumber`, `wgPageName`, `wgRevisionTimestamp`) and filters out - non-article namespaces (namespace !== 0). -4. Content root: `#mw-content-text .mw-parser-output`. Excluded sections - (References, External links, Further reading, Notes, Bibliography, Sources, - Citations) and non-article elements (navboxes, infobox metadata, edit links, - etc.) are pruned before text extraction. "See also" is intentionally - **not** excluded — it contains substantive content about related topics. + non-article namespaces (namespace !== 0). `wgNamespaceNumber` is authoritative: + URL parsing only recognizes MediaWiki's canonical (English) namespace names, which + every language edition accepts, not localized ones such as German `Diskussion:`. +4. Content root: `#mw-content-text .mw-parser-output`. Excluded sections and + non-article elements are pruned before text extraction, by one predicate shared + with the API (`shared/src/wikipedia-canonicalization.ts`): + - Non-article elements are recognized by markup every wiki emits, whatever its + language: citation superscripts, reference lists, edit links, navboxes; anything + with `role="navigation"` (navboxes, series sidebars, "main article" links); the + `navigation-not-searchable` class (what Wikimedia search leaves out: hatnotes, + navboxes, authority control); the `metadata` class (maintenance and quality + banners, sister-project boxes, person-data tables); plus a few per-wiki boxes those + conventions miss (fr.wikipedia's portal bar, nl.wikipedia's appendix and + sister-project boxes). Inline styles are never read: Wikipedia's scripts and reader + interaction change them on the live page, so they cannot agree with the Parse API. + - Excluded sections are the appendices listing citations, sources and outbound links + (English: References, Notes, Further reading, External links, Bibliography, Sources, + Citations), matched by title in English and in the equivalent titles of the largest + wikis (de, fr, es, it, pt, nl, pl, ru, ja, zh). An excluded section is its heading and + the heading's following siblings up to the next sibling heading of the same or a + higher level (subsections go with it), which holds both for the flat Parse API + output and for read views that nest each section in a `
` element. + "See also" and its equivalents are intentionally **not** excluded — they contain + substantive content about related topics. 5. Server-side canonical fetch uses the MediaWiki `action=parse` API pinned to the observed `revisionId`, ensuring content verification matches exactly the revision the user saw. 6. Wikipedia articles have no single author; no `Author` row is linked. -**Media behavior:** Same rules as other platforms — extract image URLs and -occurrences from the article body; any `has_video` article is skipped. +**Media behavior:** Same rules as other platforms — extract image occurrences from the +article body; any `has_video` article is skipped. Video is detected on the live content +root, not through the text exclusions: Wikipedia's player script wraps `