From b2ac2e505424152bd01c149e127d23199f1e9390 Mon Sep 17 00:00:00 2001 From: edersonbrilhante Date: Fri, 25 Sep 2026 15:33:51 +0200 Subject: [PATCH 1/2] fix(scale-set): implement review followups --- .github/workflows/lambda.yml | 48 ++++++++++++ .github/workflows/smoke-tests.yml | 2 + MAINTAINERS.md | 2 - ...runner-orchestration-provider-boundary.md} | 0 docs/adr/0003-scale-set-resource-ownership.md | 69 +++++++++++++++++ docs/configuration.md | 75 ++++++++++++++++++ docs/examples/index.md | 1 + docs/examples/multi-runner-scale-set.md | 14 ++++ docs/index.md | 64 ++++++++++++++++ docs/multi-runner-v1-to-v2-configuration.md | 18 +++++ docs/multi-runner-v1-v2-migration.md | 18 +++++ docs/security.md | 37 ++++++++- lambdas/functions/control-plane/src/lambda.ts | 2 +- .../src/scale-runners/github-runner.ts | 2 - .../src/scale-runners/scale-up.test.ts | 12 --- .../aws/ssm/runner-config-housekeeper.test.ts | 76 +------------------ .../aws/ssm/runner-config-housekeeper.ts | 61 +++++++++++---- lambdas/libs/storage-providers/core/index.ts | 2 +- lambdas/services/scale-set/README.md | 9 ++- .../scale-set/src/credentials.test.ts | 5 +- lambdas/services/scale-set/src/credentials.ts | 4 - lambdas/services/scale-set/src/main.ts | 1 - .../services/scale-set/src/parameter-store.ts | 14 +--- .../services/scale-set/src/reconciler.test.ts | 3 +- lambdas/services/scale-set/src/reconciler.ts | 25 +----- mkdocs.yaml | 3 + modules/multi-runner/README.md | 2 +- .../tests/config-resolution.tftest.hcl | 9 +++ ...les.experimental.orchestration-provider.tf | 4 +- .../scale-set/README.md | 36 +++++---- .../orchestration-providers/scale-set/iam.tf | 38 ++++++---- .../scale-set/locals.tf | 15 +++- .../scale-set/logging.tf | 2 +- .../tests/fixtures/computed-inputs/main.tf | 4 + .../scale-set/tests/scale-set.tftest.hcl | 28 +++++-- .../scale-set/validations.tf | 14 ++-- .../scale-set/variables.tf | 10 +-- .../runner-config/ssm-housekeeper/README.md | 2 - 38 files changed, 519 insertions(+), 212 deletions(-) rename docs/adr/{002-runner-orchestration-provider-boundary.md => 0002-runner-orchestration-provider-boundary.md} (100%) create mode 100644 docs/adr/0003-scale-set-resource-ownership.md create mode 100644 docs/examples/multi-runner-scale-set.md diff --git a/.github/workflows/lambda.yml b/.github/workflows/lambda.yml index 09d96892a1..4a2decadc5 100644 --- a/.github/workflows/lambda.yml +++ b/.github/workflows/lambda.yml @@ -32,17 +32,23 @@ jobs: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: persist-credentials: false + - name: Install dependencies run: yarn install --frozen-lockfile + - name: Run prettier run: yarn format-check + - name: Run linter run: yarn lint + - name: Run tests id: test run: yarn test + - name: Build distribution run: yarn build + - name: Upload coverage report uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 if: ${{ failure() }} @@ -50,3 +56,45 @@ jobs: name: coverage-reports path: ./**/coverage retention-days: 5 + + scale-set-container: + name: Build scale-set service container + runs-on: ubuntu-latest + steps: + - name: Harden the runner (Audit all outbound calls) + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 + with: + egress-policy: audit + + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + - name: Set up QEMU + uses: docker/setup-qemu-action@96fe6ef7f33517b61c61be40b68a1882f3264fb8 # v4.2.0 + + - name: Set up Docker Buildx + uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0 + + - name: Build scale-set service image + uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7.3.0 + with: + context: . + file: ./lambdas/services/scale-set/Dockerfile + platforms: linux/amd64,linux/arm64 + push: false + cache-from: type=gha,scope=scale-set-service + cache-to: type=gha,mode=max,scope=scale-set-service + + - name: Build scale-set service image for smoke test + uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7.3.0 + with: + context: . + file: ./lambdas/services/scale-set/Dockerfile + platforms: linux/amd64 + load: true + tags: scale-set-service:smoke-test + cache-from: type=gha,scope=scale-set-service + + - name: Run scale-set service image smoke test + run: ./.github/scripts/scale-set-container-smoke-test.sh diff --git a/.github/workflows/smoke-tests.yml b/.github/workflows/smoke-tests.yml index 761c9278e1..85ca7b44c0 100644 --- a/.github/workflows/smoke-tests.yml +++ b/.github/workflows/smoke-tests.yml @@ -93,6 +93,8 @@ jobs: image: ghcr.io/ministackorg/ministack:1.5.12@sha256:41fe1ce2e666c6cc410c6047a9db8bf1df69cd0028ebc0a6c6e5517c3a83d6e0 ports: - 4566:4566 + # MiniStack launches nested service containers for the integration smoke test. + # This digest-pinned, non-fork workflow is the narrowly scoped Docker-daemon exception. options: >- --add-host=host.docker.internal:host-gateway --volume /var/run/docker.sock:/var/run/docker.sock diff --git a/MAINTAINERS.md b/MAINTAINERS.md index b9232f8177..38a67dfec4 100644 --- a/MAINTAINERS.md +++ b/MAINTAINERS.md @@ -48,8 +48,6 @@ The following steps needs to be applied to test a PR 3. Apply the PR to the deployment. Check output for breaking changes such as destroying resources containing state. 4. Test the PR by running a workflow -Some PR tests can be run against [MiniStack](tests/ministack/README.md), including the example deployments and webhook and runner lifecycle [smoke tests](tests/ministack/README.md#webhook-and-runner-lifecycle-smoke-test). Use these tests where applicable during PR review. MiniStack test coverage is still being expanded, with additional cases in progress. - ### Security Act on security issues as soon as possible. If a security issue is reported. diff --git a/docs/adr/002-runner-orchestration-provider-boundary.md b/docs/adr/0002-runner-orchestration-provider-boundary.md similarity index 100% rename from docs/adr/002-runner-orchestration-provider-boundary.md rename to docs/adr/0002-runner-orchestration-provider-boundary.md diff --git a/docs/adr/0003-scale-set-resource-ownership.md b/docs/adr/0003-scale-set-resource-ownership.md new file mode 100644 index 0000000000..9b12268707 --- /dev/null +++ b/docs/adr/0003-scale-set-resource-ownership.md @@ -0,0 +1,69 @@ +# ADR-003: Scale-set Resource Ownership + +## Status + +Accepted + +## Date + +2026-09-23 + +## Context + +GitHub Actions runner scale sets are GitHub-side resources with their own +identity and permissions. The TypeScript controller is the component that +configures GitHub through the scale-set API. Terraform only provisions the +AWS controller substrate and supplies the controller with names and secret +references; Terraform never configures the GitHub scale set itself. + +The controller resolves and manages the runtime relationship with a scale set +and runner group from the configured GitHub scope and names. The GitHub API +lifecycle is controller-owned at runtime, not implemented by Terraform. + +## Decision + +The scale-set orchestration provider resolves GitHub scale sets by name. + +- `scale_set.name` identifies the GitHub scale set to reconcile. +- `runner.group_name` identifies the runner group used when the scale set is + resolved by name. +- Terraform creates and manages the ECS controller, IAM roles, networking, + logs, and SSM configuration required to run the reconciler. It never calls + the GitHub scale-set API. +- The controller may discover the scale-set and runner-group IDs at runtime, + but it does not write those discovered IDs back to SSM. Operators may + pre-populate optional cache parameters when they want read-side caching. +- If the named scale set is absent, the controller may register it in the + resolved runner group and reconcile its system labels. It does not delete + scale sets. +- The controller currently registers a missing scale set and reconciles its + system labels. Terraform destroy removes the AWS controller and stops future + reconciliation, but it cannot delete or rename the GitHub resources because + Terraform never configures them. + +The configured GitHub scope and scale-set name must be unique across controller +groups. A single runner configuration must not be selected by both webhook and +scale-set orchestration. + +## Consequences + +This keeps ownership boundaries explicit: the TypeScript controller is the sole +GitHub API owner, while Terraform owns only the AWS deployment and its input +references. Destroying Terraform stops the controller but does not issue a +GitHub delete. Operators must authorize the configured GitHub scope before +applying the AWS controller configuration and must handle renames as an +explicit migration. The controller task role can remain read-only for SSM +discovery and credential reads, reducing its blast radius. + +## Alternatives considered + +- **Model GitHub scale sets as Terraform resources:** not selected for the + current implementation. The TypeScript controller already owns all runtime + GitHub API operations (lookup, registration, and label reconciliation), while + Terraform owns the AWS substrate. Adding a second Terraform owner would + create competing GitHub API lifecycles and require an explicit import, + update, and destroy contract. +- **Persist every discovered ID from the controller:** rejected for now + because it would require explicit, caller-visible SSM write targets and an + expanded IAM contract. Read-side caching remains possible through + pre-provisioned parameters. diff --git a/docs/configuration.md b/docs/configuration.md index 6fd974fcc1..2a8534e162 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -296,6 +296,81 @@ In case the setup does not work as intended, trace the events through this seque ## Experimental features +### GitHub Actions runner scale-set orchestration + +Scale-set orchestration is an experimental multi-runner v2 provider for +workloads that should use GitHub's runner scale-set message protocol instead of +webhook-driven Lambda scaling. Select it inside the lane's +`multi_runner_config..orchestration_provider` block and pair it with a +compute provider that implements the scale-set capability contract. The +controller runs as one ECS Fargate task per resolved controller group and +reconciles the configured scale sets continuously. + +Use scale-set orchestration when the GitHub scale-set API and a long-lived +controller are the desired ownership model. Continue using webhook +orchestration when the existing `workflow_job` event, SQS, and Lambda lifecycle +are the better fit. The two modes must not manage the same runner lane. + +The scale-set module resolves GitHub scale sets by their configured name. When +a named scale set is absent, the controller registers it in the resolved +runner group and reconciles its system labels at runtime. The TypeScript +controller is the component that configures GitHub; Terraform never calls the +GitHub scale-set API. Terraform destroy removes the AWS controller and stops +reconciliation, but does not issue a GitHub delete. The module requires an +explicit controller image, preferably an immutable digest. The task reads GitHub App credentials +and optional discovery-cache values from SSM but does not write discovered IDs +back to SSM. See the [scale-set provider reference](https://github.com/github-aws-runners/terraform-aws-github-runner/blob/main/modules/orchestration-providers/scale-set/README.md) +for the complete input schema. + +#### Scale-set options and defaults + +Set these values under +`global_config_orchestration_provider.scale_set`. Per-lane scale-set values +under `multi_runner_config..orchestration_provider.scale_set` override +the corresponding lane settings. The controller image is represented as +optional in the Terraform type for normalization, but validation requires a +non-empty value; use an immutable digest. + +| Option | Default | Purpose | +| --- | --- | --- | +| `grouping.strategy` | `compute_provider` | Pack reconcilers by compute-provider type; use `runner_config` or `custom` to create narrower task/IAM boundaries. | +| `container.image` | none; required | Controller image reference. Prefer a release digest. | +| `container.user` | `10001:10001` | Numeric non-root UID/GID used by the application container. | +| `container.health_port` | `8080` | ECS health-check port. | +| `container.health_path` | `/healthz` | ECS liveness endpoint; `/readyz` is an application readiness signal. | +| `container.health_check_command` | `null` | Use the image health check unless an explicit ECS command is required. | +| `container.health_check_interval` / `timeout` / `retries` | `30` / `5` / `3` | ECS container health-check timing. | +| `container.health_check_start_period` | `30` | Startup grace period for the ECS health check. | +| `container.health_stale_after_seconds` | `180` | Controller health staleness threshold. | +| `container.shutdown_timeout_seconds` | `110` | Controller shutdown grace period. | +| `container.session_close_timeout_seconds` | `10` | Message-session close timeout. | +| `container.reconnect_initial_backoff_seconds` / `max` | `1` / `30` | Bounds for reconnect backoff. | +| `container.stop_timeout_seconds` | `120` | ECS container stop timeout. | +| `config_store.path_prefix` / `tier` | derived / `Standard` | SSM path prefix and parameter tier for non-secret reconciler configuration. | +| `ecs.cluster.mode` | `managed` | Create a cluster or use an external cluster. | +| `ecs.cluster.container_insights` | `true` | Enable ECS container insights on a managed cluster. | +| `ecs.task.cpu` / `memory` | `512` / `1024` | Fargate task CPU units and memory MiB. | +| `ecs.task.cpu_architecture` | `X86_64` | Fargate task architecture. | +| `ecs.task.ephemeral_storage` | `null` | Use the Fargate platform default unless a size is supplied. | +| `ecs.service.platform_version` | `LATEST` | ECS Fargate platform version. | +| `ecs.iam.path` / `permissions_boundary` | `/` / `null` | IAM role path and optional permissions boundary. | +| `network.vpc_id` / `subnet_ids` | required | Private subnets in which the controller service runs. | +| `network.https_egress.ipv4_cidrs` | `0.0.0.0/0` | Default HTTPS reachability; restrict through GitHub Meta API ranges, NAT, firewall, or proxy as required. | +| `network.https_egress.ipv6_cidrs` | `[]` | IPv6 HTTPS egress destinations. | +| `logging.retention_in_days` / `kms_key_id` | `180` / `null` | CloudWatch log retention and optional customer-managed KMS key ID, alias, or ARN. | +| `logging.log_group_class` | `STANDARD` | CloudWatch log-group class. | +| `tags` | `{}` | Tags applied to scale-set resources. | + +The scale-set lane itself defaults to `runner.group_name = "Default"`, +`runner.min_runners = 0`, `runner.max_runners = 10`, and +`runner.boot_time_in_minutes = 10`. Configure the GitHub scope, scale-set +name, runner owner, and GitHub App SSM references in the lane; credential values +are not placed in the controller manifest. + +The [multi-runner scale-set example](multi-runner-scale-set.md) shows how these +provider-specific settings coexist with webhook lanes in the same v2 +`multi_runner_config` map. + ### macOS Runners This feature is in early stage and should be considered experimental. The module supports macOS-based GitHub Actions self-hosted runners on AWS EC2 Mac instances (`mac1.metal`, `mac2.metal`, `mac2-m2.metal`). macOS runners require dedicated hosts due to Apple's licensing requirements and have longer boot times (6–20 minutes). Set `runner_os = "osx"` and `use_dedicated_host = true` to enable. See the full [macOS Runners documentation](mac-runners.md) for details. diff --git a/docs/examples/index.md b/docs/examples/index.md index f0558966bd..50aff55389 100644 --- a/docs/examples/index.md +++ b/docs/examples/index.md @@ -6,6 +6,7 @@ Examples are located in the [examples](https://github.com/github-aws-runners/ter - _[Ephemeral](ephemeral.md)_: Example usages of ephemeral runners based on the default example. - _[Multi Runner](multi-runner.md)_ : Example usage of creating a multi runner which creates multiple runners/ configurations with a single deployment. The examples including: "arm64", "windows", and "ubuntu" runners. - _[Multi Runner v2](multi-runner-v2.md)_ : Example usage of the experimental v2 multi-runner configuration interface with shared defaults and per-lane overrides. +- _[Multi Runner scale-set](multi-runner-scale-set.md)_ : Example usage of a v2 deployment combining webhook lanes with an experimental GitHub Actions scale-set lane. - _[Permissions boundary](permissions-boundary.md)_: Example usages of permissions boundaries. - _[Prebuilt Images](prebuilt.md)_: Example usages of deploying runners with a custom prebuilt image. - _[Termination watcher](termination-watcher.md)_: Example usages of termination watcher. diff --git a/docs/examples/multi-runner-scale-set.md b/docs/examples/multi-runner-scale-set.md new file mode 100644 index 0000000000..3aefbd3056 --- /dev/null +++ b/docs/examples/multi-runner-scale-set.md @@ -0,0 +1,14 @@ +# Multi-runner scale-set example + +This example combines ordinary webhook-managed lanes with one experimental +GitHub Actions runner scale-set lane. It demonstrates that v2 keeps the +deployment-wide defaults in `global_config*` and places orchestration and +compute-provider settings inside each `multi_runner_config` lane. + +The source example is available at +[examples/multi-runner-scale-set](https://github.com/github-aws-runners/terraform-aws-github-runner/tree/main/examples/multi-runner-scale-set). +Read its README before applying: the GitHub App values are sensitive, the +scale-set controller image must be supplied explicitly, and the GitHub scale +set/runner group must be authorized for the selected GitHub scope. + +--8<-- "examples/multi-runner-scale-set/README.md" diff --git a/docs/index.md b/docs/index.md index 7a7d0f70c6..9df295ee02 100644 --- a/docs/index.md +++ b/docs/index.md @@ -21,6 +21,70 @@ For ephemeral runners a pool can be configured. The pool maintains a minimum num For non ephemeral runners with the idle config the module will avoid scaling down back to zero. Instead it will maintain a minimum number of runners based on a schedule. This avoids the need to scale up when a new workflow is triggered. +### Scale-set orchestration (experimental) + +Multi-runner v2 can select the experimental scale-set orchestration provider for a +runner lane. Scale-set orchestration uses a long-running ECS Fargate controller +instead of webhook events and Lambda scale-up/scale-down handlers. The +controller maintains a message session with GitHub, reconciles the desired +capacity reported by the scale-set API, and delegates runner provisioning to the +selected compute provider. + +The controller flow is: + +1. Terraform creates one ECS service, task definition, task role, execution + role, log group, security group, and SSM configuration set for each resolved + controller group. +2. The controller loads non-secret configuration from its task manifest or SSM + group path. GitHub App values remain in caller-managed SSM parameters and + only their names are passed to the task. +3. The controller resolves the configured GitHub scale set by name, registers + it when absent, opens its message session, and reconciles capacity through + the compute provider. The TypeScript controller owns those runtime API + operations; Terraform only provisions AWS and never calls the GitHub + scale-set API. Terraform destroy stops reconciliation without issuing a + GitHub delete. +4. The compute provider creates, refreshes, and removes runner capacity. For + EC2, JIT configuration and instance lifecycle remain provider-owned. + +The request and lifecycle path is: + +```mermaid +flowchart LR + TF[Terraform\nmulti-runner v2] --> ECS[ECS Fargate\nscale-set controller] + TF --> SSM[(SSM\nconfig and secret references)] + ECS -->|GitHub App auth| API[GitHub Actions\nscale-set APIs] + API -->|desired capacity and jobs| ECS + ECS -->|JIT configuration| EC2[EC2 compute provider] + EC2 -->|runner registration| API + ECS -->|scale-up / scale-down| EC2 + EC2 -->|terminate owned capacity| EC2 +``` + +The controller is long-lived and does not receive `workflow_job` webhooks. It +opens a GitHub scale-set message session, acknowledges and processes messages, +then passes desired capacity and busy-runner information to the compute +provider. The EC2 provider publishes JIT configuration through SSM, launches +the runner instance, and later removes only its owned runner capacity when the +scale-set session reports that it is safe to scale down. + +The service is compatible with the wire behavior implemented by the upstream +[GitHub Actions scale-set client](https://github.com/actions/scaleset). Its +HTTP paths, message-session behavior, statistics handling, and User-Agent +requirements were implemented by reverse-engineering that Go client and its +protocol behavior. This is a compatibility boundary rather than a promise that +the upstream internal API is stable; validate changes in `actions/scaleset` +before upgrading the controller image. + +Controller groups can be formed per compute-provider type, per runner config, +or through explicit custom membership. Grouping shares a task and IAM policy, +so split groups when blast radius or policy size must be reduced. This provider +is experimental: callers must supply an explicit controller image, use the +scale-set configuration contract, and verify the current provider and GitHub +API limitations before production use. See the [scale-set module +documentation](https://github.com/github-aws-runners/terraform-aws-github-runner/blob/main/modules/orchestration-providers/scale-set/README.md) and the +[v1-to-v2 configuration guide](multi-runner-v1-to-v2-configuration.md). + ## Detailed design diff --git a/docs/multi-runner-v1-to-v2-configuration.md b/docs/multi-runner-v1-to-v2-configuration.md index ab968a3960..f76ec554d3 100644 --- a/docs/multi-runner-v1-to-v2-configuration.md +++ b/docs/multi-runner-v1-to-v2-configuration.md @@ -24,6 +24,7 @@ The v2 interface has two levels: The current providers are: - `orchestration_provider.webhook` +- `orchestration_provider.scale_set` (experimental) - `compute_provider.aws.ec2` Global provider blocks supply defaults and shared settings. They do not select @@ -130,6 +131,21 @@ lane only when that lane needs a different value. Each v2 lane is keyed by the same logical runner name used in v1, but its settings use canonical nested blocks: +### Choosing scale-set orchestration + +Use the webhook provider when the lane should receive `workflow_job` events and +scale EC2 runners through SQS and Lambda. Use the scale-set provider when the +lane should be reconciled by a long-running ECS controller through GitHub's +runner scale-set APIs. Scale-set lanes resolve GitHub scale sets by name and +may register a missing set at runtime. The TypeScript controller owns those +runtime API operations; Terraform only provisions AWS and never calls the +GitHub scale-set API. + +The scale-set provider is experimental and requires an explicit immutable +controller image plus a compute-provider scale-set capability. Its controller +groups share task resources and IAM permissions, so choose `runner_config` or +`custom` grouping when lanes need separate failure or permission boundaries. + ```hcl multi_runner_config = { large = { @@ -213,6 +229,8 @@ either a root module variable or an attribute under | `enable_ami_housekeeper` and related settings | `global_config_compute_provider.aws.ec2.ami.housekeeper` | | termination watcher settings | `global_config_compute_provider.aws.ec2.instance_termination_watcher` | | `log_level`, `log_class`, `logging_retention_in_days`, and tracing/metrics settings | `global_config_observability` | +| webhook orchestration settings such as `enable_jit_config`, queues, and matcher rules | `multi_runner_config..orchestration_provider.webhook` | +| scale-set orchestration and controller settings | `multi_runner_config..orchestration_provider.scale_set` and the internal scale-set module inputs | Settings that are specific to one lane should remain in that lane instead of being copied into a global block. diff --git a/docs/multi-runner-v1-v2-migration.md b/docs/multi-runner-v1-v2-migration.md index 9671613d90..b7735b43a9 100644 --- a/docs/multi-runner-v1-v2-migration.md +++ b/docs/multi-runner-v1-v2-migration.md @@ -10,6 +10,24 @@ changed to v2. This procedure is intended for an existing deployment that uses the multi-runner v1 configuration. +## Choose the orchestration provider before migrating + +State migration only changes Terraform addresses; it does not convert a +webhook lane into a scale-set lane or create a GitHub scale set. Keep a lane on +`orchestration_provider.webhook` when it should continue using `workflow_job` +events, SQS, and Lambda scaling. Select the experimental scale-set provider when +the lane should use a long-running ECS controller and GitHub's runner scale-set +message protocol. + +For a scale-set lane, authorize the GitHub scale-set API for the configured +scope and configure the scale-set and runner-group names in v2. The controller +resolves those names and may register a missing scale set. The TypeScript +controller owns the GitHub API operations; Terraform only provisions AWS and +never calls the GitHub scale-set API. +Provide an explicit immutable controller image, verify private networking and +HTTPS egress, and ensure the selected compute provider implements the scale-set +capability contract before applying the migrated configuration. + ## Before you start - Use the migration script from the same repository revision as the v2 module diff --git a/docs/security.md b/docs/security.md index 4ef4d17b94..07b3d3fe96 100644 --- a/docs/security.md +++ b/docs/security.md @@ -16,7 +16,7 @@ The examples are using standard AMI's for different operating systems. Instances The module is released using GitHub Actions and the Lambda artifacts are attached to the release. The release pipeline creates provenance attestations for those artifacts. You can find a link to the attestation in the GitHub release. The attestation only provides provenance information about the release; it is not a security guarantee. We recommend verifying the attestation after downloading the Lambda artifacts. -Releases also publish the multi-architecture scale-set service image to the GitHub Container Registry with an SBOM, build provenance, and a registry attestation. The convenience image default follows the latest module release. Production deployments should override it with the immutable image digest printed in the release notes, then verify that image with: +Releases also publish the multi-architecture scale-set service image to the GitHub Container Registry with an SBOM, build provenance, and a registry attestation. The scale-set module requires callers to select the controller image explicitly; use the immutable image digest printed in the release notes and verify that image with: ```bash gh attestation verify \ @@ -24,4 +24,39 @@ gh attestation verify \ --repo github-aws-runners/terraform-aws-github-runner ``` +## Scale-set security boundaries + +The experimental scale-set provider has separate trust boundaries for the ECS +controller, the EC2 compute provider, and GitHub: + +- The ECS task role reads only the SSM parameter names supplied for its group, + including GitHub App references and optional discovery-cache values. The + controller does not write discovered IDs back to SSM. The compute role owns + the separate SSM write/delete permissions needed to publish and consume + runner JIT configuration. +- GitHub App private keys and tokens remain in SSM and are never serialized into + the controller manifest. Use caller-managed KMS keys and parameter policies + when the default AWS-owned SSM key is insufficient for your boundary. +- Tasks run in private subnets without public IPs, with no managed security + group ingress and TCP/443 egress only. The default `0.0.0.0/0` route is a + reachability default, not a GitHub allowlist. GitHub publishes current + outbound ranges through [`api.github.com/meta`](https://api.github.com/meta); + use those ranges or a controlled NAT, firewall, or HTTPS proxy where needed. +- ECS hardening includes a numeric non-root user, a read-only root filesystem, + dropped Linux capabilities, no privilege escalation, and no Docker socket. + The task image should be digest-pinned and independently verified. +- Terraform owns the AWS controller substrate and the EC2 runner capacity + contract. The controller owns the runtime GitHub scale-set API operations, + including resolving a named set, registering a missing set, and reconciling + system labels. Terraform only provisions the AWS substrate and never calls + the GitHub scale-set API; destroying it stops reconciliation without issuing + a GitHub delete. Avoid configuring the same runner lane in both webhook and + scale-set orchestration. + +The scale-set service follows the message-session and HTTP behavior of the +upstream [`actions/scaleset`](https://github.com/actions/scaleset) Go client. +That protocol was reverse-engineered for compatibility, so treat upstream +changes and GitHub API behavior as operational dependencies and validate image +updates before production rollout. + --8<-- "SECURITY.md:mkdocsrunners" diff --git a/lambdas/functions/control-plane/src/lambda.ts b/lambdas/functions/control-plane/src/lambda.ts index ea4ddc56fc..4594a1289e 100644 --- a/lambdas/functions/control-plane/src/lambda.ts +++ b/lambdas/functions/control-plane/src/lambda.ts @@ -124,7 +124,7 @@ export async function runnerConfigHousekeeper(event: unknown, context: Context): const housekeeper = createRunnerConfigHousekeeper(); try { - await housekeeper.houseKeeper(() => context.getRemainingTimeInMillis()); + await housekeeper.houseKeeper(); } catch (e) { logger.error(`${(e as Error).message}`, { error: e as Error }); } diff --git a/lambdas/functions/control-plane/src/scale-runners/github-runner.ts b/lambdas/functions/control-plane/src/scale-runners/github-runner.ts index e957ef6f88..f968fa3008 100644 --- a/lambdas/functions/control-plane/src/scale-runners/github-runner.ts +++ b/lambdas/functions/control-plane/src/scale-runners/github-runner.ts @@ -69,8 +69,6 @@ async function getGithubRunnerRegistrationToken(githubRunnerConfig: CreateGitHub repo: githubRunnerConfig.runnerOwner.split('/')[1], }); - metricGitHubAppRateLimit(registrationToken.headers, githubRunnerConfig.appIndex); - return registrationToken.data.token; } diff --git a/lambdas/functions/control-plane/src/scale-runners/scale-up.test.ts b/lambdas/functions/control-plane/src/scale-runners/scale-up.test.ts index f83a984ec2..a8ef79c3a9 100644 --- a/lambdas/functions/control-plane/src/scale-runners/scale-up.test.ts +++ b/lambdas/functions/control-plane/src/scale-runners/scale-up.test.ts @@ -3,7 +3,6 @@ import { performance } from 'perf_hooks'; import { controlPlaneProviderRegistry } from '../control-plane-providers'; import * as ghAuth from '../github/auth'; -import * as rateLimitModule from '../github/rate-limit'; import { createStartRunnerConfig } from './github-runner'; import { publishRetryMessage } from './job-retry'; import * as scaleUpModule from './scale-up'; @@ -341,13 +340,6 @@ describe('scaleUp with GHES', () => { expect(mockOctokit.actions.createRegistrationTokenForRepo).not.toBeCalled(); }); - it('reports the GitHub App rate limit from the registration token response', async () => { - process.env.ENABLE_EPHEMERAL_RUNNERS = 'false'; - const metricSpy = vi.spyOn(rateLimitModule, 'metricGitHubAppRateLimit'); - await scaleUpModule.scaleUp(TEST_DATA); - expect(metricSpy).toHaveBeenCalledWith({ 'x-ratelimit-remaining': '4999', 'x-ratelimit-limit': '5000' }, 0); - }); - it('creates a runner with labels in a specific group', async () => { process.env.RUNNER_LABELS = 'label1,label2'; process.env.RUNNER_GROUP_NAME = 'TEST_GROUP'; @@ -2339,10 +2331,6 @@ function defaultOctokitMockImpl() { data: { token: '1234abcd', }, - headers: { - 'x-ratelimit-remaining': '4999', - 'x-ratelimit-limit': '5000', - }, }; const mockInstallationIdReturnValueOrgs = { data: { diff --git a/lambdas/libs/storage-providers/aws/ssm/runner-config-housekeeper.test.ts b/lambdas/libs/storage-providers/aws/ssm/runner-config-housekeeper.test.ts index e8b8b11c1c..1c16607333 100644 --- a/lambdas/libs/storage-providers/aws/ssm/runner-config-housekeeper.test.ts +++ b/lambdas/libs/storage-providers/aws/ssm/runner-config-housekeeper.test.ts @@ -2,7 +2,7 @@ import { DeleteParameterCommand, GetParametersByPathCommand, SSMClient } from '@ import { mockClient } from 'aws-sdk-client-mock'; import 'aws-sdk-client-mock-jest/vitest'; import { cleanSSMTokens } from './runner-config-housekeeper'; -import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest'; +import { describe, it, expect, beforeEach } from 'vitest'; process.env.AWS_REGION = 'eu-east-1'; @@ -16,7 +16,6 @@ dateOld.setDate(dateOld.getDate() - deleteAmisOlderThenDays - 1); const tokenPath = '/path/to/tokens/'; describe('clean SSM tokens / JIT config', () => { - afterEach(() => vi.unstubAllEnvs()); beforeEach(() => { mockSSMClient.reset(); mockSSMClient.on(GetParametersByPathCommand).resolves({ @@ -54,79 +53,6 @@ describe('clean SSM tokens / JIT config', () => { expect(mockSSMClient).not.toHaveReceivedCommandWith(DeleteParameterCommand, { Name: tokenPath + 'i-new-01' }); }); - it.each([undefined, []])('keeps later pages when the first page has no parameters (%s)', async (firstPage) => { - mockSSMClient.reset(); - mockSSMClient - .on(GetParametersByPathCommand) - .resolvesOnce({ Parameters: firstPage, NextToken: 'empty-page' }) - .resolvesOnce({ NextToken: 'last-page' }) - .resolvesOnce({ Parameters: [{ Name: tokenPath + 'i-old-later', LastModifiedDate: dateOld }] }); - - await cleanSSMTokens({ dryRun: false, minimumDaysOld: 1, tokenPath }); - - expect(mockSSMClient).toHaveReceivedCommandTimes(GetParametersByPathCommand, 3); - expect(mockSSMClient).toHaveReceivedCommandWith(GetParametersByPathCommand, { - Path: tokenPath, - NextToken: 'last-page', - }); - expect(mockSSMClient).toHaveReceivedCommandWith(DeleteParameterCommand, { Name: tokenPath + 'i-old-later' }); - }); - - it('keeps deletions from earlier pages when a later listing page fails', async () => { - mockSSMClient - .on(GetParametersByPathCommand, { Path: tokenPath, NextToken: 'next' }) - .rejects(new Error('SSM unavailable')); - - await expect(cleanSSMTokens({ dryRun: false, minimumDaysOld: 1, tokenPath })).rejects.toThrow('SSM unavailable'); - - expect(mockSSMClient).toHaveReceivedCommandWith(DeleteParameterCommand, { Name: tokenPath + 'i-old-01' }); - }); - - it('starts fresh against remaining parameters after an interrupted invocation', async () => { - let remaining = 60000; - const inventory = [ - { Name: tokenPath + 'first', LastModifiedDate: dateOld }, - { Name: tokenPath + 'second', LastModifiedDate: dateOld }, - ]; - mockSSMClient.reset(); - mockSSMClient.on(GetParametersByPathCommand).callsFake(() => ({ Parameters: [...inventory] })); - mockSSMClient.on(DeleteParameterCommand).callsFake((input) => { - inventory.splice( - inventory.findIndex((item) => item.Name === input.Name), - 1, - ); - remaining = 0; - return {}; - }); - await cleanSSMTokens({ dryRun: false, minimumDaysOld: 1, tokenPath }, () => remaining); - expect(inventory).toHaveLength(1); - mockSSMClient.resetHistory(); - await cleanSSMTokens({ dryRun: false, minimumDaysOld: 1, tokenPath }); - expect(mockSSMClient.commandCalls(GetParametersByPathCommand)[0].args[0].input.NextToken).toBeUndefined(); - expect(mockSSMClient).not.toHaveReceivedCommandWith(DeleteParameterCommand, { Name: tokenPath + 'first' }); - expect(inventory).toHaveLength(0); - }); - - it('continues past a failed deletion within the same invocation', async () => { - mockSSMClient.on(GetParametersByPathCommand, { Path: tokenPath }).resolves({ - Parameters: [ - { Name: tokenPath + 'failed', LastModifiedDate: dateOld }, - { Name: tokenPath + 'healthy', LastModifiedDate: dateOld }, - ], - }); - mockSSMClient.on(DeleteParameterCommand, { Name: tokenPath + 'failed' }).rejects(new Error('Denied')); - await cleanSSMTokens({ dryRun: false, minimumDaysOld: 1, tokenPath }); - expect(mockSSMClient).toHaveReceivedCommandWith(DeleteParameterCommand, { Name: tokenPath + 'healthy' }); - }); - - it('deletes a page before requesting the next page', async () => { - mockSSMClient.on(GetParametersByPathCommand, { Path: tokenPath, NextToken: 'next' }).callsFake(() => { - expect(mockSSMClient).toHaveReceivedCommandWith(DeleteParameterCommand, { Name: tokenPath + 'i-old-01' }); - return {}; - }); - await cleanSSMTokens({ dryRun: false, minimumDaysOld: 1, tokenPath }); - }); - it('should not delete when dry run is activated', async () => { await cleanSSMTokens({ dryRun: true, diff --git a/lambdas/libs/storage-providers/aws/ssm/runner-config-housekeeper.ts b/lambdas/libs/storage-providers/aws/ssm/runner-config-housekeeper.ts index b548384b30..8bd657b22a 100644 --- a/lambdas/libs/storage-providers/aws/ssm/runner-config-housekeeper.ts +++ b/lambdas/libs/storage-providers/aws/ssm/runner-config-housekeeper.ts @@ -1,4 +1,9 @@ -import { DeleteParameterCommand, GetParametersByPathCommand, SSMClient } from '@aws-sdk/client-ssm'; +import { + DeleteParameterCommand, + GetParametersByPathCommand, + SSMClient, + type GetParametersByPathCommandOutput, +} from '@aws-sdk/client-ssm'; import { getTracedAWSV3Client } from '@aws-github-runner/aws-powertools-util'; import type { RunnerConfigHousekeeper } from '../../core'; @@ -16,7 +21,7 @@ export function createAwsSsmRunnerConfigHousekeeper(options?: SSMCleanupOptions) return new AwsSsmRunnerConfigHousekeeper(options ?? loadCleanupOptions()); } -export async function cleanSSMTokens(options: SSMCleanupOptions, remainingTime = () => Infinity): Promise { +export async function cleanSSMTokens(options: SSMCleanupOptions): Promise { validateOptions(options); logger.info('Cleaning expired runner configurations', { minimumDaysOld: options.minimumDaysOld, @@ -25,39 +30,63 @@ export async function cleanSSMTokens(options: SSMCleanupOptions, remainingTime = }); const client = getTracedAWSV3Client(new SSMClient({ region: process.env.AWS_REGION })); - let nextToken: string | undefined; + let parameters: GetParametersByPathCommandOutput; + try { + parameters = await client.send(new GetParametersByPathCommand({ Path: options.tokenPath })); + while (parameters.NextToken) { + const nextParameters = await client.send( + new GetParametersByPathCommand({ Path: options.tokenPath, NextToken: parameters.NextToken }), + ); + parameters.Parameters?.push(...(nextParameters.Parameters ?? [])); + parameters.NextToken = nextParameters.NextToken; + } + } catch (error) { + logger.error('Failed to list runner configurations', { + tokenPath: options.tokenPath, + errorNames: getErrorNames(error), + }); + throw error; + } + logger.info('Found runner configurations', { + tokenPath: options.tokenPath, + parameterCount: parameters.Parameters?.length ?? 0, + }); + const minimumDate = new Date(); minimumDate.setDate(minimumDate.getDate() - options.minimumDaysOld); - do { - if (remainingTime() < 10000) return; - const page = await client.send(new GetParametersByPathCommand({ Path: options.tokenPath, NextToken: nextToken })); - for (const parameter of page.Parameters ?? []) { - if (remainingTime() < 10000) return; - if (!parameter.Name || !parameter.LastModifiedDate || !(new Date(parameter.LastModifiedDate) < minimumDate)) - continue; - logger.info('Deleting expired runner configuration', { parameterName: parameter.Name, dryRun: options.dryRun }); + + for (const parameter of parameters.Parameters ?? []) { + if (parameter.LastModifiedDate && new Date(parameter.LastModifiedDate) < minimumDate) { + logger.info('Deleting expired runner configuration', { + parameterName: parameter.Name, + lastModifiedDate: parameter.LastModifiedDate, + dryRun: options.dryRun, + }); try { if (!options.dryRun) { await new Promise((resolve) => setTimeout(resolve, 50)); await client.send(new DeleteParameterCommand({ Name: parameter.Name })); } } catch (error) { - // Failed items remain in the inventory for the next complete sweep. logger.warn('Failed to delete expired runner configuration', { parameterName: parameter.Name, errorNames: getErrorNames(error), }); } + } else { + logger.debug('Skipping runner configuration that is not expired', { + parameterName: parameter.Name, + lastModifiedDate: parameter.LastModifiedDate, + }); } - nextToken = page.NextToken; - } while (nextToken); + } } class AwsSsmRunnerConfigHousekeeper implements RunnerConfigHousekeeper { constructor(private readonly options: SSMCleanupOptions) {} - houseKeeper(remainingTime?: () => number): Promise { - return cleanSSMTokens(this.options, remainingTime); + houseKeeper(): Promise { + return cleanSSMTokens(this.options); } } diff --git a/lambdas/libs/storage-providers/core/index.ts b/lambdas/libs/storage-providers/core/index.ts index 86950eb2c7..e6060eae2a 100644 --- a/lambdas/libs/storage-providers/core/index.ts +++ b/lambdas/libs/storage-providers/core/index.ts @@ -17,7 +17,7 @@ export interface RunnerConfigStore { } export interface RunnerConfigHousekeeper { - houseKeeper(remainingTime?: () => number): Promise; + houseKeeper(): Promise; } export interface GitHubAppCredential { diff --git a/lambdas/services/scale-set/README.md b/lambdas/services/scale-set/README.md index 5cfe30ef9e..af4946402e 100644 --- a/lambdas/services/scale-set/README.md +++ b/lambdas/services/scale-set/README.md @@ -63,7 +63,7 @@ The service reads every direct child under the SSM path with paginated `GetParam } ``` -`scaleSetName` and `runnerGroupName` are the GitHub names supplied by the operator. `runnerGroupIdParameterName` is an optional SSM cache path. When present, the service reads the runner-group ID from that parameter; if it is missing, the service resolves the name through the configured GitHub Actions service endpoint and writes the ID back as a non-secret `String` parameter with overwrite enabled. The service resolves the scale-set ID from the group and scale-set names; if the named scale set does not exist, it registers it in the resolved runner group and uses the ID returned by GitHub. `scaleSetId` is only an optional legacy pin for an already-known ID. `expectedRunnerGroupId` can be omitted or null; when supplied, it is treated as an additional consistency check. Optional fields are `scaleSetId`, `runnerGroupIdParameterName`, `expectedRunnerGroupId`, `sessionOwner`, `workFolder`, `forceGhes`, `sslVerify`, and `userAgent`. `sslVerify` defaults to `true`; when false, the service uses a reconciler-scoped Undici dispatcher for both GitHub App token and scale-set requests without changing `NODE_TLS_REJECT_UNAUTHORIZED` or the global dispatcher. `userAgent` becomes the `system` identity inside the required structured scale-set protocol User-Agent rather than replacing that header. `bootTimeoutMinutes` defaults to `10`; it is orchestration-owned and is passed to the selected compute provider on every reconciliation. +`scaleSetName` and `runnerGroupName` are the GitHub names supplied by the operator. `runnerGroupIdParameterName` is an optional read-only SSM cache path. When present, the service reads the runner-group ID from that parameter; if it is missing, the service resolves the name through the configured GitHub Actions service endpoint without writing the discovered ID back to SSM. The service resolves the scale-set ID from the group and scale-set names; if the named scale set does not exist, it registers it in the resolved runner group and uses the ID returned by GitHub. `scaleSetId` is only an optional legacy pin for an already-known ID. `expectedRunnerGroupId` can be omitted or null; when supplied, it is treated as an additional consistency check. Optional fields are `scaleSetId`, `runnerGroupIdParameterName`, `expectedRunnerGroupId`, `sessionOwner`, `workFolder`, `forceGhes`, `sslVerify`, and `userAgent`. `sslVerify` defaults to `true`; when false, the service uses a reconciler-scoped Undici dispatcher for both GitHub App token and scale-set requests without changing `NODE_TLS_REJECT_UNAUTHORIZED` or the global dispatcher. `userAgent` becomes the `system` identity inside the required structured scale-set protocol User-Agent rather than replacing that header. `bootTimeoutMinutes` defaults to `10`; it is orchestration-owned and is passed to the selected compute provider on every reconciliation. GitHub App ID and private-key values are reloaded from SSM whenever an installation token is requested, so key rotation does not require a task restart. `installationIdParameterName` is optional; when it is absent or its parameter is not present, the service creates a short-lived App JWT and discovers the installation by matching the configured organization or enterprise account through `GET /app/installations`. This works with GitHub.com, GHES, and GitHub Enterprise Cloud data-residency API hosts derived from `githubConfigUrl`. A SHA-256 credential fingerprint keeps the same Octokit auth instance—and its token cache—while the values remain unchanged, and replaces it after rotation. Private keys, installation tokens, App JWTs, message-session tokens, message bodies, and JIT configurations are never accepted as manifest values and are redacted from logs. @@ -104,4 +104,9 @@ docker build --target runtime -f lambdas/services/scale-set/Dockerfile -t scale- The image supports `linux/amd64` and `linux/arm64`, uses a digest-pinned multi-stage Node image, runs as the unprivileged `node` user, includes a Node-based health check, and does not require filesystem writes. Deploy with a read-only root filesystem, all Linux capabilities dropped, no Docker socket, and only the task-role permissions required by the selected group. -The module's official GHCR package must allow anonymous pulls so the default image works without registry credentials. Production deployments should select a released image by digest and verify its provenance/attestation. A private ECR override requires `container.ecr_repository.arn`; private non-ECR registry credentials are not currently exposed by the Terraform orchestration module. +The Terraform module requires an explicit controller image. Public GHCR images +need to allow anonymous pulls when no registry credentials are configured; +production deployments should select a released image by digest and verify its +provenance/attestation. A private ECR image is supported with repository-scoped +execution-role permissions; private non-ECR registry credentials are not +currently exposed by the Terraform orchestration module. diff --git a/lambdas/services/scale-set/src/credentials.test.ts b/lambdas/services/scale-set/src/credentials.test.ts index 54ba60faff..3cffadcfd9 100644 --- a/lambdas/services/scale-set/src/credentials.test.ts +++ b/lambdas/services/scale-set/src/credentials.test.ts @@ -31,7 +31,6 @@ describe('GitHub App credentials', () => { ['/app/key', encodedKey('abc')], ]), ), - put: vi.fn(), }; await expect(loadGitHubAppCredentials(references, store)).resolves.toMatchObject({ appId: '123', @@ -86,7 +85,6 @@ describe('GitHub App credentials', () => { ['/app/key', encodedKey('abc')], ]), ), - put: vi.fn(), }; const appAuth = vi.fn().mockResolvedValue({ token: 'app-jwt' }); const installationAuth = vi @@ -114,7 +112,7 @@ describe('GitHub App credentials', () => { expect(installationAuth).toHaveBeenCalledWith({ type: 'installation', installationId: 456 }); expect(appAuth).toHaveBeenCalledTimes(1); expect(fetchImplementation).toHaveBeenCalledTimes(1); - expect(store.put).toHaveBeenCalledWith('/app/installation', '456'); + expect(store.get).toHaveBeenCalledWith(['/app/id', '/app/installation', '/app/key']); expect(fetchImplementation).toHaveBeenCalledWith( expect.objectContaining({ href: 'https://api.github.com/app/installations?per_page=100&page=1', @@ -133,7 +131,6 @@ describe('GitHub App credentials', () => { ['/app/key', encodedKey('abc')], ]), ), - put: vi.fn(), }; const appAuth = vi.fn().mockResolvedValue({ token: 'app-jwt' }); const installationAuth = vi diff --git a/lambdas/services/scale-set/src/credentials.ts b/lambdas/services/scale-set/src/credentials.ts index 7e4cacecd0..2ac3cb23c2 100644 --- a/lambdas/services/scale-set/src/credentials.ts +++ b/lambdas/services/scale-set/src/credentials.ts @@ -13,7 +13,6 @@ import { ScaleSetConfigurationError, type GitHubAppParameterReferences } from '. export interface ParameterStore { get(names: readonly string[]): Promise>; - put?(name: string, value: string): Promise; } interface GitHubAppCredentials { @@ -185,9 +184,6 @@ export async function createGitHubAppAccessTokenProvider( } else { installationId = await discoverGitHubAppInstallationId(credentials, target!, apiBaseUrl, fetchImplementation); discovered = { fingerprint: credentialFingerprint, installationId }; - if (references.installationIdParameterName !== undefined && parameterStore.put !== undefined) { - await parameterStore.put(references.installationIdParameterName, String(installationId)); - } } } const fingerprint = `${credentialFingerprint}\u0000${installationId}`; diff --git a/lambdas/services/scale-set/src/main.ts b/lambdas/services/scale-set/src/main.ts index 3f3f08389f..4cf94c5d4c 100644 --- a/lambdas/services/scale-set/src/main.ts +++ b/lambdas/services/scale-set/src/main.ts @@ -23,7 +23,6 @@ async function main(): Promise { groupName: manifest.groupName, revision: manifest.revision, reconcilerCount: manifest.reconcilers.length, - runnerConfigNames: manifest.reconcilers.map(({ runnerConfigName }) => runnerConfigName), }); logger.debug('scale_set_controller_reconcilers_loaded', { groupName: manifest.groupName, diff --git a/lambdas/services/scale-set/src/parameter-store.ts b/lambdas/services/scale-set/src/parameter-store.ts index 3df49818b8..ce07141bd7 100644 --- a/lambdas/services/scale-set/src/parameter-store.ts +++ b/lambdas/services/scale-set/src/parameter-store.ts @@ -1,6 +1,6 @@ -import { GetParametersByPathCommand, PutParameterCommand, SSMClient } from '@aws-sdk/client-ssm'; +import { GetParametersByPathCommand, SSMClient } from '@aws-sdk/client-ssm'; -import { getParameters, ssmClient } from '@aws-github-runner/aws-ssm-util'; +import { getParameters } from '@aws-github-runner/aws-ssm-util'; import { MAX_MANIFEST_BYTES, @@ -20,16 +20,6 @@ const MAX_GROUP_BYTES = 4 * 1024 * 1024; export const defaultParameterStore: ParameterStore = { get: async (names) => await getParameters([...names]), - put: async (name, value) => { - await ssmClient().send( - new PutParameterCommand({ - Name: name, - Value: value, - Type: 'String', - Overwrite: true, - }), - ); - }, }; export interface ControllerManifestLoader { diff --git a/lambdas/services/scale-set/src/reconciler.test.ts b/lambdas/services/scale-set/src/reconciler.test.ts index 2575816e9e..edaf4800ab 100644 --- a/lambdas/services/scale-set/src/reconciler.test.ts +++ b/lambdas/services/scale-set/src/reconciler.test.ts @@ -130,7 +130,7 @@ function fixture(options: { }), random: () => 0, closeSignal: () => new AbortController().signal, - parameterStore: { get: vi.fn().mockResolvedValue(new Map()), put: vi.fn() }, + parameterStore: { get: vi.fn().mockResolvedValue(new Map()) }, createComputeProviderCredentials: vi.fn(), }; return { client, computeProvider, dependencies }; @@ -170,7 +170,6 @@ describe('ScaleSetReconciler', () => { expect(client.getRunnerGroupByName).toHaveBeenCalledWith('runner-group', { signal: abort.signal }); expect(client.getRunnerScaleSet).toHaveBeenCalledWith(7, 'linux', { signal: abort.signal }); expect(client.setSystemInfo).toHaveBeenCalledWith(expect.objectContaining({ scaleSetId: 42 })); - expect(dependencies.parameterStore.put).toHaveBeenCalledWith('/runner/group-id', '7'); expect(dependencies.logger.info).toHaveBeenCalledWith( 'scale_set_compute_provider_created', expect.objectContaining({ computeProviderType: 'ec2' }), diff --git a/lambdas/services/scale-set/src/reconciler.ts b/lambdas/services/scale-set/src/reconciler.ts index 05163f92b8..a44036ecf3 100644 --- a/lambdas/services/scale-set/src/reconciler.ts +++ b/lambdas/services/scale-set/src/reconciler.ts @@ -143,7 +143,7 @@ export class ScaleSetReconciler { this.log('error', 'scale_set_reconciler_initialization_failed', { computeProviderType: this.config.computeProvider.type, ...httpErrorLogAttributes(error), - error: errorLogAttributes(error), + error, }); return; } @@ -210,11 +210,11 @@ export class ScaleSetReconciler { this.log('info', 'scale_set_reconciler_retry_stopped', { retryable: false, reason: 'fatal_error', - error: errorLogAttributes(error), + error, }); this.log('error', 'scale_set_reconciler_failed', { ...httpErrorLogAttributes(error), - error: errorLogAttributes(error), + error, }); return; } @@ -257,9 +257,6 @@ export class ScaleSetReconciler { ); } runnerGroupId = runnerGroup.id; - if (cachedRunnerGroupId === undefined && this.config.runnerGroupIdParameterName !== undefined) { - await this.dependencies.parameterStore.put?.(this.config.runnerGroupIdParameterName, String(runnerGroupId)); - } this.log('info', 'scale_set_runner_group_resolved', { runnerConfigName: this.config.runnerConfigName, runnerGroupName: this.config.runnerGroupName, @@ -508,7 +505,7 @@ export class ScaleSetReconciler { computeProviderType: this.config.computeProvider.type, desiredRunners: request.desiredRunners, busyRunners: request.busyRunners, - error: errorLogAttributes(error), + error, }); throw new ScaleSetProviderReconciliationError(undefined, { cause: error }); } @@ -727,17 +724,3 @@ function httpErrorLogAttributes(error: unknown): Record { requestCode: error.code, }; } - -function errorLogAttributes(error: unknown, depth = 0): Record { - if (!(error instanceof Error)) return { message: String(error) }; - if (depth >= 3) return { name: error.name, message: error.message, cause: '[TRUNCATED]' }; - - const errorWithMetadata = error as Error & { code?: unknown; status?: unknown; cause?: unknown }; - return { - name: error.name, - message: error.message, - ...(typeof errorWithMetadata.code === 'string' ? { code: errorWithMetadata.code } : {}), - ...(typeof errorWithMetadata.status === 'number' ? { status: errorWithMetadata.status } : {}), - ...(errorWithMetadata.cause === undefined ? {} : { cause: errorLogAttributes(errorWithMetadata.cause, depth + 1) }), - }; -} diff --git a/mkdocs.yaml b/mkdocs.yaml index 4026d30d6e..4553b6f71f 100644 --- a/mkdocs.yaml +++ b/mkdocs.yaml @@ -62,6 +62,8 @@ nav: - Security: security.md - Architecture decisions: - MiniStack for integration tests: adr/0001-use-ministack-for-terraform-integration-tests.md + - Runner orchestration provider boundary: adr/002-runner-orchestration-provider-boundary.md + - Scale-set resource ownership: adr/003-scale-set-resource-ownership.md - Modules: - Runners (main): modules/runners.md - Submodules (public): @@ -80,6 +82,7 @@ nav: - Default: examples/default.md - Multi Runner: examples/multi-runner.md - Multi Runner v2: examples/multi-runner-v2.md + - Multi Runner scale-set: examples/multi-runner-scale-set.md - Ephemeral: examples/ephemeral.md - External managed secrets: examples/external-managed-ssm-secrets.md - Custom AMI: examples/prebuilt.md diff --git a/modules/multi-runner/README.md b/modules/multi-runner/README.md index 0f253eb809..8183b103e0 100644 --- a/modules/multi-runner/README.md +++ b/modules/multi-runner/README.md @@ -167,7 +167,7 @@ module "multi-runner" { | [global\_config\_github](#input\_global\_config\_github) | Global GitHub configuration shared by all runner lanes.

global\_config\_github = {
app: {
key\_base64: "Base64-encoded GitHub App private key."
key\_base64\_ssm: "SSM parameter containing the Base64-encoded GitHub App private key."
key\_base64\_ssm.arn: "ARN of the SSM parameter containing the GitHub App private key."
key\_base64\_ssm.name: "Name of the SSM parameter containing the GitHub App private key."
id: "GitHub App ID."
id\_ssm: "SSM parameter containing the GitHub App ID."
id\_ssm.arn: "ARN of the SSM parameter containing the GitHub App ID."
id\_ssm.name: "Name of the SSM parameter containing the GitHub App ID."
installation\_id: "GitHub App installation ID for the primary scale-set installation."
installation\_id\_ssm: "SSM parameter containing the primary GitHub App installation ID."
installation\_id\_ssm.arn: "ARN of the SSM parameter containing the primary GitHub App installation ID."
installation\_id\_ssm.name: "Name of the SSM parameter containing the primary GitHub App installation ID."
webhook\_secret: "GitHub App webhook secret."
webhook\_secret\_ssm: "SSM parameter containing the GitHub App webhook secret."
webhook\_secret\_ssm.arn: "ARN of the SSM parameter containing the GitHub App webhook secret."
webhook\_secret\_ssm.name: "Name of the SSM parameter containing the GitHub App webhook secret."
}
additional\_apps: "Additional GitHub Apps used to distribute GitHub API requests."
additional\_apps.key\_base64: "Base64-encoded private key for an additional GitHub App."
additional\_apps.key\_base64\_ssm: "SSM parameter containing an additional App private key."
additional\_apps.key\_base64\_ssm.arn: "ARN of the SSM parameter containing an additional App private key."
additional\_apps.key\_base64\_ssm.name: "Name of the SSM parameter containing an additional App private key."
additional\_apps.id: "ID of an additional GitHub App."
additional\_apps.id\_ssm: "SSM parameter containing an additional GitHub App ID."
additional\_apps.id\_ssm.arn: "ARN of the SSM parameter containing an additional GitHub App ID."
additional\_apps.id\_ssm.name: "Name of the SSM parameter containing an additional GitHub App ID."
additional\_apps.installation\_id: "Optional installation ID for an additional GitHub App."
additional\_apps.installation\_id\_ssm: "SSM parameter containing an additional App installation ID."
additional\_apps.installation\_id\_ssm.arn: "ARN of the SSM parameter containing an additional App installation ID."
additional\_apps.installation\_id\_ssm.name: "Name of the SSM parameter containing an additional App installation ID."
enterprise\_server.url: "GitHub Enterprise Server URL."
enterprise\_server.ssl\_verify: "Whether to verify the GitHub Enterprise Server TLS certificate."
runner\_owner: "GitHub organization or owner/repository path for organization- or repository-level scale-set registration."
runner\_registration\_level: "GitHub scale-set registration scope: organization or repository."
user\_agent: "User-Agent value sent with GitHub API requests."
} |
object({
app = optional(object({
key_base64 = optional(string)
key_base64_ssm = optional(object({
arn = string
name = string
}))
id = optional(string)
id_ssm = optional(object({
arn = string
name = string
}))
installation_id = optional(string)
installation_id_ssm = optional(object({
arn = string
name = string
}))
webhook_secret = optional(string)
webhook_secret_ssm = optional(object({
arn = string
name = string
}))
}), null)
additional_apps = optional(list(object({
key_base64 = optional(string)
key_base64_ssm = optional(object({ arn = string, name = string }))
id = optional(string)
id_ssm = optional(object({ arn = string, name = string }))
installation_id = optional(string)
installation_id_ssm = optional(object({ arn = string, name = string }))
})), [])
enterprise_server = optional(object({
url = optional(string, null)
ssl_verify = optional(bool, true)
}), {})
runner_owner = optional(string, null)
runner_registration_level = optional(string, "organization")
user_agent = optional(string, "github-aws-runners")
})
| `{}` | no | | [global\_config\_lambda](#input\_global\_config\_lambda) | Global Lambda configuration shared by all runner lanes.

global\_config\_lambda = {
artifact.s3.bucket: "S3 bucket containing Lambda deployment artifacts."
runtime: "Default Lambda runtime."
architecture: "Default Lambda instruction-set architecture."
principals: "Additional AWS principals allowed to invoke the Lambda functions."
principals.type: "Principal type, such as AWS account, service, or organization."
principals.identifiers: "Identifiers allowed for the principal type."
subnet\_ids: "Subnets used by Lambda functions."
security\_group\_ids: "Security groups attached to Lambda functions."
tags: "Tags applied to Lambda functions and related resources."
role.path: "IAM path used for Lambda execution roles."
role.permissions\_boundary: "Optional IAM permissions boundary ARN for Lambda execution roles."
} |
object({
artifact = optional(object({
s3 = optional(object({
bucket = optional(string, null)
}), {})
}), {})
runtime = optional(string, "nodejs24.x")
architecture = optional(string, "arm64")
principals = optional(list(object({
type = string
identifiers = list(string)
})), [])
subnet_ids = optional(list(string), [])
security_group_ids = optional(list(string), [])
tags = optional(map(string), {})
role = optional(object({
path = optional(string, null)
permissions_boundary = optional(string, null)
}), {})
})
| `{}` | no | | [global\_config\_observability](#input\_global\_config\_observability) | Global observability configuration shared by all runner lanes.

global\_config\_observability = {
logs.level: "Log level for module resources."
logs.retention\_in\_days: "CloudWatch log retention period in days."
logs.kms\_key\_id: "KMS key ID used to encrypt CloudWatch log groups."
logs.class: "CloudWatch log group class."
logs.tags: "Tags applied to CloudWatch log groups."
tracing.mode: "Tracing mode used by instrumented resources."
tracing.capture\_http\_requests: "Whether HTTP requests are captured by tracing."
tracing.capture\_error: "Whether errors are captured by tracing."
metrics.enabled: "Whether module metrics are enabled."
metrics.namespace: "CloudWatch namespace used for module metrics."
metrics.metric.github\_app\_rate\_limit.enabled: "Whether GitHub App rate-limit metrics are emitted."
metrics.metric.job\_retry.enabled: "Whether job-retry metrics are emitted."
metrics.metric.spot\_termination\_warning.enabled: "Whether spot-termination warning metrics are emitted."
} |
object({
logs = optional(object({
level = optional(string, "info")
retention_in_days = optional(number, 180)
kms_key_id = optional(string, null)
class = optional(string, "STANDARD")
tags = optional(map(string), {})
}), {})
tracing = optional(object({
mode = optional(string, null)
capture_http_requests = optional(bool, false)
capture_error = optional(bool, false)
}), {})
metrics = optional(object({
enabled = optional(bool, false)
namespace = optional(string, "GitHub Runners")
metric = optional(object({
github_app_rate_limit = optional(object({
enabled = optional(bool, true)
}), {})
job_retry = optional(object({
enabled = optional(bool, true)
}), {})
spot_termination_warning = optional(object({
enabled = optional(bool, true)
}), {})
}), {})
}), {})
})
| `{}` | no | -| [global\_config\_orchestration\_provider](#input\_global\_config\_orchestration\_provider) | Global orchestration-provider configuration shared by all runner lanes.

global\_config\_orchestration\_provider = {
webhook: {
queue\_selection\_strategy: "Strategy used to select the build queue for a webhook event."
eventbridge.enabled: "Whether EventBridge integration is enabled for webhook events."
eventbridge.accept\_events: "Event types accepted by the EventBridge integration."
matcher\_config\_parameter\_store\_tier: "SSM Parameter Store tier used for matcher configuration."
runner.boot\_time\_in\_minutes: "Expected runner boot time used by orchestration."
runner.ephemeral: "Whether runners created by the orchestration provider are ephemeral."
runner.jit\_config\_enabled: "Whether JIT runner configuration is enabled."
runner.maximum\_count: "Maximum number of runners that orchestration may create."
github.repository\_white\_list: "Repositories allowed to use the webhook configuration."
lambda.artifact.zip: "Local ZIP artifact used for orchestration Lambda functions."
lambda.artifact.s3.key: "S3 object key for the orchestration Lambda artifact."
lambda.artifact.s3.object\_version: "Optional S3 object version for the orchestration Lambda artifact."
lambda.scale.up.memory\_size: "Memory allocated to the scale-up Lambda."
lambda.scale.up.timeout: "Timeout in seconds for the scale-up Lambda."
lambda.scale.up.reserved\_concurrent\_executions: "Reserved concurrent executions for the scale-up Lambda."
lambda.scale.up.job\_queued\_check\_enabled: "Whether the scale-up Lambda checks queued jobs."
lambda.scale.up.event\_source\_mapping.batch\_size: "Maximum records passed to one scale-up Lambda invocation."
lambda.scale.up.event\_source\_mapping.maximum\_batching\_window\_in\_seconds: "Maximum time to batch records before invoking the scale-up Lambda."
lambda.scale.up.tags: "Tags applied to the scale-up Lambda."
lambda.scale.down.memory\_size: "Memory allocated to the scale-down Lambda."
lambda.scale.down.timeout: "Timeout in seconds for the scale-down Lambda."
lambda.scale.down.schedule\_expression: "Schedule expression for scale-down processing."
lambda.scale.down.minimum\_running\_time\_in\_minutes: "Minimum runner lifetime before scale-down."
lambda.scale.down.idle\_confirmation\_seconds: "Seconds a runner must consistently report not-busy before scale-down terminates it; 0 disables the confirmation window."
lambda.scale.down.idle\_config: "Scheduled minimum idle-runner pool settings."
lambda.scale.down.idle\_config.cron: "Cron expression defining when the idle-runner count applies."
lambda.scale.down.idle\_config.timeZone: "Time zone used to evaluate the idle-runner schedule."
lambda.scale.down.idle\_config.idleCount: "Minimum number of idle runners maintained during the schedule."
lambda.scale.down.idle\_config.evictionStrategy: "Strategy used when evicting idle runners."
lambda.scale.down.tags: "Tags applied to the scale-down Lambda."
lambda.webhook.artifact.zip: "Local ZIP artifact used for the webhook Lambda."
lambda.webhook.artifact.s3.key: "S3 object key for the webhook Lambda artifact."
lambda.webhook.artifact.s3.object\_version: "Optional S3 object version for the webhook Lambda artifact."
lambda.webhook.api\_gateway\_access\_log\_settings: "API Gateway access-log destination and format."
lambda.webhook.api\_gateway\_access\_log\_settings.destination\_arn: "ARN of the API Gateway access-log destination."
lambda.webhook.api\_gateway\_access\_log\_settings.format: "API Gateway access-log format."
lambda.webhook.memory\_size: "Memory allocated to the webhook Lambda."
lambda.webhook.timeout: "Timeout in seconds for the webhook Lambda."
lambda.webhook.tags: "Tags applied to the webhook Lambda."
lambda.pool.memory\_size: "Memory allocated to the pool Lambda."
lambda.pool.timeout: "Timeout in seconds for the pool Lambda."
lambda.pool.reserved\_concurrent\_executions: "Reserved concurrent executions for the pool Lambda."
lambda.pool.config: "Scheduled runner-pool size configuration."
lambda.pool.config.schedule\_expression: "Schedule expression for the pool size."
lambda.pool.config.schedule\_expression\_timezone: "Time zone used to evaluate the pool schedule."
lambda.pool.config.size: "Runner pool size applied by the schedule."
lambda.pool.include\_busy\_runners: "Whether busy runners are included in pool sizing."
lambda.pool.runner\_owner: "GitHub organization that owns the runner pool."
lambda.pool.tags: "Tags applied to the pool Lambda."
queue.delay\_webhook\_event: "Seconds a webhook event remains invisible in the build queue before processing."
queue.job\_queue\_retention\_in\_seconds: "Seconds a queued job is retained before it is purged."
queue.visibility\_timeout\_seconds: "Build queue visibility timeout in seconds."
queue.redrive\_build\_queue.enabled: "Whether the build queue dead-letter queue is enabled."
queue.redrive\_build\_queue.maxReceiveCount: "Maximum receives before a message is moved to the dead-letter queue."
queue.tags: "Tags applied to build queues."
queue.encryption.kms\_data\_key\_reuse\_period\_seconds: "KMS data-key reuse period for queue encryption."
queue.encryption.kms\_master\_key\_id: "KMS key ID used for queue encryption."
queue.encryption.sqs\_managed\_sse\_enabled: "Whether SQS-managed server-side encryption is enabled."
}
} |
object({
webhook = optional(object({
queue_selection_strategy = optional(string, "first")
eventbridge = optional(object({
enabled = optional(bool, true)
accept_events = optional(list(string), [])
}), {})
matcher_config_parameter_store_tier = optional(string, "Standard")
runner = optional(object({
boot_time_in_minutes = optional(number, 5)
ephemeral = optional(bool, false)
jit_config_enabled = optional(bool, null)
maximum_count = optional(number, null)
}), {})

github = optional(object({
repository_white_list = optional(list(string), [])
}), {})

lambda = optional(object({
artifact = optional(object({
zip = optional(string, null)
s3 = optional(object({
key = string
object_version = optional(string, null)
}), null)
}), {})
scale = optional(object({
up = optional(object({
memory_size = optional(number, 512)
timeout = optional(number, 30)
reserved_concurrent_executions = optional(number, 1)
job_queued_check_enabled = optional(bool, null)
event_source_mapping = optional(object({
batch_size = optional(number, 10)
maximum_batching_window_in_seconds = optional(number, 0)
}), {})
tags = optional(map(string), {})
}), {})
down = optional(object({
memory_size = optional(number, 512)
timeout = optional(number, 60)
schedule_expression = optional(string, "cron(*/5 * * * ? *)")
minimum_running_time_in_minutes = optional(number, null)
idle_confirmation_seconds = optional(number, 0)
idle_config = optional(list(object({
cron = string
timeZone = string
idleCount = number
evictionStrategy = optional(string, "oldest_first")
})), [])
tags = optional(map(string), {})
}), {})
}), {})
webhook = optional(object({
artifact = optional(object({
zip = optional(string, null)
s3 = optional(object({
key = string
object_version = optional(string, null)
}), null)
}), {})
api_gateway_access_log_settings = optional(object({
destination_arn = string
format = string
}), null)
memory_size = optional(number, 256)
timeout = optional(number, 10)
tags = optional(map(string), {})
}), {})
pool = optional(object({
memory_size = optional(number, 512)
timeout = optional(number, 60)
reserved_concurrent_executions = optional(number, 1)
config = optional(list(object({
schedule_expression = string
schedule_expression_timezone = optional(string)
size = number
})), [])
include_busy_runners = optional(bool, false)
runner_owner = optional(string, null)
tags = optional(map(string), {})
}), {})
}), {})

queue = optional(object({
delay_webhook_event = optional(number, 30)
job_queue_retention_in_seconds = optional(number, 86400)
visibility_timeout_seconds = optional(number, 180)
redrive_build_queue = optional(object({
enabled = optional(bool, false)
maxReceiveCount = optional(number, null)
}), {
enabled = false
maxReceiveCount = null
})
tags = optional(map(string), {})
encryption = optional(object({
kms_data_key_reuse_period_seconds = number
kms_master_key_id = string
sqs_managed_sse_enabled = bool
}), {
kms_data_key_reuse_period_seconds = null
kms_master_key_id = null
sqs_managed_sse_enabled = true
})
}), {})

}), {})

scale_set = optional(object({
grouping = optional(object({
strategy = optional(string, "compute_provider")
custom = optional(object({
groups = map(object({
runner_configs = set(string)
}))
}), null)
}), {})
container = optional(object({
image = optional(string, null)
user = optional(string, "10001:10001")
health_port = optional(number, 8080)
health_path = optional(string, "/healthz")
health_check_command = optional(list(string), null)
health_check_interval = optional(number, 30)
health_check_timeout = optional(number, 5)
health_check_retries = optional(number, 3)
health_check_start_period = optional(number, 30)
health_stale_after_seconds = optional(number, 180)
shutdown_timeout_seconds = optional(number, 110)
session_close_timeout_seconds = optional(number, 10)
reconnect_initial_backoff_seconds = optional(number, 1)
reconnect_max_backoff_seconds = optional(number, 30)
stop_timeout_seconds = optional(number, 120)
}), {})
config_store = optional(object({
path_prefix = optional(string, null)
tier = optional(string, "Standard")
tags = optional(map(string), {})
}), {})
ecs = optional(object({
cluster = optional(object({
mode = optional(string, "managed")
arn = optional(string, null)
name = optional(string, null)
container_insights = optional(bool, true)
}), {})
task = optional(object({
cpu = optional(number, 512)
memory = optional(number, 1024)
cpu_architecture = optional(string, "X86_64")
ephemeral_storage = optional(object({
size_in_gib = number
}), null)
}), {})
service = optional(object({
platform_version = optional(string, "LATEST")
}), {})
iam = optional(object({
path = optional(string, "/")
permissions_boundary = optional(string, null)
}), {})
}), {})
network = optional(object({
vpc_id = optional(string, null)
subnet_ids = optional(set(string), null)
https_egress = optional(object({
ipv4_cidrs = optional(set(string), ["0.0.0.0/0"])
ipv6_cidrs = optional(set(string), [])
}), {})
}), {})
logging = optional(object({
retention_in_days = optional(number, 30)
kms_key_arn = optional(string, null)
log_group_class = optional(string, "STANDARD")
tags = optional(map(string), {})
}), {})
tags = optional(map(string), {})
}), {})
})
| `{}` | no | +| [global\_config\_orchestration\_provider](#input\_global\_config\_orchestration\_provider) | Global orchestration-provider configuration shared by all runner lanes.

global\_config\_orchestration\_provider = {
webhook: {
queue\_selection\_strategy: "Strategy used to select the build queue for a webhook event."
eventbridge.enabled: "Whether EventBridge integration is enabled for webhook events."
eventbridge.accept\_events: "Event types accepted by the EventBridge integration."
matcher\_config\_parameter\_store\_tier: "SSM Parameter Store tier used for matcher configuration."
runner.boot\_time\_in\_minutes: "Expected runner boot time used by orchestration."
runner.ephemeral: "Whether runners created by the orchestration provider are ephemeral."
runner.jit\_config\_enabled: "Whether JIT runner configuration is enabled."
runner.maximum\_count: "Maximum number of runners that orchestration may create."
github.repository\_white\_list: "Repositories allowed to use the webhook configuration."
lambda.artifact.zip: "Local ZIP artifact used for orchestration Lambda functions."
lambda.artifact.s3.key: "S3 object key for the orchestration Lambda artifact."
lambda.artifact.s3.object\_version: "Optional S3 object version for the orchestration Lambda artifact."
lambda.scale.up.memory\_size: "Memory allocated to the scale-up Lambda."
lambda.scale.up.timeout: "Timeout in seconds for the scale-up Lambda."
lambda.scale.up.reserved\_concurrent\_executions: "Reserved concurrent executions for the scale-up Lambda."
lambda.scale.up.job\_queued\_check\_enabled: "Whether the scale-up Lambda checks queued jobs."
lambda.scale.up.event\_source\_mapping.batch\_size: "Maximum records passed to one scale-up Lambda invocation."
lambda.scale.up.event\_source\_mapping.maximum\_batching\_window\_in\_seconds: "Maximum time to batch records before invoking the scale-up Lambda."
lambda.scale.up.tags: "Tags applied to the scale-up Lambda."
lambda.scale.down.memory\_size: "Memory allocated to the scale-down Lambda."
lambda.scale.down.timeout: "Timeout in seconds for the scale-down Lambda."
lambda.scale.down.schedule\_expression: "Schedule expression for scale-down processing."
lambda.scale.down.minimum\_running\_time\_in\_minutes: "Minimum runner lifetime before scale-down."
lambda.scale.down.idle\_confirmation\_seconds: "Seconds a runner must consistently report not-busy before scale-down terminates it; 0 disables the confirmation window."
lambda.scale.down.idle\_config: "Scheduled minimum idle-runner pool settings."
lambda.scale.down.idle\_config.cron: "Cron expression defining when the idle-runner count applies."
lambda.scale.down.idle\_config.timeZone: "Time zone used to evaluate the idle-runner schedule."
lambda.scale.down.idle\_config.idleCount: "Minimum number of idle runners maintained during the schedule."
lambda.scale.down.idle\_config.evictionStrategy: "Strategy used when evicting idle runners."
lambda.scale.down.tags: "Tags applied to the scale-down Lambda."
lambda.webhook.artifact.zip: "Local ZIP artifact used for the webhook Lambda."
lambda.webhook.artifact.s3.key: "S3 object key for the webhook Lambda artifact."
lambda.webhook.artifact.s3.object\_version: "Optional S3 object version for the webhook Lambda artifact."
lambda.webhook.api\_gateway\_access\_log\_settings: "API Gateway access-log destination and format."
lambda.webhook.api\_gateway\_access\_log\_settings.destination\_arn: "ARN of the API Gateway access-log destination."
lambda.webhook.api\_gateway\_access\_log\_settings.format: "API Gateway access-log format."
lambda.webhook.memory\_size: "Memory allocated to the webhook Lambda."
lambda.webhook.timeout: "Timeout in seconds for the webhook Lambda."
lambda.webhook.tags: "Tags applied to the webhook Lambda."
lambda.pool.memory\_size: "Memory allocated to the pool Lambda."
lambda.pool.timeout: "Timeout in seconds for the pool Lambda."
lambda.pool.reserved\_concurrent\_executions: "Reserved concurrent executions for the pool Lambda."
lambda.pool.config: "Scheduled runner-pool size configuration."
lambda.pool.config.schedule\_expression: "Schedule expression for the pool size."
lambda.pool.config.schedule\_expression\_timezone: "Time zone used to evaluate the pool schedule."
lambda.pool.config.size: "Runner pool size applied by the schedule."
lambda.pool.include\_busy\_runners: "Whether busy runners are included in pool sizing."
lambda.pool.runner\_owner: "GitHub organization that owns the runner pool."
lambda.pool.tags: "Tags applied to the pool Lambda."
queue.delay\_webhook\_event: "Seconds a webhook event remains invisible in the build queue before processing."
queue.job\_queue\_retention\_in\_seconds: "Seconds a queued job is retained before it is purged."
queue.visibility\_timeout\_seconds: "Build queue visibility timeout in seconds."
queue.redrive\_build\_queue.enabled: "Whether the build queue dead-letter queue is enabled."
queue.redrive\_build\_queue.maxReceiveCount: "Maximum receives before a message is moved to the dead-letter queue."
queue.tags: "Tags applied to build queues."
queue.encryption.kms\_data\_key\_reuse\_period\_seconds: "KMS data-key reuse period for queue encryption."
queue.encryption.kms\_master\_key\_id: "KMS key ID used for queue encryption."
queue.encryption.sqs\_managed\_sse\_enabled: "Whether SQS-managed server-side encryption is enabled."
}
} |
object({
webhook = optional(object({
queue_selection_strategy = optional(string, "first")
eventbridge = optional(object({
enabled = optional(bool, true)
accept_events = optional(list(string), [])
}), {})
matcher_config_parameter_store_tier = optional(string, "Standard")
runner = optional(object({
boot_time_in_minutes = optional(number, 5)
ephemeral = optional(bool, false)
jit_config_enabled = optional(bool, null)
maximum_count = optional(number, null)
}), {})

github = optional(object({
repository_white_list = optional(list(string), [])
}), {})

lambda = optional(object({
artifact = optional(object({
zip = optional(string, null)
s3 = optional(object({
key = string
object_version = optional(string, null)
}), null)
}), {})
scale = optional(object({
up = optional(object({
memory_size = optional(number, 512)
timeout = optional(number, 30)
reserved_concurrent_executions = optional(number, 1)
job_queued_check_enabled = optional(bool, null)
event_source_mapping = optional(object({
batch_size = optional(number, 10)
maximum_batching_window_in_seconds = optional(number, 0)
}), {})
tags = optional(map(string), {})
}), {})
down = optional(object({
memory_size = optional(number, 512)
timeout = optional(number, 60)
schedule_expression = optional(string, "cron(*/5 * * * ? *)")
minimum_running_time_in_minutes = optional(number, null)
idle_confirmation_seconds = optional(number, 0)
idle_config = optional(list(object({
cron = string
timeZone = string
idleCount = number
evictionStrategy = optional(string, "oldest_first")
})), [])
tags = optional(map(string), {})
}), {})
}), {})
webhook = optional(object({
artifact = optional(object({
zip = optional(string, null)
s3 = optional(object({
key = string
object_version = optional(string, null)
}), null)
}), {})
api_gateway_access_log_settings = optional(object({
destination_arn = string
format = string
}), null)
memory_size = optional(number, 256)
timeout = optional(number, 10)
tags = optional(map(string), {})
}), {})
pool = optional(object({
memory_size = optional(number, 512)
timeout = optional(number, 60)
reserved_concurrent_executions = optional(number, 1)
config = optional(list(object({
schedule_expression = string
schedule_expression_timezone = optional(string)
size = number
})), [])
include_busy_runners = optional(bool, false)
runner_owner = optional(string, null)
tags = optional(map(string), {})
}), {})
}), {})

queue = optional(object({
delay_webhook_event = optional(number, 30)
job_queue_retention_in_seconds = optional(number, 86400)
visibility_timeout_seconds = optional(number, 180)
redrive_build_queue = optional(object({
enabled = optional(bool, false)
maxReceiveCount = optional(number, null)
}), {
enabled = false
maxReceiveCount = null
})
tags = optional(map(string), {})
encryption = optional(object({
kms_data_key_reuse_period_seconds = number
kms_master_key_id = string
sqs_managed_sse_enabled = bool
}), {
kms_data_key_reuse_period_seconds = null
kms_master_key_id = null
sqs_managed_sse_enabled = true
})
}), {})

}), {})

scale_set = optional(object({
grouping = optional(object({
strategy = optional(string, "compute_provider")
custom = optional(object({
groups = map(object({
runner_configs = set(string)
}))
}), null)
}), {})
container = optional(object({
image = optional(string, null)
user = optional(string, "10001:10001")
health_port = optional(number, 8080)
health_path = optional(string, "/healthz")
health_check_command = optional(list(string), null)
health_check_interval = optional(number, 30)
health_check_timeout = optional(number, 5)
health_check_retries = optional(number, 3)
health_check_start_period = optional(number, 30)
health_stale_after_seconds = optional(number, 180)
shutdown_timeout_seconds = optional(number, 110)
session_close_timeout_seconds = optional(number, 10)
reconnect_initial_backoff_seconds = optional(number, 1)
reconnect_max_backoff_seconds = optional(number, 30)
stop_timeout_seconds = optional(number, 120)
}), {})
config_store = optional(object({
path_prefix = optional(string, null)
tier = optional(string, "Standard")
tags = optional(map(string), {})
}), {})
ecs = optional(object({
cluster = optional(object({
mode = optional(string, "managed")
arn = optional(string, null)
name = optional(string, null)
container_insights = optional(bool, true)
}), {})
task = optional(object({
cpu = optional(number, 512)
memory = optional(number, 1024)
cpu_architecture = optional(string, "X86_64")
ephemeral_storage = optional(object({
size_in_gib = number
}), null)
}), {})
service = optional(object({
platform_version = optional(string, "LATEST")
}), {})
iam = optional(object({
path = optional(string, "/")
permissions_boundary = optional(string, null)
}), {})
}), {})
network = optional(object({
vpc_id = optional(string, null)
subnet_ids = optional(set(string), null)
https_egress = optional(object({
ipv4_cidrs = optional(set(string), ["0.0.0.0/0"])
ipv6_cidrs = optional(set(string), [])
}), {})
}), {})
logging = optional(object({
retention_in_days = optional(number, 180)
kms_key_id = optional(string, null)
log_group_class = optional(string, "STANDARD")
tags = optional(map(string), {})
}), {})
tags = optional(map(string), {})
}), {})
})
| `{}` | no | | [global\_config\_storage\_provider](#input\_global\_config\_storage\_provider) | Global storage-provider configuration shared by all runner lanes.

global\_config\_storage\_provider = {
aws.ssm.paths.root: "Root path for SSM parameters."
aws.ssm.paths.app: "Path segment for application parameters."
aws.ssm.paths.webhook: "Path segment for webhook parameters."
aws.ssm.paths.tokens: "Path segment for runner token parameters."
aws.ssm.paths.config: "Path segment for runner configuration parameters."
aws.ssm.kms\_key\_id: "KMS key ID used to encrypt SSM parameters."
aws.ssm.tags: "Tags applied to SSM resources."
aws.ssm.parameters.tags: "Tags applied to runner configuration parameters."
aws.ssm.housekeeper.schedule\_expression: "Schedule for the SSM parameter housekeeper."
aws.ssm.housekeeper.state: "EventBridge rule state for the SSM housekeeper."
aws.ssm.housekeeper.tags: "Tags applied to the SSM housekeeper resources."
aws.ssm.housekeeper.lambda.artifact.zip: "Local ZIP artifact used for the SSM housekeeper Lambda."
aws.ssm.housekeeper.lambda.artifact.s3.key: "S3 object key for the SSM housekeeper Lambda."
aws.ssm.housekeeper.lambda.artifact.s3.object\_version: "Optional S3 object version for the SSM housekeeper artifact."
aws.ssm.housekeeper.lambda.memory\_size: "Memory allocated to the SSM housekeeper Lambda."
aws.ssm.housekeeper.lambda.timeout: "Timeout in seconds for the SSM housekeeper Lambda."
aws.ssm.housekeeper.config.tokenPath: "Parameter path containing runner tokens to clean up."
aws.ssm.housekeeper.config.minimumDaysOld: "Minimum age in days before an old token is eligible for cleanup."
aws.ssm.housekeeper.config.dryRun: "Whether the SSM housekeeper reports cleanup without deleting parameters."
} |
object({
aws = optional(object({
ssm = optional(object({
paths = optional(object({
root = optional(string, null)
app = optional(string, "app")
webhook = optional(string, "webhook")
tokens = optional(string, "runners/tokens")
config = optional(string, "runners/config")
}), {})
kms_key_id = optional(string, null)
tags = optional(map(string), {})
parameters = optional(object({
tags = optional(map(string), {})
}), {})
housekeeper = optional(object({
schedule_expression = optional(string, "rate(1 day)")
state = optional(string, "ENABLED")
tags = optional(map(string), {})
lambda = optional(object({
artifact = optional(object({
zip = optional(string, null)
s3 = optional(object({
key = string
object_version = optional(string, null)
}), null)
}), {})
memory_size = optional(number, 512)
timeout = optional(number, 60)
}), {})
config = optional(object({
tokenPath = optional(string, null)
minimumDaysOld = optional(number, 1)
dryRun = optional(bool, false)
}), {})
}), {})
}), {})
}), {})
})
| `{}` | no | | [iam\_overrides](#input\_iam\_overrides) | This map provides the possibility to override some IAM defaults. The following attributes are supported: `instance_profile_name` overrides the instance profile name used in the launch template. `runner_role_arn` overrides the IAM role ARN used for the runner instances. |
object({
override_instance_profile = optional(bool, null)
instance_profile_name = optional(string, null)
override_runner_role = optional(bool, null)
runner_role_arn = optional(string, null)
})
|
{
"instance_profile_name": null,
"override_instance_profile": false,
"override_runner_role": false,
"runner_role_arn": null
}
| no | | [instance\_profile\_path](#input\_instance\_profile\_path) | The path that will be added to the instance\_profile, if not set the environment name will be used. | `string` | `null` | no | diff --git a/modules/multi-runner/tests/config-resolution.tftest.hcl b/modules/multi-runner/tests/config-resolution.tftest.hcl index afae1bae23..9d4e913456 100644 --- a/modules/multi-runner/tests/config-resolution.tftest.hcl +++ b/modules/multi-runner/tests/config-resolution.tftest.hcl @@ -600,6 +600,9 @@ run "scale_set_only_lane_omits_webhook_queues" { } } scale_set = { + container = { + image = "ghcr.io/github-aws-runners/terraform-aws-github-runner-scale-set-service@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + } network = { vpc_id = "vpc-scale-set" subnet_ids = ["subnet-scale-set"] @@ -708,6 +711,9 @@ run "mixed_webhook_and_scale_set_lanes_create_webhook_queues_only_for_webhook" { } } scale_set = { + container = { + image = "ghcr.io/github-aws-runners/terraform-aws-github-runner-scale-set-service@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + } network = { vpc_id = "vpc-scale-set" subnet_ids = ["subnet-scale-set"] @@ -874,6 +880,9 @@ run "scale_set_queue_for_each_keys_are_plan_known" { global_config_orchestration_provider = { scale_set = { + container = { + image = "ghcr.io/github-aws-runners/terraform-aws-github-runner-scale-set-service@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + } network = { vpc_id = "vpc-scale-set" subnet_ids = ["subnet-scale-set"] diff --git a/modules/multi-runner/variables.experimental.orchestration-provider.tf b/modules/multi-runner/variables.experimental.orchestration-provider.tf index 41a51d6204..c9da8aa0f4 100644 --- a/modules/multi-runner/variables.experimental.orchestration-provider.tf +++ b/modules/multi-runner/variables.experimental.orchestration-provider.tf @@ -239,8 +239,8 @@ variable "global_config_orchestration_provider" { }), {}) }), {}) logging = optional(object({ - retention_in_days = optional(number, 30) - kms_key_arn = optional(string, null) + retention_in_days = optional(number, 180) + kms_key_id = optional(string, null) log_group_class = optional(string, "STANDARD") tags = optional(map(string), {}) }), {}) diff --git a/modules/orchestration-providers/scale-set/README.md b/modules/orchestration-providers/scale-set/README.md index 495a8cc314..1c27bf233e 100644 --- a/modules/orchestration-providers/scale-set/README.md +++ b/modules/orchestration-providers/scale-set/README.md @@ -12,7 +12,17 @@ This internal module deploys long-running GitHub Actions runner scale-set contro Each reconciler still owns exactly one GitHub scale-set identity and one message session. Grouping only packs reconcilers into a shared task; it does not merge scale-set identity, session state, or compute-provider behavior. It does, however, intentionally union task IAM permissions and failure/deployment blast radius across all members of that controller group. -This foundation adopts scale sets that were created elsewhere. It passes the configured name to the controller, which captures the scale-set and runner-group identifiers dynamically from GitHub; it does not create or delete the GitHub scale-set resource. The complete compute-provider contract must likewise come from its Terraform adapter; until that adapter and the public runner-config selection are wired, this internal module is not an end-to-end deployment interface. +This foundation passes the configured scale-set and runner-group names to the +controller. The controller resolves the GitHub identifiers dynamically, +registers a missing named scale set, and reconciles its system labels at +runtime. The TypeScript controller is the component that calls the GitHub +scale-set API; Terraform never configures those GitHub resources. Terraform +destroy removes the AWS controller and stops reconciliation, but does not issue +a GitHub delete. Deletion and renaming remain explicit operator or +GitHub-administration operations. The complete compute-provider +contract must likewise come from its Terraform adapter; until that adapter and +the public runner-config selection are wired, this internal module is not an +end-to-end deployment interface. The normalized `(githubConfigUrl, scale_set.name)` ownership tuple must be globally unique across all groups. `runner_registration_level` selects organization or repository scope; `runner_owner` supplies the corresponding path appended to the GitHub server URL. A null enterprise-server URL resolves to GitHub.com. Enterprise-level registration is not supported by this module. Duplicate detection normalizes URL case, one trailing slash, and an explicit default `:443` port, so equivalent spellings cannot accidentally deploy two services against one GitHub message session. Scale-set names may repeat under different GitHub scopes. @@ -100,6 +110,8 @@ For the current ECS deployment, each task receives one bounded `SCALE_SET_CONTRO Each leaf is the flat `ScaleSetReconcilerConfig` consumed by the service. GitHub credential values never enter the manifest; it contains only the exact Parameter Store names used by the runtime. The manifest source is mutually exclusive with the service's SSM group-path source, so the task does not receive `SCALE_SET_CONTROLLER_GROUP_CONFIG_PATH` or `SCALE_SET_CONTROLLER_GROUP_CONFIG_REVISION`. +The controller reads the referenced SSM parameters and optional discovery-cache parameters but does not write discovered IDs back to Parameter Store. This keeps the task role read-only; callers that want cached runner-group IDs must provision them separately. + The module also keeps the per-reconciler SSM parameters available for the grouped configuration path while that delivery mode is being phased in. The task currently uses the manifest environment variable. Terraform derives `sessionOwner` locally as `.`; if that would exceed the runtime's 256-character limit, the module truncates both readable components and appends a deterministic hash. @@ -138,13 +150,7 @@ GitHub credential **values** never enter Terraform configuration, task definitio ## Container image -The convenience default is: - -```text -ghcr.io/github-aws-runners/terraform-aws-github-runner-scale-set-service:latest -``` - -ECS `versionConsistency` is enabled so all tasks in a deployment resolve a tag consistently. Production callers should set `container.image` to the digest published with a release: +`container.image` is required. The module intentionally has no mutable default image because the controller image must be published and reviewed independently of the Terraform module. Set it to an immutable release digest: ```hcl container = { @@ -152,9 +158,9 @@ container = { } ``` -Public registry images need no pull permission. For a private ECR override, set `container.image` to the ECR image URI. The module grants the ECS task execution role wildcard ECR layer-pull permissions plus the unavoidable resource-unscoped `ecr:GetAuthorizationToken` action. The application task role is not used for image pulls. Repository-side access policy remains owned by the ECR module that owns the repository. +Public registry images need no ECR pull permission. For a private ECR image, set `container.image` to the ECR image URI. The module grants the ECS task execution role repository-scoped layer-pull permissions plus the unavoidable resource-unscoped `ecr:GetAuthorizationToken` action only for that private ECR image. The application task role is not used for image pulls. Repository-side access policy remains owned by the ECR module that owns the repository. -For the official GHCR default, verify an anonymous pull after the first package publish. Package visibility may inherit repository or organization settings and must not be inferred only from a successful authenticated workflow push. +Package visibility may inherit repository or organization settings and must not be inferred only from a successful authenticated workflow push. Verify an anonymous pull before deploying a public-registry image. ## ECS and security behavior @@ -162,10 +168,10 @@ For the official GHCR default, verify an anonymous pull after the first package - Every group gets a separate service, task definition, task role, execution role, log group, and security group. - `desired_count` is fixed at one. Deployment percentages are `minimum = 0` and `maximum = 100`, preventing old and new tasks from overlapping while session leasing is unavailable. - The ECS deployment circuit breaker and rollback are enabled. -- Tasks run in supplied private subnets with public IP assignment disabled. Managed security groups have no ingress and allow only TCP/443 egress. The IPv4 Internet default is intended for controlled NAT/firewall paths and can be narrowed. +- Tasks run in supplied private subnets with public IP assignment disabled. Managed security groups have no ingress and allow only TCP/443 egress. The default `0.0.0.0/0` is a reachability convenience, not a narrow GitHub allowlist: GitHub publishes outbound ranges at [`https://api.github.com/meta`](https://api.github.com/meta), and adopters should restrict egress to those ranges, a NAT gateway, firewall, or proxy when their security posture requires it. - The application container runs with a numeric non-root UID/GID, a read-only root filesystem, init enabled, no privilege, and all Linux capabilities dropped. - ECS probes `/healthz` for liveness, and `container.health_path` accepts only that endpoint. `/readyz` remains an application readiness signal; reconnecting to GitHub should not cause ECS to restart every reconciler in a group. -- CloudWatch encrypts logs at rest with an AWS-owned key by default. Set `logging.kms_key_arn` for a customer-managed key and ensure its key policy allows the regional CloudWatch Logs service. +- CloudWatch encrypts logs at rest with an AWS-owned key by default. Set `logging.kms_key_id` to a customer-managed key ID, alias, or ARN and ensure its key policy allows the regional CloudWatch Logs service. ## Plan-shape requirements @@ -235,12 +241,12 @@ No modules. | Name | Description | Type | Default | Required | |------|-------------|------|---------|:--------:| | [config\_store](#input\_config\_store) | Non-secret controller configuration storage. The module writes one SSM String parameter per reconciler below `path_prefix//`. The task receives only its group path and a SHA-256 revision, then loads the group with `GetParametersByPath`.

Standard parameters are limited to 4096 encoded bytes and Advanced parameters to 8192 encoded bytes. Null `path_prefix` resolves to `//scale-set-controller`. |
object({
path_prefix = optional(string, null)
tier = optional(string, "Standard")
tags = optional(map(string), {})
})
| `{}` | no | -| [container](#input\_container) | Scale-set controller image and runtime settings. A null image uses the internal official convenience image; production callers should use the release digest. Filesystem and Linux capability hardening are enforced by the module; health\_path is fixed at /healthz, the ECS liveness endpoint. |
object({
image = optional(string, null)
user = optional(string, "10001:10001")
health_port = optional(number, 8080)
health_path = optional(string, "/healthz")
health_check_command = optional(list(string), null)
health_check_interval = optional(number, 30)
health_check_timeout = optional(number, 5)
health_check_retries = optional(number, 3)
health_check_start_period = optional(number, 30)
health_stale_after_seconds = optional(number, 180)
shutdown_timeout_seconds = optional(number, 110)
session_close_timeout_seconds = optional(number, 10)
reconnect_initial_backoff_seconds = optional(number, 1)
reconnect_max_backoff_seconds = optional(number, 30)
stop_timeout_seconds = optional(number, 120)
})
| `{}` | no | +| [container](#input\_container) | Scale-set controller image and runtime settings. image is required because the module has no mutable default image; use an immutable release digest whenever possible. Filesystem and Linux capability hardening are enforced by the module; health\_path is fixed at /healthz, the ECS liveness endpoint. |
object({
image = optional(string, null)
user = optional(string, "10001:10001")
health_port = optional(number, 8080)
health_path = optional(string, "/healthz")
health_check_command = optional(list(string), null)
health_check_interval = optional(number, 30)
health_check_timeout = optional(number, 5)
health_check_retries = optional(number, 3)
health_check_start_period = optional(number, 30)
health_stale_after_seconds = optional(number, 180)
shutdown_timeout_seconds = optional(number, 110)
session_close_timeout_seconds = optional(number, 10)
reconnect_initial_backoff_seconds = optional(number, 1)
reconnect_max_backoff_seconds = optional(number, 30)
stop_timeout_seconds = optional(number, 120)
})
| `{}` | no | | [ecs](#input\_ecs) | ECS substrate configuration. A managed cluster is created by default. For an external cluster, set `cluster.mode = "external"` and pass its ARN; the mode must be plan-known while the ARN may be computed. |
object({
cluster = optional(object({
mode = optional(string, "managed")
arn = optional(string, null)
name = optional(string, null)
container_insights = optional(bool, true)
}), {})
task = optional(object({
cpu = optional(number, 512)
memory = optional(number, 1024)
cpu_architecture = optional(string, "X86_64")
ephemeral_storage = optional(object({
size_in_gib = number
}), null)
}), {})
service = optional(object({
platform_version = optional(string, "LATEST")
}), {})
iam = optional(object({
path = optional(string, "/")
permissions_boundary = optional(string, null)
}), {})
})
| `{}` | no | | [grouping](#input\_grouping) | Packing strategy for scale-set reconcilers. `compute_provider` creates one controller group per compute-provider type and is the default. `runner_config` creates one group per runner config. `custom` uses `custom.groups`; custom membership must cover every runner config exactly once.

The strategy, custom group keys, and memberships select Terraform `for_each` instances and must be known during planning. |
object({
strategy = optional(string, "compute_provider")
custom = optional(object({
groups = map(object({
runner_configs = set(string)
}))
}), null)
})
| `{}` | no | | [log\_level](#input\_log\_level) | Logging level for the scale-set controller container. | `string` | `"info"` | no | -| [logging](#input\_logging) | CloudWatch Logs configuration. CloudWatch encrypts logs at rest with an AWS-owned key by default; set `kms_key_arn` to use a customer-managed key. |
object({
retention_in_days = optional(number, 30)
kms_key_arn = optional(string, null)
log_group_class = optional(string, "STANDARD")
tags = optional(map(string), {})
})
| `{}` | no | -| [network](#input\_network) | Private Fargate networking. Tasks never receive public IP addresses and the managed security groups have no ingress. HTTPS egress defaults to IPv4 Internet access because GitHub endpoints cannot be represented as security-group destinations; route it through controlled NAT, firewall, or proxy infrastructure when required. |
object({
vpc_id = string
subnet_ids = set(string)
https_egress = optional(object({
ipv4_cidrs = optional(set(string), ["0.0.0.0/0"])
ipv6_cidrs = optional(set(string), [])
}), {})
})
| n/a | yes | +| [logging](#input\_logging) | CloudWatch Logs configuration. CloudWatch encrypts logs at rest with an AWS-owned key by default; set `kms_key_id` to a customer-managed key ID or ARN. |
object({
retention_in_days = optional(number, 180)
kms_key_id = optional(string, null)
log_group_class = optional(string, "STANDARD")
tags = optional(map(string), {})
})
| `{}` | no | +| [network](#input\_network) | Private Fargate networking. Tasks never receive public IP addresses and the managed security groups have no ingress. HTTPS egress defaults to full IPv4 Internet access for reachability. GitHub publishes outbound ranges at `https://api.github.com/meta`; restrict egress to those ranges, a NAT gateway, firewall, or proxy when your security posture requires it. |
object({
vpc_id = string
subnet_ids = set(string)
https_egress = optional(object({
ipv4_cidrs = optional(set(string), ["0.0.0.0/0"])
ipv6_cidrs = optional(set(string), [])
}), {})
})
| n/a | yes | | [prefix](#input\_prefix) | Stable prefix used for scale-set controller resources. | `string` | `"github-actions"` | no | | [runner\_configs](#input\_runner\_configs) | Normalized scale-set runner configurations keyed by stable runner-config name.

Map keys must be known during planning. Credential values are never accepted: `github.app` contains only the exact GitHub App Parameter Store references used by the runtime. `github.enterprise_server` and `github.user_agent` carry the global GitHub settings needed to render each reconciler configuration. `scale_set.runner.group_name` selects the GitHub runner group. `runner_registration_level` selects organization or repository registration, and `runner_owner` supplies the corresponding organization or owner/repository path. Enterprise-level registration is not supported by this module. `compute_provider` carries the provider-neutral scale-set capability contract for this runner configuration. Parameter and optional KMS ARNs, scale-set names, and other inner values may remain unknown until apply. |
map(object({
github = object({
enterprise_server = object({
url = optional(string, null)
ssl_verify = optional(bool, true)
})
app = object({
app_id = object({
name = string
arn = string
kms_key_arn = optional(string, null)
})
private_key = object({
name = string
arn = string
kms_key_arn = optional(string, null)
})
installation_id = object({
name = string
arn = string
kms_key_arn = optional(string, null)
})
})
runner_owner = string
runner_registration_level = string
user_agent = string
})
scale_set = object({
name = string
runner = optional(object({
labels = optional(list(string), [])
group_name = optional(string, "Default")
min_runners = optional(number, 0)
max_runners = optional(number, 10)
boot_time_in_minutes = optional(number, 10)
}), {})
})
compute_provider = object({
type = string
capabilities = object({
scale_set = object({
role_arn = optional(string, null)
configuration_json = optional(string, "{}")
environment_variables = optional(map(string), {})
iam_statements = optional(map(object({
actions = set(string)
resources = set(string)
conditions = optional(list(object({
test = string
variable = string
values = set(string)
})), [])
})), {})
})
})
})
}))
| n/a | yes | | [tags](#input\_tags) | Tags applied to scale-set orchestration resources. | `map(string)` | `{}` | no | diff --git a/modules/orchestration-providers/scale-set/iam.tf b/modules/orchestration-providers/scale-set/iam.tf index 6065fff396..d3a38524b0 100644 --- a/modules/orchestration-providers/scale-set/iam.tf +++ b/modules/orchestration-providers/scale-set/iam.tf @@ -188,23 +188,31 @@ data "aws_iam_policy_document" "execution" { resources = ["${aws_cloudwatch_log_group.controller[each.key].arn}:*"] } - statement { - sid = "PullPrivateEcrImage" - effect = "Allow" - actions = [ - "ecr:BatchCheckLayerAvailability", - "ecr:BatchGetImage", - "ecr:GetDownloadUrlForLayer", - ] - resources = ["*"] + dynamic "statement" { + for_each = local.uses_private_ecr ? [local.private_ecr_repository_arn] : [] + + content { + sid = "PullPrivateEcrImage" + effect = "Allow" + actions = [ + "ecr:BatchCheckLayerAvailability", + "ecr:BatchGetImage", + "ecr:GetDownloadUrlForLayer", + ] + resources = [statement.value] + } } - statement { - # ECR does not support resource-level permissions for authorization tokens. - sid = "AuthorizePrivateEcrPull" - effect = "Allow" - actions = ["ecr:GetAuthorizationToken"] - resources = ["*"] + dynamic "statement" { + for_each = local.uses_private_ecr ? [true] : [] + + content { + # ECR does not support resource-level permissions for authorization tokens. + sid = "AuthorizePrivateEcrPull" + effect = "Allow" + actions = ["ecr:GetAuthorizationToken"] + resources = ["*"] + } } } diff --git a/modules/orchestration-providers/scale-set/locals.tf b/modules/orchestration-providers/scale-set/locals.tf index 523b6b5a26..0dd4f79ad2 100644 --- a/modules/orchestration-providers/scale-set/locals.tf +++ b/modules/orchestration-providers/scale-set/locals.tf @@ -40,8 +40,19 @@ locals { ) } - official_container_image = "ghcr.io/github-aws-runners/terraform-aws-github-runner-scale-set-service:latest" - resolved_container_image = coalesce(var.container.image, local.official_container_image) + resolved_container_image = var.container.image + private_ecr_image_match = try(regex( + "^([0-9]{12})\\.dkr\\.ecr\\.([a-z0-9-]+)\\.amazonaws\\.com(\\.cn)?/([A-Za-z0-9._/-]+)([:@].+)?$", + var.container.image, + ), []) + uses_private_ecr = length(local.private_ecr_image_match) > 0 + private_ecr_repository_arn = local.uses_private_ecr ? format( + "arn:%s:ecr:%s:%s:repository/%s", + data.aws_partition.current.partition, + local.private_ecr_image_match[1], + local.private_ecr_image_match[0], + local.private_ecr_image_match[3], + ) : null resolved_health_check_command = var.container.health_check_command != null ? var.container.health_check_command : [ "CMD", diff --git a/modules/orchestration-providers/scale-set/logging.tf b/modules/orchestration-providers/scale-set/logging.tf index 974106316a..0c04c00fac 100644 --- a/modules/orchestration-providers/scale-set/logging.tf +++ b/modules/orchestration-providers/scale-set/logging.tf @@ -3,7 +3,7 @@ resource "aws_cloudwatch_log_group" "controller" { name = "/aws/ecs/${local.group_resource_names[each.key]}" retention_in_days = var.logging.retention_in_days - kms_key_id = var.logging.kms_key_arn + kms_key_id = var.logging.kms_key_id log_group_class = var.logging.log_group_class tags = merge( diff --git a/modules/orchestration-providers/scale-set/tests/fixtures/computed-inputs/main.tf b/modules/orchestration-providers/scale-set/tests/fixtures/computed-inputs/main.tf index db8ae2b225..a9d116e911 100644 --- a/modules/orchestration-providers/scale-set/tests/fixtures/computed-inputs/main.tf +++ b/modules/orchestration-providers/scale-set/tests/fixtures/computed-inputs/main.tf @@ -16,6 +16,10 @@ module "subject" { prefix = "computed-test" + container = { + image = "ghcr.io/github-aws-runners/terraform-aws-github-runner-scale-set-service@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + } + runner_configs = { computed = { github = { diff --git a/modules/orchestration-providers/scale-set/tests/scale-set.tftest.hcl b/modules/orchestration-providers/scale-set/tests/scale-set.tftest.hcl index b939c0310b..1bf135461a 100644 --- a/modules/orchestration-providers/scale-set/tests/scale-set.tftest.hcl +++ b/modules/orchestration-providers/scale-set/tests/scale-set.tftest.hcl @@ -39,6 +39,10 @@ mock_provider "aws" { variables { prefix = "scale-set-test" + container = { + image = "ghcr.io/github-aws-runners/terraform-aws-github-runner-scale-set-service@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + } + runner_configs = { linux-small = { github = { @@ -224,7 +228,7 @@ variables { } logging = { - kms_key_arn = "arn:aws:kms:eu-west-1:123456789012:key/22222222-2222-2222-2222-222222222222" + kms_key_id = "arn:aws:kms:eu-west-1:123456789012:key/22222222-2222-2222-2222-222222222222" } tags = { @@ -283,7 +287,7 @@ run "groups_by_compute_provider_and_hardens_each_task" { condition = alltrue([ for task in values(aws_ecs_task_definition.controller) : ( length(jsondecode(task.container_definitions)) == 1 && - jsondecode(task.container_definitions)[0].image == "ghcr.io/github-aws-runners/terraform-aws-github-runner-scale-set-service:latest" && + jsondecode(task.container_definitions)[0].image == "ghcr.io/github-aws-runners/terraform-aws-github-runner-scale-set-service@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" && jsondecode(task.container_definitions)[0].versionConsistency == "enabled" && jsondecode(task.container_definitions)[0].readonlyRootFilesystem && !jsondecode(task.container_definitions)[0].privileged && @@ -363,7 +367,7 @@ run "groups_by_compute_provider_and_hardens_each_task" { length(aws_security_group.controller["ec2"].egress) == 1 && one(aws_security_group.controller["ec2"].egress).from_port == 443 && one(aws_security_group.controller["ec2"].egress).to_port == 443 && - aws_cloudwatch_log_group.controller["ec2"].kms_key_id == var.logging.kms_key_arn + aws_cloudwatch_log_group.controller["ec2"].kms_key_id == var.logging.kms_key_id ) error_message = "Controller networking must have no ingress and only HTTPS egress, and logs must honor customer-managed encryption." } @@ -396,7 +400,9 @@ run "groups_by_compute_provider_and_hardens_each_task" { && contains(flatten([for statement in data.aws_iam_policy_document.task["ec2"].statement : statement.actions]), "sts:AssumeRole") && !contains(flatten([for statement in data.aws_iam_policy_document.task["ec2"].statement : statement.resources]), "arn:aws:ssm:eu-west-1:123456789012:parameter/scale-set-test/runners/config/ami_id") && contains(flatten([for statement in data.aws_iam_policy_document.compute["ec2/linux-small"].statement : statement.actions]), "ssm:GetParameters") && - contains(flatten([for statement in data.aws_iam_policy_document.compute["ec2/linux-small"].statement : statement.resources]), "arn:aws:ssm:eu-west-1:123456789012:parameter/scale-set-test/runners/config/ami_id") + contains(flatten([for statement in data.aws_iam_policy_document.compute["ec2/linux-small"].statement : statement.resources]), "arn:aws:ssm:eu-west-1:123456789012:parameter/scale-set-test/runners/config/ami_id") && + !contains(flatten([for statement in data.aws_iam_policy_document.execution["ec2"].statement : statement.actions]), "ecr:GetAuthorizationToken") && + !contains(flatten([for statement in data.aws_iam_policy_document.execution["ec2"].statement : statement.actions]), "ecr:BatchGetImage") ) error_message = "Controller IAM must contain only controller permissions, while provider permissions such as AMI SSM reads must be attached to the compute role." } @@ -437,7 +443,7 @@ run "grants_execution_role_ecr_pull_permissions" { condition = ( contains(flatten([ for statement in data.aws_iam_policy_document.execution["ec2"].statement : statement.resources - ]), "*") && + ]), "arn:aws:ecr:eu-west-1:999999999999:repository/scale-set-controller") && contains(flatten([ for statement in data.aws_iam_policy_document.execution["ec2"].statement : statement.actions ]), "ecr:GetAuthorizationToken") && @@ -451,8 +457,18 @@ run "grants_execution_role_ecr_pull_permissions" { for statement in data.aws_iam_policy_document.execution["ec2"].statement : statement.actions ]), "ecr:GetDownloadUrlForLayer") ) - error_message = "The ECS execution role must have wildcard ECR pull permissions, including the authorization-token permission." + error_message = "Private ECR images must receive repository-scoped layer-pull permissions and the unavoidable wildcard authorization-token permission." + } +} + +run "requires_explicit_container_image" { + command = plan + + variables { + container = {} } + + expect_failures = [terraform_data.validate_runtime] } run "supports_exact_custom_groups" { diff --git a/modules/orchestration-providers/scale-set/validations.tf b/modules/orchestration-providers/scale-set/validations.tf index 9553581597..446de1022e 100644 --- a/modules/orchestration-providers/scale-set/validations.tf +++ b/modules/orchestration-providers/scale-set/validations.tf @@ -266,11 +266,11 @@ resource "terraform_data" "validate_runtime" { lifecycle { precondition { condition = ( - var.container.image == null ? true : ( - length(trimspace(var.container.image)) > 0 && - length(regexall("[[:space:]]", var.container.image)) == 0 - )) - error_message = "Container image references must be non-empty and cannot contain whitespace." + var.container.image != null && + length(trimspace(var.container.image)) > 0 && + length(regexall("[[:space:]]", var.container.image)) == 0 + ) + error_message = "container.image must be set to a non-empty image reference without whitespace; the module has no mutable default image." } precondition { @@ -364,9 +364,9 @@ resource "terraform_data" "validate_runtime" { condition = ( contains(["STANDARD", "INFREQUENT_ACCESS"], var.logging.log_group_class) && contains([1, 3, 5, 7, 14, 30, 60, 90, 120, 150, 180, 365, 400, 545, 731, 1096, 1827, 2192, 2557, 2922, 3288, 3653], var.logging.retention_in_days) && - (var.logging.kms_key_arn == null ? true : can(regex("^arn:[^:]+:kms:[^:]+:[0-9]{12}:key/.+$", var.logging.kms_key_arn))) + (var.logging.kms_key_id == null ? true : can(regex("^arn:[^:]+:kms:[^:]+:[0-9]{12}:key/.+$", var.logging.kms_key_id))) ) - error_message = "logging must use a supported class and retention period; kms_key_arn must be a KMS key ARN when set." + error_message = "logging must use a supported class and retention period; kms_key_id must be a KMS ARN when set." } } } diff --git a/modules/orchestration-providers/scale-set/variables.tf b/modules/orchestration-providers/scale-set/variables.tf index 236dfc8213..5627ec0fd4 100644 --- a/modules/orchestration-providers/scale-set/variables.tf +++ b/modules/orchestration-providers/scale-set/variables.tf @@ -97,7 +97,7 @@ variable "grouping" { } variable "container" { - description = "Scale-set controller image and runtime settings. A null image uses the internal official convenience image; production callers should use the release digest. Filesystem and Linux capability hardening are enforced by the module; health_path is fixed at /healthz, the ECS liveness endpoint." + description = "Scale-set controller image and runtime settings. image is required because the module has no mutable default image; use an immutable release digest whenever possible. Filesystem and Linux capability hardening are enforced by the module; health_path is fixed at /healthz, the ECS liveness endpoint." type = object({ image = optional(string, null) user = optional(string, "10001:10001") @@ -167,7 +167,7 @@ variable "ecs" { variable "network" { description = <<-EOT - Private Fargate networking. Tasks never receive public IP addresses and the managed security groups have no ingress. HTTPS egress defaults to IPv4 Internet access because GitHub endpoints cannot be represented as security-group destinations; route it through controlled NAT, firewall, or proxy infrastructure when required. + Private Fargate networking. Tasks never receive public IP addresses and the managed security groups have no ingress. HTTPS egress defaults to full IPv4 Internet access for reachability. GitHub publishes outbound ranges at `https://api.github.com/meta`; restrict egress to those ranges, a NAT gateway, firewall, or proxy when your security posture requires it. EOT type = object({ vpc_id = string @@ -181,10 +181,10 @@ variable "network" { } variable "logging" { - description = "CloudWatch Logs configuration. CloudWatch encrypts logs at rest with an AWS-owned key by default; set `kms_key_arn` to use a customer-managed key." + description = "CloudWatch Logs configuration. CloudWatch encrypts logs at rest with an AWS-owned key by default; set `kms_key_id` to a customer-managed key ID or ARN." type = object({ - retention_in_days = optional(number, 30) - kms_key_arn = optional(string, null) + retention_in_days = optional(number, 180) + kms_key_id = optional(string, null) log_group_class = optional(string, "STANDARD") tags = optional(map(string), {}) }) diff --git a/modules/runner-config/ssm-housekeeper/README.md b/modules/runner-config/ssm-housekeeper/README.md index a878605eda..b8899e3f43 100644 --- a/modules/runner-config/ssm-housekeeper/README.md +++ b/modules/runner-config/ssm-housekeeper/README.md @@ -1,7 +1,5 @@ # SSM housekeeper module -Cleanup is stateless: each invocation lists current parameters and deletes eligible items page by page. It starts deleting before listing the next page, including after empty pages. Individual deletion failures do not block other items, and a later listing failure leaves earlier deletions completed. A deadline guard stops new work with ten seconds remaining. The next scheduled invocation starts a fresh scan; deleted parameters are no longer listed. No scan cursor or completed-item list is stored. Age and dry-run protections remain in place. - > This module is treated as an internal module; breaking changes do not trigger a major release bump. This provider-neutral child module owns the Lambda function, EventBridge schedule, IAM policies, and CloudWatch log group used to remove expired runner registration parameters from Parameter Store. From 1dc7bdbaff0b425404b192fe1993086f7946f158 Mon Sep 17 00:00:00 2001 From: edersonbrilhante Date: Fri, 25 Sep 2026 16:11:06 +0200 Subject: [PATCH 2/2] ci: adjust conflict from merge --- .github/workflows/lambda.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/lambda.yml b/.github/workflows/lambda.yml index 4a2decadc5..4fdbc150c7 100644 --- a/.github/workflows/lambda.yml +++ b/.github/workflows/lambda.yml @@ -97,4 +97,4 @@ jobs: cache-from: type=gha,scope=scale-set-service - name: Run scale-set service image smoke test - run: ./.github/scripts/scale-set-container-smoke-test.sh + run: ./tests/scale-set-container-smoke-test.sh