diff --git a/.github/dependabot.yml b/.github/dependabot.yml index 23719d296..a87bbf6a1 100644 --- a/.github/dependabot.yml +++ b/.github/dependabot.yml @@ -6,6 +6,16 @@ updates: interval: "weekly" commit-message: prefix: "chore" + groups: + # init, autobuild and analyze are separate dependencies to Dependabot, so + # ungrouped they land in separate pull requests. They share one CodeQL + # bundle — init writes a configuration file that the other two read back — + # so whichever merges first leaves the workflow on mismatched versions and + # the run fails with "Loaded a configuration file for version X, but + # running version Y". Keep them in one pull request. + codeql-action: + patterns: + - "github/codeql-action*" - package-ecosystem: "docker" directory: "/docker" diff --git a/.github/workflows/chart.yml b/.github/workflows/chart.yml deleted file mode 100644 index 600df71d6..000000000 --- a/.github/workflows/chart.yml +++ /dev/null @@ -1,100 +0,0 @@ -name: Helm Chart Publisher - -on: - push: - # Pre-release tags (e.g. v0.4.0-rc.1) build images via release.yml but - # must not land in the public Helm index. The negative pattern below - # filters them out; workflow_dispatch can still publish a specific - # tag manually if ever needed. - tags: - - "v*.*.*" - - "!v*-rc.*" - workflow_dispatch: - inputs: - tag: - description: "Release tag (e.g., v1.0.0)" - required: true - type: string -permissions: - contents: write - packages: write - -env: - REGISTRY: ghcr.io - -jobs: - export-registry: - uses: ./.github/workflows/setup-release.yml - with: - tag: ${{ inputs.tag || github.ref_name }} - - publish-github-pages: - needs: export-registry - runs-on: ubuntu-latest - # Only the gh-pages publish needs serialization: helm-gh-pages always - # rewrites the gh-pages branch, so concurrent runs for different tags - # would race. The OCI publish below pushes immutable per-tag blobs and - # is safe to run in parallel across tags, so it stays unguarded. - concurrency: - group: helm-chart-publish-gh-pages - cancel-in-progress: false - steps: - - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - with: - submodules: true - fetch-depth: 0 - - name: Publish Helm chart to GitHub Pages - uses: stefanprodan/helm-gh-pages@0ad2bb377311d61ac04ad9eb6f252fb68e207260 # v1.7.0 - with: - token: ${{ secrets.GITHUB_TOKEN }} - charts_dir: charts - target_dir: charts - linting: on - - publish-oci: - needs: export-registry - runs-on: ubuntu-latest - steps: - - name: Checkout code - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - - - name: Login to GitHub Container Registry - uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0 - with: - registry: ${{ env.REGISTRY }} - username: ${{ github.actor }} - password: ${{ secrets.GITHUB_TOKEN }} - - - name: Package and push Helm charts to GHCR via Makefile - run: | - set -euo pipefail - - RELEASE_TAG="${{ needs.export-registry.outputs.tag }}" - CHART_VERSION="${{ needs.export-registry.outputs.version }}" - OCI_REGISTRY="${{ needs.export-registry.outputs.registry }}/charts" - - make helm-push REGISTRY="${OCI_REGISTRY}" TAG="${RELEASE_TAG}" CHART_VERSION="${CHART_VERSION}" - - - name: Verify chart appVersion matches release tag - run: | - set -euo pipefail - - RELEASE_TAG="${{ needs.export-registry.outputs.tag }}" - CHART_VERSION="${{ needs.export-registry.outputs.version }}" - EXPECTED_APP_VERSION="${RELEASE_TAG}" - - rm -rf .helm-verify - mkdir -p .helm-verify - - for chart in hub-agent member-agent; do - helm pull "oci://${{ needs.export-registry.outputs.registry }}/charts/${chart}" --version "${CHART_VERSION}" --destination .helm-verify >/dev/null - packaged=".helm-verify/${chart}-${CHART_VERSION}.tgz" - actual_app_version=$(tar -xOf "${packaged}" "${chart}/Chart.yaml" | awk -F': ' '/^appVersion:/ {gsub(/"/, "", $2); print $2}') - if [[ "${actual_app_version}" != "${EXPECTED_APP_VERSION}" ]]; then - echo "ERROR: ${chart} appVersion (${actual_app_version}) does not match release tag (${EXPECTED_APP_VERSION})" - exit 1 - fi - echo "✅ ${chart} appVersion=${actual_app_version} matches release tag=${EXPECTED_APP_VERSION}" - done - - rm -rf .helm-verify diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 4d3d8d6b7..02c3c2f9f 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -76,7 +76,7 @@ jobs: KUBEFLEET_CI_TEST_RUNNER_NAME: 'ginkgo' - name: Upload Codecov report - uses: codecov/codecov-action@fb8b3582c8e4def4969c97caa2f19720cb33a72f # v7.0.0 + uses: codecov/codecov-action@0b35c9ecc4f0529d0eb674914510c22f85b196b4 # v7.1.0 with: ## Repository upload token - get it from codecov.io. Required only for private repositories token: ${{ secrets.CODECOV_TOKEN }} diff --git a/.github/workflows/codeql-analysis.yml b/.github/workflows/codeql-analysis.yml index 9651d240c..b0566085d 100644 --- a/.github/workflows/codeql-analysis.yml +++ b/.github/workflows/codeql-analysis.yml @@ -42,7 +42,7 @@ jobs: # Initializes the CodeQL tools for scanning. - name: Initialize CodeQL - uses: github/codeql-action/init@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v4 + uses: github/codeql-action/init@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4 with: languages: ${{ matrix.language }} # If you wish to specify custom queries, you can do so here or in a config file. @@ -56,7 +56,7 @@ jobs: # Autobuild attempts to build any compiled languages (C/C++, C#, or Java). # If this step fails, then you should remove it and run the build manually (see below) - name: Autobuild - uses: github/codeql-action/autobuild@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v4 + uses: github/codeql-action/autobuild@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4 # ℹ️ Command-line programs to run using the OS shell. # 📚 See https://docs.github.com/en/actions/using-workflows/workflow-syntax-for-github-actions#jobsjob_idstepsrun @@ -69,4 +69,4 @@ jobs: # ./location_of_script_within_repo/buildscript.sh - name: Perform CodeQL Analysis - uses: github/codeql-action/analyze@db488ddef3bf6cb639b32c2e9a7c0a7ea8271d28 # v4 + uses: github/codeql-action/analyze@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4 diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 030e9b91b..12176fce5 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -1,4 +1,14 @@ -name: Release Images +name: Release + +# One workflow owns the whole release. Every artifact a tag produces - images, +# the CRD bundle, the Helm charts - is built by a job in this graph, and the +# GitHub Release stays a draft until all of them have succeeded. +# +# This replaces the previous split between release.yml and chart.yml, which had +# no ordering between them: a chart publish could fail (or simply never run) +# while the release was already public, leaving a release that advertised charts +# nobody could pull, and images the charts pointed at could be published after +# the charts that referenced them. on: push: @@ -11,50 +21,73 @@ on: required: true type: string -permissions: - contents: read - packages: write - -# Serialize releases per ref so concurrent tag pushes can't race on image -# pushes to the same ${REGISTRY}/${IMAGE}:${TAG}. Different tags can still -# run in parallel. We never want cancel-in-progress here: aborting a -# half-pushed image is worse than letting it finish. +# Serialize per release tag: a re-run must not race the original run on the same +# registry paths or the same draft release. Distinct tags still run in parallel. +# Never cancel-in-progress - aborting a half-pushed image or a half-uploaded +# release asset leaves more mess than letting the run finish. concurrency: - group: release-images-${{ github.ref }} + group: release-${{ inputs.tag || github.ref_name }} cancel-in-progress: false +# Least privilege by default; each job widens only what it needs. +permissions: + contents: read + env: - REGISTRY: ghcr.io HUB_AGENT_IMAGE_NAME: hub-agent MEMBER_AGENT_IMAGE_NAME: member-agent REFRESH_TOKEN_IMAGE_NAME: refresh-token - GO_VERSION: "1.26.6" jobs: - export-registry: + # Validates the tag shape and derives every value the rest of the graph keys + # off (registry path, tag, version, prerelease). A malformed tag fails here, + # before anything is published. + setup: uses: ./.github/workflows/setup-release.yml with: tag: ${{ inputs.tag || github.ref_name }} - build-and-publish: - needs: export-registry - env: - REGISTRY: ${{ needs.export-registry.outputs.registry }} - TAG: ${{ needs.export-registry.outputs.tag }} + # Create the release up front, as a draft, so the producer jobs have somewhere + # to upload while the release stays invisible to consumers. publish-release + # flips it at the end; until then a failed run leaves only a draft. + create-draft-release: + needs: setup runs-on: ubuntu-latest + permissions: + contents: write + env: + TAG: ${{ needs.setup.outputs.tag }} + PRERELEASE: ${{ needs.setup.outputs.prerelease }} + GH_TOKEN: ${{ github.token }} + GH_REPO: ${{ github.repository }} steps: - - name: Set up Go ${{ env.GO_VERSION }} - uses: actions/setup-go@b7ad1dad31e06c5925ef5d2fc7ad053ef454303e # v7 + - name: Checkout code + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: - go-version: ${{ env.GO_VERSION }} + ref: ${{ needs.setup.outputs.tag }} + - name: Create or reuse the draft release + run: ./hack/release/create-draft-release.sh + + publish-images: + needs: [setup, create-draft-release] + runs-on: ubuntu-latest + permissions: + contents: read + packages: write + env: + REGISTRY: ${{ needs.setup.outputs.registry }} + TAG: ${{ needs.setup.outputs.tag }} + VERSION: ${{ needs.setup.outputs.version }} + PRERELEASE: ${{ needs.setup.outputs.prerelease }} + steps: - name: Checkout code uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: - ref: ${{ needs.export-registry.outputs.tag }} + ref: ${{ needs.setup.outputs.tag }} - name: Login to ghcr.io - uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f + uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0 with: registry: ghcr.io username: ${{ github.actor }} @@ -66,12 +99,10 @@ jobs: # published under the long form ("v0.4.0-rc.1") for testers only: the # short-tag namespace is deliberately reserved for stable releases that # consumers can safely pin to, so RC tags get no short alias. - - name: Build and push images with tag ${{ env.TAG }} - env: - VERSION: ${{ needs.export-registry.outputs.version }} + - name: Build and push images with tag ${{ needs.setup.outputs.tag }} run: | set -euo pipefail - if [[ "${TAG}" == *-rc.* ]]; then + if [ "${PRERELEASE}" = "true" ]; then make push else make push IMAGE_EXTRA_TAG="${VERSION}" @@ -82,18 +113,16 @@ jobs: # architecture would otherwise go unnoticed until a consumer on the other # architecture failed to pull. Stable releases also carry the short alias. - name: Verify images are multi-arch - env: - VERSION: ${{ needs.export-registry.outputs.version }} run: | set -euo pipefail tags="${TAG}" - if [[ "${TAG}" != *-rc.* ]]; then + if [ "${PRERELEASE}" != "true" ]; then tags="${tags} ${VERSION}" fi echo "✅ Verifying published images:" - for IMAGE in ${{ env.HUB_AGENT_IMAGE_NAME }} ${{ env.MEMBER_AGENT_IMAGE_NAME }} ${{ env.REFRESH_TOKEN_IMAGE_NAME }}; do + for IMAGE in "${HUB_AGENT_IMAGE_NAME}" "${MEMBER_AGENT_IMAGE_NAME}" "${REFRESH_TOKEN_IMAGE_NAME}"; do for tag in ${tags}; do - ref="${{ env.REGISTRY }}/${IMAGE}:${tag}" + ref="${REGISTRY}/${IMAGE}:${tag}" echo " - ${ref}" manifest="$(docker buildx imagetools inspect "${ref}")" for platform in linux/amd64 linux/arm64; do @@ -106,51 +135,216 @@ jobs: # Publish the raw CRDs as a standalone release asset so consumers can install # them without pulling a Helm chart. The bundle carries the unmodified CRDs the # charts install (no downstream-specific labels), split into crds/hub and - # crds/member so each set can be applied to the right cluster. Runs after the - # images are published so a release is only created once the build succeeds. + # crds/member so each set can be applied to the right cluster. publish-crds: - needs: [export-registry, build-and-publish] + needs: [setup, create-draft-release] runs-on: ubuntu-latest permissions: contents: write env: - TAG: ${{ needs.export-registry.outputs.tag }} + TAG: ${{ needs.setup.outputs.tag }} steps: - name: Checkout code uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: - ref: ${{ needs.export-registry.outputs.tag }} + ref: ${{ needs.setup.outputs.tag }} - name: Package CRDs run: make crd-package TAG="${TAG}" - - name: Create or update the release and upload the CRD bundle + # --clobber makes the upload idempotent so a re-run replaces the asset + # rather than failing on a name collision. The token is scoped to this + # step so it is not in the environment of the packaging step above. + - name: Upload the CRD bundle to the draft release env: GH_TOKEN: ${{ github.token }} run: | set -euo pipefail - created_draft="" - if ! gh release view "${TAG}" >/dev/null 2>&1; then - # Any semver pre-release suffix (-rc.N, -alpha, -beta, ...) is a prerelease. - prerelease="" - case "${TAG}" in *-*) prerelease="--prerelease" ;; esac - # Create as a draft first so a partially uploaded release is never public. - # --verify-tag: `gh release create` creates the tag itself when it is - # missing, pointing at the default branch's head. A tag that does not - # exist at all already fails earlier, at build-and-publish's checkout, - # so this guards the narrower case where the ref resolved to something - # that is not the tag - the release would then point at a different - # commit than the images were built from. - gh release create "${TAG}" --title "${TAG}" --generate-notes --draft --verify-tag ${prerelease} - created_draft="true" - fi gh release upload "${TAG}" \ "_crd-package/kubefleet-crds-${TAG}.tgz" \ "_crd-package/kubefleet-crds-${TAG}.tgz.sha256" \ --clobber - # Only publish releases this job created; never flip a maintainer's existing release. - if [ "${created_draft}" = "true" ]; then - gh release edit "${TAG}" --draft=false - elif [ "$(gh release view "${TAG}" --json isDraft --jq .isDraft)" = "true" ]; then - echo "::warning::Release ${TAG} already existed as a draft; the CRD bundle was uploaded but the release was left unpublished. Publish it manually." - fi + + # Charts are published only for stable releases: an RC must be installable by + # testers from its images, but must never land in the public chart index that + # `helm repo update` resolves. Charts wait on publish-images because a chart + # whose appVersion points at images that do not exist yet is broken on arrival. + publish-charts-oci: + needs: [setup, publish-images] + if: ${{ needs.setup.outputs.prerelease == 'false' }} + runs-on: ubuntu-latest + permissions: + contents: read + packages: write + env: + REGISTRY: ${{ needs.setup.outputs.registry }} + TAG: ${{ needs.setup.outputs.tag }} + CHART_VERSION: ${{ needs.setup.outputs.version }} + steps: + - name: Checkout code + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: ${{ needs.setup.outputs.tag }} + + - name: Login to GitHub Container Registry + uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0 + with: + registry: ghcr.io + username: ${{ github.actor }} + password: ${{ secrets.GITHUB_TOKEN }} + + # Pin Helm rather than inheriting whatever the runner image ships, so the + # version that packages a release is the same one code-lint.yml lints the + # charts with. + - name: Set up Helm + uses: azure/setup-helm@9bc31f4ebc9c6b171d7bfbaa5d006ae7abdb4310 # v5 + with: + version: v3.17.0 + + - name: Package and push Helm charts to GHCR + run: | + set -euo pipefail + make helm-push REGISTRY="${REGISTRY}/charts" TAG="${TAG}" CHART_VERSION="${CHART_VERSION}" + + - name: Verify chart appVersion matches release tag + run: | + set -euo pipefail + rm -rf .helm-verify + mkdir -p .helm-verify + + for chart in hub-agent member-agent; do + helm pull "oci://${REGISTRY}/charts/${chart}" --version "${CHART_VERSION}" --destination .helm-verify >/dev/null + packaged=".helm-verify/${chart}-${CHART_VERSION}.tgz" + actual_app_version="$(tar -xOf "${packaged}" "${chart}/Chart.yaml" | awk -F': ' '/^appVersion:/ {gsub(/"/, "", $2); print $2}')" + if [ "${actual_app_version}" != "${TAG}" ]; then + echo "::error::${chart} appVersion (${actual_app_version}) does not match release tag (${TAG})" + exit 1 + fi + echo "✅ ${chart} appVersion=${actual_app_version} matches release tag=${TAG}" + done + + rm -rf .helm-verify + + publish-charts-pages: + needs: [setup, publish-images] + if: ${{ needs.setup.outputs.prerelease == 'false' }} + runs-on: ubuntu-latest + permissions: + contents: write + env: + TAG: ${{ needs.setup.outputs.tag }} + CHART_VERSION: ${{ needs.setup.outputs.version }} + # helm-gh-pages rewrites the whole gh-pages branch, so this job serializes + # across every release rather than per tag like the rest of the workflow. + # + # Only one run may be *pending* on a group by default, so a third overlapping + # release cancels the one already waiting and strands it as a draft until + # someone re-runs the job. `queue: max` is the real fix, but actionlint (this + # repo's lint gate, pinned at 1.7.12) rejects the key as unknown - support is + # merged upstream but unreleased. Until it ships, RELEASING.md documents the + # symptom and its one-click recovery. + concurrency: + group: helm-chart-publish-gh-pages + cancel-in-progress: false + steps: + # This job hands a contents:write token to a third-party Docker action + # whose base image is a floating tag (see the note in RELEASING.md and the + # follow-up issue). Auditing egress at least records what it reaches out + # to until that action is replaced. + - name: Harden Runner + uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1 + with: + egress-policy: audit + + - name: Checkout code + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: ${{ needs.setup.outputs.tag }} + fetch-depth: 0 + + # chart_version/app_version are what make the index carry the release + # being cut. Without them the action packages charts/*/Chart.yaml + # verbatim, and those are pinned at 0.1.0/v0.1.0 in-tree - so every + # release republished "hub-agent 0.1.0" pointing at image tag v0.1.0, + # overwriting the previous entry. The OCI path already overrides both + # (see make helm-push); this brings the index in line with it. + - name: Publish Helm chart to GitHub Pages + uses: stefanprodan/helm-gh-pages@0ad2bb377311d61ac04ad9eb6f252fb68e207260 # v1.7.0 + with: + token: ${{ secrets.GITHUB_TOKEN }} + charts_dir: charts + target_dir: charts + chart_version: ${{ needs.setup.outputs.version }} + app_version: ${{ needs.setup.outputs.tag }} + linting: on + + # Unknown action inputs are a warning, not an error, so a rename or typo + # in the two above would silently reinstate the 0.1.0 bug with a green + # build. Check the branch the action just wrote rather than trusting that + # it accepted them; the published index is not queryable until Pages + # redeploys, but the commit is there immediately. + - name: Verify the published index carries this release + run: | + set -euo pipefail + git fetch --depth=1 origin gh-pages + for chart in hub-agent member-agent; do + packaged="charts/${chart}-${CHART_VERSION}.tgz" + if ! git cat-file -e "FETCH_HEAD:${packaged}" 2>/dev/null; then + echo "::error::gh-pages has no ${packaged}; the chart index was not updated for this release." + exit 1 + fi + app_version="$(git cat-file blob "FETCH_HEAD:${packaged}" \ + | tar -xzO "${chart}/Chart.yaml" \ + | awk -F': ' '/^appVersion:/ {gsub(/"/, "", $2); print $2}')" + if [ "${app_version}" != "${TAG}" ]; then + echo "::error::gh-pages ${chart} appVersion (${app_version}) does not match release tag (${TAG})" + exit 1 + fi + echo "✅ gh-pages carries ${packaged} with appVersion=${app_version}" + done + + # The atomic commit point: the release becomes visible only after every + # producer that was supposed to run has succeeded. + # + # The condition has to override the implicit `success()` on `needs`, because + # the chart jobs are legitimately skipped for release candidates and a skipped + # dependency would otherwise skip this job too. It uses `!cancelled()` rather + # than `always()`: `always()` runs even when the workflow is cancelled, so a + # maintainer hitting Cancel after the producers had finished would still get a + # published release. + # + # Every producer is then required explicitly. `skipped` is only acceptable for + # the chart jobs, and only on a pre-release - otherwise anything that made + # their `if:` evaluate false would silently publish a stable release with no + # charts, which is the exact failure this workflow exists to prevent. + publish-release: + needs: + - setup + - create-draft-release + - publish-images + - publish-crds + - publish-charts-oci + - publish-charts-pages + if: >- + ${{ !cancelled() + && needs.publish-images.result == 'success' + && needs.publish-crds.result == 'success' + && (needs.setup.outputs.prerelease == 'true' + || (needs.publish-charts-oci.result == 'success' + && needs.publish-charts-pages.result == 'success')) }} + runs-on: ubuntu-latest + permissions: + contents: write + env: + TAG: ${{ needs.setup.outputs.tag }} + PRERELEASE: ${{ needs.setup.outputs.prerelease }} + GH_TOKEN: ${{ github.token }} + GH_REPO: ${{ github.repository }} + steps: + - name: Checkout code + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: ${{ needs.setup.outputs.tag }} + + - name: Verify the release assets, then publish + run: ./hack/release/publish-release.sh diff --git a/.github/workflows/setup-release.yml b/.github/workflows/setup-release.yml index 65cd8984b..3405510c4 100644 --- a/.github/workflows/setup-release.yml +++ b/.github/workflows/setup-release.yml @@ -17,6 +17,9 @@ on: version: description: "Release version without v prefix (e.g., 1.0.0)" value: ${{ jobs.export.outputs.version }} + prerelease: + description: "\"true\" when the tag is a pre-release (e.g., v1.0.0-rc.1), \"false\" otherwise" + value: ${{ jobs.export.outputs.prerelease }} env: REGISTRY: ghcr.io @@ -32,6 +35,7 @@ jobs: registry: ${{ steps.setup.outputs.registry }} tag: ${{ steps.setup.outputs.tag }} version: ${{ steps.setup.outputs.version }} + prerelease: ${{ steps.setup.outputs.prerelease }} steps: - id: setup # The tag arrives as an environment variable rather than being @@ -53,10 +57,18 @@ jobs: exit 1 fi + # The regex above admits exactly one pre-release form, so the + # suffix test is a complete classification. Callers gate on + # this single output instead of re-deriving "is this an RC?" + # in every consuming job. + PRERELEASE=false + case "${TAG}" in *-rc.*) PRERELEASE=true ;; esac + # registry must be in lowercase { echo "registry=$(echo "${{ env.REGISTRY }}/${{ github.repository }}" | tr '[:upper:]' '[:lower:]')" echo "tag=${TAG}" echo "version=${TAG#v}" + echo "prerelease=${PRERELEASE}" } >> "$GITHUB_OUTPUT" - echo "Release tag: ${TAG}, version: ${TAG#v}" + echo "Release tag: ${TAG}, version: ${TAG#v}, prerelease: ${PRERELEASE}" diff --git a/.github/workflows/workflow-lint.yml b/.github/workflows/workflow-lint.yml index 532818b2c..903630348 100644 --- a/.github/workflows/workflow-lint.yml +++ b/.github/workflows/workflow-lint.yml @@ -6,11 +6,13 @@ on: paths: - ".github/workflows/**" - ".github/release.yml" + - "hack/release/**" pull_request: branches: [main, "release-*"] paths: - ".github/workflows/**" - ".github/release.yml" + - "hack/release/**" permissions: contents: read @@ -43,3 +45,11 @@ jobs: - name: Run actionlint run: actionlint -color + + # The release scripts live outside the workflow files, so actionlint's + # embedded-shell checking does not reach them. + - name: Shellcheck the release scripts + run: shellcheck hack/release/*.sh hack/release/testdata/gh + + - name: Test the release scripts + run: ./hack/release/test-release-scripts.sh diff --git a/RELEASING.md b/RELEASING.md new file mode 100644 index 000000000..5bab9bb4f --- /dev/null +++ b/RELEASING.md @@ -0,0 +1,204 @@ +# Releasing + +This is the operational runbook for cutting a KubeFleet release and for +recovering when a release run fails partway through. It covers *how* a release +is produced; what the version numbers mean and how long each release is +supported are covered in [VERSIONING.md](VERSIONING.md) and +[SECURITY.md](SECURITY.md). + +## What a release publishes + +| Artifact | Location | Stable (`v0.4.0`) | Release candidate (`v0.4.0-rc.1`) | +| --- | --- | --- | --- | +| Agent images (`hub-agent`, `member-agent`, `refresh-token`) | `ghcr.io/kubefleet-dev/kubefleet/` | `:v0.4.0` and `:0.4.0` | `:v0.4.0-rc.1` only | +| CRD bundle (`kubefleet-crds-.tgz` + `.sha256`) | GitHub Release asset | Yes | Yes | +| Helm charts (OCI) | `oci://ghcr.io/kubefleet-dev/kubefleet/charts/` | Yes | No | +| Helm charts (index) | `https://kubefleet-dev.github.io/kubefleet/charts` | Yes | No | +| GitHub Release | Releases page | Published | Published, flagged pre-release | + +Release candidates deliberately get no short image alias and never enter the +public chart index: the short-tag namespace and the `helm repo` index are +reserved for releases users can safely pin to. Testers install an RC from its +full tag. + +## Cutting a release + +All images and charts are built from the tag itself, so everything that ships +must be merged before the tag is pushed. + +1. Confirm `main` (or the `release-0.Y` branch) is green and carries every + change intended for the release, including backports — see + [CONTRIBUTING.md](CONTRIBUTING.md#backporting-to-release-branches). +2. Tag and push. The tag must match `vMAJOR.MINOR.PATCH` or + `vMAJOR.MINOR.PATCH-rc.N`; any other shape is rejected before anything is + published. + + ```bash + git tag -a v0.4.0-rc.1 -m "v0.4.0-rc.1" + git push upstream v0.4.0-rc.1 + ``` + +3. Watch the `Release` workflow. It publishes the GitHub Release only after + every artifact has been produced. +4. For a stable release, repeat with the final tag (for example `v0.4.0`) once + the RC has soaked. + +A release can also be started from the Actions tab via **Run workflow** on the +`Release` workflow, passing the tag as an input. Two preconditions apply: + +- The tag must already exist. The workflow passes `--verify-tag`, so a dispatch + naming a tag that was never pushed fails instead of inventing one at the head + of the default branch. +- The tag must be one cut after this workflow landed. A dispatch runs the + workflow definition from the selected branch but checks out the *tag*, and the + jobs call scripts under `hack/release/`; against an older tag that predates + them, the first job fails immediately with "No such file or directory". + Nothing is published when it does. Tags on a `release-0.Y` branch that predates + this workflow are unaffected — they carry their own contemporary workflow. + +## The release pipeline + +`.github/workflows/release.yml` is the single owner of a release. Its job graph: + +```text +setup validate the tag; derive registry, version, prerelease + └── create-draft-release create (or reuse) the GitHub Release as a draft + ├── publish-images multi-arch buildx push, then verify both platforms + │ ├── publish-charts-oci stable only; helm push + appVersion check + │ └── publish-charts-pages stable only; rewrite the gh-pages index + └── publish-crds package the CRDs, upload the bundle to the draft + +publish-release needs ALL of the jobs above; verifies the release's + assets, then flips the draft to published +``` + +Two properties matter when something goes wrong: + +- **The GitHub Release is a draft until the very end.** No release page, release + notes, or release asset is visible until every producer has succeeded. Note + the scope: this covers the *release*, not the registry. Images and charts are + publicly pullable the moment their own job succeeds, so a run that fails after + `publish-images` has left `ghcr.io/.../hub-agent:v0.4.0` reachable even though + no release mentions it. +- **Charts wait for images.** A chart whose `appVersion` points at images that + do not exist yet is broken on arrival, so the chart jobs run only after the + images are pushed and verified. + +The two jobs with real branching logic — `create-draft-release` and +`publish-release` — live in [`hack/release/`](hack/release) rather than inline +in the workflow, and are covered by `hack/release/test-release-scripts.sh`, +which CI runs on every change to either. + +## Recovering from a failed run + +The normal recovery is **Re-run failed jobs** on the workflow run. Jobs that +already succeeded are not re-run, and every job is safe to repeat: the CRD +upload uses `--clobber`, `create-draft-release` reuses the draft it created the +first time, and image pushes rewrite the same tags. + +Re-running `publish-images` rebuilds from source rather than reproducing the +earlier build byte-for-byte, so the tag ends up pointing at a *new* digest. That +is harmless while the release is still a draft — nothing has been announced yet +— but it is why a re-run is not an option once a release has been published. + +| Where it failed | What is already public | What to do | +| --- | --- | --- | +| `setup` | Nothing | The tag is malformed. Delete it, fix, re-tag. | +| `create-draft-release` | Nothing | See [Re-releasing an existing tag](#re-releasing-an-existing-tag) if it refused because the release is already published. | +| `publish-images` | Any images pushed before the failure (`make push` builds hub-agent, member-agent, then refresh-token in order) | Fix, then re-run failed jobs. | +| `publish-crds` | Possibly the images — it runs in parallel with `publish-images`, not after it | Fix, then re-run failed jobs. | +| `publish-charts-oci` / `publish-charts-pages` | Images; CRD bundle is attached to the still-hidden draft | Fix, then re-run failed jobs. The release stays a draft until the charts land. | +| `publish-release` | Images, charts | The asset check found the draft incomplete or its bundle failed its own checksum. Inspect `gh release view `, re-upload, re-run failed jobs. | + +`publish-charts-pages` serializes across *all* releases, because the action it +uses rewrites the whole `gh-pages` branch. GitHub keeps at most one pending +entry per concurrency group, so if three stable releases overlap, the middle +one's pages job is **cancelled** rather than queued. That leaves its release as +a draft with everything else done; **Re-run failed jobs** finishes it. Cutting +stable releases one at a time avoids the situation entirely. + +If the fix requires a code change, the tag must move or be replaced — see +below. Do **not** rebuild a different commit under a tag that already pushed +images. + +### Re-releasing an existing tag + +`create-draft-release` refuses to run against a release that is already +published. This is deliberate: consumers may already have pinned the images and +charts that release advertises, and a second run would replace them in place +with a different build. + +- **If the release should not have gone out** (wrong commit, broken build): + delete the GitHub Release and the tag, then cut the *next* tag rather than + reusing the old one — `-rc.N+1` for a release candidate, or the next patch + version for a stable release. Container tags that have been pulled are not + safely reusable, and the CRD bundle checksum users recorded would change under + them. Because the bad images stay pullable under their original tag (see + [Abandoning a release](#abandoning-a-release)), also delete those package + versions if the build was actually broken rather than merely superseded. +- **If only one artifact is missing** (for example a chart publish that was + fixed after the release went public): publish that artifact manually rather + than re-running the whole workflow. Note that the OCI registry and the Pages + index are published by two different jobs and need two different fixes: + + ```bash + # OCI charts + make helm-push REGISTRY=ghcr.io/kubefleet-dev/kubefleet/charts \ + TAG=v0.4.0 CHART_VERSION=0.4.0 + + # Pages index: re-run the publish-charts-pages job from the workflow run, + # which is the only thing that rewrites the gh-pages branch. + ``` + +### Abandoning a release + +If a release is called off, delete the draft and the tag so the next attempt +starts clean: + +```bash +gh release delete v0.4.0-rc.1 --yes +git push upstream :refs/tags/v0.4.0-rc.1 +git tag -d v0.4.0-rc.1 +``` + +That removes the release and the tag, but **not** the artifacts the producer +jobs already published. Those outlive the release and have to be cleaned up +deliberately: + +- **Images.** `ghcr.io/kubefleet-dev/kubefleet/:v0.4.0` — and the short + alias `:0.4.0` for a stable tag — stay publicly pullable. The next tag is a + different version, so it never supersedes them. Delete the package versions + (`gh api --method DELETE /orgs/kubefleet-dev/packages/container//versions/`) + if the build was broken rather than merely renumbered. +- **Charts.** If `publish-charts-oci` ran, the chart is in the OCI registry and + needs the same treatment. If `publish-charts-pages` ran, the `gh-pages` index + already advertises the abandoned version, and the next stable release will not + remove it — the entry has to be dropped from `charts/index.yaml` on the + `gh-pages` branch by hand. + +Abandoning a *stable* release after the chart jobs have run is therefore not +cleanly reversible. Soak on release candidates, which publish neither chart. + +## After a release + +- At the first RC of a new minor, cut the matching `release-0.Y` branch. From + then on, fixes land on `main` and are backported with the `cherry-pick/0.Y` + labels described in + [CONTRIBUTING.md](CONTRIBUTING.md#backporting-to-release-branches). +- Verify the published release page lists the CRD bundle and its checksum, and + that the generated notes look right — they come from the `release-note/*` + labels on the PRs in the release. +- **One-time, at the first stable release cut by this workflow:** the `gh-pages` + chart index carries stale `hub-agent 0.1.0` and `member-agent 0.1.0` entries + from before the index was given the real release version. The publish step + merges into the existing index rather than replacing it, so those entries + survive and `helm search repo kubefleet --versions` keeps offering `0.1.0`. + Delete them from `charts/index.yaml` on the `gh-pages` branch (and the + matching `charts/*-0.1.0.tgz`) once a correctly-versioned entry exists. + +## See also + +- [VERSIONING.md](VERSIONING.md) — versioning scheme, agent skew, upgrade order. +- [SECURITY.md](SECURITY.md) — supported versions and security-patch policy. +- [CONTRIBUTING.md](CONTRIBUTING.md) — PR conventions, release-note labels, and + backport policy. diff --git a/VERSIONING.md b/VERSIONING.md index c6e6d3dfa..16a4d253f 100644 --- a/VERSIONING.md +++ b/VERSIONING.md @@ -146,6 +146,8 @@ installs need no separate step: KubeFleet ships its CRDs under ## See also +- [RELEASING.md](RELEASING.md) — how a release is cut and how to recover a + failed release run. - [SECURITY.md](SECURITY.md) — supported versions and security-patch policy. - [CONTRIBUTING.md](CONTRIBUTING.md) — PR conventions and release-note labels. - [Kubernetes version skew policy](https://kubernetes.io/releases/version-skew-policy/) diff --git a/apis/cluster/v1beta1/zz_generated.deepcopy.go b/apis/cluster/v1beta1/zz_generated.deepcopy.go index cec52aa39..a51641c3f 100644 --- a/apis/cluster/v1beta1/zz_generated.deepcopy.go +++ b/apis/cluster/v1beta1/zz_generated.deepcopy.go @@ -21,7 +21,7 @@ limitations under the License. package v1beta1 import ( - v1 "k8s.io/api/core/v1" + "k8s.io/api/core/v1" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" runtime "k8s.io/apimachinery/pkg/runtime" ) diff --git a/apis/kubefleet.dev/placement/v1alpha1/common.go b/apis/kubefleet.dev/placement/v1alpha1/common.go index e1b991dc1..9ccad8432 100644 --- a/apis/kubefleet.dev/placement/v1alpha1/common.go +++ b/apis/kubefleet.dev/placement/v1alpha1/common.go @@ -16,6 +16,18 @@ limitations under the License. package v1alpha1 +const ( + // The Kinds of API resource types in this package. + ClusterClaimKind = "ClusterClaim" + PlacementPolicyKind = "PlacementPolicy" + ClusterPlacementPolicyKind = "ClusterPlacementPolicy" + PlacementBindingKind = "PlacementBinding" + ClusterPlacementBindingKind = "ClusterPlacementBinding" + PlacementResourceSnapshotKind = "PlacementResourceSnapshot" + ClusterPlacementResourceSnapshotKind = "ClusterPlacementResourceSnapshot" + WorkKind = "Work" +) + type ObjectReference struct { // The namespace of the referenced object. // diff --git a/apis/kubefleet.dev/placement/v1alpha1/placementbinding_types.go b/apis/kubefleet.dev/placement/v1alpha1/placementbinding_types.go index 39e2c21b8..e3976ceef 100644 --- a/apis/kubefleet.dev/placement/v1alpha1/placementbinding_types.go +++ b/apis/kubefleet.dev/placement/v1alpha1/placementbinding_types.go @@ -30,9 +30,11 @@ const ( const ( PlacementBindingSynchronizedCondReasonAllResourcesSynchronized = "AllResourcesSynchronized" PlacementBindingSynchronizedCondReasonFailedToSynchronizeSomeResources = "FailedToSynchronizeSomeResources" + PlacementBindingSynchronizedCondReasonWaitingForSynchronization = "WaitingForSynchronization" - PlacementBindingAvailableCondReasonAllResourcesAvailable = "AllResourcesAvailable" - PlacementBindingAvailableCondReasonSomeResourcesUnavailable = "SomeResourcesUnavailable" + PlacementBindingAvailableCondReasonAllResourcesAvailable = "AllResourcesAvailable" + PlacementBindingAvailableCondReasonSomeResourcesUnavailable = "SomeResourcesUnavailable" + PlacementBindingAvailableCondReasonWaitingForAvailabilityCheck = "WaitingForAvailabilityCheck" ) // PlacementBinding is the KubeFleet API that binds the resources selected by a placement @@ -163,6 +165,13 @@ type PlacementBindingStatus struct { // +kubebuilder:validation:Optional // +kubebuilder:validation:MaxItems=50 FailedResources []FailedResource `json:"failedResources,omitempty"` + + // The name of the placement resource snapshot that KubeFleet has last processed for this binding. + // This field helps KubeFleet track the processing progress; it also reveals whether the reported status + // is up to date. + // + // +kubebuilder:validation:Optional + LastProcessedResourceSnapshotName *string `json:"lastProcessedResourceSnapshotName,omitempty"` } type FailedResource struct { diff --git a/apis/kubefleet.dev/placement/v1alpha1/placementpolicy_types.go b/apis/kubefleet.dev/placement/v1alpha1/placementpolicy_types.go index f29abb1cf..2d54fc82f 100644 --- a/apis/kubefleet.dev/placement/v1alpha1/placementpolicy_types.go +++ b/apis/kubefleet.dev/placement/v1alpha1/placementpolicy_types.go @@ -564,8 +564,9 @@ type BindingManager struct { // A list of references to the objects that are currently managing the bindings for this placement, // under the reconciliation of the specified controller. // - // +kubebuilder:validation:Optional - ObjectRefs []ObjectReference `json:"objectRefs,omitempty"` + // +kubebuilder:validation:Required + // +kubebuilder:validation:MinItems=1 + ObjectRefs []ObjectReference `json:"objectRefs"` } // The list objects for the PlacementPolicy and ClusterPlacementPolicy APIs. diff --git a/apis/kubefleet.dev/placement/v1alpha1/placementresourcesnapshot_types.go b/apis/kubefleet.dev/placement/v1alpha1/placementresourcesnapshot_types.go index 0f2d284d7..f98db8ed8 100644 --- a/apis/kubefleet.dev/placement/v1alpha1/placementresourcesnapshot_types.go +++ b/apis/kubefleet.dev/placement/v1alpha1/placementresourcesnapshot_types.go @@ -21,6 +21,51 @@ import ( runtime "k8s.io/apimachinery/pkg/runtime" ) +const ( + // When users create a placement policy to place resources across member clusters, KubeFleet will capture the + // resources selected by the placement policy at a specific point in time in the form of placement resource + // snapshots. This enables KubeFleet to roll out resources to member clusters in a consistent manner. + // + // As resources change over time, there might be a time series of placement resource snapshots associated + // with a placement policy. KubeFleet assigns these snapshots with a monotonically increasing index based on + // their creation timestamp, starting from 0 with a step of 1. + // + // Due to sizing limitations in Kubernetes, when there are too many resources being selected at a time, or + // when some resources are too large, KubeFleet will capture them using multiple placement resource snapshots. + // These snapshots share the same index (as they are snapshots from the same point in time), and KubeFleet + // will further assign them each with a sub-index to tell them apart, also starting from 0 with a step of 1. + // The snapshot of the sub-index 0 is considered the primary snapshot of the same index. + + // PlacementResourceSnapshotOwnedByLabelKey is a label key that denotes the owner placement policy of + // a placement resource snapshot. Its value is the name of the owner placement policy. KubeFleet + // might truncate the name and add a hash suffix as needed. + // + // This label is set on all placement resource snapshots. + PlacementResourceSnapshotOwnedByLabelKey = "placement.kubefleet.dev/placement-resource-snapshot-owned-by" + // PlacementResourceSnapshotIndexLabelKey is a label key that denotes the index of a placement resource snapshot. + // Its value is the index integer formatted as a string. + // + // This label is set on all placement resource snapshots. + PlacementResourceSnapshotIndexLabelKey = "placement.kubefleet.dev/placement-resource-snapshot-index" + // PlacementResourceSnapshotSubIndexLabelKey is a label key that denotes the sub-index of a placement resource snapshot. + // Its value is the sub-index integer formatted as a string. + // + // This label is set on all placement resource snapshots. + PlacementResourceSnapshotSubIndexLabelKey = "placement.kubefleet.dev/placement-resource-snapshot-sub-index" + // SubIndexedPlacementResourceSnapshotCountLabelKey is a label key that denotes the total number of sub-indexed + // placement resource snapshots associated with the same index. Its value is the count integer + // formatted as a string. + // + // This label is set only on resource placement snapshots with the sub-index of 0. + SubIndexedPlacementResourceSnapshotCountLabelKey = "placement.kubefleet.dev/sub-indexed-placement-resource-snapshot-count" + + // PlacementResourceSnapshotContentsHashAnnotationKey is an annotation key that denotes the hash of the contents + // of a placement resource snapshot. Its value is the hash string. + // + // This annotation is set on all placement resource snapshots. + PlacementResourceSnapshotContentsHashAnnotationKey = "placement.kubefleet.dev/placement-resource-snapshot-contents-hash" +) + // PlacementResourceSnapshot is the KubeFleet API that captures the resources selected by a placement policy // as seen on the hub cluster at a specific point in time. It is referenced by other KubeFleet APIs // to enable consistent rollouts of resources across multiple member clusters in the fleet. diff --git a/apis/kubefleet.dev/placement/v1alpha1/work_types.go b/apis/kubefleet.dev/placement/v1alpha1/work_types.go index f282ab694..eb78ba88c 100644 --- a/apis/kubefleet.dev/placement/v1alpha1/work_types.go +++ b/apis/kubefleet.dev/placement/v1alpha1/work_types.go @@ -21,6 +21,39 @@ import ( "k8s.io/apimachinery/pkg/runtime" ) +const ( + // The label key that denotes the namespace of a work object's owner placement policy and placement binding + // objects. For cluster-scoped owners, the label has an empty value. + WorkOwnerNamespaceLabelKey = "placement.kubefleet.dev/owner-namespace" + // The label key that denotes the name of a work object's owner placement policy object. + // + // KubeFleet might truncate the name value and append a hash to satisfy Kubernetes' label value length limit + // (63 characters). + WorkOwnedByPlacementPolicyLabelKey = "placement.kubefleet.dev/owned-by-placement-policy" + // The label key that denotes the name of a work object's owner placement binding object. + // + // KubeFleet might truncate the name value and append a hash to satisfy Kubernetes' label value length limit + // (63 characters). + WorkOwnedByPlacementBindingLabelKey = "placement.kubefleet.dev/owned-by-placement-binding" + + // The annotation key that denotes the name of a work object's owner placement policy object. + // + // The name will saved as the annotation value as it is. + WorkOwnedByPlacementPolicyAnnotationKey = "placement.kubefleet.dev/owned-by-placement-policy" + // The annotation key that denotes the name of a work object's owner placement binding object. + // + // The name will saved as the annotation value as it is. + WorkOwnedByPlacementBindingAnnotationKey = "placement.kubefleet.dev/owned-by-placement-binding" + // The annotation key that denotes the name of a work object's primary placement resource snapshot. + // + // The name will saved as the annotation value as it is. + WorkLinkedToPrimaryPlacementResourceSnapshotAnnotationKey = "placement.kubefleet.dev/linked-to-primary-placement-resource-snapshot" + // The annotation key that denotes the number of linked work objects. + LinkedWorkCountAnnotationKey = "placement.kubefleet.dev/linked-work-count" + // The annotation key that denotes the source from which the work object is derived. + WorkDerivedFromSourceAnnotationKey = "placement.kubefleet.dev/derived-from" +) + const ( // The condition types for the Work API. WorkCondTypeApplied = "Applied" diff --git a/apis/kubefleet.dev/placement/v1alpha1/zz_generated.deepcopy.go b/apis/kubefleet.dev/placement/v1alpha1/zz_generated.deepcopy.go index b483d46f4..8f5a15a5d 100644 --- a/apis/kubefleet.dev/placement/v1alpha1/zz_generated.deepcopy.go +++ b/apis/kubefleet.dev/placement/v1alpha1/zz_generated.deepcopy.go @@ -21,7 +21,7 @@ limitations under the License. package v1alpha1 import ( - v1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/runtime" "k8s.io/apimachinery/pkg/util/intstr" ) @@ -710,6 +710,11 @@ func (in *PlacementBindingStatus) DeepCopyInto(out *PlacementBindingStatus) { (*in)[i].DeepCopyInto(&(*out)[i]) } } + if in.LastProcessedResourceSnapshotName != nil { + in, out := &in.LastProcessedResourceSnapshotName, &out.LastProcessedResourceSnapshotName + *out = new(string) + **out = **in + } } // DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new PlacementBindingStatus. diff --git a/apis/placement/v1/clusterresourceplacement_types.go b/apis/placement/v1/clusterresourceplacement_types.go index e4fb3a2f3..b8620ae9a 100644 --- a/apis/placement/v1/clusterresourceplacement_types.go +++ b/apis/placement/v1/clusterresourceplacement_types.go @@ -969,7 +969,8 @@ type RollingUpdateConfig struct { // Defaults to 25%. // +kubebuilder:default="25%" // +kubebuilder:validation:XIntOrString - // +kubebuilder:validation:Pattern="^((100|[0-9]{1,2})%|[0-9]+)$" + // +kubebuilder:validation:Pattern="^((100|[0-9]{1,2})%|[0-9]{1,9})$" + // +kubebuilder:validation:XValidation:rule="type(self) == int ? self >= 0 : true",message="maxUnavailable must be a non-negative integer or a percentage" // +kubebuilder:validation:Optional MaxUnavailable *intstr.IntOrString `json:"maxUnavailable,omitempty"` @@ -983,7 +984,8 @@ type RollingUpdateConfig struct { // Defaults to 25%. // +kubebuilder:default="25%" // +kubebuilder:validation:XIntOrString - // +kubebuilder:validation:Pattern="^((100|[0-9]{1,2})%|[0-9]+)$" + // +kubebuilder:validation:Pattern="^((100|[0-9]{1,2})%|[0-9]{1,9})$" + // +kubebuilder:validation:XValidation:rule="type(self) == int ? self >= 0 : true",message="maxSurge must be a non-negative integer or a percentage" // +kubebuilder:validation:Optional MaxSurge *intstr.IntOrString `json:"maxSurge,omitempty"` diff --git a/apis/placement/v1/disruptionbudget_types.go b/apis/placement/v1/disruptionbudget_types.go new file mode 100644 index 000000000..4e482b2b1 --- /dev/null +++ b/apis/placement/v1/disruptionbudget_types.go @@ -0,0 +1,121 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package v1 + +import ( + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/util/intstr" +) + +// +kubebuilder:object:root=true +// +kubebuilder:resource:scope=Cluster,categories={fleet,fleet-placement},shortName=crpdb + +// ClusterResourcePlacementDisruptionBudget is the policy applied to a ClusterResourcePlacement +// object that specifies its disruption budget, i.e., how many placements (clusters) can be +// down at the same time due to voluntary disruptions (e.g., evictions). Involuntary +// disruptions are not subject to this budget, but will still count against it. +// +// To apply a ClusterResourcePlacementDisruptionBudget to a ClusterResourcePlacement, use the +// same name for the ClusterResourcePlacementDisruptionBudget object as the ClusterResourcePlacement +// object. This guarantees a 1:1 link between the two objects. +type ClusterResourcePlacementDisruptionBudget struct { + metav1.TypeMeta `json:",inline"` + metav1.ObjectMeta `json:"metadata,omitempty"` + + // Spec is the desired state of the ClusterResourcePlacementDisruptionBudget. + // +kubebuilder:validation:XValidation:rule="!(has(self.maxUnavailable) && has(self.minAvailable))",message="Both MaxUnavailable and MinAvailable cannot be specified" + // +required + Spec PlacementDisruptionBudgetSpec `json:"spec"` +} + +// PlacementDisruptionBudgetSpec is the desired state of the PlacementDisruptionBudget. +type PlacementDisruptionBudgetSpec struct { + // MaxUnavailable is the maximum number of placements (clusters) that can be down at the + // same time due to voluntary disruptions. For example, a setting of 1 would imply that + // a voluntary disruption (e.g., an eviction) can only happen if all placements (clusters) + // from the linked Placement object are applied and available. + // + // This can be either an absolute value (e.g., 1) or a percentage (e.g., 10%). + // + // If a percentage is specified, Fleet will calculate the corresponding absolute values + // as follows: + // * if the linked Placement object is of the PickFixed placement type, + // we don't perform any calculation because eviction is not allowed for PickFixed CRP. + // * if the linked Placement object is of the PickAll placement type, MaxUnavailable cannot + // be specified since we cannot derive the total number of clusters selected. + // * if the linked Placement object is of the PickN placement type, + // the percentage is against the number of clusters specified in the placement (i.e., the + // value of the NumberOfClusters fields in the placement policy). + // The end result will be rounded up to the nearest integer if applicable. + // + // One may use a value of 0 for this field; in this case, no voluntary disruption would be + // allowed. + // + // This field is mutually exclusive with the MinAvailable field in the spec; set only one of them + // at a time. If none is set, no disruption is allowed for the target placement. + // + // +kubebuilder:validation:XIntOrString + // +kubebuilder:validation:XValidation:rule="type(self) == string ? self.matches('^(100|[0-9]{1,2})%$') : self >= 0",message="If supplied value is String should match regex '^(100|[0-9]{1,2})%$' or If supplied value is Integer must be greater than or equal to 0" + // +optional + MaxUnavailable *intstr.IntOrString `json:"maxUnavailable,omitempty"` + + // MinAvailable is the minimum number of placements (clusters) that must be available at any + // time despite voluntary disruptions. For example, a setting of 10 would imply that + // a voluntary disruption (e.g., an eviction) can only happen if there are at least 11 + // placements (clusters) from the linked Placement object are applied and available. + // + // This can be either an absolute value (e.g., 1) or a percentage (e.g., 10%). + // + // If a percentage is specified, Fleet will calculate the corresponding absolute values + // as follows: + // * if the linked Placement object is of the PickFixed placement type, + // we don't perform any calculation because eviction is not allowed for PickFixed CRP. + // * if the linked Placement object is of the PickAll placement type, MinAvailable can be + // specified but only as an integer since we cannot derive the total number of clusters selected. + // * if the linked Placement object is of the PickN placement type, + // the percentage is against the number of clusters specified in the placement (i.e., the + // value of the NumberOfClusters fields in the placement policy). + // The end result will be rounded up to the nearest integer if applicable. + // + // One may use a value of 0 for this field; in this case, voluntary disruption would be + // allowed at any time. + // + // This field is mutually exclusive with the MaxUnavailable field in the spec; set only one of them + // at a time. If none is set, no disruption is allowed for the target placement. + // + // +kubebuilder:validation:XIntOrString + // +kubebuilder:validation:XValidation:rule="type(self) == string ? self.matches('^(100|[0-9]{1,2})%$') : self >= 0",message="If supplied value is String should match regex '^(100|[0-9]{1,2})%$' or If supplied value is Integer must be greater than or equal to 0" + // +optional + MinAvailable *intstr.IntOrString `json:"minAvailable,omitempty"` +} + +// ClusterResourcePlacementDisruptionBudgetList contains a list of ClusterResourcePlacementDisruptionBudget objects. +// +kubebuilder:resource:scope=Cluster +// +k8s:deepcopy-gen:interfaces=k8s.io/apimachinery/pkg/runtime.Object +type ClusterResourcePlacementDisruptionBudgetList struct { + metav1.TypeMeta `json:",inline"` + metav1.ListMeta `json:"metadata,omitempty"` + + // Items is the list of PlacementDisruptionBudget objects. + Items []ClusterResourcePlacementDisruptionBudget `json:"items"` +} + +func init() { + SchemeBuilder.Register( + &ClusterResourcePlacementDisruptionBudget{}, + &ClusterResourcePlacementDisruptionBudgetList{}) +} diff --git a/apis/placement/v1/eviction_types.go b/apis/placement/v1/eviction_types.go new file mode 100644 index 000000000..edaf9ad2d --- /dev/null +++ b/apis/placement/v1/eviction_types.go @@ -0,0 +1,154 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package v1 + +import ( + "k8s.io/apimachinery/pkg/api/meta" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" +) + +// +kubebuilder:object:root=true +// +kubebuilder:resource:scope=Cluster,categories={fleet,fleet-placement},shortName=crpe +// +kubebuilder:subresource:status +// +kubebuilder:printcolumn:JSONPath=`.status.conditions[?(@.type=="Valid")].status`,name="Valid",type=string +// +kubebuilder:printcolumn:JSONPath=`.status.conditions[?(@.type=="Executed")].status`,name="Executed",type=string + +// ClusterResourcePlacementEviction is an eviction attempt on a specific placement from +// a ClusterResourcePlacement object; one may use this API to force the removal of specific +// resources from a cluster. +// +// An eviction is a voluntary disruption; its execution is subject to the disruption budget +// linked with the target ClusterResourcePlacement object (if present). +// +// Beware that an eviction alone does not guarantee that a placement will not re-appear; i.e., +// after an eviction, the Fleet scheduler might still pick the previous target cluster for +// placement. To prevent this, consider adding proper taints to the target cluster before running +// an eviction that will exclude it from future placements; this is especially true in scenarios +// where one would like to perform a cluster replacement. +// +// For safety reasons, Fleet will only execute an eviction once; the spec in this object is immutable, +// and once executed, the object will be ignored after. To trigger another eviction attempt on the +// same placement from the same ClusterResourcePlacement object, one must re-create (delete and +// create) the same Eviction object. Note also that an Eviction object will be +// ignored once it is deemed invalid (e.g., such an object might be targeting a CRP object or +// a placement that does not exist yet), even if it does become valid later +// (e.g., the CRP object or the placement appears later). To fix the situation, re-create the +// Eviction object. +// +// Note: Eviction of resources from a cluster propagated by a PickFixed CRP is not allowed. +// If the user wants to remove resources from a cluster propagated by a PickFixed CRP simply +// remove the cluster name from cluster names field from the CRP spec. +// +// Executed evictions might be kept around for a while for auditing purposes; the Fleet controllers might +// have a TTL set up for such objects and will garbage collect them automatically. For further +// information, see the Fleet documentation. +type ClusterResourcePlacementEviction struct { + metav1.TypeMeta `json:",inline"` + metav1.ObjectMeta `json:"metadata,omitempty"` + + // Spec is the desired state of the ClusterResourcePlacementEviction. + // + // Note that all fields in the spec are immutable. + // +required + Spec PlacementEvictionSpec `json:"spec"` + + // Status is the observed state of the ClusterResourcePlacementEviction. + // +optional + Status PlacementEvictionStatus `json:"status,omitempty"` +} + +// PlacementEvictionSpec is the desired state of the parent PlacementEviction. +type PlacementEvictionSpec struct { + // PlacementName is the name of the Placement object which + // the Eviction object targets. + // +kubebuilder:validation:Required + // +kubebuilder:validation:XValidation:rule="self == oldSelf",message="The PlacementName field is immutable" + // +kubebuilder:validation:MaxLength=255 + PlacementName string `json:"placementName"` + + // ClusterName is the name of the cluster that the Eviction object targets. + // +kubebuilder:validation:Required + // +kubebuilder:validation:XValidation:rule="self == oldSelf",message="The ClusterName field is immutable" + // +kubebuilder:validation:MaxLength=255 + ClusterName string `json:"clusterName"` +} + +// PlacementEvictionStatus is the observed state of the parent PlacementEviction. +type PlacementEvictionStatus struct { + // Conditions is the list of currently observed conditions for the + // PlacementEviction object. + // + // Available condition types include: + // * Valid: whether the Eviction object is valid, i.e., it targets at a valid placement. + // * Executed: whether the Eviction object has been executed. + // +optional + Conditions []metav1.Condition `json:"conditions,omitempty"` +} + +// PlacementEvictionConditionType identifies a specific condition of the +// PlacementEviction. +type PlacementEvictionConditionType string + +const ( + // PlacementEvictionConditionTypeValid indicates whether the Eviction object is valid. + // + // The following values are possible: + // * True: the Eviction object is valid. + // * False: the Eviction object is invalid; it might be targeting a CRP object or a placement + // that does not exist yet. + // Note that this is a terminal state; once an Eviction object is deemed invalid, it will + // not be evaluated again, even if the target appears later. + PlacementEvictionConditionTypeValid PlacementEvictionConditionType = "Valid" + + // PlacementEvictionConditionTypeExecuted indicates whether the Eviction object has been executed. + // + // The following values are possible: + // * True: the Eviction object has been executed. + // Note that this is a terminal state; once an Eviction object is executed, it will not be + // executed again. + // * False: the Eviction object has not been executed yet. + PlacementEvictionConditionTypeExecuted PlacementEvictionConditionType = "Executed" +) + +// ClusterResourcePlacementEvictionList contains a list of ClusterResourcePlacementEviction objects. +// +kubebuilder:resource:scope=Cluster +// +k8s:deepcopy-gen:interfaces=k8s.io/apimachinery/pkg/runtime.Object +type ClusterResourcePlacementEvictionList struct { + metav1.TypeMeta `json:",inline"` + metav1.ListMeta `json:"metadata,omitempty"` + + // Items is the list of ClusterResourcePlacementEviction objects. + Items []ClusterResourcePlacementEviction `json:"items"` +} + +// SetConditions set the given conditions on the ClusterResourcePlacementEviction. +func (e *ClusterResourcePlacementEviction) SetConditions(conditions ...metav1.Condition) { + for _, c := range conditions { + meta.SetStatusCondition(&e.Status.Conditions, c) + } +} + +// GetCondition returns the condition of the given ClusterResourcePlacementEviction. +func (e *ClusterResourcePlacementEviction) GetCondition(conditionType string) *metav1.Condition { + return meta.FindStatusCondition(e.Status.Conditions, conditionType) +} + +func init() { + SchemeBuilder.Register( + &ClusterResourcePlacementEviction{}, + &ClusterResourcePlacementEvictionList{}) +} diff --git a/apis/placement/v1/zz_generated.deepcopy.go b/apis/placement/v1/zz_generated.deepcopy.go index 6d83557c5..27ba690bb 100644 --- a/apis/placement/v1/zz_generated.deepcopy.go +++ b/apis/placement/v1/zz_generated.deepcopy.go @@ -633,6 +633,123 @@ func (in *ClusterResourcePlacement) DeepCopyObject() runtime.Object { return nil } +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *ClusterResourcePlacementDisruptionBudget) DeepCopyInto(out *ClusterResourcePlacementDisruptionBudget) { + *out = *in + out.TypeMeta = in.TypeMeta + in.ObjectMeta.DeepCopyInto(&out.ObjectMeta) + in.Spec.DeepCopyInto(&out.Spec) +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new ClusterResourcePlacementDisruptionBudget. +func (in *ClusterResourcePlacementDisruptionBudget) DeepCopy() *ClusterResourcePlacementDisruptionBudget { + if in == nil { + return nil + } + out := new(ClusterResourcePlacementDisruptionBudget) + in.DeepCopyInto(out) + return out +} + +// DeepCopyObject is an autogenerated deepcopy function, copying the receiver, creating a new runtime.Object. +func (in *ClusterResourcePlacementDisruptionBudget) DeepCopyObject() runtime.Object { + if c := in.DeepCopy(); c != nil { + return c + } + return nil +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *ClusterResourcePlacementDisruptionBudgetList) DeepCopyInto(out *ClusterResourcePlacementDisruptionBudgetList) { + *out = *in + out.TypeMeta = in.TypeMeta + in.ListMeta.DeepCopyInto(&out.ListMeta) + if in.Items != nil { + in, out := &in.Items, &out.Items + *out = make([]ClusterResourcePlacementDisruptionBudget, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new ClusterResourcePlacementDisruptionBudgetList. +func (in *ClusterResourcePlacementDisruptionBudgetList) DeepCopy() *ClusterResourcePlacementDisruptionBudgetList { + if in == nil { + return nil + } + out := new(ClusterResourcePlacementDisruptionBudgetList) + in.DeepCopyInto(out) + return out +} + +// DeepCopyObject is an autogenerated deepcopy function, copying the receiver, creating a new runtime.Object. +func (in *ClusterResourcePlacementDisruptionBudgetList) DeepCopyObject() runtime.Object { + if c := in.DeepCopy(); c != nil { + return c + } + return nil +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *ClusterResourcePlacementEviction) DeepCopyInto(out *ClusterResourcePlacementEviction) { + *out = *in + out.TypeMeta = in.TypeMeta + in.ObjectMeta.DeepCopyInto(&out.ObjectMeta) + out.Spec = in.Spec + in.Status.DeepCopyInto(&out.Status) +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new ClusterResourcePlacementEviction. +func (in *ClusterResourcePlacementEviction) DeepCopy() *ClusterResourcePlacementEviction { + if in == nil { + return nil + } + out := new(ClusterResourcePlacementEviction) + in.DeepCopyInto(out) + return out +} + +// DeepCopyObject is an autogenerated deepcopy function, copying the receiver, creating a new runtime.Object. +func (in *ClusterResourcePlacementEviction) DeepCopyObject() runtime.Object { + if c := in.DeepCopy(); c != nil { + return c + } + return nil +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *ClusterResourcePlacementEvictionList) DeepCopyInto(out *ClusterResourcePlacementEvictionList) { + *out = *in + out.TypeMeta = in.TypeMeta + in.ListMeta.DeepCopyInto(&out.ListMeta) + if in.Items != nil { + in, out := &in.Items, &out.Items + *out = make([]ClusterResourcePlacementEviction, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new ClusterResourcePlacementEvictionList. +func (in *ClusterResourcePlacementEvictionList) DeepCopy() *ClusterResourcePlacementEvictionList { + if in == nil { + return nil + } + out := new(ClusterResourcePlacementEvictionList) + in.DeepCopyInto(out) + return out +} + +// DeepCopyObject is an autogenerated deepcopy function, copying the receiver, creating a new runtime.Object. +func (in *ClusterResourcePlacementEvictionList) DeepCopyObject() runtime.Object { + if c := in.DeepCopy(); c != nil { + return c + } + return nil +} + // DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. func (in *ClusterResourcePlacementList) DeepCopyInto(out *ClusterResourcePlacementList) { *out = *in @@ -1397,6 +1514,68 @@ func (in *PerClusterPlacementStatus) DeepCopy() *PerClusterPlacementStatus { return out } +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *PlacementDisruptionBudgetSpec) DeepCopyInto(out *PlacementDisruptionBudgetSpec) { + *out = *in + if in.MaxUnavailable != nil { + in, out := &in.MaxUnavailable, &out.MaxUnavailable + *out = new(intstr.IntOrString) + **out = **in + } + if in.MinAvailable != nil { + in, out := &in.MinAvailable, &out.MinAvailable + *out = new(intstr.IntOrString) + **out = **in + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new PlacementDisruptionBudgetSpec. +func (in *PlacementDisruptionBudgetSpec) DeepCopy() *PlacementDisruptionBudgetSpec { + if in == nil { + return nil + } + out := new(PlacementDisruptionBudgetSpec) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *PlacementEvictionSpec) DeepCopyInto(out *PlacementEvictionSpec) { + *out = *in +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new PlacementEvictionSpec. +func (in *PlacementEvictionSpec) DeepCopy() *PlacementEvictionSpec { + if in == nil { + return nil + } + out := new(PlacementEvictionSpec) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *PlacementEvictionStatus) DeepCopyInto(out *PlacementEvictionStatus) { + *out = *in + if in.Conditions != nil { + in, out := &in.Conditions, &out.Conditions + *out = make([]metav1.Condition, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new PlacementEvictionStatus. +func (in *PlacementEvictionStatus) DeepCopy() *PlacementEvictionStatus { + if in == nil { + return nil + } + out := new(PlacementEvictionStatus) + in.DeepCopyInto(out) + return out +} + // DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. func (in *PlacementPolicy) DeepCopyInto(out *PlacementPolicy) { *out = *in diff --git a/apis/placement/v1alpha1/eviction_types.go b/apis/placement/v1alpha1/eviction_types.go index ba2d52913..f05d716e2 100644 --- a/apis/placement/v1alpha1/eviction_types.go +++ b/apis/placement/v1alpha1/eviction_types.go @@ -34,7 +34,7 @@ import ( // // Beware that an eviction alone does not guarantee that a placement will not re-appear; i.e., // after an eviction, the Fleet scheduler might still pick the previous target cluster for -// placement. To prevent this, considering adding proper taints to the target cluster before running +// placement. To prevent this, consider adding proper taints to the target cluster before running // an eviction that will exclude it from future placements; this is especially true in scenarios // where one would like to perform a cluster replacement. // diff --git a/apis/placement/v1beta1/clusterresourceplacement_types.go b/apis/placement/v1beta1/clusterresourceplacement_types.go index f571d0a5a..3cd137cf6 100644 --- a/apis/placement/v1beta1/clusterresourceplacement_types.go +++ b/apis/placement/v1beta1/clusterresourceplacement_types.go @@ -984,7 +984,8 @@ type RollingUpdateConfig struct { // Defaults to 25%. // +kubebuilder:default="25%" // +kubebuilder:validation:XIntOrString - // +kubebuilder:validation:Pattern="^((100|[0-9]{1,2})%|[0-9]+)$" + // +kubebuilder:validation:Pattern="^((100|[0-9]{1,2})%|[0-9]{1,9})$" + // +kubebuilder:validation:XValidation:rule="type(self) == int ? self >= 0 : true",message="maxUnavailable must be a non-negative integer or a percentage" // +kubebuilder:validation:Optional MaxUnavailable *intstr.IntOrString `json:"maxUnavailable,omitempty"` @@ -998,7 +999,8 @@ type RollingUpdateConfig struct { // Defaults to 25%. // +kubebuilder:default="25%" // +kubebuilder:validation:XIntOrString - // +kubebuilder:validation:Pattern="^((100|[0-9]{1,2})%|[0-9]+)$" + // +kubebuilder:validation:Pattern="^((100|[0-9]{1,2})%|[0-9]{1,9})$" + // +kubebuilder:validation:XValidation:rule="type(self) == int ? self >= 0 : true",message="maxSurge must be a non-negative integer or a percentage" // +kubebuilder:validation:Optional MaxSurge *intstr.IntOrString `json:"maxSurge,omitempty"` diff --git a/apis/placement/v1beta1/commons.go b/apis/placement/v1beta1/commons.go index 3800d8817..6631cb7f0 100644 --- a/apis/placement/v1beta1/commons.go +++ b/apis/placement/v1beta1/commons.go @@ -79,6 +79,23 @@ const ( // by name in ResourceOverride and ClusterResourceOverride via labelSelector. MemberNameLabel = FleetPrefix + "member-name" + // KubeFleetPrefix is the prefix used for the labels/annotations of the kubefleet.dev APIs. Like + // FleetPrefix, it is reserved for KubeFleet's own keys, per the Kubernetes convention that + // leaves unprefixed keys to end users. + // + // The two prefixes are not exempted alike by the member cluster label guard: where that guard + // is on, a FleetPrefix label may be modified by any user, while a KubeFleetPrefix one may be + // modified only by a service account, so that the hub agent can seed ClusterAliasLabel without + // opening a scheduling label to everyone. Users in system:masters bypass the guard entirely and + // may modify either; note that this is narrower than the administrator checks elsewhere in the + // webhook, so a kubeadm:cluster-admins user is still subject to it. + KubeFleetPrefix = "kubefleet.dev/" + + // ClusterAliasLabel is a label on MemberCluster objects that names the cluster for placement by + // role rather than by name. It is seeded from the MemberCluster's name when absent and never + // reasserted, so an admin can move an alias to another cluster. + ClusterAliasLabel = KubeFleetPrefix + "cluster-alias" + // WorkFinalizer is used by the work generator to make sure that the binding is not deleted until the work objects // it generates are all deleted, or used by the work controller to make sure the work has been deleted in the member // cluster. diff --git a/apis/placement/v1beta1/disruptionbudget_types.go b/apis/placement/v1beta1/disruptionbudget_types.go index dab63c8de..ad3ce879c 100644 --- a/apis/placement/v1beta1/disruptionbudget_types.go +++ b/apis/placement/v1beta1/disruptionbudget_types.go @@ -66,8 +66,8 @@ type PlacementDisruptionBudgetSpec struct { // One may use a value of 0 for this field; in this case, no voluntary disruption would be // allowed. // - // This field is mutually exclusive with the MinAvailable field in the spec; exactly one - // of them can be set at a time. + // This field is mutually exclusive with the MinAvailable field in the spec; set only one of them + // at a time. If none is set, no disruption is allowed for the target placement. // // +kubebuilder:validation:XIntOrString // +kubebuilder:validation:XValidation:rule="type(self) == string ? self.matches('^(100|[0-9]{1,2})%$') : self >= 0",message="If supplied value is String should match regex '^(100|[0-9]{1,2})%$' or If supplied value is Integer must be greater than or equal to 0" @@ -95,8 +95,8 @@ type PlacementDisruptionBudgetSpec struct { // One may use a value of 0 for this field; in this case, voluntary disruption would be // allowed at any time. // - // This field is mutually exclusive with the MaxUnavailable field in the spec; exactly one - // of them can be set at a time. + // This field is mutually exclusive with the MaxUnavailable field in the spec; set only one of them + // at a time. If none is set, no disruption is allowed for the target placement. // // +kubebuilder:validation:XIntOrString // +kubebuilder:validation:XValidation:rule="type(self) == string ? self.matches('^(100|[0-9]{1,2})%$') : self >= 0",message="If supplied value is String should match regex '^(100|[0-9]{1,2})%$' or If supplied value is Integer must be greater than or equal to 0" diff --git a/apis/placement/v1beta1/eviction_types.go b/apis/placement/v1beta1/eviction_types.go index 11637d628..eff440ce4 100644 --- a/apis/placement/v1beta1/eviction_types.go +++ b/apis/placement/v1beta1/eviction_types.go @@ -37,7 +37,7 @@ import ( // // Beware that an eviction alone does not guarantee that a placement will not re-appear; i.e., // after an eviction, the Fleet scheduler might still pick the previous target cluster for -// placement. To prevent this, considering adding proper taints to the target cluster before running +// placement. To prevent this, consider adding proper taints to the target cluster before running // an eviction that will exclude it from future placements; this is especially true in scenarios // where one would like to perform a cluster replacement. // diff --git a/apis/placement/v1beta1/zz_generated.deepcopy.go b/apis/placement/v1beta1/zz_generated.deepcopy.go index b9ff2e710..73d66c8fa 100644 --- a/apis/placement/v1beta1/zz_generated.deepcopy.go +++ b/apis/placement/v1beta1/zz_generated.deepcopy.go @@ -21,7 +21,7 @@ limitations under the License. package v1beta1 import ( - v1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/runtime" "k8s.io/apimachinery/pkg/util/intstr" ) diff --git a/charts/README.md b/charts/README.md index 7456a8759..e125e2cf7 100644 --- a/charts/README.md +++ b/charts/README.md @@ -124,15 +124,21 @@ helm upgrade member-agent kubefleet/member-agent --namespace fleet-system ## Chart Publishing -Charts are automatically published to both locations when: -- Changes are pushed to the `main` branch affecting chart files -- A version tag (e.g., `v1.0.0`) is created +Charts are published to both locations when a stable version tag (e.g. +`v1.0.0`) is pushed, carrying that release's version and appVersion. +Release-candidate tags (e.g. `v1.0.0-rc.1`) build and publish images but +deliberately do not publish charts, so no pre-release version reaches the chart +index. **Published Locations:** - **OCI Registry**: `oci://ghcr.io/kubefleet-dev/kubefleet/charts/{chart-name}` - **GitHub Pages**: `https://kubefleet-dev.github.io/kubefleet/charts` -The publishing workflow is defined in `.github/workflows/chart.yml`. +Chart publishing is part of the release workflow in +`.github/workflows/release.yml`, which publishes the GitHub Release only after +the charts and every other release artifact have been published. See +[RELEASING.md](../RELEASING.md) for the full pipeline and its recovery +procedure. ## Development diff --git a/cmd/hubagent/workload/setup.go b/cmd/hubagent/workload/setup.go index 536ccfad0..2f4fcfbad 100644 --- a/cmd/hubagent/workload/setup.go +++ b/cmd/hubagent/workload/setup.go @@ -172,7 +172,7 @@ func SetupControllers(ctx context.Context, wg *sync.WaitGroup, mgr ctrl.Manager, resourceSnapshotResolver.Config = controller.NewResourceSnapshotConfig(opts.PlacementMgmtOpts.ResourceSnapshotCreationMinimumInterval, opts.PlacementMgmtOpts.ResourceChangesCollectionDuration) pc := &placement.Reconciler{ Client: mgr.GetClient(), - Recorder: mgr.GetEventRecorderFor(placementControllerName), + Recorder: mgr.GetEventRecorder(placementControllerName), Scheme: mgr.GetScheme(), UncachedReader: mgr.GetAPIReader(), ResourceSelectorResolver: resourceSelectorResolver, @@ -514,7 +514,7 @@ func SetupControllers(ctx context.Context, wg *sync.WaitGroup, mgr ctrl.Manager, klog.Info("Setting up resource change controller") rcr := &resourcechange.Reconciler{ DynamicClient: dynamicClient, - Recorder: mgr.GetEventRecorderFor(resourceChangeControllerName), + Recorder: mgr.GetEventRecorder(resourceChangeControllerName), RestMapper: mgr.GetRESTMapper(), InformerManager: dynamicInformerManager, PlacementControllerV1Beta1: clusterResourcePlacementControllerV1Beta1, diff --git a/cmd/memberagent/main.go b/cmd/memberagent/main.go index 94745b021..9581d7686 100644 --- a/cmd/memberagent/main.go +++ b/cmd/memberagent/main.go @@ -387,7 +387,7 @@ func Start(ctx context.Context, hubCfg, memberConfig *rest.Config, hubOpts, memb spokeDynamicClient, memberMgr.GetClient(), restMapper, - hubMgr.GetEventRecorderFor("work_applier"), + hubMgr.GetEventRecorder("work_applier"), // The number of concurrent reconcilations. This is set to 5 to boost performance in // resource processing. 5, diff --git a/config/crd/bases/placement.kubefleet.dev_clusterplacementbindings.yaml b/config/crd/bases/placement.kubefleet.dev_clusterplacementbindings.yaml index a78e895bc..87dfb4ef9 100644 --- a/config/crd/bases/placement.kubefleet.dev_clusterplacementbindings.yaml +++ b/config/crd/bases/placement.kubefleet.dev_clusterplacementbindings.yaml @@ -560,6 +560,12 @@ spec: type: object maxItems: 50 type: array + lastProcessedResourceSnapshotName: + description: |- + The name of the placement resource snapshot that KubeFleet has last processed for this binding. + This field helps KubeFleet track the processing progress; it also reveals whether the reported status + is up to date. + type: string selectedResources: description: The number of resources that are included in the currently associated resource snapshot(s). diff --git a/config/crd/bases/placement.kubefleet.dev_clusterplacementpolicies.yaml b/config/crd/bases/placement.kubefleet.dev_clusterplacementpolicies.yaml index d528f88bd..6e18f5a7e 100644 --- a/config/crd/bases/placement.kubefleet.dev_clusterplacementpolicies.yaml +++ b/config/crd/bases/placement.kubefleet.dev_clusterplacementpolicies.yaml @@ -588,9 +588,11 @@ spec: - kind - name type: object + minItems: 1 type: array required: - controllerName + - objectRefs type: object conditions: description: A list of conditions that describe the workload placement. diff --git a/config/crd/bases/placement.kubefleet.dev_placementbindings.yaml b/config/crd/bases/placement.kubefleet.dev_placementbindings.yaml index b45d48f4e..e6c135e67 100644 --- a/config/crd/bases/placement.kubefleet.dev_placementbindings.yaml +++ b/config/crd/bases/placement.kubefleet.dev_placementbindings.yaml @@ -560,6 +560,12 @@ spec: type: object maxItems: 50 type: array + lastProcessedResourceSnapshotName: + description: |- + The name of the placement resource snapshot that KubeFleet has last processed for this binding. + This field helps KubeFleet track the processing progress; it also reveals whether the reported status + is up to date. + type: string selectedResources: description: The number of resources that are included in the currently associated resource snapshot(s). diff --git a/config/crd/bases/placement.kubefleet.dev_placementpolicies.yaml b/config/crd/bases/placement.kubefleet.dev_placementpolicies.yaml index 578a6eb5f..36a9c326e 100644 --- a/config/crd/bases/placement.kubefleet.dev_placementpolicies.yaml +++ b/config/crd/bases/placement.kubefleet.dev_placementpolicies.yaml @@ -588,9 +588,11 @@ spec: - kind - name type: object + minItems: 1 type: array required: - controllerName + - objectRefs type: object conditions: description: A list of conditions that describe the workload placement. diff --git a/config/crd/bases/placement.kubernetes-fleet.io_clusterresourceplacementdisruptionbudgets.yaml b/config/crd/bases/placement.kubernetes-fleet.io_clusterresourceplacementdisruptionbudgets.yaml index a3af13ea4..3886d6dfc 100644 --- a/config/crd/bases/placement.kubernetes-fleet.io_clusterresourceplacementdisruptionbudgets.yaml +++ b/config/crd/bases/placement.kubernetes-fleet.io_clusterresourceplacementdisruptionbudgets.yaml @@ -19,6 +19,118 @@ spec: singular: clusterresourceplacementdisruptionbudget scope: Cluster versions: + - name: v1 + schema: + openAPIV3Schema: + description: |- + ClusterResourcePlacementDisruptionBudget is the policy applied to a ClusterResourcePlacement + object that specifies its disruption budget, i.e., how many placements (clusters) can be + down at the same time due to voluntary disruptions (e.g., evictions). Involuntary + disruptions are not subject to this budget, but will still count against it. + + To apply a ClusterResourcePlacementDisruptionBudget to a ClusterResourcePlacement, use the + same name for the ClusterResourcePlacementDisruptionBudget object as the ClusterResourcePlacement + object. This guarantees a 1:1 link between the two objects. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: Spec is the desired state of the ClusterResourcePlacementDisruptionBudget. + properties: + maxUnavailable: + anyOf: + - type: integer + - type: string + description: |- + MaxUnavailable is the maximum number of placements (clusters) that can be down at the + same time due to voluntary disruptions. For example, a setting of 1 would imply that + a voluntary disruption (e.g., an eviction) can only happen if all placements (clusters) + from the linked Placement object are applied and available. + + This can be either an absolute value (e.g., 1) or a percentage (e.g., 10%). + + If a percentage is specified, Fleet will calculate the corresponding absolute values + as follows: + * if the linked Placement object is of the PickFixed placement type, + we don't perform any calculation because eviction is not allowed for PickFixed CRP. + * if the linked Placement object is of the PickAll placement type, MaxUnavailable cannot + be specified since we cannot derive the total number of clusters selected. + * if the linked Placement object is of the PickN placement type, + the percentage is against the number of clusters specified in the placement (i.e., the + value of the NumberOfClusters fields in the placement policy). + The end result will be rounded up to the nearest integer if applicable. + + One may use a value of 0 for this field; in this case, no voluntary disruption would be + allowed. + + This field is mutually exclusive with the MinAvailable field in the spec; set only one of them + at a time. If none is set, no disruption is allowed for the target placement. + x-kubernetes-int-or-string: true + x-kubernetes-validations: + - message: If supplied value is String should match regex '^(100|[0-9]{1,2})%$' + or If supplied value is Integer must be greater than or equal + to 0 + rule: 'type(self) == string ? self.matches(''^(100|[0-9]{1,2})%$'') + : self >= 0' + minAvailable: + anyOf: + - type: integer + - type: string + description: |- + MinAvailable is the minimum number of placements (clusters) that must be available at any + time despite voluntary disruptions. For example, a setting of 10 would imply that + a voluntary disruption (e.g., an eviction) can only happen if there are at least 11 + placements (clusters) from the linked Placement object are applied and available. + + This can be either an absolute value (e.g., 1) or a percentage (e.g., 10%). + + If a percentage is specified, Fleet will calculate the corresponding absolute values + as follows: + * if the linked Placement object is of the PickFixed placement type, + we don't perform any calculation because eviction is not allowed for PickFixed CRP. + * if the linked Placement object is of the PickAll placement type, MinAvailable can be + specified but only as an integer since we cannot derive the total number of clusters selected. + * if the linked Placement object is of the PickN placement type, + the percentage is against the number of clusters specified in the placement (i.e., the + value of the NumberOfClusters fields in the placement policy). + The end result will be rounded up to the nearest integer if applicable. + + One may use a value of 0 for this field; in this case, voluntary disruption would be + allowed at any time. + + This field is mutually exclusive with the MaxUnavailable field in the spec; set only one of them + at a time. If none is set, no disruption is allowed for the target placement. + x-kubernetes-int-or-string: true + x-kubernetes-validations: + - message: If supplied value is String should match regex '^(100|[0-9]{1,2})%$' + or If supplied value is Integer must be greater than or equal + to 0 + rule: 'type(self) == string ? self.matches(''^(100|[0-9]{1,2})%$'') + : self >= 0' + type: object + x-kubernetes-validations: + - message: Both MaxUnavailable and MinAvailable cannot be specified + rule: '!(has(self.maxUnavailable) && has(self.minAvailable))' + required: + - spec + type: object + served: true + storage: false - name: v1alpha1 schema: openAPIV3Schema: @@ -190,8 +302,8 @@ spec: One may use a value of 0 for this field; in this case, no voluntary disruption would be allowed. - This field is mutually exclusive with the MinAvailable field in the spec; exactly one - of them can be set at a time. + This field is mutually exclusive with the MinAvailable field in the spec; set only one of them + at a time. If none is set, no disruption is allowed for the target placement. x-kubernetes-int-or-string: true x-kubernetes-validations: - message: If supplied value is String should match regex '^(100|[0-9]{1,2})%$' @@ -225,8 +337,8 @@ spec: One may use a value of 0 for this field; in this case, voluntary disruption would be allowed at any time. - This field is mutually exclusive with the MaxUnavailable field in the spec; exactly one - of them can be set at a time. + This field is mutually exclusive with the MaxUnavailable field in the spec; set only one of them + at a time. If none is set, no disruption is allowed for the target placement. x-kubernetes-int-or-string: true x-kubernetes-validations: - message: If supplied value is String should match regex '^(100|[0-9]{1,2})%$' diff --git a/config/crd/bases/placement.kubernetes-fleet.io_clusterresourceplacementevictions.yaml b/config/crd/bases/placement.kubernetes-fleet.io_clusterresourceplacementevictions.yaml index ae40aae6f..30e1ba64b 100644 --- a/config/crd/bases/placement.kubernetes-fleet.io_clusterresourceplacementevictions.yaml +++ b/config/crd/bases/placement.kubernetes-fleet.io_clusterresourceplacementevictions.yaml @@ -19,6 +19,165 @@ spec: singular: clusterresourceplacementeviction scope: Cluster versions: + - additionalPrinterColumns: + - jsonPath: .status.conditions[?(@.type=="Valid")].status + name: Valid + type: string + - jsonPath: .status.conditions[?(@.type=="Executed")].status + name: Executed + type: string + name: v1 + schema: + openAPIV3Schema: + description: |- + ClusterResourcePlacementEviction is an eviction attempt on a specific placement from + a ClusterResourcePlacement object; one may use this API to force the removal of specific + resources from a cluster. + + An eviction is a voluntary disruption; its execution is subject to the disruption budget + linked with the target ClusterResourcePlacement object (if present). + + Beware that an eviction alone does not guarantee that a placement will not re-appear; i.e., + after an eviction, the Fleet scheduler might still pick the previous target cluster for + placement. To prevent this, consider adding proper taints to the target cluster before running + an eviction that will exclude it from future placements; this is especially true in scenarios + where one would like to perform a cluster replacement. + + For safety reasons, Fleet will only execute an eviction once; the spec in this object is immutable, + and once executed, the object will be ignored after. To trigger another eviction attempt on the + same placement from the same ClusterResourcePlacement object, one must re-create (delete and + create) the same Eviction object. Note also that an Eviction object will be + ignored once it is deemed invalid (e.g., such an object might be targeting a CRP object or + a placement that does not exist yet), even if it does become valid later + (e.g., the CRP object or the placement appears later). To fix the situation, re-create the + Eviction object. + + Note: Eviction of resources from a cluster propagated by a PickFixed CRP is not allowed. + If the user wants to remove resources from a cluster propagated by a PickFixed CRP simply + remove the cluster name from cluster names field from the CRP spec. + + Executed evictions might be kept around for a while for auditing purposes; the Fleet controllers might + have a TTL set up for such objects and will garbage collect them automatically. For further + information, see the Fleet documentation. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: |- + Spec is the desired state of the ClusterResourcePlacementEviction. + + Note that all fields in the spec are immutable. + properties: + clusterName: + description: ClusterName is the name of the cluster that the Eviction + object targets. + maxLength: 255 + type: string + x-kubernetes-validations: + - message: The ClusterName field is immutable + rule: self == oldSelf + placementName: + description: |- + PlacementName is the name of the Placement object which + the Eviction object targets. + maxLength: 255 + type: string + x-kubernetes-validations: + - message: The PlacementName field is immutable + rule: self == oldSelf + required: + - clusterName + - placementName + type: object + status: + description: Status is the observed state of the ClusterResourcePlacementEviction. + properties: + conditions: + description: |- + Conditions is the list of currently observed conditions for the + PlacementEviction object. + + Available condition types include: + * Valid: whether the Eviction object is valid, i.e., it targets at a valid placement. + * Executed: whether the Eviction object has been executed. + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + type: object + required: + - spec + type: object + served: true + storage: false + subresources: + status: {} - name: v1alpha1 schema: openAPIV3Schema: @@ -32,7 +191,7 @@ spec: Beware that an eviction alone does not guarantee that a placement will not re-appear; i.e., after an eviction, the Fleet scheduler might still pick the previous target cluster for - placement. To prevent this, considering adding proper taints to the target cluster before running + placement. To prevent this, consider adding proper taints to the target cluster before running an eviction that will exclude it from future placements; this is especially true in scenarios where one would like to perform a cluster replacement. @@ -191,7 +350,7 @@ spec: Beware that an eviction alone does not guarantee that a placement will not re-appear; i.e., after an eviction, the Fleet scheduler might still pick the previous target cluster for - placement. To prevent this, considering adding proper taints to the target cluster before running + placement. To prevent this, consider adding proper taints to the target cluster before running an eviction that will exclude it from future placements; this is especially true in scenarios where one would like to perform a cluster replacement. diff --git a/config/crd/bases/placement.kubernetes-fleet.io_clusterresourceplacements.yaml b/config/crd/bases/placement.kubernetes-fleet.io_clusterresourceplacements.yaml index 7740c8c48..9aa1b1e41 100644 --- a/config/crd/bases/placement.kubernetes-fleet.io_clusterresourceplacements.yaml +++ b/config/crd/bases/placement.kubernetes-fleet.io_clusterresourceplacements.yaml @@ -952,8 +952,11 @@ spec: This does not apply to the case that we do in-place update of resources on the same cluster. This can not be 0 if MaxUnavailable is 0. Defaults to 25%. - pattern: ^((100|[0-9]{1,2})%|[0-9]+)$ + pattern: ^((100|[0-9]{1,2})%|[0-9]{1,9})$ x-kubernetes-int-or-string: true + x-kubernetes-validations: + - message: maxSurge must be a non-negative integer or a percentage + rule: 'type(self) == int ? self >= 0 : true' maxUnavailable: anyOf: - type: integer @@ -971,8 +974,12 @@ spec: The minimum of MaxUnavailable is 0 to allow no downtime moving a placement from one cluster to another. Please set it to be greater than 0 to avoid rolling out stuck during in-place resource update. Defaults to 25%. - pattern: ^((100|[0-9]{1,2})%|[0-9]+)$ + pattern: ^((100|[0-9]{1,2})%|[0-9]{1,9})$ x-kubernetes-int-or-string: true + x-kubernetes-validations: + - message: maxUnavailable must be a non-negative integer or + a percentage + rule: 'type(self) == int ? self >= 0 : true' unavailablePeriodSeconds: default: 60 description: |- @@ -2631,8 +2638,11 @@ spec: This does not apply to the case that we do in-place update of resources on the same cluster. This can not be 0 if MaxUnavailable is 0. Defaults to 25%. - pattern: ^((100|[0-9]{1,2})%|[0-9]+)$ + pattern: ^((100|[0-9]{1,2})%|[0-9]{1,9})$ x-kubernetes-int-or-string: true + x-kubernetes-validations: + - message: maxSurge must be a non-negative integer or a percentage + rule: 'type(self) == int ? self >= 0 : true' maxUnavailable: anyOf: - type: integer @@ -2650,8 +2660,12 @@ spec: The minimum of MaxUnavailable is 0 to allow no downtime moving a placement from one cluster to another. Please set it to be greater than 0 to avoid rolling out stuck during in-place resource update. Defaults to 25%. - pattern: ^((100|[0-9]{1,2})%|[0-9]+)$ + pattern: ^((100|[0-9]{1,2})%|[0-9]{1,9})$ x-kubernetes-int-or-string: true + x-kubernetes-validations: + - message: maxUnavailable must be a non-negative integer or + a percentage + rule: 'type(self) == int ? self >= 0 : true' unavailablePeriodSeconds: default: 60 description: |- diff --git a/config/crd/bases/placement.kubernetes-fleet.io_resourceplacements.yaml b/config/crd/bases/placement.kubernetes-fleet.io_resourceplacements.yaml index ee3855bce..df15dd56a 100644 --- a/config/crd/bases/placement.kubernetes-fleet.io_resourceplacements.yaml +++ b/config/crd/bases/placement.kubernetes-fleet.io_resourceplacements.yaml @@ -944,8 +944,11 @@ spec: This does not apply to the case that we do in-place update of resources on the same cluster. This can not be 0 if MaxUnavailable is 0. Defaults to 25%. - pattern: ^((100|[0-9]{1,2})%|[0-9]+)$ + pattern: ^((100|[0-9]{1,2})%|[0-9]{1,9})$ x-kubernetes-int-or-string: true + x-kubernetes-validations: + - message: maxSurge must be a non-negative integer or a percentage + rule: 'type(self) == int ? self >= 0 : true' maxUnavailable: anyOf: - type: integer @@ -963,8 +966,12 @@ spec: The minimum of MaxUnavailable is 0 to allow no downtime moving a placement from one cluster to another. Please set it to be greater than 0 to avoid rolling out stuck during in-place resource update. Defaults to 25%. - pattern: ^((100|[0-9]{1,2})%|[0-9]+)$ + pattern: ^((100|[0-9]{1,2})%|[0-9]{1,9})$ x-kubernetes-int-or-string: true + x-kubernetes-validations: + - message: maxUnavailable must be a non-negative integer or + a percentage + rule: 'type(self) == int ? self >= 0 : true' unavailablePeriodSeconds: default: 60 description: |- @@ -2608,8 +2615,11 @@ spec: This does not apply to the case that we do in-place update of resources on the same cluster. This can not be 0 if MaxUnavailable is 0. Defaults to 25%. - pattern: ^((100|[0-9]{1,2})%|[0-9]+)$ + pattern: ^((100|[0-9]{1,2})%|[0-9]{1,9})$ x-kubernetes-int-or-string: true + x-kubernetes-validations: + - message: maxSurge must be a non-negative integer or a percentage + rule: 'type(self) == int ? self >= 0 : true' maxUnavailable: anyOf: - type: integer @@ -2627,8 +2637,12 @@ spec: The minimum of MaxUnavailable is 0 to allow no downtime moving a placement from one cluster to another. Please set it to be greater than 0 to avoid rolling out stuck during in-place resource update. Defaults to 25%. - pattern: ^((100|[0-9]{1,2})%|[0-9]+)$ + pattern: ^((100|[0-9]{1,2})%|[0-9]{1,9})$ x-kubernetes-int-or-string: true + x-kubernetes-validations: + - message: maxUnavailable must be a non-negative integer or + a percentage + rule: 'type(self) == int ? self >= 0 : true' unavailablePeriodSeconds: default: 60 description: |- diff --git a/go.mod b/go.mod index 3c01dda70..e427eaf49 100644 --- a/go.mod +++ b/go.mod @@ -3,91 +3,107 @@ module go.goms.io/fleet go 1.26.6 require ( - github.com/Azure/azure-sdk-for-go/sdk/azcore v1.18.0 - github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.10.1 - github.com/Azure/karpenter-provider-azure v1.5.1 - github.com/crossplane/crossplane-runtime/v2 v2.1.0 + github.com/Azure/azure-sdk-for-go/sdk/azcore v1.23.1 + github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.14.1 + github.com/Azure/karpenter-provider-azure v1.14.2 + github.com/crossplane/crossplane-runtime/v2 v2.4.0 github.com/evanphx/json-patch/v5 v5.9.11 - github.com/go-logr/logr v1.4.3 + github.com/go-logr/logr v1.4.4 github.com/gofrs/uuid v4.4.0+incompatible github.com/google/go-cmp v0.7.0 - github.com/grpc-ecosystem/grpc-gateway/v2 v2.26.3 - github.com/onsi/ginkgo/v2 v2.23.4 - github.com/onsi/gomega v1.37.0 - github.com/prometheus/client_golang v1.22.0 - github.com/prometheus/client_model v0.6.2 + github.com/grpc-ecosystem/grpc-gateway/v2 v2.29.0 + github.com/onsi/ginkgo/v2 v2.32.2 + github.com/onsi/gomega v1.43.0 + github.com/prometheus/client_golang v1.24.1 + github.com/prometheus/client_model v0.6.3 github.com/qri-io/jsonpointer v0.1.1 - github.com/spf13/cobra v1.9.1 - github.com/spf13/pflag v1.0.6 - github.com/stretchr/testify v1.11.1 - github.com/wI2L/jsondiff v0.6.0 - go.goms.io/fleet-networking v0.3.3 + github.com/spf13/cobra v1.10.2 + github.com/spf13/pflag v1.0.10 + github.com/stretchr/testify v1.12.1 + github.com/wI2L/jsondiff v0.7.1 + go.goms.io/fleet-networking v0.3.44 go.uber.org/atomic v1.11.0 - go.uber.org/zap v1.27.0 - golang.org/x/sync v0.22.0 - golang.org/x/time v0.11.0 - gomodules.xyz/jsonpatch/v2 v2.4.0 + go.uber.org/zap v1.28.0 + golang.org/x/sync v0.23.0 + golang.org/x/time v0.16.0 + gomodules.xyz/jsonpatch/v2 v2.5.0 google.golang.org/grpc v1.83.2 - google.golang.org/protobuf v1.36.11 - k8s.io/api v0.34.1 - k8s.io/apiextensions-apiserver v0.34.1 - k8s.io/apimachinery v0.34.1 - k8s.io/client-go v0.34.1 - k8s.io/component-helpers v0.32.3 - k8s.io/klog/v2 v2.130.1 - k8s.io/kubectl v0.32.3 - k8s.io/metrics v0.32.3 - k8s.io/utils v0.0.0-20250604170112-4c0f3b243397 - sigs.k8s.io/cloud-provider-azure v1.32.4 - sigs.k8s.io/cloud-provider-azure/pkg/azclient v0.5.20 - sigs.k8s.io/cluster-inventory-api v0.0.0-20251028164203-2e3fabb46733 - sigs.k8s.io/controller-runtime v0.22.4 + google.golang.org/protobuf v1.36.12 + k8s.io/api v0.35.8 + k8s.io/apiextensions-apiserver v0.35.8 + k8s.io/apimachinery v0.35.8 + k8s.io/client-go v0.35.8 + k8s.io/component-helpers v0.35.8 + k8s.io/klog/v2 v2.140.0 + k8s.io/kubectl v0.35.8 + k8s.io/metrics v0.35.8 + k8s.io/utils v0.0.0-20260210185600-b8788abfbbc2 + sigs.k8s.io/cloud-provider-azure v1.35.9 + sigs.k8s.io/cloud-provider-azure/pkg/azclient v0.19.0 + sigs.k8s.io/cluster-inventory-api v0.1.3 + sigs.k8s.io/controller-runtime v0.23.3 sigs.k8s.io/yaml v1.6.0 ) require ( - dario.cat/mergo v1.0.2 // indirect - github.com/Azure/azure-sdk-for-go/sdk/internal v1.11.1 // indirect + github.com/Azure/azure-sdk-for-go/sdk/internal v1.12.0 // indirect github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/authorization/armauthorization/v2 v2.2.0 // indirect - github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/compute/armcompute/v6 v6.4.0 // indirect + github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/compute/armcompute/v7 v7.3.0 // indirect github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/containerregistry/armcontainerregistry v1.2.0 // indirect - github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/containerservice/armcontainerservice/v6 v6.5.0 // indirect + github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/containerservice/armcontainerservice/v6 v6.6.0 // indirect github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/keyvault/armkeyvault v1.5.0 // indirect - github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/msi/armmsi v1.2.0 // indirect - github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/network/armnetwork/v6 v6.2.0 // indirect + github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/msi/armmsi v1.3.0 // indirect + github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/network/armnetwork/v9 v9.0.0 // indirect github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/privatedns/armprivatedns v1.3.0 // indirect github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/resources/armresources v1.2.0 // indirect - github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/storage/armstorage v1.7.0 // indirect - github.com/Azure/azure-sdk-for-go/sdk/security/keyvault/azsecrets v1.3.1 // indirect - github.com/Azure/azure-sdk-for-go/sdk/security/keyvault/internal v1.1.1 // indirect + github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/storage/armstorage/v2 v2.0.0 // indirect + github.com/Azure/azure-sdk-for-go/sdk/security/keyvault/azsecrets v1.4.0 // indirect + github.com/Azure/azure-sdk-for-go/sdk/security/keyvault/internal v1.2.0 // indirect + github.com/Azure/go-autorest v14.2.0+incompatible // indirect + github.com/Azure/go-autorest/autorest v0.11.30 // indirect + github.com/Azure/go-autorest/autorest/adal v0.9.24 // indirect + github.com/Azure/go-autorest/autorest/date v0.3.1 // indirect + github.com/Azure/go-autorest/logger v0.2.2 // indirect + github.com/Azure/go-autorest/tracing v0.6.1 // indirect github.com/Azure/msi-dataplane v0.4.3 // indirect - github.com/AzureAD/microsoft-authentication-library-for-go v1.4.2 // indirect + github.com/AzureAD/microsoft-authentication-library-for-go v1.8.0 // indirect + github.com/Masterminds/semver/v3 v3.5.0 // indirect github.com/antlr4-go/antlr/v4 v4.13.1 // indirect + github.com/awslabs/operatorpkg v0.0.0-20260708223819-4da4c353c5fa // indirect github.com/beorn7/perks v1.0.1 // indirect github.com/blang/semver/v4 v4.0.0 // indirect github.com/cespare/xxhash/v2 v2.3.0 // indirect + github.com/crossplane/crossplane/apis/v2 v2.3.4 // indirect github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc // indirect - github.com/emicklei/go-restful/v3 v3.12.2 // indirect - github.com/fsnotify/fsnotify v1.9.0 // indirect + github.com/emicklei/go-restful/v3 v3.13.0 // indirect + github.com/fsnotify/fsnotify v1.10.1 // indirect github.com/fxamacker/cbor/v2 v2.9.0 // indirect github.com/go-errors/errors v1.4.2 // indirect github.com/go-logr/zapr v1.3.0 // indirect - github.com/go-openapi/jsonpointer v0.21.1 // indirect - github.com/go-openapi/jsonreference v0.21.0 // indirect - github.com/go-openapi/swag v0.23.1 // indirect + github.com/go-openapi/jsonpointer v1.0.0 // indirect + github.com/go-openapi/jsonreference v1.0.0 // indirect + github.com/go-openapi/swag v0.28.0 // indirect + github.com/go-openapi/swag/cmdutils v0.28.0 // indirect + github.com/go-openapi/swag/conv v0.28.0 // indirect + github.com/go-openapi/swag/fileutils v0.28.0 // indirect + github.com/go-openapi/swag/jsonutils v0.28.0 // indirect + github.com/go-openapi/swag/loading v0.28.0 // indirect + github.com/go-openapi/swag/mangling v0.28.0 // indirect + github.com/go-openapi/swag/netutils v0.28.0 // indirect + github.com/go-openapi/swag/pools v0.28.0 // indirect + github.com/go-openapi/swag/stringutils v0.28.0 // indirect + github.com/go-openapi/swag/typeutils v0.28.0 // indirect + github.com/go-openapi/swag/yamlutils v0.28.0 // indirect github.com/go-task/slim-sprig/v3 v3.0.0 // indirect - github.com/gogo/protobuf v1.3.2 // indirect - github.com/golang-jwt/jwt/v5 v5.2.2 // indirect + github.com/golang-jwt/jwt/v4 v4.5.2 // indirect + github.com/golang-jwt/jwt/v5 v5.3.1 // indirect github.com/google/btree v1.1.3 // indirect - github.com/google/gnostic-models v0.7.0 // indirect - github.com/google/pprof v0.0.0-20250403155104-27863c87afa6 // indirect - github.com/google/shlex v0.0.0-20191202100458-e7afc7fbc510 // indirect + github.com/google/gnostic-models v0.7.1 // indirect + github.com/google/pprof v0.0.0-20260402051712-545e8a4df936 // indirect github.com/google/uuid v1.6.0 // indirect github.com/inconshreveable/mousetrap v1.1.0 // indirect - github.com/josharian/intern v1.0.0 // indirect github.com/json-iterator/go v1.1.12 // indirect github.com/kylelemons/godebug v1.1.0 // indirect - github.com/mailru/easyjson v0.9.0 // indirect github.com/mitchellh/hashstructure/v2 v2.0.2 // indirect github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd // indirect github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee // indirect @@ -95,51 +111,47 @@ require ( github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 // indirect github.com/patrickmn/go-cache v2.1.0+incompatible // indirect github.com/pkg/browser v0.0.0-20240102092130-5ac0b6a4141c // indirect - github.com/pkg/errors v0.9.1 // indirect github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2 // indirect - github.com/prometheus/common v0.62.0 // indirect - github.com/prometheus/procfs v0.15.1 // indirect - github.com/samber/lo v1.51.0 // indirect + github.com/prometheus/common v0.70.1 // indirect + github.com/prometheus/procfs v0.21.1 // indirect + github.com/robfig/cron/v3 v3.0.1 // indirect + github.com/samber/lo v1.53.0 // indirect github.com/tidwall/gjson v1.18.0 // indirect github.com/tidwall/match v1.1.1 // indirect github.com/tidwall/pretty v1.2.1 // indirect github.com/tidwall/sjson v1.2.5 // indirect github.com/x448/float16 v0.8.4 // indirect github.com/xlab/treeprint v1.2.0 // indirect - go.opentelemetry.io/otel v1.44.0 // indirect - go.opentelemetry.io/otel/metric v1.44.0 // indirect - go.uber.org/automaxprocs v1.6.0 // indirect - go.uber.org/mock v0.5.1 // indirect + go.opentelemetry.io/otel v1.46.0 // indirect + go.opentelemetry.io/otel/metric v1.46.0 // indirect + go.uber.org/mock v0.6.0 // indirect go.uber.org/multierr v1.11.0 // indirect - go.yaml.in/yaml/v2 v2.4.2 // indirect - go.yaml.in/yaml/v3 v3.0.4 // indirect + go.yaml.in/yaml/v2 v2.4.4 // indirect + go.yaml.in/yaml/v3 v3.0.5 // indirect golang.org/x/crypto v0.56.0 // indirect - golang.org/x/exp v0.0.0-20250305212735-054e65f0b394 // indirect + golang.org/x/exp v0.0.0-20260112195511-716be5621a96 // indirect + golang.org/x/mod v0.40.0 // indirect golang.org/x/net v0.58.0 // indirect golang.org/x/oauth2 v0.36.0 // indirect golang.org/x/sys v0.47.0 // indirect golang.org/x/term v0.45.0 // indirect golang.org/x/text v0.41.0 // indirect - golang.org/x/tools v0.48.0 // indirect + golang.org/x/tools v0.49.0 // indirect google.golang.org/genproto/googleapis/api v0.0.0-20260526163538-3dc84a4a5aaa // indirect google.golang.org/genproto/googleapis/rpc v0.0.0-20260526163538-3dc84a4a5aaa // indirect - gopkg.in/evanphx/json-patch.v4 v4.12.0 // indirect + gopkg.in/evanphx/json-patch.v4 v4.13.0 // indirect gopkg.in/inf.v0 v0.9.1 // indirect - gopkg.in/yaml.v3 v3.0.1 // indirect - k8s.io/cli-runtime v0.32.3 // indirect - k8s.io/kube-openapi v0.0.0-20250710124328-f3f2b991d03b // indirect - sigs.k8s.io/json v0.0.0-20241014173422-cfa47c3a1cc8 // indirect - sigs.k8s.io/karpenter v1.5.0 // indirect - sigs.k8s.io/kustomize/api v0.18.0 // indirect - sigs.k8s.io/kustomize/kyaml v0.18.1 // indirect + k8s.io/cli-runtime v0.35.8 // indirect + k8s.io/cloud-provider v0.35.8 // indirect + k8s.io/component-base v0.35.8 // indirect + k8s.io/csi-translation-lib v0.35.0 // indirect + k8s.io/kube-openapi v0.0.0-20260319004828-5883c5ee87b9 // indirect + sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730 // indirect + sigs.k8s.io/karpenter v1.14.0 // indirect + sigs.k8s.io/kustomize/api v0.20.1 // indirect + sigs.k8s.io/kustomize/kyaml v0.20.1 // indirect sigs.k8s.io/randfill v1.0.0 // indirect - sigs.k8s.io/structured-merge-diff/v6 v6.3.0 // indirect + sigs.k8s.io/structured-merge-diff/v6 v6.3.2 // indirect ) -replace ( - // fix CVE-2023-47108 introduced by k8s.io/apiextensions-apiserver - go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc => go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc v0.46.0 - - k8s.io/kube-scheduler => k8s.io/kube-scheduler v0.30.2 // weird bug that the goland won't compile without this - sigs.k8s.io/work-api => github.com/Azure/k8s-work-api v0.5.0 -) +replace sigs.k8s.io/work-api => github.com/Azure/k8s-work-api v0.5.0 diff --git a/go.sum b/go.sum index b15f748b6..3f8d6b4ac 100644 --- a/go.sum +++ b/go.sum @@ -1,35 +1,33 @@ -dario.cat/mergo v1.0.2 h1:85+piFYR1tMbRrLcDwR18y4UKJ3aH1Tbzi24VRW1TK8= -dario.cat/mergo v1.0.2/go.mod h1:E/hbnu0NxMFBjpMIE34DRGLWqDy0g5FuKDhCb31ngxA= -github.com/Azure/aks-middleware v0.0.40 h1:eFRuAxCcIAZoy/6+FvumDl2KOWnSPxXcAeCSOA4+aTo= -github.com/Azure/aks-middleware v0.0.40/go.mod h1:7Y+wxZmS7p1K0FPreiO3+6Wr8YhYjWz9c50YohDQIQ4= +cel.dev/expr v0.25.2 h1:K6j46C81hXtZQfuX60cVWQFBJahKSE2gfRbNuvr5bFs= +cel.dev/expr v0.25.2/go.mod h1:hrXvqGP6G6gyx8UAHSHJ5RGk//1Oj5nXQ2NI02Nrsg4= +github.com/Azure/aks-middleware v0.0.42 h1:StRGz6OuQi6mht5LV9uwhWn74kEsFP1wvpYQnyUOKHM= +github.com/Azure/aks-middleware v0.0.42/go.mod h1:7Y+wxZmS7p1K0FPreiO3+6Wr8YhYjWz9c50YohDQIQ4= github.com/Azure/azure-kusto-go v0.16.1 h1:vCBWcQghmC1qIErUUgVNWHxGhZVStu1U/hki6iBA14k= github.com/Azure/azure-kusto-go v0.16.1/go.mod h1:9F2zvXH8B6eWzgI1S4k1ZXAIufnBZ1bv1cW1kB1n3D0= github.com/Azure/azure-sdk-for-go v68.0.0+incompatible h1:fcYLmCpyNYRnvJbPerq7U0hS+6+I79yEDJBqVNcqUzU= github.com/Azure/azure-sdk-for-go v68.0.0+incompatible/go.mod h1:9XXNKU+eRnpl9moKnB4QOLf1HestfXbmab5FXxiDBjc= -github.com/Azure/azure-sdk-for-go-extensions v0.1.9 h1:bLtHrA9ZKx6TIvAzj45IXOmcTDVGYWe0AWMuxyo4ung= -github.com/Azure/azure-sdk-for-go-extensions v0.1.9/go.mod h1:jUub1P7aPW+gzLHrY0vhXjjFv9tZhCRKI1+zp2d56oA= -github.com/Azure/azure-sdk-for-go/sdk/azcore v1.18.0 h1:Gt0j3wceWMwPmiazCa8MzMA0MfhmPIz0Qp0FJ6qcM0U= -github.com/Azure/azure-sdk-for-go/sdk/azcore v1.18.0/go.mod h1:Ot/6aikWnKWi4l9QB7qVSwa8iMphQNqkWALMoNT3rzM= -github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.10.1 h1:B+blDbyVIG3WaikNxPnhPiJ1MThR03b3vKGtER95TP4= -github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.10.1/go.mod h1:JdM5psgjfBf5fo2uWOZhflPWyDBZ/O/CNAH9CtsuZE4= -github.com/Azure/azure-sdk-for-go/sdk/azidentity/cache v0.3.2 h1:yz1bePFlP5Vws5+8ez6T3HWXPmwOK7Yvq8QxDBD3SKY= -github.com/Azure/azure-sdk-for-go/sdk/azidentity/cache v0.3.2/go.mod h1:Pa9ZNPuoNu/GztvBSKk9J1cDJW6vk/n0zLtV4mgd8N8= -github.com/Azure/azure-sdk-for-go/sdk/internal v1.11.1 h1:FPKJS1T+clwv+OLGt13a8UjqeRuh0O4SJ3lUriThc+4= -github.com/Azure/azure-sdk-for-go/sdk/internal v1.11.1/go.mod h1:j2chePtV91HrC22tGoRX3sGY42uF13WzmmV80/OdVAA= +github.com/Azure/azure-sdk-for-go-extensions v0.6.0 h1:LzJ4iAk3ZBZ0Y27uUm66XBQntbgMr3QXn2KIDb4Mx04= +github.com/Azure/azure-sdk-for-go-extensions v0.6.0/go.mod h1:f/wRrqvvh197V5r4jGADV7528UdO/zfL+/Ud92BMSag= +github.com/Azure/azure-sdk-for-go/sdk/azcore v1.23.1 h1:zvXfGJCWvywnCA814d8ZiVyt+fm9nnTE8xSb99zRyfo= +github.com/Azure/azure-sdk-for-go/sdk/azcore v1.23.1/go.mod h1:iptorS+VYKFL2N6PnebpS91dubG35eAOEERnT4PJbQU= +github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.14.1 h1:u93s+zU2JD62im61Bm5CZIc1ZrOJaIAWEg0WOrMVkEo= +github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.14.1/go.mod h1:oXtinPO4OLj9d1DOTrqrL1oRwGhcqadvAmrl6wTeGlk= +github.com/Azure/azure-sdk-for-go/sdk/azidentity/cache v0.4.0 h1:xFaZZ+IubdftrDHnGGwZ6QvQ3KHTtWl2MCK+GMt2vxs= +github.com/Azure/azure-sdk-for-go/sdk/azidentity/cache v0.4.0/go.mod h1:mCBhUhlMjLLJKr5aqw2TNS/VqJOie8MzWq3DAMJeKso= +github.com/Azure/azure-sdk-for-go/sdk/internal v1.12.0 h1:fhqpLE3UEXi9lPaBRpQ6XuRW0nU7hgg4zlmZZa+a9q4= +github.com/Azure/azure-sdk-for-go/sdk/internal v1.12.0/go.mod h1:7dCRMLwisfRH3dBupKeNCioWYUZ4SS09Z14H+7i8ZoY= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/authorization/armauthorization/v2 v2.2.0 h1:Hp+EScFOu9HeCbeW8WU2yQPJd4gGwhMgKxWe+G6jNzw= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/authorization/armauthorization/v2 v2.2.0/go.mod h1:/pz8dyNQe+Ey3yBp/XuYz7oqX8YDNWVpPB0hH3XWfbc= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/compute/armcompute v1.0.0 h1:/Di3vB4sNeQ+7A8efjUVENvyB945Wruvstucqp7ZArg= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/compute/armcompute v1.0.0/go.mod h1:gM3K25LQlsET3QR+4V74zxCsFAy0r6xMNN9n80SZn+4= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/compute/armcompute/v5 v5.7.0 h1:LkHbJbgF3YyvC53aqYGR+wWQDn2Rdp9AQdGndf9QvY4= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/compute/armcompute/v5 v5.7.0/go.mod h1:QyiQdW4f4/BIfB8ZutZ2s+28RAgfa/pT+zS++ZHyM1I= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/compute/armcompute/v6 v6.4.0 h1:z7Mqz6l0EFH549GvHEqfjKvi+cRScxLWbaoeLm9wxVQ= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/compute/armcompute/v6 v6.4.0/go.mod h1:v6gbfH+7DG7xH2kUNs+ZJ9tF6O3iNnR85wMtmr+F54o= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/compute/armcompute/v7 v7.3.0 h1:nyxugFxG2uhbMeJVCFFuD2j9wu+6KgeabITdINraQsE= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/compute/armcompute/v7 v7.3.0/go.mod h1:e4RAYykLIz73CF52KhSooo4whZGXvXrD09m0jkgnWiU= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/containerregistry/armcontainerregistry v1.2.0 h1:DWlwvVV5r/Wy1561nZ3wrpI1/vDIBRY/Wd1HWaRBZWA= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/containerregistry/armcontainerregistry v1.2.0/go.mod h1:E7ltexgRDmeJ0fJWv0D/HLwY2xbDdN+uv+X2uZtOx3w= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/containerservice/armcontainerservice/v5 v5.0.0 h1:5n7dPVqsWfVKw+ZiEKSd3Kzu7gwBkbEBkeXb8rgaE9Q= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/containerservice/armcontainerservice/v5 v5.0.0/go.mod h1:HcZY0PHPo/7d75p99lB6lK0qYOP4vLRJUBpiehYXtLQ= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/containerservice/armcontainerservice/v6 v6.5.0 h1:8deM0E7Il/6jxRU9Kgv8kKm3uq3O6Gh6NVNqADa4zbU= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/containerservice/armcontainerservice/v6 v6.5.0/go.mod h1:PhSVsfd99UdSWx7VAnbHr5i1O4WQ3YkYBFqQpSOx7oA= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/containerservice/armcontainerservice/v6 v6.6.0 h1:xkWEcbsnJWid3rOf/S/LOHy1I55JA+4kw/f8Tnm+Onc= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/containerservice/armcontainerservice/v6 v6.6.0/go.mod h1:OWKfCmX4X3Vp2w7GSx1LZn8566tOHJBA6K0IAUVNYx0= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/containerservice/armcontainerservice/v9 v9.5.0-beta.1 h1:fnsRu+aUmY9LqHMiBm+OYYwiUNp/dUHUGtjNkz0j5OY= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/containerservice/armcontainerservice/v9 v9.5.0-beta.1/go.mod h1:VD8lsnWhQBWSA9/kT+3DI20fjha1dBy11S5XzP8dlWE= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/internal/v2 v2.0.0 h1:PTFGRSlMKCQelWwxUyYVEUqseBJVemLyqWJjvMyt0do= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/internal/v2 v2.0.0/go.mod h1:LRr2FzBTQlONPPa5HREE5+RjSCTXl7BwOvYOaWTqCaI= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/internal/v3 v3.1.0 h1:2qsIIvxVT+uE6yrNldntJKlLRgxGbZ85kgtz5SNBhMw= @@ -38,58 +36,73 @@ github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/keyvault/armkeyvault v1.5. github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/keyvault/armkeyvault v1.5.0/go.mod h1:4YIVtzMFVsPwBvitCDX7J9sqthSj43QD1sP6fYc1egc= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/managementgroups/armmanagementgroups v1.0.0 h1:pPvTJ1dY0sA35JOeFq6TsY2xj6Z85Yo23Pj4wCCvu4o= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/managementgroups/armmanagementgroups v1.0.0/go.mod h1:mLfWfj8v3jfWKsL9G4eoBoXVcsqcIUTapmdKy7uGOp0= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/msi/armmsi v1.2.0 h1:z4YeiSXxnUI+PqB46Yj6MZA3nwb1CcJIkEMDrzUd8Cs= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/msi/armmsi v1.2.0/go.mod h1:rko9SzMxcMk0NJsNAxALEGaTYyy79bNRwxgJfrH0Spw= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/msi/armmsi v1.3.0 h1:L7G3dExHBgUxsO3qpTGhk/P2dgnYyW48yn7AO33Tbek= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/msi/armmsi v1.3.0/go.mod h1:Ms6gYEy0+A2knfKrwdatsggTXYA2+ICKug8w7STorFw= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/network/armnetwork v1.1.0 h1:QM6sE5k2ZT/vI5BEe0r7mqjsUSnhVBFbOsVkEuaEfiA= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/network/armnetwork v1.1.0/go.mod h1:243D9iHbcQXoFUtgHJwL7gl2zx1aDuDMjvBZVGr2uW0= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/network/armnetwork/v6 v6.2.0 h1:HYGD75g0bQ3VO/Omedm54v4LrD3B1cGImuRF3AJ5wLo= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/network/armnetwork/v6 v6.2.0/go.mod h1:ulHyBFJOI0ONiRL4vcJTmS7rx18jQQlEPmAgo80cRdM= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/network/armnetwork/v9 v9.0.0 h1:CbHDMVJhcJSmXenq+UDWyIjumzVkZIb5pVUGzsCok5M= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/network/armnetwork/v9 v9.0.0/go.mod h1:raqbEXrok4aycS74XoU6p9Hne1dliAFpHLizlp+qJoM= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/privatedns/armprivatedns v1.3.0 h1:yzrctSl9GMIQ5lHu7jc8olOsGjWDCsBpJhWqfGa/YIM= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/privatedns/armprivatedns v1.3.0/go.mod h1:GE4m0rnnfwLGX0Y9A9A25Zx5N/90jneT5ABevqzhuFQ= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/resourcegraph/armresourcegraph v0.9.0 h1:zLzoX5+W2l95UJoVwiyNS4dX8vHyQ6x2xRLoBBL9wMk= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/resourcegraph/armresourcegraph v0.9.0/go.mod h1:wVEOJfGTj0oPAUGA1JuRAvz/lxXQsWW16axmHPP47Bk= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/resourcegraph/armresourcegraph v0.10.0 h1:+1fJwTilk/X7inNqwREnYEOgFCdg8ut7GULxARDbu34= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/resourcegraph/armresourcegraph v0.10.0/go.mod h1:EGwSLlGqrrfYQhtCi9JcIkPQKl9WxsL6ZPJd+63Vy1A= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/resources/armresources v1.2.0 h1:Dd+RhdJn0OTtVGaeDLZpcumkIVCtA/3/Fo42+eoYvVM= github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/resources/armresources v1.2.0/go.mod h1:5kakwfW5CjC9KK+Q4wjXAg+ShuIm2mBMua0ZFj2C8PE= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/storage/armstorage v1.7.0 h1:D3pGIZLYN7MnksIkMkeRylz13YPetz6/H8rc5S9Vllg= -github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/storage/armstorage v1.7.0/go.mod h1:kJn8QL2DCyKnbDFMdi4SZiK0OOetns2eeKv+cJql0Yw= -github.com/Azure/azure-sdk-for-go/sdk/security/keyvault/azsecrets v1.3.1 h1:mrkDCdkMsD4l9wjFGhofFHFrV43Y3c53RSLKOCJ5+Ow= -github.com/Azure/azure-sdk-for-go/sdk/security/keyvault/azsecrets v1.3.1/go.mod h1:hPv41DbqMmnxcGralanA/kVlfdH5jv3T4LxGku2E1BY= -github.com/Azure/azure-sdk-for-go/sdk/security/keyvault/internal v1.1.1 h1:bFWuoEKg+gImo7pvkiQEFAc8ocibADgXeiLAxWhWmkI= -github.com/Azure/azure-sdk-for-go/sdk/security/keyvault/internal v1.1.1/go.mod h1:Vih/3yc6yac2JzU4hzpaDupBJP0Flaia9rXXrU8xyww= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/resources/armsubscriptions v1.3.0 h1:wxQx2Bt4xzPIKvW59WQf1tJNx/ZZKPfN+EhPX3Z6CYY= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/resources/armsubscriptions v1.3.0/go.mod h1:TpiwjwnW/khS0LKs4vW5UmmT9OWcxaveS8U7+tlknzo= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/storage/armstorage v1.8.1 h1:/Zt+cDPnpC3OVDm/JKLOs7M2DKmLRIIp3XIx9pHHiig= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/storage/armstorage v1.8.1/go.mod h1:Ng3urmn6dYe8gnbCMoHHVl5APYz2txho3koEkV2o2HA= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/storage/armstorage/v2 v2.0.0 h1:+vh02EiRx2UmL9NDoA36U18Bgwl9luxs6ia0GAI9Rzg= +github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/storage/armstorage/v2 v2.0.0/go.mod h1:iKOtU3WyuNvNc4L1Z4IxHaoO0dGq5tg+uhLix/KRmzE= +github.com/Azure/azure-sdk-for-go/sdk/security/keyvault/azsecrets v1.4.0 h1:/g8S6wk65vfC6m3FIxJ+i5QDyN9JWwXI8Hb0Img10hU= +github.com/Azure/azure-sdk-for-go/sdk/security/keyvault/azsecrets v1.4.0/go.mod h1:gpl+q95AzZlKVI3xSoseF9QPrypk0hQqBiJYeB/cR/I= +github.com/Azure/azure-sdk-for-go/sdk/security/keyvault/internal v1.2.0 h1:nCYfgcSyHZXJI8J0IWE5MsCGlb2xp9fJiXyxWgmOFg4= +github.com/Azure/azure-sdk-for-go/sdk/security/keyvault/internal v1.2.0/go.mod h1:ucUjca2JtSZboY8IoUqyQyuuXvwbMBVwFOm0vdQPNhA= github.com/Azure/go-autorest v14.2.0+incompatible h1:V5VMDjClD3GiElqLWO7mz2MxNAK/vTfRHdAubSIPRgs= github.com/Azure/go-autorest v14.2.0+incompatible/go.mod h1:r+4oMnoxhatjLLJ6zxSWATqVooLgysK6ZNox3g/xq24= github.com/Azure/go-autorest/autorest v0.11.30 h1:iaZ1RGz/ALZtN5eq4Nr1SOFSlf2E4pDI3Tcsl+dZPVE= github.com/Azure/go-autorest/autorest v0.11.30/go.mod h1:t1kpPIOpIVX7annvothKvb0stsrXa37i7b+xpmBW8Fs= +github.com/Azure/go-autorest/autorest/adal v0.9.22/go.mod h1:XuAbAEUv2Tta//+voMI038TrJBqjKam0me7qR+L8Cmk= github.com/Azure/go-autorest/autorest/adal v0.9.24 h1:BHZfgGsGwdkHDyZdtQRQk1WeUdW0m2WPAwuHZwUi5i4= github.com/Azure/go-autorest/autorest/adal v0.9.24/go.mod h1:7T1+g0PYFmACYW5LlG2fcoPiPlFHjClyRGL7dRlP5c8= -github.com/Azure/go-autorest/autorest/date v0.3.0 h1:7gUk1U5M/CQbp9WoqinNzJar+8KY+LPI6wiWrP/myHw= github.com/Azure/go-autorest/autorest/date v0.3.0/go.mod h1:BI0uouVdmngYNUzGWeSYnokU+TrmwEsOqdt8Y6sso74= +github.com/Azure/go-autorest/autorest/date v0.3.1 h1:o9Z8Jyt+VJJTCZ/UORishuHOusBwolhjokt9s5k8I4w= +github.com/Azure/go-autorest/autorest/date v0.3.1/go.mod h1:Dz/RDmXlfiFFS/eW+b/xMUSFs1tboPVy6UjgADToWDM= +github.com/Azure/go-autorest/autorest/mocks v0.4.1/go.mod h1:LTp+uSrOhSkaKrUy935gNZuuIPPVsHlr9DSOxSayd+k= +github.com/Azure/go-autorest/autorest/mocks v0.4.2 h1:PGN4EDXnuQbojHbU0UWoNvmu9AGVwYHG9/fkDYhtAfw= +github.com/Azure/go-autorest/autorest/mocks v0.4.2/go.mod h1:Vy7OitM9Kei0i1Oj+LvyAWMXJHeKH1MVlzFugfVrmyU= github.com/Azure/go-autorest/autorest/to v0.4.1 h1:CxNHBqdzTr7rLtdrtb5CMjJcDut+WNGCVv7OmS5+lTc= github.com/Azure/go-autorest/autorest/to v0.4.1/go.mod h1:EtaofgU4zmtvn1zT2ARsjRFdq9vXx0YWtmElwL+GZ9M= github.com/Azure/go-autorest/autorest/validation v0.3.1 h1:AgyqjAd94fwNAoTjl/WQXg4VvFeRFpO+UhNyRXqF1ac= github.com/Azure/go-autorest/autorest/validation v0.3.1/go.mod h1:yhLgjC0Wda5DYXl6JAsWyUe4KVNffhoDhG0zVzUMo3E= -github.com/Azure/go-autorest/logger v0.2.1 h1:IG7i4p/mDa2Ce4TRyAO8IHnVhAVF3RFU+ZtXWSmf4Tg= github.com/Azure/go-autorest/logger v0.2.1/go.mod h1:T9E3cAhj2VqvPOtCYAvby9aBXkZmbF5NWuPV8+WeEW8= -github.com/Azure/go-autorest/tracing v0.6.0 h1:TYi4+3m5t6K48TGI9AUdb+IzbnSxvnvUMfuitfgcfuo= +github.com/Azure/go-autorest/logger v0.2.2 h1:hYqBsEBywrrOSW24kkOCXRcKfKhK76OzLTfF+MYDE2o= +github.com/Azure/go-autorest/logger v0.2.2/go.mod h1:I5fg9K52o+iuydlWfa9T5K6WFos9XYr9dYTFzpqgibw= github.com/Azure/go-autorest/tracing v0.6.0/go.mod h1:+vhtPC754Xsa23ID7GlGsrdKBpUA79WCAKPPZVC2DeU= -github.com/Azure/karpenter-provider-azure v1.5.1 h1:CH92k7EgLyufVk16c4EsCTUJKrVBBgbWJg85sjbQAHE= -github.com/Azure/karpenter-provider-azure v1.5.1/go.mod h1:Sc2rQ+qqzv9J1Wr9jTpTpzDYsy0MJoNfqPromvH87n8= +github.com/Azure/go-autorest/tracing v0.6.1 h1:YUMSrC/CeD1ZnnXcNYU4a/fzsO35u2Fsful9L/2nyR0= +github.com/Azure/go-autorest/tracing v0.6.1/go.mod h1:/3EgjbsjraOqiicERAeu3m7/z0x1TzjQGAwDrJrXGkc= +github.com/Azure/karpenter-provider-azure v1.14.2 h1:ahAvrgrlHNbPJxPiusamAES7cGkGOJYse2mGlRA2Hao= +github.com/Azure/karpenter-provider-azure v1.14.2/go.mod h1:awuyp3bH6x0CnNnstjB5kVn6DevjMM3kwUz7rGi6Onc= github.com/Azure/msi-dataplane v0.4.3 h1:dWPWzY4b54tLIR9T1Q014Xxd/1DxOsMIp6EjRFAJlQY= github.com/Azure/msi-dataplane v0.4.3/go.mod h1:yAfxdJyvcnvSDfSyOFV9qm4fReEQDl+nZLGeH2ZWSmw= -github.com/Azure/skewer v0.0.20 h1:+dy82zLboRcAlSPH1Tj1z/I/vywBOE+tgW4GpZGgqBw= -github.com/Azure/skewer v0.0.20/go.mod h1:LVH7jmduRKmPj8YcIz7V4f53xJEntjweL4aoLyChkwk= +github.com/Azure/skewer v0.0.24 h1:6u+/o+LO6ufDnpk08KruOAeUDWm62Fi1DcbAY8EFdPA= +github.com/Azure/skewer v0.0.24/go.mod h1:KwKqaVxdP/XSYDvWiIxbvBRDcm+cQ8lRAwKRPNGsxMg= github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1 h1:WJTmL004Abzc5wDB5VtZG2PJk5ndYDgVacGqfirKxjM= github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1/go.mod h1:tCcJZ0uHAmvjsVYzEFivsRTN00oz5BEsRgQHu5JZ9WE= -github.com/AzureAD/microsoft-authentication-library-for-go v1.4.2 h1:oygO0locgZJe7PpYPXT5A29ZkwJaPqcva7BVeemZOZs= -github.com/AzureAD/microsoft-authentication-library-for-go v1.4.2/go.mod h1:wP83P5OoQ5p6ip3ScPr0BAq0BvuPAvacpEuSzyouqAI= -github.com/alecthomas/units v0.0.0-20211218093645-b94a6e3cc137 h1:s6gZFSlWYmbqAuRjVTiNNhvNRfY2Wxp9nhfyel4rklc= -github.com/alecthomas/units v0.0.0-20211218093645-b94a6e3cc137/go.mod h1:OMCwj8VM1Kc9e19TLln2VL61YJF0x1XFtfdL4JdbSyE= +github.com/AzureAD/microsoft-authentication-library-for-go v1.8.0 h1:Nljr4q1GRA/5vCrMONS+g4u4LRHNgOXVSh3O43J2CnI= +github.com/AzureAD/microsoft-authentication-library-for-go v1.8.0/go.mod h1:Y33QHnf0FfdVewFFISOGe20mkZbxX4H839o955/PoeI= +github.com/Masterminds/semver/v3 v3.5.0 h1:kQceYJfbupGfZOKZQg0kou0DgAKhzDg2NZPAwZ/2OOE= +github.com/Masterminds/semver/v3 v3.5.0/go.mod h1:4V+yj/TJE1HU9XfppCwVMZq3I84lprf4nC11bSS5beM= +github.com/Pallinder/go-randomdata v1.2.0 h1:DZ41wBchNRb/0GfsePLiSwb0PHZmT67XY00lCDlaYPg= +github.com/Pallinder/go-randomdata v1.2.0/go.mod h1:yHmJgulpD2Nfrm0cR9tI/+oAgRqCQQixsA8HyRZfV9Y= +github.com/alecthomas/units v0.0.0-20240927000941-0f3dac36c52b h1:mimo19zliBX/vSQ6PWWSL9lK8qwHozUj03+zLoEB8O0= +github.com/alecthomas/units v0.0.0-20240927000941-0f3dac36c52b/go.mod h1:fvzegU4vN3H1qMT+8wDmzjAcDONcgo2/SZ/TyfdUOFs= github.com/antlr4-go/antlr/v4 v4.13.1 h1:SqQKkuVZ+zWkMMNkjy5FZe5mr5WURWnlpmOuzYWrPrQ= github.com/antlr4-go/antlr/v4 v4.13.1/go.mod h1:GKmUxMtwp6ZgGwZSva4eWPC5mS6vUAmOABFgjdkM7Nw= -github.com/asaskevich/govalidator v0.0.0-20230301143203-a9d515a09cc2 h1:DklsrG3dyBCFEj5IhUbnKptjxatkF07cF2ak3yi77so= -github.com/asaskevich/govalidator v0.0.0-20230301143203-a9d515a09cc2/go.mod h1:WaHUgvxTVq04UNunO+XhnAqY/wQc+bxr74GqbsZ/Jqw= -github.com/awslabs/operatorpkg v0.0.0-20250425180727-b22281cd8057 h1:HfT+gl2sOiVU6sGWEWtWi+xuq4MLx25TibfSDMcuQi8= -github.com/awslabs/operatorpkg v0.0.0-20250425180727-b22281cd8057/go.mod h1:Ip8R3ED5KRLmiq2CmJdE+3UTlJAc5dQQBZHXU0W5bqM= +github.com/avast/retry-go v3.0.0+incompatible h1:4SOWQ7Qs+oroOTQOYnAHqelpCO0biHSxpiH9JdtuBj0= +github.com/avast/retry-go v3.0.0+incompatible/go.mod h1:XtSnn+n/sHqQIpZ10K1qAevBhOOCWBLXXy3hyiqqBrY= +github.com/awslabs/operatorpkg v0.0.0-20260708223819-4da4c353c5fa h1:SM2D/TaMO79MaSpBvRLm2Ed/SKbGCxuckRJBDDCGUck= +github.com/awslabs/operatorpkg v0.0.0-20260708223819-4da4c353c5fa/go.mod h1:G/D6aERpu0RWYGZC8wrAfrtEI6nWHAQIkXmTGnNn2jM= github.com/beorn7/perks v1.0.1 h1:VlbKKnNfV8bJzeqoa4cOKqO6bYr3WgKZxO8Z16+hsOM= github.com/beorn7/perks v1.0.1/go.mod h1:G2ZrVWU2WbWT9wwq4/hrbKbnv/1ERSJQ0ibhJ6rlkpw= github.com/blang/semver/v4 v4.0.0 h1:1PFHFE6yCCTv8C1TeyNNarDzntLi7wMI5i/pzqYIsAM= @@ -97,107 +110,147 @@ github.com/blang/semver/v4 v4.0.0/go.mod h1:IbckMUScFkM3pff0VJDNKRiT6TG/YpiHIM2y github.com/cespare/xxhash/v2 v2.3.0 h1:UL815xU9SqsFlibzuggzjXhog7bL6oX9BbNZnL2UFvs= github.com/cespare/xxhash/v2 v2.3.0/go.mod h1:VGX0DQ3Q6kWi7AoAeZDth3/j3BFtOZR5XLFGgcrjCOs= github.com/cpuguy83/go-md2man/v2 v2.0.6/go.mod h1:oOW0eioCTA6cOiMLiUPZOpcVxMig6NIQQ7OS05n1F4g= -github.com/crossplane/crossplane-runtime/v2 v2.1.0 h1:JBMhL9T+/PfyjLAQEdZWlKLvA3jJVtza8zLLwd9Gs4k= -github.com/crossplane/crossplane-runtime/v2 v2.1.0/go.mod h1:j78pmk0qlI//Ur7zHhqTr8iePHFcwJKrZnzZB+Fg4t0= +github.com/crossplane/crossplane-runtime/v2 v2.4.0 h1:m5iGVHbtEh9nZYz7w/wUWDsXVlpG2c8v5i0SZScwHh8= +github.com/crossplane/crossplane-runtime/v2 v2.4.0/go.mod h1:Pf+hg5A/46QaGjKi7Pk+HdsrFOA/THvST9qM4El8Rzs= +github.com/crossplane/crossplane/apis/v2 v2.3.4 h1:qB1YqF300msWEkDQTFUhqfeOq3HtVfSZ1+3wNz1JStA= +github.com/crossplane/crossplane/apis/v2 v2.3.4/go.mod h1:s0cfMZA+Y/dwQ577PLQn5SBWyUg9k6vKjGAkHEwBShg= github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc h1:U9qPSI2PIWSS1VwoXQT9A3Wy9MM3WgvqSxFWenqJduM= github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= -github.com/dgryski/go-rendezvous v0.0.0-20200823014737-9f7001d12a5f h1:lO4WD4F/rVNCu3HqELle0jiPLLBs70cWOduZpkS1E78= -github.com/dgryski/go-rendezvous v0.0.0-20200823014737-9f7001d12a5f/go.mod h1:cuUVRXasLTGF7a8hSLbxyZXjz+1KgoB3wDUb6vlszIc= -github.com/emicklei/go-restful/v3 v3.12.2 h1:DhwDP0vY3k8ZzE0RunuJy8GhNpPL6zqLkDf9B/a0/xU= -github.com/emicklei/go-restful/v3 v3.12.2/go.mod h1:6n3XBCmQQb25CM2LCACGz8ukIrRry+4bhvbpWn3mrbc= +github.com/emicklei/go-restful/v3 v3.13.0 h1:C4Bl2xDndpU6nJ4bc1jXd+uTmYPVUwkD6bFY/oTyCes= +github.com/emicklei/go-restful/v3 v3.13.0/go.mod h1:6n3XBCmQQb25CM2LCACGz8ukIrRry+4bhvbpWn3mrbc= github.com/evanphx/json-patch v5.9.11+incompatible h1:ixHHqfcGvxhWkniF1tWxBHA0yb4Z+d1UQi45df52xW8= github.com/evanphx/json-patch v5.9.11+incompatible/go.mod h1:50XU6AFN0ol/bzJsmQLiYLvXMP4fmwYFNcr97nuDLSk= github.com/evanphx/json-patch/v5 v5.9.11 h1:/8HVnzMq13/3x9TPvjG08wUGqBTmZBsCWzjTM0wiaDU= github.com/evanphx/json-patch/v5 v5.9.11/go.mod h1:3j+LviiESTElxA4p3EMKAB9HXj3/XEtnUf6OZxqIQTM= github.com/felixge/httpsnoop v1.0.4 h1:NFTV2Zj1bL4mc9sqWACXbQFVBBg2W3GPvqp8/ESS2Wg= github.com/felixge/httpsnoop v1.0.4/go.mod h1:m8KPJKqk1gH5J9DgRY2ASl2lWCfGKXixSwevea8zH2U= -github.com/fsnotify/fsnotify v1.9.0 h1:2Ml+OJNzbYCTzsxtv8vKSFD9PbJjmhYF14k/jKC7S9k= -github.com/fsnotify/fsnotify v1.9.0/go.mod h1:8jBTzvmWwFyi3Pb8djgCCO5IBqzKJ/Jwo8TRcHyHii0= +github.com/fsnotify/fsnotify v1.10.1 h1:b0/UzAf9yR5rhf3RPm9gf3ehBPpf0oZKIjtpKrx59Ho= +github.com/fsnotify/fsnotify v1.10.1/go.mod h1:TLheqan6HD6GBK6PrDWyDPBaEV8LspOxvPSjC+bVfgo= github.com/fxamacker/cbor/v2 v2.9.0 h1:NpKPmjDBgUfBms6tr6JZkTHtfFGcMKsw3eGcmD/sapM= github.com/fxamacker/cbor/v2 v2.9.0/go.mod h1:vM4b+DJCtHn+zz7h3FFp/hDAI9WNWCsZj23V5ytsSxQ= -github.com/gabriel-vasile/mimetype v1.4.8 h1:FfZ3gj38NjllZIeJAmMhr+qKL8Wu+nOoI3GqacKw1NM= -github.com/gabriel-vasile/mimetype v1.4.8/go.mod h1:ByKUIKGjh1ODkGM1asKUbQZOLGrPjydw3hYPU2YU9t8= +github.com/gabriel-vasile/mimetype v1.4.13 h1:46nXokslUBsAJE/wMsp5gtO500a4F3Nkz9Ufpk2AcUM= +github.com/gabriel-vasile/mimetype v1.4.13/go.mod h1:d+9Oxyo1wTzWdyVUPMmXFvp4F9tea18J8ufA774AB3s= +github.com/gkampitakis/ciinfo v0.3.2 h1:JcuOPk8ZU7nZQjdUhctuhQofk7BGHuIy0c9Ez8BNhXs= +github.com/gkampitakis/ciinfo v0.3.2/go.mod h1:1NIwaOcFChN4fa/B0hEBdAb6npDlFL8Bwx4dfRLRqAo= +github.com/gkampitakis/go-diff v1.3.2 h1:Qyn0J9XJSDTgnsgHRdz9Zp24RaJeKMUHg2+PDZZdC4M= +github.com/gkampitakis/go-diff v1.3.2/go.mod h1:LLgOrpqleQe26cte8s36HTWcTmMEur6OPYerdAAS9tk= +github.com/gkampitakis/go-snaps v0.5.15 h1:amyJrvM1D33cPHwVrjo9jQxX8g/7E2wYdZ+01KS3zGE= +github.com/gkampitakis/go-snaps v0.5.15/go.mod h1:HNpx/9GoKisdhw9AFOBT1N7DBs9DiHo/hGheFGBZ+mc= github.com/go-errors/errors v1.4.2 h1:J6MZopCL4uSllY1OfXM374weqZFFItUbrImctkmUxIA= github.com/go-errors/errors v1.4.2/go.mod h1:sIVyrIiJhuEF+Pj9Ebtd6P/rEYROXFi3BopGUQ5a5Og= -github.com/go-faker/faker/v4 v4.6.0 h1:6aOPzNptRiDwD14HuAnEtlTa+D1IfFuEHO8+vEFwjTs= -github.com/go-faker/faker/v4 v4.6.0/go.mod h1:ZmrHuVtTTm2Em9e0Du6CJ9CADaLEzGXW62z1YqFH0m0= -github.com/go-logr/logr v1.4.3 h1:CjnDlHq8ikf6E492q6eKboGOC0T8CDaOvkHCIg8idEI= -github.com/go-logr/logr v1.4.3/go.mod h1:9T104GzyrTigFIr8wt5mBrctHMim0Nb2HLGrmQ40KvY= +github.com/go-faker/faker/v4 v4.7.0 h1:VboC02cXHl/NuQh5lM2W8b87yp4iFXIu59x4w0RZi4E= +github.com/go-faker/faker/v4 v4.7.0/go.mod h1:u1dIRP5neLB6kTzgyVjdBOV5R1uP7BdxkcWk7tiKQXk= +github.com/go-logr/logr v1.4.4 h1:tG4xh9yMsRCAiodLVTxyrkzSZ9+o0L1Kg/+cPVcbP/8= +github.com/go-logr/logr v1.4.4/go.mod h1:9T104GzyrTigFIr8wt5mBrctHMim0Nb2HLGrmQ40KvY= github.com/go-logr/stdr v1.2.2 h1:hSWxHoqTgW2S2qGc0LTAI563KZ5YKYRhT3MFKZMbjag= github.com/go-logr/stdr v1.2.2/go.mod h1:mMo/vtBO5dYbehREoey6XUKy/eSumjCCveDpRre4VKE= github.com/go-logr/zapr v1.3.0 h1:XGdV8XW8zdwFiwOA2Dryh1gj2KRQyOOoNmBy4EplIcQ= github.com/go-logr/zapr v1.3.0/go.mod h1:YKepepNBd1u/oyhd/yQmtjVXmm9uML4IXUgMOwR8/Gg= -github.com/go-openapi/analysis v0.23.0 h1:aGday7OWupfMs+LbmLZG4k0MYXIANxcuBTYUC03zFCU= -github.com/go-openapi/analysis v0.23.0/go.mod h1:9mz9ZWaSlV8TvjQHLl2mUW2PbZtemkE8yA5v22ohupo= -github.com/go-openapi/errors v0.22.1 h1:kslMRRnK7NCb/CvR1q1VWuEQCEIsBGn5GgKD9e+HYhU= -github.com/go-openapi/errors v0.22.1/go.mod h1:+n/5UdIqdVnLIJ6Q9Se8HNGUXYaY6CN8ImWzfi/Gzp0= -github.com/go-openapi/jsonpointer v0.21.1 h1:whnzv/pNXtK2FbX/W9yJfRmE2gsmkfahjMKB0fZvcic= -github.com/go-openapi/jsonpointer v0.21.1/go.mod h1:50I1STOfbY1ycR8jGz8DaMeLCdXiI6aDteEdRNNzpdk= -github.com/go-openapi/jsonreference v0.21.0 h1:Rs+Y7hSXT83Jacb7kFyjn4ijOuVGSvOdF2+tg1TRrwQ= -github.com/go-openapi/jsonreference v0.21.0/go.mod h1:LmZmgsrTkVg9LG4EaHeY8cBDslNPMo06cago5JNLkm4= -github.com/go-openapi/loads v0.22.0 h1:ECPGd4jX1U6NApCGG1We+uEozOAvXvJSF4nnwHZ8Aco= -github.com/go-openapi/loads v0.22.0/go.mod h1:yLsaTCS92mnSAZX5WWoxszLj0u+Ojl+Zs5Stn1oF+rs= -github.com/go-openapi/runtime v0.28.0 h1:gpPPmWSNGo214l6n8hzdXYhPuJcGtziTOgUpvsFWGIQ= -github.com/go-openapi/runtime v0.28.0/go.mod h1:QN7OzcS+XuYmkQLw05akXk0jRH/eZ3kb18+1KwW9gyc= -github.com/go-openapi/spec v0.21.0 h1:LTVzPc3p/RzRnkQqLRndbAzjY0d0BCL72A6j3CdL9ZY= -github.com/go-openapi/spec v0.21.0/go.mod h1:78u6VdPw81XU44qEWGhtr982gJ5BWg2c0I5XwVMotYk= -github.com/go-openapi/strfmt v0.23.0 h1:nlUS6BCqcnAk0pyhi9Y+kdDVZdZMHfEKQiS4HaMgO/c= -github.com/go-openapi/strfmt v0.23.0/go.mod h1:NrtIpfKtWIygRkKVsxh7XQMDQW5HKQl6S5ik2elW+K4= -github.com/go-openapi/swag v0.23.1 h1:lpsStH0n2ittzTnbaSloVZLuB5+fvSY/+hnagBjSNZU= -github.com/go-openapi/swag v0.23.1/go.mod h1:STZs8TbRvEQQKUA+JZNAm3EWlgaOBGpyFDqQnDHMef0= -github.com/go-openapi/validate v0.24.0 h1:LdfDKwNbpB6Vn40xhTdNZAnfLECL81w+VX3BumrGD58= -github.com/go-openapi/validate v0.24.0/go.mod h1:iyeX1sEufmv3nPbBdX3ieNviWnOZaJ1+zquzJEf2BAQ= +github.com/go-openapi/analysis v0.25.5 h1:xPYEvTb90o1y0epuiOPAoG4QqahjP3cdp5xNlHeKJRI= +github.com/go-openapi/analysis v0.25.5/go.mod h1:d3UGtQC5uq5Kqqqis2VH09Km/v3vwsWrYkbp4gdm+Rc= +github.com/go-openapi/errors v0.22.8 h1:oP7sW7TWc3wFFjrzzj0nI83H2qMBkNjNfSd+XRejk/I= +github.com/go-openapi/errors v0.22.8/go.mod h1:BuUoHcYrU6E7V9gfj1I5wLQqgtIHnup/alXZ8KdgQ0w= +github.com/go-openapi/jsonpointer v1.0.0 h1:kR9tHqY0CtZaOPVFm622dPVNhrvYpwr4uCxgL3h1H8s= +github.com/go-openapi/jsonpointer v1.0.0/go.mod h1:Z3rw7dWu1p9IgitXCFamSlA5lmDiklEB6vkaxcNZW5Y= +github.com/go-openapi/jsonreference v1.0.0 h1:jlmTr6torcd1YgDQvSfNmRtKzYDO4FGBkrAdlAVWnpY= +github.com/go-openapi/jsonreference v1.0.0/go.mod h1:jtwdyGbJk0Xhe5Y+rwtglQP6Sb1WZST4rT32LWB+sv0= +github.com/go-openapi/loads v0.25.0 h1:74Bc2snfaVlsHzwdQj/3gsA9XJz3daXTJVs+4ZaK7jI= +github.com/go-openapi/loads v0.25.0/go.mod h1:JFBw4SIB9+PTIFHDfcXuSSy5h6aWzjtUCrPYyx3qWU8= +github.com/go-openapi/runtime v0.33.0 h1:Dd3Oj2ig+WH8ckK95l0Wn2V8a4bH/UqWPRZVT0vc8yU= +github.com/go-openapi/runtime v0.33.0/go.mod h1:+rsupH3+TFKqmFysqkmgBOTxpVJV8eV+j9myvvea2Xw= +github.com/go-openapi/runtime/server-middleware v0.30.0 h1:8rPoJ/xv7JL8BsovaqboKETlpWBArVh8n+0L/GyePog= +github.com/go-openapi/runtime/server-middleware v0.30.0/go.mod h1:OYNT/TxNvB/VK5oe4htM2jDTwlEXuejVJmu0DVZfAMs= +github.com/go-openapi/spec v0.22.9 h1:/vKIFDcGKp0ktZWGbym/tJEWbk6/XOEmAVU0kqKMH+w= +github.com/go-openapi/spec v0.22.9/go.mod h1:b/mNUYIOQOyIiUzUzXEE8xzyZqf93KvM9hQGP91yfl0= +github.com/go-openapi/strfmt v0.27.0 h1:kbcTeaD9TXuXD0hhMXzuYa1sdTo6+dWGvwjW93E80IM= +github.com/go-openapi/strfmt v0.27.0/go.mod h1:s/qhDqfY72irigXUGJmtgid2Rm+3tnz3k8hZaRmvWYc= +github.com/go-openapi/swag v0.28.0 h1:xkgbOSKj6DZziNpyqRRAOt3GJGtgjgsd2RoyT30VWuw= +github.com/go-openapi/swag v0.28.0/go.mod h1:4qYnT3Cqr1p1VknOdPo70evN4rgQnAg6jwApHyxSGIg= +github.com/go-openapi/swag/cmdutils v0.28.0 h1:7TOeNtkYru1SG8Y34tDh9WBbLsMqGnptuxWiHREPZ4Q= +github.com/go-openapi/swag/cmdutils v0.28.0/go.mod h1:Sm1MVFMkF6guJJ+pQqHnQA3N0j9qALV3NxzDSv6bETM= +github.com/go-openapi/swag/conv v0.28.0 h1:GtqqbyFe7vR5Y7ehxG9W6/OvrSFdf1OLeTGp40TqxH8= +github.com/go-openapi/swag/conv v0.28.0/go.mod h1:mbUE+mzctnhxi864m0Q07SpN8OowD9JhxmxuYvZZD/k= +github.com/go-openapi/swag/fileutils v0.28.0 h1:Z04XWQD7R8Eq+7GnOrjovBxPPmZzsS4gt2H2GPGIViU= +github.com/go-openapi/swag/fileutils v0.28.0/go.mod h1:VvJFZLTZS0AI854gEQz5tk7dBESdLjiNUMSZ/th2ry8= +github.com/go-openapi/swag/jsonutils v0.28.0 h1:YIch6FwO7RXzeAnbO8Tu7dWBZeUEH+4nA0HXltVTnv4= +github.com/go-openapi/swag/jsonutils v0.28.0/go.mod h1:CYM3WlTUcagR2ZoHdz54di/cbBqt82tuxuXgAjxw+mg= +github.com/go-openapi/swag/jsonutils/fixtures_test v0.28.0 h1:qV+VVUAx5Oro8WjVWpZeql7YReTKhT4smR4zhcOQZr0= +github.com/go-openapi/swag/jsonutils/fixtures_test v0.28.0/go.mod h1:mofwUWx70wvskwESqRJ//k/9kURmCgyJl5m5Ppoh5kY= +github.com/go-openapi/swag/loading v0.28.0 h1:td8QZdZC9MIYGGSnSPKShKiK22I2tU5UQvuUhIBPRLU= +github.com/go-openapi/swag/loading v0.28.0/go.mod h1:rXB0QiQX5mMveXEA7ouM4KiiM9jVJe4K6BVbwhD1M4k= +github.com/go-openapi/swag/mangling v0.28.0 h1:pH8eyeNO9SLYsTMWJrurnNfKmDa28XrlA+HePVD53VM= +github.com/go-openapi/swag/mangling v0.28.0/go.mod h1:jtBE2+V+3pILxOR7Vgce+Cwp6A2PgZbvVqfNntbVs0w= +github.com/go-openapi/swag/netutils v0.28.0 h1:YXN6TALEi2pzts8/8GNm6T61HTAZsieukGZidap989k= +github.com/go-openapi/swag/netutils v0.28.0/go.mod h1:J+WYyFMLtvtCGqa6jLv+YNUmIKI3ZRQRrvfNDMoQoEQ= +github.com/go-openapi/swag/pools v0.28.0 h1:HPMZWSAfce3rdVTFcjFiCIBtDg9h4x2QlRrHipwhxeU= +github.com/go-openapi/swag/pools v0.28.0/go.mod h1:kVQefhSK5RWuRe7BXsL8htgBPAMpN7HDGpGEknqugeE= +github.com/go-openapi/swag/stringutils v0.28.0 h1:ixsc9iYgDPubHL/8nSkbnryEHpD2VRlBMLKpQyPXcDU= +github.com/go-openapi/swag/stringutils v0.28.0/go.mod h1:lzRN95CxXmA03XcDWHLOb6nOMcxCqR5rGY0lOgsfRoM= +github.com/go-openapi/swag/typeutils v0.28.0 h1:nRBKSBXjDgf01VDPB3fWeD9nQuhCOVeIYAkUx2tbkyY= +github.com/go-openapi/swag/typeutils v0.28.0/go.mod h1:Srm0xFNRZ1Y+vCxJclo5qzx8aj+1pAKda/YfFPrG0dQ= +github.com/go-openapi/swag/yamlutils v0.28.0 h1:TV3JXH6DS46KUroDtMLAYHGkdWf5VDq3wVWFirmzROY= +github.com/go-openapi/swag/yamlutils v0.28.0/go.mod h1:x0q/yndZHEgk9Rx3DyDqzFUmHy55KTvIZldvF2dTJXs= +github.com/go-openapi/testify/enable/yaml/v2 v2.6.0 h1:gGHwAJ0R/5jU8BEGDbfRNR3hL68dAVi84WuOApp29B0= +github.com/go-openapi/testify/enable/yaml/v2 v2.6.0/go.mod h1:tY+St1SGq4NFl0QIqdTY4aEdbChAHxhyB77XQi9iJCo= +github.com/go-openapi/testify/v2 v2.6.0 h1:5PKH2HE7YJ/LuRPQGvSxBRlFXNQhSetBLlGAgUEu3ug= +github.com/go-openapi/testify/v2 v2.6.0/go.mod h1:SgsVHtfooshd0tublTtJ50FPKhujf47YRqauXXOUxfw= +github.com/go-openapi/validate v0.26.1 h1:pZSbvtRO8G2R2FpWTYRn3w8LrsNwbtaVhP2dWiBa0Us= +github.com/go-openapi/validate v0.26.1/go.mod h1:B8UMgXiQiwwQWIbmuROlwJZDPGlikPuh7iHV1vPX9Oo= github.com/go-playground/locales v0.14.1 h1:EWaQ/wswjilfKLTECiXz7Rh+3BjFhfDFKv/oXslEjJA= github.com/go-playground/locales v0.14.1/go.mod h1:hxrqLVvrK65+Rwrd5Fc6F2O76J/NuW9t0sjnWqG1slY= github.com/go-playground/universal-translator v0.18.1 h1:Bcnm0ZwsGyWbCzImXv+pAJnYK9S473LQFuzCbDbfSFY= github.com/go-playground/universal-translator v0.18.1/go.mod h1:xekY+UJKNuX9WP91TpwSH2VMlDf28Uj24BCp08ZFTUY= -github.com/go-playground/validator/v10 v10.26.0 h1:SP05Nqhjcvz81uJaRfEV0YBSSSGMc/iMaVtFbr3Sw2k= -github.com/go-playground/validator/v10 v10.26.0/go.mod h1:I5QpIEbmr8On7W0TktmJAumgzX4CA1XNl4ZmDuVHKKo= +github.com/go-playground/validator/v10 v10.30.3 h1:4MU6YkEwx7GbcPJOZxrtbu+QfF3pJLJuaYTeAH0DYy8= +github.com/go-playground/validator/v10 v10.30.3/go.mod h1:4Axh7oCNGcoGkqLoE4YWt6n20mcEIsPRlB7vPk3lpyc= github.com/go-task/slim-sprig/v3 v3.0.0 h1:sUs3vkvUymDpBKi3qH1YSqBQk9+9D/8M2mN1vB6EwHI= github.com/go-task/slim-sprig/v3 v3.0.0/go.mod h1:W848ghGpv3Qj3dhTPRyJypKRiqCdHZiAzKg9hl15HA8= +github.com/go-viper/mapstructure/v2 v2.5.0 h1:vM5IJoUAy3d7zRSVtIwQgBj7BiWtMPfmPEgAXnvj1Ro= +github.com/go-viper/mapstructure/v2 v2.5.0/go.mod h1:oJDH3BJKyqBA2TXFhDsKDGDTlndYOZ6rGS0BRZIxGhM= +github.com/goccy/go-yaml v1.18.0 h1:8W7wMFS12Pcas7KU+VVkaiCng+kG8QiFeFwzFb+rwuw= +github.com/goccy/go-yaml v1.18.0/go.mod h1:XBurs7gK8ATbW4ZPGKgcbrY1Br56PdM69F7LkFRi1kA= github.com/gofrs/uuid v4.4.0+incompatible h1:3qXRTX8/NbyulANqlc0lchS1gqAVxRgsuW1YrTJupqA= github.com/gofrs/uuid v4.4.0+incompatible/go.mod h1:b2aQJv3Z4Fp6yNu3cdSllBxTCLRxnplIgP/c0N/04lM= -github.com/gogo/protobuf v1.3.2 h1:Ov1cvc58UF3b5XjBnZv7+opcTcQFZebYjWzi34vdm4Q= -github.com/gogo/protobuf v1.3.2/go.mod h1:P1XiOD3dCwIKUDQYPy72D8LYyHL2YPYrpS2s69NZV8Q= +github.com/golang-jwt/jwt/v4 v4.0.0/go.mod h1:/xlHOz8bRuivTWchD4jCa+NbatV+wEUSzwAxVc6locg= +github.com/golang-jwt/jwt/v4 v4.5.0/go.mod h1:m21LjoU+eqJr34lmDMbreY2eSTRJ1cv77w39/MY0Ch0= github.com/golang-jwt/jwt/v4 v4.5.2 h1:YtQM7lnr8iZ+j5q71MGKkNw9Mn7AjHM68uc9g5fXeUI= github.com/golang-jwt/jwt/v4 v4.5.2/go.mod h1:m21LjoU+eqJr34lmDMbreY2eSTRJ1cv77w39/MY0Ch0= -github.com/golang-jwt/jwt/v5 v5.2.2 h1:Rl4B7itRWVtYIHFrSNd7vhTiz9UpLdi6gZhZ3wEeDy8= -github.com/golang-jwt/jwt/v5 v5.2.2/go.mod h1:pqrtFR0X4osieyHYxtmOUWsAWrfe1Q5UVIyoH402zdk= +github.com/golang-jwt/jwt/v5 v5.3.1 h1:kYf81DTWFe7t+1VvL7eS+jKFVWaUnK9cB1qbwn63YCY= +github.com/golang-jwt/jwt/v5 v5.3.1/go.mod h1:fxCRLWMO43lRc8nhHWY6LGqRcf+1gQWArsqaEUEa5bE= github.com/golang/protobuf v1.5.4 h1:i7eJL8qZTpSEXOPTxNKhASYpMn+8e5Q6AdndVa1dWek= github.com/golang/protobuf v1.5.4/go.mod h1:lnTiLA8Wa4RWRcIUkrtSVa5nRhsEGBg48fD6rSs7xps= github.com/google/btree v1.1.3 h1:CVpQJjYgC4VbzxeGVHfvZrv1ctoYCAI8vbl07Fcxlyg= github.com/google/btree v1.1.3/go.mod h1:qOPhT0dTNdNzV6Z/lhRX0YXUafgPLFUh+gZMl761Gm4= -github.com/google/gnostic-models v0.7.0 h1:qwTtogB15McXDaNqTZdzPJRHvaVJlAl+HVQnLmJEJxo= -github.com/google/gnostic-models v0.7.0/go.mod h1:whL5G0m6dmc5cPxKc5bdKdEN3UjI7OUGxBlw57miDrQ= +github.com/google/cel-go v0.30.0 h1:ll54AkzKunWkBn9wSoiUXbFZXYZTkdJGNXTBXUoolGo= +github.com/google/cel-go v0.30.0/go.mod h1:X0bD6iVNR8pkROSOoHVdgTkzmRcosof7WQqCD6wcMc8= +github.com/google/gnostic-models v0.7.1 h1:SisTfuFKJSKM5CPZkffwi6coztzzeYUhc3v4yxLWH8c= +github.com/google/gnostic-models v0.7.1/go.mod h1:whL5G0m6dmc5cPxKc5bdKdEN3UjI7OUGxBlw57miDrQ= github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8= github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU= github.com/google/gofuzz v1.0.0/go.mod h1:dBl0BpW6vV/+mYPU4Po3pmUjxk6FQPldtuIdl/M65Eg= github.com/google/gofuzz v1.2.0 h1:xRy4A+RhZaiKjJ1bPfwQ8sedCA+YS2YcCHW6ec7JMi0= github.com/google/gofuzz v1.2.0/go.mod h1:dBl0BpW6vV/+mYPU4Po3pmUjxk6FQPldtuIdl/M65Eg= -github.com/google/pprof v0.0.0-20250403155104-27863c87afa6 h1:BHT72Gu3keYf3ZEu2J0b1vyeLSOYI8bm5wbJM/8yDe8= -github.com/google/pprof v0.0.0-20250403155104-27863c87afa6/go.mod h1:boTsfXsheKC2y+lKOCMpSfarhxDeIzfZG1jqGcPl3cA= -github.com/google/shlex v0.0.0-20191202100458-e7afc7fbc510 h1:El6M4kTTCOh6aBiKaUGG7oYTSPP8MxqL4YI3kZKwcP4= -github.com/google/shlex v0.0.0-20191202100458-e7afc7fbc510/go.mod h1:pupxD2MaaD3pAXIBCelhxNneeOaAeabZDe5s4K6zSpQ= +github.com/google/pprof v0.0.0-20260402051712-545e8a4df936 h1:EwtI+Al+DeppwYX2oXJCETMO23COyaKGP6fHVpkpWpg= +github.com/google/pprof v0.0.0-20260402051712-545e8a4df936/go.mod h1:MxpfABSjhmINe3F1It9d+8exIHFvUqtLIRCdOGNXqiI= github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo= -github.com/grpc-ecosystem/grpc-gateway/v2 v2.26.3 h1:5ZPtiqj0JL5oKWmcsq4VMaAW5ukBEgSGXEN89zeH1Jo= -github.com/grpc-ecosystem/grpc-gateway/v2 v2.26.3/go.mod h1:ndYquD05frm2vACXE1nsccT4oJzjhw2arTS2cpUD1PI= +github.com/grpc-ecosystem/grpc-gateway/v2 v2.29.0 h1:5VipnvEpbqr2gA2VbM+nYVbkIF28c5ZQfqCBQ5g2xfk= +github.com/grpc-ecosystem/grpc-gateway/v2 v2.29.0/go.mod h1:Hyl3n6Twe1hvtd9XUXDec4pTvgMSEixRuQKPTMH2bNs= +github.com/imdario/mergo v0.3.16 h1:wwQJbIsHYGMUyLSPrEq1CT16AhnhNJQ51+4fdHUnCl4= +github.com/imdario/mergo v0.3.16/go.mod h1:WBLT9ZmE3lPoWsEzCh9LPo3TiwVN+ZKEjmz+hD27ysY= github.com/inconshreveable/mousetrap v1.1.0 h1:wN+x4NVGpMsO7ErUn/mUI3vEoE6Jt13X2s0bqwp9tc8= github.com/inconshreveable/mousetrap v1.1.0/go.mod h1:vpF70FUmC8bwa3OWnCshd2FqLfsEA9PFc4w1p2J65bw= github.com/jongio/azidext/go/azidext v0.5.0 h1:uPInXD4NZ3J0k79FPwIA0YXknFn+WcqZqSgs3/jPgvQ= github.com/jongio/azidext/go/azidext v0.5.0/go.mod h1:TVRX/hJhzbsCKaOIzicH6a8IvOH0hpjWk/JwZZgtXeU= -github.com/josharian/intern v1.0.0 h1:vlS4z54oSdjm0bgjRigI+G1HpF+tI+9rE5LLzOg8HmY= -github.com/josharian/intern v1.0.0/go.mod h1:5DoeVV0s6jJacbCEi61lwdGj/aVlrQvzHFFd8Hwg//Y= +github.com/joshdk/go-junit v1.0.0 h1:S86cUKIdwBHWwA6xCmFlf3RTLfVXYQfvanM5Uh+K6GE= +github.com/joshdk/go-junit v1.0.0/go.mod h1:TiiV0PqkaNfFXjEiyjWM3XXrhVyCa1K4Zfga6W52ung= github.com/json-iterator/go v1.1.12 h1:PV8peI4a0ysnczrg+LtxykD8LfKY9ML6u2jnxaEnrnM= github.com/json-iterator/go v1.1.12/go.mod h1:e30LSqwooZae/UwlEbR2852Gd8hjQvJoHmT4TnhNGBo= github.com/keybase/go-keychain v0.0.1 h1:way+bWYa6lDppZoZcgMbYsvC7GxljxrskdNInRtuthU= github.com/keybase/go-keychain v0.0.1/go.mod h1:PdEILRW3i9D8JcdM+FmY6RwkHGnhHxXwkPPMeUgOK1k= -github.com/kisielk/errcheck v1.5.0/go.mod h1:pFxgyoBC7bSaBwPgfKdkLd5X25qrDl4LWUI2bnpBCr8= -github.com/kisielk/gotool v1.0.0/go.mod h1:XhKaO+MFFWcvkIS/tQcRk01m1F5IRFswLeQ+oQHNcck= -github.com/klauspost/compress v1.18.0 h1:c/Cqfb0r+Yi+JtIEq73FWXVkRonBlf0CRNYc8Zttxdo= -github.com/klauspost/compress v1.18.0/go.mod h1:2Pp+KzxcywXVXMr50+X0Q/Lsb43OQHYWRCY2AiWywWQ= +github.com/klauspost/compress v1.19.1 h1:VsB4HPswih7mmZ8WleSFQ75c/Ui1M4trX5oAsJnhSlk= +github.com/klauspost/compress v1.19.1/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ= github.com/kr/pretty v0.3.1 h1:flRD4NNwYAUpkphVc1HcthR4KEIFJ65n8Mw5qdRn3LE= github.com/kr/pretty v0.3.1/go.mod h1:hoEshYVHaxMs3cyo3Yncou5ZscifuDolrwPKZanG3xk= github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY= @@ -206,12 +259,12 @@ github.com/kylelemons/godebug v1.1.0 h1:RPNrshWIDI6G2gRW9EHilWtl7Z6Sb1BR0xunSBf0 github.com/kylelemons/godebug v1.1.0/go.mod h1:9/0rRGxNHcop5bhtWyNeEfOS8JIWk580+fNqagV/RAw= github.com/leodido/go-urn v1.4.0 h1:WT9HwE9SGECu3lg4d/dIA+jxlljEa1/ffXKmRjqdmIQ= github.com/leodido/go-urn v1.4.0/go.mod h1:bvxc+MVxLKB4z00jd1z+Dvzr47oO32F/QSNjSBOlFxI= -github.com/mailru/easyjson v0.9.0 h1:PrnmzHw7262yW8sTBwxi1PdJA3Iw/EKBa8psRf7d9a4= -github.com/mailru/easyjson v0.9.0/go.mod h1:1+xMtQp2MRNVL/V1bOzuP3aP8VNwRW55fQUto+XFtTU= +github.com/maruel/natural v1.1.1 h1:Hja7XhhmvEFhcByqDoHz9QZbkWey+COd9xWfCfn1ioo= +github.com/maruel/natural v1.1.1/go.mod h1:v+Rfd79xlw1AgVBjbO0BEQmptqb5HvL/k9GRHB7ZKEg= +github.com/mfridman/tparse v0.18.0 h1:wh6dzOKaIwkUGyKgOntDW4liXSo37qg5AXbIhkMV3vE= +github.com/mfridman/tparse v0.18.0/go.mod h1:gEvqZTuCgEhPbYk/2lS3Kcxg1GmTxxU7kTC8DvP0i/A= github.com/mitchellh/hashstructure/v2 v2.0.2 h1:vGKWl0YJqUNxE8d+h8f6NJLcCJrgbhC4NcD46KavDd4= github.com/mitchellh/hashstructure/v2 v2.0.2/go.mod h1:MG3aRVU/N29oo/V/IhBX8GR/zz4kQkprJgF2EVszyDE= -github.com/mitchellh/mapstructure v1.5.0 h1:jeMsZIYE/09sWLaz43PL7Gy6RuMjD2eJVyuac5Z2hdY= -github.com/mitchellh/mapstructure v1.5.0/go.mod h1:bFUtVrKA4DC2yAKiSyO/QUcy7e+RRV2QTWOzhPopBRo= github.com/modern-go/concurrent v0.0.0-20180228061459-e0a39a4cb421/go.mod h1:6dJC0mAP4ikYIbvyc7fijjWJddQyLn8Ig3JB5CqoB9Q= github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd h1:TRLaZ9cD/w8PVh93nsPXa1VrQ6jlwL5oN8l14QlcNfg= github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd/go.mod h1:6dJC0mAP4ikYIbvyc7fijjWJddQyLn8Ig3JB5CqoB9Q= @@ -222,14 +275,12 @@ github.com/monochromegane/go-gitignore v0.0.0-20200626010858-205db1a8cc00 h1:n6/ github.com/monochromegane/go-gitignore v0.0.0-20200626010858-205db1a8cc00/go.mod h1:Pm3mSP3c5uWn86xMLZ5Sa7JB9GsEZySvHYXCTK4E9q4= github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA= github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ= -github.com/oklog/ulid v1.3.1 h1:EGfNDEx6MqHz8B3uNV6QAib1UR2Lm97sHi3ocA6ESJ4= -github.com/oklog/ulid v1.3.1/go.mod h1:CirwcVhetQ6Lv90oh/F+FBtV6XMibvdAFo93nm5qn4U= -github.com/onsi/ginkgo/v2 v2.23.4 h1:ktYTpKJAVZnDT4VjxSbiBenUjmlL/5QkBEocaWXiQus= -github.com/onsi/ginkgo/v2 v2.23.4/go.mod h1:Bt66ApGPBFzHyR+JO10Zbt0Gsp4uWxu5mIOTusL46e8= -github.com/onsi/gomega v1.37.0 h1:CdEG8g0S133B4OswTDC/5XPSzE1OeP29QOioj2PID2Y= -github.com/onsi/gomega v1.37.0/go.mod h1:8D9+Txp43QWKhM24yyOBEdpkzN8FvJyAwecBgsU4KU0= -github.com/opentracing/opentracing-go v1.2.0 h1:uEJPy/1a5RIPAJ0Ov+OIO8OxWu77jEv+1B0VhjKrZUs= -github.com/opentracing/opentracing-go v1.2.0/go.mod h1:GxEUsuufX4nBwe+T+Wl9TAgYrxe9dPLANfrWvHYVTgc= +github.com/oklog/ulid/v2 v2.1.1 h1:suPZ4ARWLOJLegGFiZZ1dFAkqzhMjL3J1TzI+5wHz8s= +github.com/oklog/ulid/v2 v2.1.1/go.mod h1:rcEKHmBBKfef9DhnvX7y1HZBYxjXb0cP5ExxNsTT1QQ= +github.com/onsi/ginkgo/v2 v2.32.2 h1:2o6vyFvR6snrJWgRVztC+OwuqqPEMI1UzYl2s2iU7Cg= +github.com/onsi/ginkgo/v2 v2.32.2/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44= +github.com/onsi/gomega v1.43.0 h1:VlG/1FxqNxhSO+lq/OHBNaaqwiBK/mO8JbVkX9Y+FeU= +github.com/onsi/gomega v1.43.0/go.mod h1:REff/hsDsodHoKlWsP2mAPhu1+5/6hVYNf9rIEBpeSg= github.com/patrickmn/go-cache v2.1.0+incompatible h1:HRMgzkcYKYpi3C8ajMPV8OFXaaRUnok+kx1WdO15EQc= github.com/patrickmn/go-cache v2.1.0+incompatible/go.mod h1:3Qf8kWWT7OJRJbdiICTKqZju1ZixQ/KpMGzzAfe6+WQ= github.com/pkg/browser v0.0.0-20240102092130-5ac0b6a4141c h1:+mdjkGKdHQG3305AYmdv1U2eRNDiU2ErMBj1gwrq8eQ= @@ -239,42 +290,44 @@ github.com/pkg/errors v0.9.1/go.mod h1:bwawxfHBFNV+L2hUp1rHADufV3IMtnDRdf1r5NINE github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4= github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2 h1:Jamvg5psRIccs7FGNTlIRMkT8wgtp5eCXdBlqhYGL6U= github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4= -github.com/prashantv/gostub v1.1.0 h1:BTyx3RfQjRHnUWaGF9oQos79AlQ5k8WNktv7VGvVH4g= -github.com/prashantv/gostub v1.1.0/go.mod h1:A5zLQHz7ieHGG7is6LLXLz7I8+3LZzsrV0P1IAHhP5U= -github.com/prometheus/client_golang v1.22.0 h1:rb93p9lokFEsctTys46VnV1kLCDpVZ0a/Y92Vm0Zc6Q= -github.com/prometheus/client_golang v1.22.0/go.mod h1:R7ljNsLXhuQXYZYtw6GAE9AZg8Y7vEW5scdCXrWRXC0= -github.com/prometheus/client_model v0.6.2 h1:oBsgwpGs7iVziMvrGhE53c/GrLUsZdHnqNwqPLxwZyk= -github.com/prometheus/client_model v0.6.2/go.mod h1:y3m2F6Gdpfy6Ut/GBsUqTWZqCUvMVzSfMLjcu6wAwpE= -github.com/prometheus/common v0.62.0 h1:xasJaQlnWAeyHdUBeGjXmutelfJHWMRr+Fg4QszZ2Io= -github.com/prometheus/common v0.62.0/go.mod h1:vyBcEuLSvWos9B1+CyL7JZ2up+uFzXhkqml0W5zIY1I= -github.com/prometheus/procfs v0.15.1 h1:YagwOFzUgYfKKHX6Dr+sHT7km/hxC76UB0learggepc= -github.com/prometheus/procfs v0.15.1/go.mod h1:fB45yRUv8NstnjriLhBQLuOUt+WW4BsoGhij/e3PBqk= +github.com/prometheus/client_golang v1.24.1 h1:JnJkREXzWxUdCuPFpIWZiPispT9xVV59uiuyR2bPlnU= +github.com/prometheus/client_golang v1.24.1/go.mod h1:F+oSRECHg4sse5ucfYpYDeIv/hu68Zo0uoHKetWnzcE= +github.com/prometheus/client_model v0.6.3 h1:O0jaTVAYNxTHYInEPFJt5I3+sN8zqBtVMPTB1qyxiEo= +github.com/prometheus/client_model v0.6.3/go.mod h1:gpN5P9S7Rr6Yr92PiQ+Ixvhf6JZEkF1dnxsYL2aPBEM= +github.com/prometheus/common v0.70.1 h1:1HvjP4D5oL3t8RsPlwxA9onvvStjtIHYE5XuuwOi/PY= +github.com/prometheus/common v0.70.1/go.mod h1:VdFUQDMZK3VLkurFUVhia6uys/0suUp86TJz5qbJRhc= +github.com/prometheus/procfs v0.21.1 h1:GljZCt+zSTS+NZq88cyQ1LjZ+RCHp3uVuabBWA5+OJI= +github.com/prometheus/procfs v0.21.1/go.mod h1:aB55Cww9pdSJVHk0hUf0inxWyyjPogFIjmHKYgMKmtY= github.com/qri-io/jsonpointer v0.1.1 h1:prVZBZLL6TW5vsSB9fFHFAMBLI4b0ri5vribQlTJiBA= github.com/qri-io/jsonpointer v0.1.1/go.mod h1:DnJPaYgiKu56EuDp8TU5wFLdZIcAnb/uH9v37ZaMV64= -github.com/redis/go-redis/v9 v9.8.0 h1:q3nRvjrlge/6UD7eTu/DSg2uYiU2mCL0G/uzBWqhicI= -github.com/redis/go-redis/v9 v9.8.0/go.mod h1:huWgSWd8mW6+m0VPhJjSSQ+d6Nh1VICQ6Q5lHuCH/Iw= github.com/robfig/cron/v3 v3.0.1 h1:WdRxkvbJztn8LMz/QEvLN5sBU+xKpSqwwUO1Pjr4qDs= github.com/robfig/cron/v3 v3.0.1/go.mod h1:eQICP3HwyT7UooqI/z+Ov+PtYAWygg1TEWWzGIFLtro= -github.com/rogpeppe/go-internal v1.13.1 h1:KvO1DLK/DRN07sQ1LQKScxyZJuNnedQ5/wKSR38lUII= -github.com/rogpeppe/go-internal v1.13.1/go.mod h1:uMEvuHeurkdAXX61udpOXGD/AzZDWNMNyH2VO9fmH0o= +github.com/rogpeppe/go-internal v1.14.1 h1:UQB4HGPB6osV0SQTLymcB4TgvyWu6ZyliaW0tI/otEQ= +github.com/rogpeppe/go-internal v1.14.1/go.mod h1:MaRKkUm5W0goXpeCfT7UZI6fk/L7L7so1lCWt35ZSgc= github.com/russross/blackfriday/v2 v2.1.0/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM= -github.com/samber/lo v1.51.0 h1:kysRYLbHy/MB7kQZf5DSN50JHmMsNEdeY24VzJFu7wI= -github.com/samber/lo v1.51.0/go.mod h1:4+MXEGsJzbKGaUEQFKBq2xtfuznW9oz/WrgyzMzRoM0= +github.com/samber/lo v1.53.0 h1:t975lj2py4kJPQ6haz1QMgtId2gtmfktACxIXArw3HM= +github.com/samber/lo v1.53.0/go.mod h1:4+MXEGsJzbKGaUEQFKBq2xtfuznW9oz/WrgyzMzRoM0= github.com/sergi/go-diff v1.2.0 h1:XU+rvMAioB0UC3q1MFrIQy4Vo5/4VsRDQQXHsEya6xQ= github.com/sergi/go-diff v1.2.0/go.mod h1:STckp+ISIX8hZLjrqAeVduY0gWCT9IjLuqbuNXdaHfM= github.com/shopspring/decimal v1.4.0 h1:bxl37RwXBklmTi0C79JfXCEBD1cqqHt0bbgBAGFp81k= github.com/shopspring/decimal v1.4.0/go.mod h1:gawqmDU56v4yIKSwfBSFip1HdCCXN8/+DMd9qYNcwME= -github.com/spf13/cobra v1.9.1 h1:CXSaggrXdbHK9CF+8ywj8Amf7PBRmPCOJugH954Nnlo= -github.com/spf13/cobra v1.9.1/go.mod h1:nDyEzZ8ogv936Cinf6g1RU9MRY64Ir93oCnqb9wxYW0= -github.com/spf13/pflag v1.0.6 h1:jFzHGLGAlb3ruxLB8MhbI6A8+AQX/2eW4qeyNZXNp2o= -github.com/spf13/pflag v1.0.6/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg= +github.com/spf13/cobra v1.10.2 h1:DMTTonx5m65Ic0GOoRY2c16WCbHxOOw6xxezuLaBpcU= +github.com/spf13/cobra v1.10.2/go.mod h1:7C1pvHqHw5A4vrJfjNwvOdzYu0Gml16OCs2GRiTUUS4= +github.com/spf13/pflag v1.0.9/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg= +github.com/spf13/pflag v1.0.10 h1:4EBh2KAYBwaONj6b2Ye1GiHfwjqyROoF4RwYO+vPwFk= +github.com/spf13/pflag v1.0.10/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg= github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME= -github.com/stretchr/objx v0.5.2 h1:xuMeJ0Sdp5ZMRXx/aWO6RZxdr3beISkG5/G/aIRr3pY= -github.com/stretchr/objx v0.5.2/go.mod h1:FRsXN1f5AsAjCGJKqEizvkpNtU+EGNCLh3NxZ/8L+MA= +github.com/stretchr/objx v0.4.0/go.mod h1:YvHI0jy2hoMjB+UWwv71VJQ9isScKT/TqJzVSSt89Yw= +github.com/stretchr/objx v0.5.0/go.mod h1:Yh+to48EsGEfYuaHDzXPcE3xhTkx73EhmCGUpEOglKo= +github.com/stretchr/objx v0.5.3 h1:jmXUvGomnU1o3W/V5h2VEradbpJDwGrzugQQvL0POH4= +github.com/stretchr/objx v0.5.3/go.mod h1:rDQraq+vQZU7Fde9LOZLr8Tax6zZvy4kuNKF+QYS+U0= github.com/stretchr/testify v1.3.0/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI= github.com/stretchr/testify v1.7.0/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg= -github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U= -github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U= +github.com/stretchr/testify v1.7.1/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg= +github.com/stretchr/testify v1.8.0/go.mod h1:yNjHg4UonilssWZ8iaSj1OCr/vHnekPRkoO+kdMU+MU= +github.com/stretchr/testify v1.8.2/go.mod h1:w2LPCIKwWwSfY2zedu0+kehJoqGctiVI29o6fzry7u4= +github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE= +github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg= github.com/tidwall/gjson v1.14.2/go.mod h1:/wbyibRr2FHMks5tjHJ5F8dMZh3AcwJEMf5vlfC0lxk= github.com/tidwall/gjson v1.18.0 h1:FIDeeyB800efLX89e5a8Y0BNH+LOngJyGrIWxG2FKQY= github.com/tidwall/gjson v1.18.0/go.mod h1:/wbyibRr2FHMks5tjHJ5F8dMZh3AcwJEMf5vlfC0lxk= @@ -285,98 +338,110 @@ github.com/tidwall/pretty v1.2.1 h1:qjsOFOWWQl+N3RsoF5/ssm1pHmJJwhjlSbZ51I6wMl4= github.com/tidwall/pretty v1.2.1/go.mod h1:ITEVvHYasfjBbM0u2Pg8T2nJnzm8xPwvNhhsoaGGjNU= github.com/tidwall/sjson v1.2.5 h1:kLy8mja+1c9jlljvWTlSazM7cKDRfJuR/bOJhcY5NcY= github.com/tidwall/sjson v1.2.5/go.mod h1:Fvgq9kS/6ociJEDnK0Fk1cpYF4FIW6ZF7LAe+6jwd28= -github.com/wI2L/jsondiff v0.6.0 h1:zrsH3FbfVa3JO9llxrcDy/XLkYPLgoMX6Mz3T2PP2AI= -github.com/wI2L/jsondiff v0.6.0/go.mod h1:D6aQ5gKgPF9g17j+E9N7aasmU1O+XvfmWm1y8UMmNpw= +github.com/wI2L/jsondiff v0.7.1 h1:Fg9+yj+1/x3UtPBJhR91TKEzRkrEEWcAcLbg9dzEaNM= +github.com/wI2L/jsondiff v0.7.1/go.mod h1:yAt2W7U6Jd4HK0RA8DGSGk0zDtfEtOUUJVnH/xICpjo= github.com/x448/float16 v0.8.4 h1:qLwI1I70+NjRFUR3zs1JPUCgaCXSh3SW62uAKT1mSBM= github.com/x448/float16 v0.8.4/go.mod h1:14CWIYCyZA/cWjXOioeEpHeN/83MdbZDRQHoFcYsOfg= github.com/xlab/treeprint v1.2.0 h1:HzHnuAF1plUN2zGlAFHbSQP2qJ0ZAD3XF5XD7OesXRQ= github.com/xlab/treeprint v1.2.0/go.mod h1:gj5Gd3gPdKtR1ikdDK6fnFLdmIS0X30kTTuNd/WEJu0= -github.com/yuin/goldmark v1.1.27/go.mod h1:3hX8gzYuyVAZsxl0MRgGTJEmQBFcNTphYh9decYSb74= -github.com/yuin/goldmark v1.2.1/go.mod h1:3hX8gzYuyVAZsxl0MRgGTJEmQBFcNTphYh9decYSb74= -go.goms.io/fleet-networking v0.3.3 h1:5rwBntaUoLF+E1CzaWAEL4GdvLJPQorKhjgkbLlllPE= -go.goms.io/fleet-networking v0.3.3/go.mod h1:Qgbi8M1fGaz/p5rtb6HJPmTDATWRnMt9HD1gz57WKUc= -go.mongodb.org/mongo-driver v1.14.0 h1:P98w8egYRjYe3XDjxhYJagTokP/H6HzlsnojRgZRd80= -go.mongodb.org/mongo-driver v1.14.0/go.mod h1:Vzb0Mk/pa7e6cWw85R4F/endUC3u0U9jGcNU603k65c= +github.com/yuin/goldmark v1.4.13/go.mod h1:6yULJ656Px+3vBD8DxQVa3kxgyrAnzto9xy5taEt/CY= +go.goms.io/fleet-networking v0.3.44 h1:cPjWY2Gge0Y8+/iL4TFxyiC7YWmE87CYA0g6JKKIA8s= +go.goms.io/fleet-networking v0.3.44/go.mod h1:AgPNZEIabvhOlmjW3wasOU1HqKfe/SdcuyDBtGHINsM= go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ64= go.opentelemetry.io/auto/sdk v1.2.1/go.mod h1:KRTj+aOaElaLi+wW1kO/DZRXwkF4C5xPbEe3ZiIhN7Y= -go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.61.0 h1:F7Jx+6hwnZ41NSFTO5q4LYDtJRXBf2PD0rNBkeB/lus= -go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.61.0/go.mod h1:UHB22Z8QsdRDrnAtX4PntOl36ajSxcdUMt1sF7Y6E7Q= -go.opentelemetry.io/otel v1.44.0 h1:JjwHmHpA4iZ3wBxluu2fbbE7j4kqlE8jXyAyPXH7HqU= -go.opentelemetry.io/otel v1.44.0/go.mod h1:BMgjTHL9WPRlRjL2oZCBTL4whCGtXch2H4BhOPIAyYc= -go.opentelemetry.io/otel/exporters/prometheus v0.57.0 h1:AHh/lAP1BHrY5gBwk8ncc25FXWm/gmmY3BX258z5nuk= -go.opentelemetry.io/otel/exporters/prometheus v0.57.0/go.mod h1:QpFWz1QxqevfjwzYdbMb4Y1NnlJvqSGwyuU0B4iuc9c= -go.opentelemetry.io/otel/metric v1.44.0 h1:1w0gILTcHdr3YI+ixLyjemwrVnsMURbTZFrSYCdDdmc= -go.opentelemetry.io/otel/metric v1.44.0/go.mod h1:8O7hanEPBNgEMmybD3s2VBKcgWOCsA6tzHBPODAiquo= -go.opentelemetry.io/otel/sdk v1.44.0 h1:nHYwb9lK+fJPU/dnT6s7W7Z8itMWyqrnVfbheVYrZ58= -go.opentelemetry.io/otel/sdk v1.44.0/go.mod h1:Osuydd3Se74nqjAKxid74N5eC+jfEqfTegHRnq58oK0= -go.opentelemetry.io/otel/sdk/metric v1.44.0 h1:3LlKgI+VjbVsjNRFZJZAJ30WjXC5VkNRks6si09iEfI= -go.opentelemetry.io/otel/sdk/metric v1.44.0/go.mod h1:5B5pMARnXxKhltooO4xUuCBorl65a4EpnTalObqOigA= -go.opentelemetry.io/otel/trace v1.44.0 h1:jxF5CsGYCe74MCRx2X4g7WsY/VBKRqqpNvXlX/6gtIk= -go.opentelemetry.io/otel/trace v1.44.0/go.mod h1:oLl1jrMQAVo6v3GAggN+1VH9VIz9iUSvW53sW1Q8PIE= +go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.65.0 h1:7iP2uCb7sGddAr30RRS6xjKy7AZ2JtTOPA3oolgVSw8= +go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.65.0/go.mod h1:c7hN3ddxs/z6q9xwvfLPk+UHlWRQyaeR1LdgfL/66l0= +go.opentelemetry.io/otel v1.46.0 h1:FHt5/CDyVxi/8IM1CH7VE/rRgq3kLHa2mSTVMO8AWyc= +go.opentelemetry.io/otel v1.46.0/go.mod h1:Gj3SEScelsNC45tp4nSxRYlS+f5iez7W8XPMCt905kE= +go.opentelemetry.io/otel/metric v1.46.0 h1:yBnkXvgV7AXFILZc5K6IZe/CBFF3OS7BJ8ov6/lj0K8= +go.opentelemetry.io/otel/metric v1.46.0/go.mod h1:iPmdWqifKUdzziPkvvzIJXITl56fQx2mGM/DHLB3/2o= +go.opentelemetry.io/otel/sdk v1.46.0 h1:h5CNQQjEbuQXY/JfZtgt3i7HVFV3aHPO2OAwO2eTYPI= +go.opentelemetry.io/otel/sdk v1.46.0/go.mod h1:GAERFXFt5SYCEB+YiKUbMBeza6UaDH7GmGOZEfh2gSM= +go.opentelemetry.io/otel/sdk/metric v1.46.0 h1:0piZ26EG4RBfebb2jhDH6ERCYHoVWduc3kLgPCwSnSE= +go.opentelemetry.io/otel/sdk/metric v1.46.0/go.mod h1:I1PbKrdVc8Qu8HYVDNtqVIwLwjNrhsV/uFuxfwg8mO4= +go.opentelemetry.io/otel/trace v1.46.0 h1:OULy7ccdJnZtJ0UDYFOIGaCmiWzJ8Vi2G/Rsu60qs1c= +go.opentelemetry.io/otel/trace v1.46.0/go.mod h1:J7GAXweO77XSFkB/rmAqk9D6ihszhFjLU+d9WuUxDLI= go.uber.org/atomic v1.11.0 h1:ZvwS0R+56ePWxUNi+Atn9dWONBPp/AUETXlHW0DxSjE= go.uber.org/atomic v1.11.0/go.mod h1:LUxbIzbOniOlMKjJjyPfpl4v+PKK2cNJn91OQbhoJI0= -go.uber.org/automaxprocs v1.6.0 h1:O3y2/QNTOdbF+e/dpXNNW7Rx2hZ4sTIPyybbxyNqTUs= -go.uber.org/automaxprocs v1.6.0/go.mod h1:ifeIMSnPZuznNm6jmdzmU3/bfk01Fe2fotchwEFJ8r8= go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto= go.uber.org/goleak v1.3.0/go.mod h1:CoHD4mav9JJNrW/WLlf7HGZPjdw8EucARQHekz1X6bE= -go.uber.org/mock v0.5.1 h1:ASgazW/qBmR+A32MYFDB6E2POoTgOwT509VP0CT/fjs= -go.uber.org/mock v0.5.1/go.mod h1:ge71pBPLYDk7QIi1LupWxdAykm7KIEFchiOqd6z7qMM= +go.uber.org/mock v0.6.0 h1:hyF9dfmbgIX5EfOdasqLsWD6xqpNZlXblLB/Dbnwv3Y= +go.uber.org/mock v0.6.0/go.mod h1:KiVJ4BqZJaMj4svdfmHM0AUx4NJYO8ZNpPnZn1Z+BBU= go.uber.org/multierr v1.11.0 h1:blXXJkSxSSfBVBlC76pxqeO+LN3aDfLQo+309xJstO0= go.uber.org/multierr v1.11.0/go.mod h1:20+QtiLqy0Nd6FdQB9TLXag12DsQkrbs3htMFfDN80Y= -go.uber.org/zap v1.27.0 h1:aJMhYGrd5QSmlpLMr2MftRKl7t8J8PTZPA732ud/XR8= -go.uber.org/zap v1.27.0/go.mod h1:GB2qFLM7cTU87MWRP2mPIjqfIDnGu+VIO4V/SdhGo2E= -go.yaml.in/yaml/v2 v2.4.2 h1:DzmwEr2rDGHl7lsFgAHxmNz/1NlQ7xLIrlN2h5d1eGI= -go.yaml.in/yaml/v2 v2.4.2/go.mod h1:081UH+NErpNdqlCXm3TtEran0rJZGxAYx9hb/ELlsPU= -go.yaml.in/yaml/v3 v3.0.4 h1:tfq32ie2Jv2UxXFdLJdh3jXuOzWiL1fo0bu/FbuKpbc= +go.uber.org/zap v1.28.0 h1:IZzaP1Fv73/T/pBMLk4VutPl36uNC+OSUh3JLG3FIjo= +go.uber.org/zap v1.28.0/go.mod h1:rDLpOi171uODNm/mxFcuYWxDsqWSAVkFdX4XojSKg/Q= +go.yaml.in/yaml/v2 v2.4.4 h1:tuyd0P+2Ont/d6e2rl3be67goVK4R6deVxCUX5vyPaQ= +go.yaml.in/yaml/v2 v2.4.4/go.mod h1:gMZqIpDtDqOfM0uNfy0SkpRhvUryYH0Z6wdMYcacYXQ= go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg= +go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw= +go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg= golang.org/x/crypto v0.0.0-20190308221718-c2843e01d9a2/go.mod h1:djNgcEr1/C05ACkg1iLfiJU5Ep61QUkGW8qpdssI0+w= -golang.org/x/crypto v0.0.0-20191011191535-87dc89f01550/go.mod h1:yigFU9vqHzYiE8UmvKecakEJjdnWj3jj499lnFckfCI= -golang.org/x/crypto v0.0.0-20200622213623-75b288015ac9/go.mod h1:LzIPMQfyMNhhGPhUkYOs5KpL4U8rLKemX1yGLhDgUto= +golang.org/x/crypto v0.0.0-20210921155107-089bfa567519/go.mod h1:GvvjBRRGRdwPK5ydBHafDWAxML/pGHZbMvKqRZ5+Abc= +golang.org/x/crypto v0.0.0-20220722155217-630584e8d5aa/go.mod h1:IxCIyHEi3zRg3s0A5j5BB6A9Jmi73HwBIUl50j+osU4= +golang.org/x/crypto v0.17.0/go.mod h1:gCAAfMLgwOJRpTjQ2zCCt2OcSfYMTeZVSRtQlPC7Nq4= golang.org/x/crypto v0.56.0 h1:GUh5Ii4J5jtcseSMiRqr1jXCNHoxjeV9Fmekc2oLy6Y= golang.org/x/crypto v0.56.0/go.mod h1:OMW5y6CY9l38uPLmxU6l6pwcXp1obtLo3e6gT7gQR2I= -golang.org/x/exp v0.0.0-20250305212735-054e65f0b394 h1:nDVHiLt8aIbd/VzvPWN6kSOPE7+F/fNFDSXLVYkE/Iw= -golang.org/x/exp v0.0.0-20250305212735-054e65f0b394/go.mod h1:sIifuuw/Yco/y6yb6+bDNfyeQ/MdPUy/hKEMYQV17cM= -golang.org/x/mod v0.2.0/go.mod h1:s0Qsj1ACt9ePp/hMypM3fl4fZqREWJwdYDEqhRiZZUA= -golang.org/x/mod v0.3.0/go.mod h1:s0Qsj1ACt9ePp/hMypM3fl4fZqREWJwdYDEqhRiZZUA= -golang.org/x/net v0.0.0-20190404232315-eb5bcb51f2a3/go.mod h1:t9HGtf8HONx5eT2rtn7q6eTqICYqUVnKs3thJo3Qplg= +golang.org/x/exp v0.0.0-20260112195511-716be5621a96 h1:Z/6YuSHTLOHfNFdb8zVZomZr7cqNgTJvA8+Qz75D8gU= +golang.org/x/exp v0.0.0-20260112195511-716be5621a96/go.mod h1:nzimsREAkjBCIEFtHiYkrJyT+2uy9YZJB7H1k68CXZU= +golang.org/x/mod v0.6.0-dev.0.20220419223038-86c51ed26bb4/go.mod h1:jJ57K6gSWd91VN4djpZkiMVwK6gcyfeH4XE8wZrZaV4= +golang.org/x/mod v0.8.0/go.mod h1:iBbtSCu2XBx23ZKBPSOrRkjjQPZFPuis4dIYUhu/chs= +golang.org/x/mod v0.40.0 h1:hUv+3cXcdRHz08UmSiOob7sadHig73uo5bkXxQ/tvUs= +golang.org/x/mod v0.40.0/go.mod h1:0/weTWkPWGBikyTWAX3dkjVztMmBA5hM0DH6BElSupE= golang.org/x/net v0.0.0-20190620200207-3b0461eec859/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s= -golang.org/x/net v0.0.0-20200226121028-0de0cce0169b/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s= -golang.org/x/net v0.0.0-20201021035429-f5854403a974/go.mod h1:sp8m0HH+o8qH0wwXwYZr8TS3Oi6o0r6Gce1SSxlDquU= +golang.org/x/net v0.0.0-20210226172049-e18ecbb05110/go.mod h1:m0MpNAwzfU5UDzcl9v0D8zg8gWTRqZa9RBIspLL5mdg= +golang.org/x/net v0.0.0-20211112202133-69e39bad7dc2/go.mod h1:9nx3DQGgdP8bBQD5qxJ1jj9UTztislL4KSBs9R2vV5Y= +golang.org/x/net v0.0.0-20220722155237-a158d28d115b/go.mod h1:XRhObCWvk6IyKnWLug+ECip1KBveYUHfp+8e9klMJ9c= +golang.org/x/net v0.6.0/go.mod h1:2Tu9+aMcznHK/AK1HMvgo6xiTLG5rD5rZLDS+rp2Bjs= +golang.org/x/net v0.10.0/go.mod h1:0qNGK6F8kojg2nk9dLZ2mShWaEBan6FAoqfSigmmuDg= golang.org/x/net v0.58.0 h1:ynWG7rqYi4ccpTEuPZ2QGWHktVEM9DMCj9yzDE0Q7To= golang.org/x/net v0.58.0/go.mod h1:YwCddHnFlT7eLQqVprV19OnhLGtc5xOKgE0RyqgfWAU= golang.org/x/oauth2 v0.36.0 h1:peZ/1z27fi9hUOFCAZaHyrpWG5lwe0RJEEEeH0ThlIs= golang.org/x/oauth2 v0.36.0/go.mod h1:YDBUJMTkDnJS+A4BP4eZBjCqtokkg1hODuPjwiGPO7Q= golang.org/x/sync v0.0.0-20190423024810-112230192c58/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= -golang.org/x/sync v0.0.0-20190911185100-cd5d95a43a6e/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= -golang.org/x/sync v0.0.0-20201020160332-67f06af15bc9/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= -golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek= -golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0= +golang.org/x/sync v0.0.0-20220722155255-886fb9371eb4/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= +golang.org/x/sync v0.1.0/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= +golang.org/x/sync v0.23.0 h1:KameEIfc1IkluZyXWLn39Wd4tURc6GbCiISGiZm2bQk= +golang.org/x/sync v0.23.0/go.mod h1:sUUOizhqBxiL6pEWpqNLUiaJn1ShEbZ6BBqskPbjZm0= golang.org/x/sys v0.0.0-20190215142949-d0b11bdaac8a/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY= -golang.org/x/sys v0.0.0-20190412213103-97732733099d/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= -golang.org/x/sys v0.0.0-20200930185726-fdedc70b468f/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20201119102817-f84b799fce68/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20210423082822-04245dca01da/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20210615035016-665e8c7367d1/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.0.0-20220520151302-bc2c85ada10a/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.0.0-20220722155257-8c9f86f7a55f/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.1.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.5.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.8.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.15.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA= golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs= golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= +golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo= +golang.org/x/term v0.0.0-20210927222741-03fcf44c2211/go.mod h1:jbD1KX2456YbFQfuXm/mYQcufACuNUgVhRMnK/tPxf8= +golang.org/x/term v0.5.0/go.mod h1:jMB1sMXY+tzblOD4FWmEbocvup2/aLOaQEp7JmGp78k= +golang.org/x/term v0.8.0/go.mod h1:xPskH00ivmX89bAKVGSKKtLOWNx2+17Eiy94tnKShWo= +golang.org/x/term v0.15.0/go.mod h1:BDl952bC7+uMoWR75FIrCDx79TPU9oHkTZ9yRbYOrX0= golang.org/x/term v0.45.0 h1:NwWyBmoJCbfTHpxrWoZ9C6/VxOf7ic219I8xZZFdrf0= golang.org/x/term v0.45.0/go.mod h1:9aqxs0blBcrm/n0L9QW0aRVD+ktan8ssZromtqJC43w= golang.org/x/text v0.3.0/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ= golang.org/x/text v0.3.3/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ= +golang.org/x/text v0.3.6/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ= +golang.org/x/text v0.3.7/go.mod h1:u+2+/6zg+i71rQMx5EYifcz6MCKuco9NR6JIITiCfzQ= +golang.org/x/text v0.7.0/go.mod h1:mrYo+phRRbMaCq/xk9113O4dZlRixOauAjOtrjsXDZ8= +golang.org/x/text v0.9.0/go.mod h1:e1OnstbJyHTd6l/uOt8jFFHp6TRDWZR/bV3emEE/zU8= +golang.org/x/text v0.14.0/go.mod h1:18ZOQIKpY8NJVqYksKHtTdi31H5itFRjB5/qKTNYzSU= golang.org/x/text v0.41.0 h1:vz/seA0lnX87Othu2f/0L24RcgrXD9/YFTSuGjj3rH8= golang.org/x/text v0.41.0/go.mod h1:jvf1O8ajNzZqhSrQBPbutR/EB83Cc0CFrezNQIwbb5M= -golang.org/x/time v0.11.0 h1:/bpjEDfN9tkoN/ryeYHnv5hcMlc8ncjMcM4XBk5NWV0= -golang.org/x/time v0.11.0/go.mod h1:CDIdPxbZBQxdj6cxyCIdrNogrJKMJ7pr37NYpMcMDSg= +golang.org/x/time v0.16.0 h1:vMb6ptszcQMkcwiRTAuNNU50gom6++Q/6gY2hDM6VDE= +golang.org/x/time v0.16.0/go.mod h1:rVKOqvZeKvrDKTQiAHJ7wmwP0RzleSphoEA9RcdLA0s= golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ= golang.org/x/tools v0.0.0-20191119224855-298f0cb1881e/go.mod h1:b+2E5dAYhXwXZwtnZ6UAqBI28+e2cm9otk0dWdXHAEo= -golang.org/x/tools v0.0.0-20200619180055-7c47624df98f/go.mod h1:EkVYQZoAsY45+roYkvgYkIh4xh/qjgUK9TdY2XT94GE= -golang.org/x/tools v0.0.0-20210106214847-113979e3529a/go.mod h1:emZCQorbCU4vsT4fOWvOPXz4eW1wZW4PmDk9uLelYpA= -golang.org/x/tools v0.48.0 h1:3+hClM1aLL5mjMKm5ovokw9epgRXPuu2tILgismM6RE= -golang.org/x/tools v0.48.0/go.mod h1:08xX0orndb/F7jJxGDicx061tyd5pcMto75YMAXr6lk= +golang.org/x/tools v0.1.12/go.mod h1:hNGJHUnrk76NpqgfD5Aqm5Crs+Hm0VOH/i9J2+nxYbc= +golang.org/x/tools v0.6.0/go.mod h1:Xwgl3UAJ/d3gWutnCtw505GrjyAbvKui8lOU390QaIU= +golang.org/x/tools v0.49.0 h1:3NI7VXzL9+1WZD52Dx2ttoPwD5DWrFGpl9mFZDlmisI= +golang.org/x/tools v0.49.0/go.mod h1:SJNXV9DBKT0UbdttsQjbfJlAE/q+y36++zo3uL3N0Oo= golang.org/x/xerrors v0.0.0-20190717185122-a985d3407aa7/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= -golang.org/x/xerrors v0.0.0-20191011141410-1b5146add898/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= -golang.org/x/xerrors v0.0.0-20191204190536-9bdfabe68543/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= -golang.org/x/xerrors v0.0.0-20200804184101-5ec99f83aff1/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= -gomodules.xyz/jsonpatch/v2 v2.4.0 h1:Ci3iUJyx9UeRx7CeFN8ARgGbkESwJK+KB9lLcWxY/Zw= -gomodules.xyz/jsonpatch/v2 v2.4.0/go.mod h1:AH3dM2RI6uoBZxn3LVrfvJ3E0/9dG4cSrbuBJT4moAY= +gomodules.xyz/jsonpatch/v2 v2.5.0 h1:JELs8RLM12qJGXU4u/TO3V25KW8GreMKl9pdkk14RM0= +gomodules.xyz/jsonpatch/v2 v2.5.0/go.mod h1:AH3dM2RI6uoBZxn3LVrfvJ3E0/9dG4cSrbuBJT4moAY= gonum.org/v1/gonum v0.17.0 h1:VbpOemQlsSMrYmn7T2OUvQ4dqxQXU+ouZFQsZOx50z4= gonum.org/v1/gonum v0.17.0/go.mod h1:El3tOrEuMpv2UdMrbNlKEh9vd86bmQ6vqIcDwxEOc1E= google.golang.org/genproto/googleapis/api v0.0.0-20260526163538-3dc84a4a5aaa h1:Kjn0N0tCrDgiAFW+lGO4JZ3ck44CehvJQMAwj9QF0G8= @@ -385,71 +450,69 @@ google.golang.org/genproto/googleapis/rpc v0.0.0-20260526163538-3dc84a4a5aaa h1: google.golang.org/genproto/googleapis/rpc v0.0.0-20260526163538-3dc84a4a5aaa/go.mod h1:4Hqkh8ycfw05ld/3BWL7rJOSfebL2Q+DVDeRgYgxUU8= google.golang.org/grpc v1.83.2 h1:EManeRomTObA0BU7I8vXgg/78uE5MJ9M8B39EX2WscU= google.golang.org/grpc v1.83.2/go.mod h1:YPI1hK3kDked6iHvgX3tR0y+nX/qpMFKhPgFsokw1S8= -google.golang.org/protobuf v1.36.11 h1:fV6ZwhNocDyBLK0dj+fg8ektcVegBBuEolpbTQyBNVE= -google.golang.org/protobuf v1.36.11/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco= +google.golang.org/protobuf v1.36.12 h1:pJOKDDOyeXErUroCihFAd5LQuwXBSpVnKGrj5o/fwxc= +google.golang.org/protobuf v1.36.12/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco= gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= -gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk= -gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EVd6muEfDQjcINNoR0C8j2r3qZ4Q= gopkg.in/dnaeon/go-vcr.v3 v3.2.0 h1:Rltp0Vf+Aq0u4rQXgmXgtgoRDStTnFN83cWgSGSoRzM= gopkg.in/dnaeon/go-vcr.v3 v3.2.0/go.mod h1:2IMOnnlx9I6u9x+YBsM3tAMx6AlOxnJ0pWxQAzZ79Ag= -gopkg.in/evanphx/json-patch.v4 v4.12.0 h1:n6jtcsulIzXPJaxegRbvFNNrZDjbij7ny3gmSPG+6V4= -gopkg.in/evanphx/json-patch.v4 v4.12.0/go.mod h1:p8EYWUEYMpynmqDbY58zCKCFZw8pRWMG4EsWvDvM72M= +gopkg.in/evanphx/json-patch.v4 v4.13.0 h1:czT3CmqEaQ1aanPc5SdlgQrrEIb8w/wwCvWWnfEbYzo= +gopkg.in/evanphx/json-patch.v4 v4.13.0/go.mod h1:p8EYWUEYMpynmqDbY58zCKCFZw8pRWMG4EsWvDvM72M= gopkg.in/inf.v0 v0.9.1 h1:73M5CoZyi3ZLMOyDlQh031Cx6N9NDJ2Vvfl76EDAgDc= gopkg.in/inf.v0 v0.9.1/go.mod h1:cWUDdTG/fYaXco+Dcufb5Vnc6Gp2YChqWtbxRZE0mXw= -gopkg.in/yaml.v2 v2.4.0 h1:D8xgwECY7CYvx+Y2n4sBz93Jn9JRvxdiyyo8CTfuKaY= -gopkg.in/yaml.v2 v2.4.0/go.mod h1:RDklbk79AGWmwhnvt/jBztapEOGDOx6ZbXqjP6csGnQ= gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= -k8s.io/api v0.34.1 h1:jC+153630BMdlFukegoEL8E/yT7aLyQkIVuwhmwDgJM= -k8s.io/api v0.34.1/go.mod h1:SB80FxFtXn5/gwzCoN6QCtPD7Vbu5w2n1S0J5gFfTYk= -k8s.io/apiextensions-apiserver v0.34.1 h1:NNPBva8FNAPt1iSVwIE0FsdrVriRXMsaWFMqJbII2CI= -k8s.io/apiextensions-apiserver v0.34.1/go.mod h1:hP9Rld3zF5Ay2Of3BeEpLAToP+l4s5UlxiHfqRaRcMc= -k8s.io/apimachinery v0.34.1 h1:dTlxFls/eikpJxmAC7MVE8oOeP1zryV7iRyIjB0gky4= -k8s.io/apimachinery v0.34.1/go.mod h1:/GwIlEcWuTX9zKIg2mbw0LRFIsXwrfoVxn+ef0X13lw= -k8s.io/cli-runtime v0.32.3 h1:khLF2ivU2T6Q77H97atx3REY9tXiA3OLOjWJxUrdvss= -k8s.io/cli-runtime v0.32.3/go.mod h1:vZT6dZq7mZAca53rwUfdFSZjdtLyfF61mkf/8q+Xjak= -k8s.io/client-go v0.34.1 h1:ZUPJKgXsnKwVwmKKdPfw4tB58+7/Ik3CrjOEhsiZ7mY= -k8s.io/client-go v0.34.1/go.mod h1:kA8v0FP+tk6sZA0yKLRG67LWjqufAoSHA2xVGKw9Of8= -k8s.io/cloud-provider v0.32.3 h1:WC7KhWrqXsU4b0E4tjS+nBectGiJbr1wuc1TpWXvtZM= -k8s.io/cloud-provider v0.32.3/go.mod h1:/fwBfgRPuh16n8vLHT+PPT+Bc4LAEaJYj38opO2wsYY= -k8s.io/component-base v0.34.1 h1:v7xFgG+ONhytZNFpIz5/kecwD+sUhVE6HU7qQUiRM4A= -k8s.io/component-base v0.34.1/go.mod h1:mknCpLlTSKHzAQJJnnHVKqjxR7gBeHRv0rPXA7gdtQ0= -k8s.io/component-helpers v0.32.3 h1:9veHpOGTPLluqU4hAu5IPOwkOIZiGAJUhHndfVc5FT4= -k8s.io/component-helpers v0.32.3/go.mod h1:utTBXk8lhkJewBKNuNf32Xl3KT/0VV19DmiXU/SV4Ao= -k8s.io/csi-translation-lib v0.32.3 h1:fKdc9LMVEMk18xsgoPm1Ga8GjfhI7AM3UX8gnIeXZKs= -k8s.io/csi-translation-lib v0.32.3/go.mod h1:VX6+hCKgQyFnUX3VrnXZAgYYBXkrqx4BZk9vxr9qRcE= -k8s.io/klog/v2 v2.130.1 h1:n9Xl7H1Xvksem4KFG4PYbdQCQxqc/tTUyrgXaOhHSzk= -k8s.io/klog/v2 v2.130.1/go.mod h1:3Jpz1GvMt720eyJH1ckRHK1EDfpxISzJ7I9OYgaDtPE= -k8s.io/kube-openapi v0.0.0-20250710124328-f3f2b991d03b h1:MloQ9/bdJyIu9lb1PzujOPolHyvO06MXG5TUIj2mNAA= -k8s.io/kube-openapi v0.0.0-20250710124328-f3f2b991d03b/go.mod h1:UZ2yyWbFTpuhSbFhv24aGNOdoRdJZgsIObGBUaYVsts= -k8s.io/kubectl v0.32.3 h1:VMi584rbboso+yjfv0d8uBHwwxbC438LKq+dXd5tOAI= -k8s.io/kubectl v0.32.3/go.mod h1:6Euv2aso5GKzo/UVMacV6C7miuyevpfI91SvBvV9Zdg= -k8s.io/metrics v0.32.3 h1:2vsBvw0v8rIIlczZ/lZ8Kcqk9tR6Fks9h+dtFNbc2a4= -k8s.io/metrics v0.32.3/go.mod h1:9R1Wk5cb+qJpCQon9h52mgkVCcFeYxcY+YkumfwHVCU= -k8s.io/utils v0.0.0-20250604170112-4c0f3b243397 h1:hwvWFiBzdWw1FhfY1FooPn3kzWuJ8tmbZBHi4zVsl1Y= -k8s.io/utils v0.0.0-20250604170112-4c0f3b243397/go.mod h1:OLgZIPagt7ERELqWJFomSt595RzquPNLL48iOWgYOg0= -sigs.k8s.io/cloud-provider-azure v1.32.4 h1:v50uJzcE04w25Ra9EfWX/GHTTJKUC0+0Xpt+TOJ+D14= -sigs.k8s.io/cloud-provider-azure v1.32.4/go.mod h1:FbBaQt7N6/UVtK/VmIuJMLGe0gKUJ6NwoGrvH+zEa9w= -sigs.k8s.io/cloud-provider-azure/pkg/azclient v0.5.20 h1:aVSc4LFdBVlrhlldIzPo4NrcTQRdnAlqTB31sOcPIrM= -sigs.k8s.io/cloud-provider-azure/pkg/azclient v0.5.20/go.mod h1:OkkCYstvomfIwV4rvVIegymcgMnt7ZQ3+1Wi9WZmP1s= -sigs.k8s.io/cloud-provider-azure/pkg/azclient/configloader v0.5.2 h1:jjFJF0PmS9IHLokD41mM6RVoqQF3BQtVDmQd6ZMnN6E= -sigs.k8s.io/cloud-provider-azure/pkg/azclient/configloader v0.5.2/go.mod h1:7DdZ9ipIsmPLpBlfT4gueejcUlJBZQKWhdljQE5SKvc= -sigs.k8s.io/cluster-inventory-api v0.0.0-20251028164203-2e3fabb46733 h1:l90ANqblqFrE4L2QLLk+9iPjfmaLRvOFL51l/fgwUgg= -sigs.k8s.io/cluster-inventory-api v0.0.0-20251028164203-2e3fabb46733/go.mod h1:guwenlZ9iIfYlNxn7ExCfugOLTh6wjjRX3adC36YCmQ= -sigs.k8s.io/controller-runtime v0.22.4 h1:GEjV7KV3TY8e+tJ2LCTxUTanW4z/FmNB7l327UfMq9A= -sigs.k8s.io/controller-runtime v0.22.4/go.mod h1:+QX1XUpTXN4mLoblf4tqr5CQcyHPAki2HLXqQMY6vh8= -sigs.k8s.io/json v0.0.0-20241014173422-cfa47c3a1cc8 h1:gBQPwqORJ8d8/YNZWEjoZs7npUVDpVXUUOFfW6CgAqE= -sigs.k8s.io/json v0.0.0-20241014173422-cfa47c3a1cc8/go.mod h1:mdzfpAEoE6DHQEN0uh9ZbOCuHbLK5wOm7dK4ctXE9Tg= -sigs.k8s.io/karpenter v1.5.0 h1:3HaFtFvkteUJ+SjIViR1ImR0qR+GTqDulahauIuE4Qg= -sigs.k8s.io/karpenter v1.5.0/go.mod h1:YuqGoQsLti+V7ugHQVGXuT4v1QwCMiKloHLcPDfwMbY= -sigs.k8s.io/kustomize/api v0.18.0 h1:hTzp67k+3NEVInwz5BHyzc9rGxIauoXferXyjv5lWPo= -sigs.k8s.io/kustomize/api v0.18.0/go.mod h1:f8isXnX+8b+SGLHQ6yO4JG1rdkZlvhaCf/uZbLVMb0U= -sigs.k8s.io/kustomize/kyaml v0.18.1 h1:WvBo56Wzw3fjS+7vBjN6TeivvpbW9GmRaWZ9CIVmt4E= -sigs.k8s.io/kustomize/kyaml v0.18.1/go.mod h1:C3L2BFVU1jgcddNBE1TxuVLgS46TjObMwW5FT9FcjYo= +k8s.io/api v0.35.8 h1:hxpmPYdneQPKNh0cZyB09Hwd3vgXzdcJs5R3toDXsvU= +k8s.io/api v0.35.8/go.mod h1:I5gVNknFd4hfVVcMCixrenD7V38JUY78q3jtpGyC19c= +k8s.io/apiextensions-apiserver v0.35.8 h1:2lvyZ28M01/1uOPftaPXIbsOaCslzKHTEcdCjCI0VRE= +k8s.io/apiextensions-apiserver v0.35.8/go.mod h1:/ZbM1upeajFY5yHqGMKEzWTLb5Tx3kU2SkxTCtqB4Uw= +k8s.io/apimachinery v0.35.8 h1:piOyQQgse1sGztJVfy3B8f11YpT+KwK5KkD5Jie1EK0= +k8s.io/apimachinery v0.35.8/go.mod h1:z9Vq5oR1X38pkhh0wV531iKSeqmOVjqgHdYMjvzq2+o= +k8s.io/apiserver v0.35.8 h1:DuwdXkMrrNi6ovUaQk9EZhIgP7cDaWOvNP/CvrLJNfM= +k8s.io/apiserver v0.35.8/go.mod h1:97kTXpbFyeZwLcrK5hRYfc9ZfM3T7D4oYC0WXsJoXXM= +k8s.io/cli-runtime v0.35.8 h1:5rfvENhl4HklU8wUHVRPwUH/xHxgrK3Ly9c0dU4nDCg= +k8s.io/cli-runtime v0.35.8/go.mod h1:TSinz+vrk8BTO2+Kd6BePZyW40YuK9W8m+5V4X6fTjQ= +k8s.io/client-go v0.35.8 h1:tIW2sirCQMiGoCSvtOYqS059CDQ5n1nrDQa+PVt4nqY= +k8s.io/client-go v0.35.8/go.mod h1:fT8dATMU8FHMq4hlOudbsxihQ1LIQfDaLNDXBnIk6OQ= +k8s.io/cloud-provider v0.35.8 h1:/X0fmAL3343U0eJWUItXLmpOTs5oiW6QPPMS4abeDfE= +k8s.io/cloud-provider v0.35.8/go.mod h1:q/DZw6CvlTv6OTLYmXRa43fL02vn7YMuI5ednlM1vAk= +k8s.io/component-base v0.35.8 h1:71CLVx1zho3wxSpKroYAzaV5e1yexwWFQqUxIEimGGk= +k8s.io/component-base v0.35.8/go.mod h1:kM4Ide4Gh+bdZElhhgVvx5fC7oK39Smy8F3PbxEvV5o= +k8s.io/component-helpers v0.35.8 h1:+eh/NDwHD0oLhkrjmCvsDgmZypQEZhqS/PfMiUvvBbA= +k8s.io/component-helpers v0.35.8/go.mod h1:qMbFrr4rgxAtySi7IG5GTMAXgZBa/yVAWxybcfPs5co= +k8s.io/csi-translation-lib v0.35.0 h1:jdVC/9rv3lfHl5/MFQXqIVcEZEOXPbl4IPI8cczPdWw= +k8s.io/csi-translation-lib v0.35.0/go.mod h1:/6R70QdDxBCrMkrLhIBLP4mdtL35hEoJ5a/c2s1k9z8= +k8s.io/dynamic-resource-allocation v0.35.0 h1:St6dsCCylLg3HiFPcyHzFF8YQO6yziUDaVRLGdkrNH8= +k8s.io/dynamic-resource-allocation v0.35.0/go.mod h1:uaFga3VJtwyfpfZwpuJG7mlurWGQaaiGUa+QZmooz2U= +k8s.io/klog/v2 v2.140.0 h1:Tf+J3AH7xnUzZyVVXhTgGhEKnFqye14aadWv7bzXdzc= +k8s.io/klog/v2 v2.140.0/go.mod h1:o+/RWfJ6PwpnFn7OyAG3QnO47BFsymfEfrz6XyYSSp0= +k8s.io/kube-openapi v0.0.0-20260319004828-5883c5ee87b9 h1:Sztf7ESG9tAXRW/ACJZjrj5jhdOUqS2KFRQT+CTvu78= +k8s.io/kube-openapi v0.0.0-20260319004828-5883c5ee87b9/go.mod h1:uGBT7iTA6c6MvqUvSXIaYZo9ukscABYi2btjhvgKGZ0= +k8s.io/kubectl v0.35.8 h1:bhfqvUYygfEFGeGjOfjkWSZ+gSG4TySnycrwHC2OzTs= +k8s.io/kubectl v0.35.8/go.mod h1:mJgoBx+ROWm7REDEm5/KFuYfiWI/6leFtKZ9pA7l1ec= +k8s.io/metrics v0.35.8 h1:3Hy6RbhDSDUStqW4h5gJ98HEFCaxUXK6IrYgCyUcGbA= +k8s.io/metrics v0.35.8/go.mod h1:FJjRr6YRmnyMxupSlFke/C4+RwD+ssVIYXZAu6CzvDo= +k8s.io/utils v0.0.0-20260210185600-b8788abfbbc2 h1:AZYQSJemyQB5eRxqcPky+/7EdBj0xi3g0ZcxxJ7vbWU= +k8s.io/utils v0.0.0-20260210185600-b8788abfbbc2/go.mod h1:xDxuJ0whA3d0I4mf/C4ppKHxXynQ+fxnkmQH0vTHnuk= +sigs.k8s.io/cloud-provider-azure v1.35.9 h1:MRRcN8LVGdF7U9SJejLUIP3zwKZBsUjn13H3+UJ3LcQ= +sigs.k8s.io/cloud-provider-azure v1.35.9/go.mod h1:6/H5/oK2vV5+7Mn/Nc3NtqvXVLY+oBidketGgS74eO0= +sigs.k8s.io/cloud-provider-azure/pkg/azclient v0.19.0 h1:dh6aoXX1aofigfsRDZPECa4rat1NBKcE6ZQlhx2F2hY= +sigs.k8s.io/cloud-provider-azure/pkg/azclient v0.19.0/go.mod h1:qIX6mr9uEnUlC+3BQCjFBR9c4Qj+5mN0hrNZIYdMFWE= +sigs.k8s.io/cluster-inventory-api v0.1.3 h1:E7GY85hOIIPdALNdTO9Cbs2PXqjotWq6afe/NTwgCJo= +sigs.k8s.io/cluster-inventory-api v0.1.3/go.mod h1:7J3M6srZ1I4snZR+p5zxgEBdXnia3tlHo5ODMHJpEUk= +sigs.k8s.io/controller-runtime v0.23.3 h1:VjB/vhoPoA9l1kEKZHBMnQF33tdCLQKJtydy4iqwZ80= +sigs.k8s.io/controller-runtime v0.23.3/go.mod h1:B6COOxKptp+YaUT5q4l6LqUJTRpizbgf9KSRNdQGns0= +sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730 h1:IpInykpT6ceI+QxKBbEflcR5EXP7sU1kvOlxwZh5txg= +sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730/go.mod h1:mdzfpAEoE6DHQEN0uh9ZbOCuHbLK5wOm7dK4ctXE9Tg= +sigs.k8s.io/karpenter v1.14.0 h1:+SXlqYNb9FIKRpDgPPRnphVmWbrWLw4tgzKezr372Mc= +sigs.k8s.io/karpenter v1.14.0/go.mod h1:rrYIUdQk4UvFnjcm1/b3yNSTf1oJCDD4fdmkkMNda14= +sigs.k8s.io/kustomize/api v0.20.1 h1:iWP1Ydh3/lmldBnH/S5RXgT98vWYMaTUL1ADcr+Sv7I= +sigs.k8s.io/kustomize/api v0.20.1/go.mod h1:t6hUFxO+Ph0VxIk1sKp1WS0dOjbPCtLJ4p8aADLwqjM= +sigs.k8s.io/kustomize/kyaml v0.20.1 h1:PCMnA2mrVbRP3NIB6v9kYCAc38uvFLVs8j/CD567A78= +sigs.k8s.io/kustomize/kyaml v0.20.1/go.mod h1:0EmkQHRUsJxY8Ug9Niig1pUMSCGHxQ5RklbpV/Ri6po= sigs.k8s.io/randfill v1.0.0 h1:JfjMILfT8A6RbawdsK2JXGBR5AQVfd+9TbzrlneTyrU= sigs.k8s.io/randfill v1.0.0/go.mod h1:XeLlZ/jmk4i1HRopwe7/aU3H5n1zNUcX6TM94b3QxOY= -sigs.k8s.io/structured-merge-diff/v6 v6.3.0 h1:jTijUJbW353oVOd9oTlifJqOGEkUw2jB/fXCbTiQEco= -sigs.k8s.io/structured-merge-diff/v6 v6.3.0/go.mod h1:M3W8sfWvn2HhQDIbGWj3S099YozAsymCo/wrT5ohRUE= +sigs.k8s.io/structured-merge-diff/v6 v6.3.2 h1:kwVWMx5yS1CrnFWA/2QHyRVJ8jM6dBA80uLmm0wJkk8= +sigs.k8s.io/structured-merge-diff/v6 v6.3.2/go.mod h1:M3W8sfWvn2HhQDIbGWj3S099YozAsymCo/wrT5ohRUE= sigs.k8s.io/yaml v1.6.0 h1:G8fkbMSAFqgEFgh4b1wmtzDnioxFCUgTZhlbj5P9QYs= sigs.k8s.io/yaml v1.6.0/go.mod h1:796bPqUfzR/0jLAl6XjHl3Ck7MiyVv8dbTdyT3/pMf4= diff --git a/hack/release/create-draft-release.sh b/hack/release/create-draft-release.sh new file mode 100755 index 000000000..b7617b7bd --- /dev/null +++ b/hack/release/create-draft-release.sh @@ -0,0 +1,53 @@ +#!/usr/bin/env bash +# Create the GitHub Release for a tag as a draft, or reuse the draft already +# there. +# +# The release is created before any artifact exists so the producer jobs have +# somewhere to upload while the release stays invisible to consumers. +# publish-release.sh makes it public once every producer has succeeded. +# +# Environment: +# TAG release tag, e.g. v0.4.0 (required) +# PRERELEASE "true" when TAG is a release candidate (required) +# GH_TOKEN token with contents: write +# GH_REPO owner/repo + +set -euo pipefail + +: "${TAG:?TAG must be set}" +: "${PRERELEASE:?PRERELEASE must be set}" + +stderr="$(mktemp)" +trap 'rm -f "${stderr}"' EXIT + +# Distinguish "no such release" from a transient API failure. Treating an +# outage as "the release does not exist" would take the create path and bypass +# the published-release guard below. +if is_draft="$(gh release view "${TAG}" --json isDraft --jq .isDraft 2>"${stderr}")"; then + if [ "${is_draft}" != "true" ]; then + echo "::error::Release ${TAG} is already published. Re-running the full workflow for a completed release is refused; see RELEASING.md for the recovery procedure." + exit 1 + fi + # A draft may be a re-run of this workflow, or notes a maintainer pre-staged. + # Either way it is reused as-is; publish-release.sh reconciles the + # pre-release flag at publish time, so a hand-created draft cannot go out + # mislabelled. + echo "Reusing existing draft release ${TAG}." + exit 0 +fi + +if ! grep -qiE "not found|404" "${stderr}"; then + echo "::error::Could not determine the state of release ${TAG}: $(tr '\n' ' ' <"${stderr}")" + exit 1 +fi + +# --verify-tag: without it, `gh release create` happily invents the tag at the +# default branch's HEAD, which would cut a full release from whatever is on +# main under a version nobody intended. +create_args=(--title "${TAG}" --generate-notes --draft --verify-tag) +if [ "${PRERELEASE}" = "true" ]; then + create_args+=(--prerelease) +fi + +gh release create "${TAG}" "${create_args[@]}" +echo "Created draft release ${TAG}." diff --git a/hack/release/publish-release.sh b/hack/release/publish-release.sh new file mode 100755 index 000000000..ed7d91d03 --- /dev/null +++ b/hack/release/publish-release.sh @@ -0,0 +1,54 @@ +#!/usr/bin/env bash +# Publish the draft release for a tag, after checking it actually carries the +# assets it is supposed to. +# +# This is the atomic commit point of a release: everything else in the release +# workflow produces artifacts, and this is the only step that makes the release +# visible. +# +# Environment: +# TAG release tag, e.g. v0.4.0 (required) +# PRERELEASE "true" when TAG is a release candidate (required) +# GH_TOKEN token with contents: write +# GH_REPO owner/repo + +set -euo pipefail + +: "${TAG:?TAG must be set}" +: "${PRERELEASE:?PRERELEASE must be set}" + +bundle="kubefleet-crds-${TAG}.tgz" +checksum="${bundle}.sha256" + +# Only assets GitHub finished receiving count. An upload interrupted mid-stream +# leaves an asset row with the right name in a non-"uploaded" state, which a +# name-only check would accept. +assets="$(gh release view "${TAG}" --json assets \ + --jq '.assets[] | select(.state == "uploaded" and .size > 0) | .name')" + +for want in "${bundle}" "${checksum}"; do + if ! grep -qxF -- "${want}" <<<"${assets}"; then + echo "::error::Release ${TAG} is missing fully-uploaded asset ${want}; leaving it as a draft." + exit 1 + fi +done + +# Verify the bytes, not just the names: the bundle ships with a checksum, so +# confirming it here is the difference between "an asset with that name exists" +# and "the artifact users will download is intact". +workdir="$(mktemp -d)" +trap 'rm -rf "${workdir}"' EXIT +gh release download "${TAG}" --dir "${workdir}" --pattern "${bundle}" --pattern "${checksum}" + +if command -v sha256sum >/dev/null 2>&1; then + (cd "${workdir}" && sha256sum -c "${checksum}") +else + (cd "${workdir}" && shasum -a 256 -c "${checksum}") +fi + +# Set the pre-release flag here rather than only at creation time: a draft this +# workflow reused may have been created by hand, and GitHub defaults such +# drafts to "not a pre-release". Publishing an RC under that flag would make it +# the repository's "Latest release". +gh release edit "${TAG}" --draft=false --prerelease="${PRERELEASE}" +echo "✅ Published release ${TAG} (prerelease=${PRERELEASE})." diff --git a/hack/release/test-release-scripts.sh b/hack/release/test-release-scripts.sh new file mode 100755 index 000000000..a7eedda69 --- /dev/null +++ b/hack/release/test-release-scripts.sh @@ -0,0 +1,145 @@ +#!/usr/bin/env bash +# Exercise the release scripts against a stubbed gh CLI. +# +# These scripts decide whether a release goes public, so their failure modes +# matter more than most: the cases below are the ones where getting it wrong +# publishes something wrong rather than just failing the run. +# +# Run directly: ./hack/release/test-release-scripts.sh +# Requires: bash, jq. No test framework. + +set -uo pipefail + +cd "$(dirname "$0")" || exit 1 +here="${PWD}" +export PATH="${here}/testdata:${PATH}" + +create_script="${here}/create-draft-release.sh" +publish_script="${here}/publish-release.sh" + +passed=0 +failed=0 + +# Every case runs one script with a fresh stub log, then asserts on its exit +# code, its output, and the gh commands it issued. +run_case() { + FAKE_GH_LOG="$(mktemp)" + export FAKE_GH_LOG + output="" + rc=0 + output="$(env "$@" 2>&1)" || rc=$? + log="$(cat "${FAKE_GH_LOG}")" + rm -f "${FAKE_GH_LOG}" +} + +ok() { + echo " PASS $1" + passed=$((passed + 1)) +} + +bad() { + echo " FAIL $1" + echo " rc=${rc}" + echo " output: ${output}" + echo " gh calls: $(tr '\n' '|' <<<"${log}")" + failed=$((failed + 1)) +} + +expect_rc() { # expect_rc + if [ "${rc}" = "$1" ]; then ok "$2 (rc=${rc})"; else bad "$2 - want rc=$1"; fi +} + +expect_gh() { # expect_gh + if grep -qF -- "$1" <<<"${log}"; then ok "$2"; else bad "$2 - no gh call matching '$1'"; fi +} + +expect_no_gh() { # expect_no_gh + if grep -qF -- "$1" <<<"${log}"; then bad "$2 - unexpected gh call '$1'"; else ok "$2"; fi +} + +expect_output() { # expect_output + if grep -qF -- "$1" <<<"${output}"; then ok "$2"; else bad "$2 - output lacks '$1'"; fi +} + +echo "== create-draft-release.sh ==" + +run_case FAKE_GH_STATE=absent TAG=v0.4.0 PRERELEASE=false bash "${create_script}" +expect_rc 0 "no existing release, stable: succeeds" +expect_gh "gh release create v0.4.0 --title v0.4.0 --generate-notes --draft --verify-tag" \ + "no existing release, stable: creates a verified draft" +expect_no_gh "--prerelease" "stable release is not flagged as a pre-release" + +run_case FAKE_GH_STATE=absent TAG=v0.4.0-rc.1 PRERELEASE=true bash "${create_script}" +expect_rc 0 "no existing release, RC: succeeds" +expect_gh "--draft --verify-tag --prerelease" "RC is created as a pre-release" + +run_case FAKE_GH_STATE=draft TAG=v0.4.0 PRERELEASE=false bash "${create_script}" +expect_rc 0 "existing draft: succeeds" +expect_no_gh "release create" "existing draft is reused, not recreated" + +run_case FAKE_GH_STATE=published TAG=v0.4.0 PRERELEASE=false bash "${create_script}" +expect_rc 1 "already-published release: refuses" +expect_output "::error::" "already-published release: annotates the failure" +expect_no_gh "release create" "already-published release: creates nothing" + +# A GitHub outage must not be read as "the release does not exist" - that would +# take the create path and step over the published-release guard above. +run_case FAKE_GH_STATE=absent FAKE_GH_ERROR="HTTP 503: Service unavailable" \ + TAG=v0.4.0 PRERELEASE=false bash "${create_script}" +expect_rc 1 "API error that is not a 404: fails closed" +expect_no_gh "release create" "API error: creates nothing" + +echo "== publish-release.sh ==" + +both_uploaded="$(printf 'kubefleet-crds-v0.4.0.tgz;uploaded;4096\nkubefleet-crds-v0.4.0.tgz.sha256;uploaded;98')" + +run_case FAKE_GH_STATE=draft FAKE_GH_ASSETS="${both_uploaded}" TAG=v0.4.0 PRERELEASE=false \ + bash "${publish_script}" +expect_rc 0 "complete draft, stable: publishes" +expect_gh "gh release edit v0.4.0 --draft=false --prerelease=false" \ + "stable release is published without the pre-release flag" + +run_case FAKE_GH_STATE=draft \ + FAKE_GH_ASSETS="$(printf 'kubefleet-crds-v0.4.0-rc.1.tgz;uploaded;4096\nkubefleet-crds-v0.4.0-rc.1.tgz.sha256;uploaded;98')" \ + TAG=v0.4.0-rc.1 PRERELEASE=true bash "${publish_script}" +expect_rc 0 "complete draft, RC: publishes" +# A draft created by hand defaults to prerelease=false, so the flag has to be +# set at publish time or an RC becomes the repository's "Latest release". +expect_gh "--draft=false --prerelease=true" "RC is published flagged as a pre-release" + +run_case FAKE_GH_STATE=draft FAKE_GH_ASSETS="kubefleet-crds-v0.4.0.tgz;uploaded;4096" \ + TAG=v0.4.0 PRERELEASE=false bash "${publish_script}" +expect_rc 1 "missing checksum asset: refuses to publish" +expect_no_gh "release edit" "missing checksum asset: release stays a draft" + +run_case FAKE_GH_STATE=draft FAKE_GH_ASSETS="" TAG=v0.4.0 PRERELEASE=false bash "${publish_script}" +expect_rc 1 "no assets at all: refuses to publish" + +# GitHub keeps an asset row for an upload that never finished; it is present by +# name but not in the "uploaded" state. +run_case FAKE_GH_STATE=draft \ + FAKE_GH_ASSETS="$(printf 'kubefleet-crds-v0.4.0.tgz;new;0\nkubefleet-crds-v0.4.0.tgz.sha256;uploaded;98')" \ + TAG=v0.4.0 PRERELEASE=false bash "${publish_script}" +expect_rc 1 "interrupted upload (state != uploaded): refuses to publish" +expect_no_gh "release edit" "interrupted upload: release stays a draft" + +run_case FAKE_GH_STATE=draft \ + FAKE_GH_ASSETS="$(printf 'kubefleet-crds-v0.4.0.tgz;uploaded;0\nkubefleet-crds-v0.4.0.tgz.sha256;uploaded;98')" \ + TAG=v0.4.0 PRERELEASE=false bash "${publish_script}" +expect_rc 1 "zero-byte asset: refuses to publish" + +# Names alone are not proof; the bundle ships a checksum, so it gets checked. +run_case FAKE_GH_STATE=draft FAKE_GH_ASSETS="${both_uploaded}" FAKE_GH_DOWNLOAD=corrupt \ + TAG=v0.4.0 PRERELEASE=false bash "${publish_script}" +expect_rc 1 "bundle that fails its own checksum: refuses to publish" +expect_no_gh "release edit" "failed checksum: release stays a draft" + +# An asset whose name only looks right must not satisfy the check. +run_case FAKE_GH_STATE=draft \ + FAKE_GH_ASSETS="$(printf 'kubefleet-crds-v0.4.0.tgz.sha256;uploaded;98\nkubefleet-crds-v0.4.0.tgz.asc;uploaded;800')" \ + TAG=v0.4.0 PRERELEASE=false bash "${publish_script}" +expect_rc 1 "similar-but-wrong asset names: refuses to publish" + +echo +echo "passed=${passed} failed=${failed}" +[ "${failed}" -eq 0 ] diff --git a/hack/release/testdata/gh b/hack/release/testdata/gh new file mode 100755 index 000000000..7a1955fda --- /dev/null +++ b/hack/release/testdata/gh @@ -0,0 +1,76 @@ +#!/usr/bin/env bash +# Stand-in for the gh CLI, used by test-release-scripts.sh. It is put on PATH +# ahead of the real gh so the release scripts can be exercised without touching +# GitHub. +# +# Behaviour is driven by the environment: +# FAKE_GH_STATE absent | draft | published +# FAKE_GH_ERROR stderr text for the "absent" case (default: release not found) +# FAKE_GH_ASSETS asset rows as "name;state;size", one per line +# FAKE_GH_DOWNLOAD good | corrupt - whether the downloaded bundle matches its +# recorded checksum +# FAKE_GH_LOG file every invocation is appended to +# +# The --jq expressions are handed to the real jq so the scripts' own filters are +# what gets tested, not a reimplementation of them. + +set -uo pipefail + +echo "gh $*" >>"${FAKE_GH_LOG}" + +jq_expr="" +dir="" +args=("$@") +for i in "${!args[@]}"; do + case "${args[$i]}" in + --jq) jq_expr="${args[$((i + 1))]}" ;; + --dir) dir="${args[$((i + 1))]}" ;; + esac +done + +case "${1:-} ${2:-}" in + "release view") + if [ "${FAKE_GH_STATE}" = "absent" ]; then + echo "${FAKE_GH_ERROR:-release not found}" >&2 + exit 1 + fi + if [[ "$*" == *isDraft* ]]; then + [ "${FAKE_GH_STATE}" = "draft" ] && echo "true" || echo "false" + exit 0 + fi + if [[ "$*" == *assets* ]]; then + # Rebuild the assets JSON gh would return, then apply the caller's filter. + json="$( + while IFS=';' read -r name state size; do + [ -n "${name}" ] || continue + jq -n --arg n "${name}" --arg s "${state}" --argjson z "${size}" \ + '{name: $n, state: $s, size: $z}' + done <<<"${FAKE_GH_ASSETS:-}" | jq -s '{assets: .}' + )" + jq -r "${jq_expr}" <<<"${json}" + exit 0 + fi + exit 0 + ;; + "release download") + mkdir -p "${dir}" + printf 'pretend-tarball\n' >"${dir}/kubefleet-crds-${TAG}.tgz" + if command -v sha256sum >/dev/null 2>&1; then + sum="$(cd "${dir}" && sha256sum "kubefleet-crds-${TAG}.tgz")" + else + sum="$(cd "${dir}" && shasum -a 256 "kubefleet-crds-${TAG}.tgz")" + fi + if [ "${FAKE_GH_DOWNLOAD:-good}" = "corrupt" ]; then + # Overwrite the content after the checksum was taken, so the recorded + # checksum no longer describes the file - what a truncated or tampered + # upload looks like on download. + printf 'tampered\n' >"${dir}/kubefleet-crds-${TAG}.tgz" + fi + echo "${sum}" >"${dir}/kubefleet-crds-${TAG}.tgz.sha256" + exit 0 + ;; + "release create" | "release edit" | "release upload") + exit 0 + ;; +esac +exit 0 diff --git a/pkg/controllers/internalmembercluster/v1beta1/member_controller.go b/pkg/controllers/internalmembercluster/v1beta1/member_controller.go index ea8024561..0025ae297 100644 --- a/pkg/controllers/internalmembercluster/v1beta1/member_controller.go +++ b/pkg/controllers/internalmembercluster/v1beta1/member_controller.go @@ -31,7 +31,7 @@ import ( utilrand "k8s.io/apimachinery/pkg/util/rand" "k8s.io/client-go/kubernetes" "k8s.io/client-go/rest" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" "k8s.io/client-go/util/retry" "k8s.io/klog/v2" ctrl "sigs.k8s.io/controller-runtime" @@ -109,7 +109,7 @@ type Reconciler struct { // The property provider configuration. propertyProviderCfg *propertyProviderConfig - recorder record.EventRecorder + recorder events.EventRecorder } const ( @@ -342,16 +342,16 @@ func (r *Reconciler) connectToPropertyProvider(ctx context.Context, imc *cluster err := fmt.Errorf("property provider startup deadline exceeded") klog.ErrorS(err, "Failed to start property provider within the startup deadline", "internalMemberCluster", klog.KObj(imc)) reportPropertyProviderStartedCondition(imc, metav1.ConditionFalse, ClusterPropertyProviderStartedTimedOutReason, ClusterPropertyProviderStartedTimedOutMessage) - r.recorder.Event(imc, corev1.EventTypeWarning, ClusterPropertyProviderStartedTimedOutReason, ClusterPropertyProviderStartedTimedOutMessage) + r.recorder.Eventf(imc, nil, corev1.EventTypeWarning, ClusterPropertyProviderStartedTimedOutReason, "StartPropertyProvider", ClusterPropertyProviderStartedTimedOutMessage) case err := <-startedCh: if err != nil { klog.ErrorS(err, "Failed to start property provider", "internalMemberCluster", klog.KObj(imc)) reportPropertyProviderStartedCondition(imc, metav1.ConditionFalse, ClusterPropertyProviderStartedFailedReason, fmt.Sprintf(ClusterPropertyProviderStartedFailedMessage, err)) - r.recorder.Event(imc, corev1.EventTypeWarning, ClusterPropertyProviderStartedFailedReason, fmt.Sprintf(ClusterPropertyProviderStartedFailedMessage, err)) + r.recorder.Eventf(imc, nil, corev1.EventTypeWarning, ClusterPropertyProviderStartedFailedReason, "StartPropertyProvider", ClusterPropertyProviderStartedFailedMessage, err) } else { klog.V(2).InfoS("Property provider started", "internalMemberCluster", klog.KObj(imc)) reportPropertyProviderStartedCondition(imc, metav1.ConditionTrue, ClusterPropertyProviderStartedReason, ClusterPropertyProviderStartedMessage) - r.recorder.Event(imc, corev1.EventTypeNormal, ClusterPropertyProviderStartedReason, ClusterPropertyProviderStartedMessage) + r.recorder.Eventf(imc, nil, corev1.EventTypeNormal, ClusterPropertyProviderStartedReason, "StartPropertyProvider", ClusterPropertyProviderStartedMessage) r.propertyProviderCfg.isPropertyProviderStarted = true } } @@ -448,7 +448,7 @@ func (r *Reconciler) reportClusterPropertiesWithPropertyProvider(ctx context.Con ) err := fmt.Errorf("property provider collection deadline exceeded") klog.ErrorS(err, "Failed to collect cluster properties", "internalMemberCluster", klog.KObj(imc)) - r.recorder.Event(imc, corev1.EventTypeWarning, ClusterPropertyCollectionTimedOutReason, ClusterPropertyCollectionTimedOutMessage) + r.recorder.Eventf(imc, nil, corev1.EventTypeWarning, ClusterPropertyCollectionTimedOutReason, "CollectClusterProperties", ClusterPropertyCollectionTimedOutMessage) return err case <-collectedCh: // The property provider has returned the latest cluster properties; update the @@ -597,7 +597,7 @@ func (r *Reconciler) markInternalMemberClusterHealthy(imc clusterv1beta1.Conditi existingCondition := imc.GetConditionWithType(clusterv1beta1.MemberAgent, newCondition.Type) if existingCondition == nil || existingCondition.Status != newCondition.Status { klog.V(2).InfoS("InternalMemberCluster is healthy", "internalMemberCluster", klog.KObj(imc)) - r.recorder.Event(imc, corev1.EventTypeNormal, EventReasonInternalMemberClusterHealthy, "internal member cluster healthy") + r.recorder.Eventf(imc, nil, corev1.EventTypeNormal, EventReasonInternalMemberClusterHealthy, "HealthCheck", "internal member cluster healthy") } imc.SetConditionsWithType(clusterv1beta1.MemberAgent, newCondition) @@ -617,7 +617,7 @@ func (r *Reconciler) markInternalMemberClusterUnhealthy(imc clusterv1beta1.Condi existingCondition := imc.GetConditionWithType(clusterv1beta1.MemberAgent, newCondition.Type) if existingCondition == nil || existingCondition.Status != newCondition.Status { klog.V(2).InfoS("InternalMemberCluster is unhealthy", "internalMemberCluster", klog.KObj(imc)) - r.recorder.Event(imc, corev1.EventTypeWarning, EventReasonInternalMemberClusterUnhealthy, "internal member cluster unhealthy") + r.recorder.Eventf(imc, nil, corev1.EventTypeWarning, EventReasonInternalMemberClusterUnhealthy, "HealthCheck", "internal member cluster unhealthy") } imc.SetConditionsWithType(clusterv1beta1.MemberAgent, newCondition) @@ -635,7 +635,7 @@ func (r *Reconciler) markInternalMemberClusterJoined(imc clusterv1beta1.Conditio // Joined status changed. existingCondition := imc.GetConditionWithType(clusterv1beta1.MemberAgent, newCondition.Type) if existingCondition == nil || existingCondition.ObservedGeneration != imc.GetGeneration() || existingCondition.Status != newCondition.Status { - r.recorder.Event(imc, corev1.EventTypeNormal, EventReasonInternalMemberClusterJoined, "internal member cluster joined") + r.recorder.Eventf(imc, nil, corev1.EventTypeNormal, EventReasonInternalMemberClusterJoined, "Join", "internal member cluster joined") klog.V(2).InfoS("InternalMemberCluster has joined", "internalMemberCluster", klog.KObj(imc)) sharedmetrics.ReportJoinResultMetric() } @@ -656,7 +656,7 @@ func (r *Reconciler) markInternalMemberClusterJoinFailed(imc clusterv1beta1.Cond // Joined status changed. existingCondition := imc.GetConditionWithType(clusterv1beta1.MemberAgent, newCondition.Type) if existingCondition == nil || existingCondition.ObservedGeneration != imc.GetGeneration() || existingCondition.Status != newCondition.Status { - r.recorder.Event(imc, corev1.EventTypeNormal, EventReasonInternalMemberClusterFailedToJoin, "internal member cluster failed to join") + r.recorder.Eventf(imc, nil, corev1.EventTypeNormal, EventReasonInternalMemberClusterFailedToJoin, "Join", "internal member cluster failed to join") klog.ErrorS(err, "Agent failed to join", "internalMemberCluster", klog.KObj(imc)) } @@ -675,7 +675,7 @@ func (r *Reconciler) markInternalMemberClusterLeft(imc clusterv1beta1.Conditione // Joined status changed. existingCondition := imc.GetConditionWithType(clusterv1beta1.MemberAgent, newCondition.Type) if existingCondition == nil || existingCondition.ObservedGeneration != imc.GetGeneration() || existingCondition.Status != newCondition.Status { - r.recorder.Event(imc, corev1.EventTypeNormal, EventReasonInternalMemberClusterLeft, "internal member cluster left") + r.recorder.Eventf(imc, nil, corev1.EventTypeNormal, EventReasonInternalMemberClusterLeft, "Leave", "internal member cluster left") klog.V(2).InfoS("InternalMemberCluster has left", "internalMemberCluster", klog.KObj(imc)) sharedmetrics.ReportLeaveResultMetric() } @@ -695,7 +695,7 @@ func (r *Reconciler) markInternalMemberClusterLeaveFailed(imc clusterv1beta1.Con // Joined status changed. if !condition.IsConditionStatusTrue(imc.GetConditionWithType(clusterv1beta1.MemberAgent, newCondition.Type), imc.GetGeneration()) { - r.recorder.Event(imc, corev1.EventTypeNormal, EventReasonInternalMemberClusterFailedToLeave, "internal member cluster failed to leave") + r.recorder.Eventf(imc, nil, corev1.EventTypeNormal, EventReasonInternalMemberClusterFailedToLeave, "Leave", "internal member cluster failed to leave") klog.ErrorS(err, "Agent leave failed", "internalMemberCluster", klog.KObj(imc)) } @@ -704,7 +704,7 @@ func (r *Reconciler) markInternalMemberClusterLeaveFailed(imc clusterv1beta1.Con // SetupWithManager sets up the controller with the Manager. func (r *Reconciler) SetupWithManager(mgr ctrl.Manager, name string) error { - r.recorder = mgr.GetEventRecorderFor("v1beta1InternalMemberClusterController") + r.recorder = mgr.GetEventRecorder("v1beta1InternalMemberClusterController") return ctrl.NewControllerManagedBy(mgr).Named(name). For(&clusterv1beta1.InternalMemberCluster{}, builder.WithPredicates(predicate.GenerationChangedPredicate{})). Complete(r) diff --git a/pkg/controllers/internalmembercluster/v1beta1/member_controller_test.go b/pkg/controllers/internalmembercluster/v1beta1/member_controller_test.go index 15f9c22c7..742258b81 100644 --- a/pkg/controllers/internalmembercluster/v1beta1/member_controller_test.go +++ b/pkg/controllers/internalmembercluster/v1beta1/member_controller_test.go @@ -26,7 +26,6 @@ import ( "github.com/crossplane/crossplane-runtime/v2/pkg/test" "github.com/google/go-cmp/cmp" "github.com/google/go-cmp/cmp/cmpopts" - "github.com/stretchr/testify/assert" corev1 "k8s.io/api/core/v1" apierrors "k8s.io/apimachinery/pkg/api/errors" "k8s.io/apimachinery/pkg/api/resource" @@ -35,7 +34,7 @@ import ( "k8s.io/apimachinery/pkg/runtime/schema" "k8s.io/apimachinery/pkg/util/validation/field" "k8s.io/client-go/rest" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" "sigs.k8s.io/controller-runtime/pkg/client" "sigs.k8s.io/controller-runtime/pkg/client/fake" @@ -78,37 +77,89 @@ var ( ) func TestMarkInternalMemberClusterJoined(t *testing.T) { - r := Reconciler{recorder: utils.NewFakeRecorder(1)} + r := Reconciler{recorder: events.NewFakeRecorder(1)} internalMemberCluster := &clusterv1beta1.InternalMemberCluster{} r.markInternalMemberClusterJoined(internalMemberCluster) // check that the correct event is emitted - event := <-r.recorder.(*record.FakeRecorder).Events - expected := utils.GetEventString(internalMemberCluster, corev1.EventTypeNormal, EventReasonInternalMemberClusterJoined, "internal member cluster joined") - assert.Equal(t, expected, event, utils.TestCaseMsg, "TestMarkInternalMemberClusterJoined") + event := <-r.recorder.(*events.FakeRecorder).Events + expected := utils.GetEventString(corev1.EventTypeNormal, EventReasonInternalMemberClusterJoined, "internal member cluster joined") + if event != expected { + t.Errorf("markInternalMemberClusterJoined() emitted event %v, want %v", event, expected) + } // Check expected condition. expectedCondition := metav1.Condition{Type: string(clusterv1beta1.AgentJoined), Status: metav1.ConditionTrue, Reason: EventReasonInternalMemberClusterJoined} actualCondition := internalMemberCluster.GetConditionWithType(clusterv1beta1.MemberAgent, expectedCondition.Type) - assert.Equal(t, "", cmp.Diff(expectedCondition, *(actualCondition), cmpopts.IgnoreTypes(time.Time{})), utils.TestCaseMsg, "TestMarkInternalMemberClusterJoined") + if diff := cmp.Diff(*actualCondition, expectedCondition, cmpopts.IgnoreTypes(time.Time{})); diff != "" { + t.Errorf("markInternalMemberClusterJoined() condition mismatch (-got, +want):\n%s", diff) + } } func TestMarkInternalMemberClusterLeft(t *testing.T) { - r := Reconciler{recorder: utils.NewFakeRecorder(1)} + r := Reconciler{recorder: events.NewFakeRecorder(1)} internalMemberCluster := &clusterv1beta1.InternalMemberCluster{} r.markInternalMemberClusterLeft(internalMemberCluster) // check that the correct event is emitted - event := <-r.recorder.(*record.FakeRecorder).Events - expected := utils.GetEventString(internalMemberCluster, corev1.EventTypeNormal, EventReasonInternalMemberClusterLeft, "internal member cluster left") - assert.Equal(t, expected, event, utils.TestCaseMsg, "TestMarkInternalMemberClusterLeft") + event := <-r.recorder.(*events.FakeRecorder).Events + expected := utils.GetEventString(corev1.EventTypeNormal, EventReasonInternalMemberClusterLeft, "internal member cluster left") + if event != expected { + t.Errorf("markInternalMemberClusterLeft() emitted event %v, want %v", event, expected) + } // Check expected conditions. expectedCondition := metav1.Condition{Type: string(clusterv1beta1.AgentJoined), Status: metav1.ConditionFalse, Reason: EventReasonInternalMemberClusterLeft} actualCondition := internalMemberCluster.GetConditionWithType(clusterv1beta1.MemberAgent, expectedCondition.Type) - assert.Equal(t, "", cmp.Diff(expectedCondition, *(actualCondition), cmpopts.IgnoreTypes(time.Time{})), utils.TestCaseMsg, "TestMarkInternalMemberClusterLeft") + if diff := cmp.Diff(*actualCondition, expectedCondition, cmpopts.IgnoreTypes(time.Time{})); diff != "" { + t.Errorf("markInternalMemberClusterLeft() condition mismatch (-got, +want):\n%s", diff) + } +} + +func TestMarkInternalMemberClusterJoinFailed(t *testing.T) { + r := Reconciler{recorder: events.NewFakeRecorder(1)} + internalMemberCluster := &clusterv1beta1.InternalMemberCluster{} + joinErr := errors.New("join failed") + + r.markInternalMemberClusterJoinFailed(internalMemberCluster, joinErr) + + // check that the correct event is emitted + event := <-r.recorder.(*events.FakeRecorder).Events + wantEvent := utils.GetEventString(corev1.EventTypeNormal, EventReasonInternalMemberClusterFailedToJoin, "internal member cluster failed to join") + if event != wantEvent { + t.Errorf("markInternalMemberClusterJoinFailed() emitted event %v, want %v", event, wantEvent) + } + + // Check expected condition. + wantCondition := metav1.Condition{Type: string(clusterv1beta1.AgentJoined), Status: metav1.ConditionUnknown, Reason: EventReasonInternalMemberClusterFailedToJoin, Message: joinErr.Error()} + gotCondition := internalMemberCluster.GetConditionWithType(clusterv1beta1.MemberAgent, wantCondition.Type) + if diff := cmp.Diff(*gotCondition, wantCondition, cmpopts.IgnoreTypes(time.Time{})); diff != "" { + t.Errorf("markInternalMemberClusterJoinFailed() condition mismatch (-got, +want):\n%s", diff) + } +} + +func TestMarkInternalMemberClusterLeaveFailed(t *testing.T) { + r := Reconciler{recorder: events.NewFakeRecorder(1)} + internalMemberCluster := &clusterv1beta1.InternalMemberCluster{} + leaveErr := errors.New("leave failed") + + r.markInternalMemberClusterLeaveFailed(internalMemberCluster, leaveErr) + + // check that the correct event is emitted + event := <-r.recorder.(*events.FakeRecorder).Events + wantEvent := utils.GetEventString(corev1.EventTypeNormal, EventReasonInternalMemberClusterFailedToLeave, "internal member cluster failed to leave") + if event != wantEvent { + t.Errorf("markInternalMemberClusterLeaveFailed() emitted event %v, want %v", event, wantEvent) + } + + // Check expected condition. + wantCondition := metav1.Condition{Type: string(clusterv1beta1.AgentJoined), Status: metav1.ConditionUnknown, Reason: EventReasonInternalMemberClusterFailedToLeave, Message: leaveErr.Error()} + gotCondition := internalMemberCluster.GetConditionWithType(clusterv1beta1.MemberAgent, wantCondition.Type) + if diff := cmp.Diff(*gotCondition, wantCondition, cmpopts.IgnoreTypes(time.Time{})); diff != "" { + t.Errorf("markInternalMemberClusterLeaveFailed() condition mismatch (-got, +want):\n%s", diff) + } } func TestUpdateMemberAgentHeartBeat(t *testing.T) { @@ -116,46 +167,58 @@ func TestUpdateMemberAgentHeartBeat(t *testing.T) { updateMemberAgentHeartBeat(internalMemberCluster) lastReceivedHeartBeat := internalMemberCluster.Status.AgentStatus[0].LastReceivedHeartbeat - assert.NotNil(t, lastReceivedHeartBeat) + if lastReceivedHeartBeat.IsZero() { + t.Fatal("updateMemberAgentHeartBeat() left LastReceivedHeartbeat unset") + } updateMemberAgentHeartBeat(internalMemberCluster) newLastReceivedHeartBeat := internalMemberCluster.Status.AgentStatus[0].LastReceivedHeartbeat - assert.NotEqual(t, lastReceivedHeartBeat, newLastReceivedHeartBeat) + if newLastReceivedHeartBeat.Time.Equal(lastReceivedHeartBeat.Time) { + t.Errorf("updateMemberAgentHeartBeat() LastReceivedHeartbeat = %v, want a time after %v", newLastReceivedHeartBeat, lastReceivedHeartBeat) + } } func TestMarkInternalMemberClusterHealthy(t *testing.T) { - r := Reconciler{recorder: utils.NewFakeRecorder(1)} + r := Reconciler{recorder: events.NewFakeRecorder(1)} internalMemberCluster := &clusterv1beta1.InternalMemberCluster{} r.markInternalMemberClusterHealthy(internalMemberCluster) // check that the correct event is emitted - event := <-r.recorder.(*record.FakeRecorder).Events - expected := utils.GetEventString(internalMemberCluster, corev1.EventTypeNormal, EventReasonInternalMemberClusterHealthy, "internal member cluster healthy") - assert.Equal(t, expected, event, utils.TestCaseMsg, "TestMarkInternalMemberClusterHealthy") + event := <-r.recorder.(*events.FakeRecorder).Events + expected := utils.GetEventString(corev1.EventTypeNormal, EventReasonInternalMemberClusterHealthy, "internal member cluster healthy") + if event != expected { + t.Errorf("markInternalMemberClusterHealthy() emitted event %v, want %v", event, expected) + } // Check expected conditions. expectedCondition := metav1.Condition{Type: string(clusterv1beta1.AgentHealthy), Status: metav1.ConditionTrue, Reason: EventReasonInternalMemberClusterHealthy} actualCondition := internalMemberCluster.GetConditionWithType(clusterv1beta1.MemberAgent, expectedCondition.Type) - assert.Equal(t, "", cmp.Diff(expectedCondition, *(actualCondition), cmpopts.IgnoreTypes(time.Time{})), utils.TestCaseMsg, "TestMarkInternalMemberClusterHealthy") + if diff := cmp.Diff(*actualCondition, expectedCondition, cmpopts.IgnoreTypes(time.Time{})); diff != "" { + t.Errorf("markInternalMemberClusterHealthy() condition mismatch (-got, +want):\n%s", diff) + } } func TestMarkInternalMemberClusterHeartbeatUnhealthy(t *testing.T) { internalMemberCluster := &clusterv1beta1.InternalMemberCluster{} err := errors.New("rand-err-msg") - r := Reconciler{recorder: utils.NewFakeRecorder(1)} + r := Reconciler{recorder: events.NewFakeRecorder(1)} r.markInternalMemberClusterUnhealthy(internalMemberCluster, err) // check that the correct event is emitted - event := <-r.recorder.(*record.FakeRecorder).Events - expected := utils.GetEventString(internalMemberCluster, corev1.EventTypeWarning, EventReasonInternalMemberClusterUnhealthy, "internal member cluster unhealthy") - assert.Equal(t, expected, event, utils.TestCaseMsg, "TestMarkInternalMemberClusterHeartbeatUnhealthy") + event := <-r.recorder.(*events.FakeRecorder).Events + expected := utils.GetEventString(corev1.EventTypeWarning, EventReasonInternalMemberClusterUnhealthy, "internal member cluster unhealthy") + if event != expected { + t.Errorf("markInternalMemberClusterUnhealthy() emitted event %v, want %v", event, expected) + } // Check expected conditions. expectedCondition := metav1.Condition{Type: string(clusterv1beta1.AgentHealthy), Status: metav1.ConditionFalse, Reason: EventReasonInternalMemberClusterUnhealthy, Message: "rand-err-msg"} actualCondition := internalMemberCluster.GetConditionWithType(clusterv1beta1.MemberAgent, expectedCondition.Type) - assert.Equal(t, "", cmp.Diff(expectedCondition, *(actualCondition), cmpopts.IgnoreTypes(time.Time{})), utils.TestCaseMsg, "TestMarkInternalMemberClusterHeartbeatUnhealthy") + if diff := cmp.Diff(*actualCondition, expectedCondition, cmpopts.IgnoreTypes(time.Time{})); diff != "" { + t.Errorf("markInternalMemberClusterUnhealthy() condition mismatch (-got, +want):\n%s", diff) + } } func TestUpdateInternalMemberClusterWithRetry(t *testing.T) { @@ -222,7 +285,9 @@ func TestUpdateInternalMemberClusterWithRetry(t *testing.T) { for testName, testCase := range testCases { t.Run(testName, func(t *testing.T) { err := testCase.r.updateInternalMemberClusterWithRetry(context.Background(), testCase.internalMemberCluster) - assert.Equal(t, testCase.wantErr, err, utils.TestCaseMsg, testName) + if diff := cmp.Diff(err, testCase.wantErr); diff != "" { + t.Errorf("updateInternalMemberClusterWithRetry() error mismatch (-got, +want):\n%s", diff) + } }) } } @@ -321,7 +386,9 @@ func TestSetConditionWithType(t *testing.T) { for testName, testCase := range testCases { t.Run(testName, func(t *testing.T) { testCase.internalMemberCluster.SetConditionsWithType(clusterv1beta1.MemberAgent, testCase.condition) - assert.Equal(t, "", cmp.Diff(testCase.wantedAgentStatus, testCase.internalMemberCluster.GetAgentStatus(clusterv1beta1.MemberAgent), cmpopts.IgnoreTypes(time.Time{}))) + if diff := cmp.Diff(testCase.internalMemberCluster.GetAgentStatus(clusterv1beta1.MemberAgent), testCase.wantedAgentStatus, cmpopts.IgnoreTypes(time.Time{})); diff != "" { + t.Errorf("SetConditionsWithType() agent status mismatch (-got, +want):\n%s", diff) + } }) } } @@ -388,7 +455,9 @@ func TestGetConditionWithType(t *testing.T) { for testName, testCase := range testCases { t.Run(testName, func(t *testing.T) { actualCondition := testCase.internalMemberCluster.GetConditionWithType(clusterv1beta1.MemberAgent, testCase.conditionType) - assert.Equal(t, testCase.wantedCondition, actualCondition) + if diff := cmp.Diff(actualCondition, testCase.wantedCondition); diff != "" { + t.Errorf("GetConditionWithType() mismatch (-got, +want):\n%s", diff) + } }) } } @@ -454,7 +523,7 @@ func TestReportClusterPropertiesWithPropertyProviderTooManyCalls(t *testing.T) { propertyProviderCfg: &propertyProviderConfig{ propertyProvider: nrpp, }, - recorder: utils.NewFakeRecorder(maxQueuedPropertyCollectionCalls + 1), + recorder: events.NewFakeRecorder(maxQueuedPropertyCollectionCalls + 1), } for i := 0; i < maxQueuedPropertyCollectionCalls; i++ { // Invoke the method with no expectations for returns. @@ -529,7 +598,7 @@ func TestReportClusterPropertiesWithPropertyProviderTimedOut(t *testing.T) { propertyProviderCfg: &propertyProviderConfig{ propertyProvider: nrpp, }, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), } if err := r.reportClusterPropertiesWithPropertyProvider(ctx, tc.imc); err == nil { @@ -658,7 +727,7 @@ func TestReportClusterPropertiesWithPropertyProvider(t *testing.T) { propertyProviderCfg: &propertyProviderConfig{ propertyProvider: &dummyProvider{}, }, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), } if err := r.reportClusterPropertiesWithPropertyProvider(ctx, tc.imc); err != nil { @@ -1449,7 +1518,7 @@ func TestConnectToPropertyProvider(t *testing.T) { propertyProviderCfg: &propertyProviderConfig{ propertyProvider: tc.propertyProvider, }, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), } imc := imcTemplate.DeepCopy() diff --git a/pkg/controllers/membercluster/v1beta1/membercluster_controller.go b/pkg/controllers/membercluster/v1beta1/membercluster_controller.go index d22bfc159..c34bed2a6 100644 --- a/pkg/controllers/membercluster/v1beta1/membercluster_controller.go +++ b/pkg/controllers/membercluster/v1beta1/membercluster_controller.go @@ -29,7 +29,7 @@ import ( "k8s.io/apimachinery/pkg/api/meta" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/types" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" "k8s.io/client-go/util/retry" "k8s.io/klog/v2" "k8s.io/utils/ptr" @@ -68,7 +68,7 @@ const ( // Reconciler reconciles a MemberCluster object type Reconciler struct { client.Client - recorder record.EventRecorder + recorder events.EventRecorder // Need to update MC based on the IMC conditions based on the agent list. NetworkingAgentsEnabled bool // the max number of concurrent reconciles per controller. @@ -283,16 +283,29 @@ func (r *Reconciler) ensureFinalizer(ctx context.Context, mc *clusterv1beta1.Mem // ensureMemberNameLabel makes sure that the member cluster has a label with its own name. // This enables selecting clusters by name in ResourceOverride and ClusterResourceOverride via labelSelector. func (r *Reconciler) ensureMemberNameLabel(ctx context.Context, mc *clusterv1beta1.MemberCluster) error { - if mc.Labels != nil && mc.Labels[placementv1beta1.MemberNameLabel] == mc.Name { - return nil - } - + changed := false if mc.Labels == nil { mc.Labels = make(map[string]string) } - mc.Labels[placementv1beta1.MemberNameLabel] = mc.Name - klog.InfoS("Ensured the member cluster name label", "memberCluster", klog.KObj(mc)) + if mc.Labels[placementv1beta1.MemberNameLabel] != mc.Name { + mc.Labels[placementv1beta1.MemberNameLabel] = mc.Name + changed = true + } + + // The alias label is seeded from the cluster name, but only when it is absent. Unlike the name + // label above, which the controller owns and reasserts, the alias exists to be renamed by an + // admin so that a selector can follow a role rather than a fixed name; reasserting it would + // revert that rename on the next reconcile. + if _, found := mc.Labels[placementv1beta1.ClusterAliasLabel]; !found { + mc.Labels[placementv1beta1.ClusterAliasLabel] = mc.Name + changed = true + } + + if !changed { + return nil + } + klog.InfoS("Ensured the member cluster name and alias labels", "memberCluster", klog.KObj(mc)) return r.Update(ctx, mc, client.FieldOwner(utils.MCControllerFieldManagerName)) } @@ -369,7 +382,7 @@ func (r *Reconciler) syncNamespace(ctx context.Context, mc *clusterv1beta1.Membe if err = r.Client.Create(ctx, &expectedNS, client.FieldOwner(utils.MCControllerFieldManagerName)); err != nil { return "", fmt.Errorf("failed to create namespace %s: %w", namespaceName, err) } - r.recorder.Event(mc, corev1.EventTypeNormal, eventReasonNamespaceCreated, "Namespace was created") + r.recorder.Eventf(mc, nil, corev1.EventTypeNormal, eventReasonNamespaceCreated, "ReconcileNamespace", "Namespace was created") klog.V(2).InfoS("created namespace", "memberCluster", klog.KObj(mc), "namespace", namespaceName) return namespaceName, nil } @@ -382,7 +395,7 @@ func (r *Reconciler) syncNamespace(ctx context.Context, mc *clusterv1beta1.Membe if err := r.Client.Patch(ctx, ¤tNS, patch, client.FieldOwner(utils.MCControllerFieldManagerName)); err != nil { return "", fmt.Errorf("failed to patch namespace %s: %w", namespaceName, err) } - r.recorder.Event(mc, corev1.EventTypeNormal, eventReasonNamespacePatched, "Namespace was patched") + r.recorder.Eventf(mc, nil, corev1.EventTypeNormal, eventReasonNamespacePatched, "ReconcileNamespace", "Namespace was patched") klog.V(2).InfoS("patched namespace", "memberCluster", klog.KObj(mc), "namespace", namespaceName) } return namespaceName, nil @@ -399,7 +412,7 @@ func (r *Reconciler) syncRole(ctx context.Context, mc *clusterv1beta1.MemberClus Namespace: namespaceName, OwnerReferences: []metav1.OwnerReference{*toOwnerReference(mc)}, }, - Rules: []rbacv1.PolicyRule{utils.FleetClusterRule, utils.FleetPlacementRule, utils.FleetNetworkRule, utils.EventRule}, + Rules: []rbacv1.PolicyRule{utils.FleetClusterRule, utils.FleetPlacementRule, utils.FleetNetworkRule, utils.EventRule, utils.EventsK8sIoRule}, } // Creates role if not found. @@ -412,7 +425,7 @@ func (r *Reconciler) syncRole(ctx context.Context, mc *clusterv1beta1.MemberClus if err = r.Client.Create(ctx, &expectedRole, client.FieldOwner(utils.MCControllerFieldManagerName)); err != nil { return "", fmt.Errorf("failed to create role %s with rules %+v: %w", roleName, expectedRole.Rules, err) } - r.recorder.Event(mc, corev1.EventTypeNormal, eventReasonRoleCreated, "role was created") + r.recorder.Eventf(mc, nil, corev1.EventTypeNormal, eventReasonRoleCreated, "ReconcileRole", "role was created") klog.V(2).InfoS("created role", "memberCluster", klog.KObj(mc), "role", roleName) return roleName, nil } @@ -426,7 +439,7 @@ func (r *Reconciler) syncRole(ctx context.Context, mc *clusterv1beta1.MemberClus if err := r.Client.Update(ctx, ¤tRole, client.FieldOwner(utils.MCControllerFieldManagerName)); err != nil { return "", fmt.Errorf("failed to update role %s with rules %+v: %w", roleName, currentRole.Rules, err) } - r.recorder.Event(mc, corev1.EventTypeNormal, eventReasonRoleUpdated, "role was updated") + r.recorder.Eventf(mc, nil, corev1.EventTypeNormal, eventReasonRoleUpdated, "ReconcileRole", "role was updated") klog.V(2).InfoS("updated role", "memberCluster", klog.KObj(mc), "role", roleName) return roleName, nil } @@ -468,7 +481,7 @@ func (r *Reconciler) syncRoleBinding(ctx context.Context, mc *clusterv1beta1.Mem if err = r.Client.Create(ctx, &expectedRoleBinding, client.FieldOwner(utils.MCControllerFieldManagerName)); err != nil { return fmt.Errorf("failed to create role binding %s: %w", roleBindingName, err) } - r.recorder.Event(mc, corev1.EventTypeNormal, eventReasonRoleBindingCreated, "role binding was created") + r.recorder.Eventf(mc, nil, corev1.EventTypeNormal, eventReasonRoleBindingCreated, "ReconcileRoleBinding", "role binding was created") klog.V(2).InfoS("created role binding", "memberCluster", klog.KObj(mc), "subject", mc.Spec.Identity) return nil } @@ -483,7 +496,7 @@ func (r *Reconciler) syncRoleBinding(ctx context.Context, mc *clusterv1beta1.Mem if err := r.Client.Update(ctx, &expectedRoleBinding, client.FieldOwner(utils.MCControllerFieldManagerName)); err != nil { return fmt.Errorf("failed to update role binding %s: %w", roleBindingName, err) } - r.recorder.Event(mc, corev1.EventTypeNormal, eventReasonRoleBindingUpdated, "role binding was updated") + r.recorder.Eventf(mc, nil, corev1.EventTypeNormal, eventReasonRoleBindingUpdated, "ReconcileRoleBinding", "role binding was updated") klog.V(2).InfoS("updated role binding", "memberCluster", klog.KObj(mc), "subject", mc.Spec.Identity) return nil } @@ -514,7 +527,7 @@ func (r *Reconciler) syncInternalMemberCluster(ctx context.Context, mc *clusterv if err := r.Client.Create(ctx, &expectedImc, client.FieldOwner(utils.MCControllerFieldManagerName)); err != nil { return nil, controller.NewAPIServerError(false, fmt.Errorf("failed to create internal member cluster %s with spec %+v: %w", klog.KObj(&expectedImc), expectedImc.Spec, err)) } - r.recorder.Event(mc, corev1.EventTypeNormal, eventReasonIMCCreated, "Internal member cluster was created") + r.recorder.Eventf(mc, nil, corev1.EventTypeNormal, eventReasonIMCCreated, "ReconcileInternalMemberCluster", "Internal member cluster was created") klog.V(2).InfoS("created internal member cluster", "InternalMemberCluster", klog.KObj(&expectedImc), "spec", expectedImc.Spec) return &expectedImc, nil } @@ -528,7 +541,7 @@ func (r *Reconciler) syncInternalMemberCluster(ctx context.Context, mc *clusterv if err := r.Client.Update(ctx, currentImc, client.FieldOwner(utils.MCControllerFieldManagerName)); err != nil { return nil, controller.NewAPIServerError(false, fmt.Errorf("failed to update internal member cluster %s with spec %+v: %w", klog.KObj(currentImc), currentImc.Spec, err)) } - r.recorder.Event(mc, corev1.EventTypeNormal, eventReasonIMCSpecUpdated, "internal member cluster spec updated") + r.recorder.Eventf(mc, nil, corev1.EventTypeNormal, eventReasonIMCSpecUpdated, "ReconcileInternalMemberCluster", "internal member cluster spec updated") klog.V(2).InfoS("updated internal member cluster", "InternalMemberCluster", klog.KObj(currentImc), "spec", currentImc.Spec) return currentImc, nil } @@ -642,7 +655,7 @@ func (r *Reconciler) aggregateJoinedCondition(mc *clusterv1beta1.MemberCluster) } // markMemberClusterReadyToJoin is used to update the ReadyToJoin condition as true of member cluster. -func markMemberClusterReadyToJoin(recorder record.EventRecorder, mc apis.ConditionedObj) { +func markMemberClusterReadyToJoin(recorder events.EventRecorder, mc apis.ConditionedObj) { klog.V(4).InfoS("Mark the member cluster ReadyToJoin", "memberCluster", klog.KObj(mc)) newCondition := metav1.Condition{ Type: string(clusterv1beta1.ConditionTypeMemberClusterReadyToJoin), @@ -655,7 +668,7 @@ func markMemberClusterReadyToJoin(recorder record.EventRecorder, mc apis.Conditi // Joined status changed. existingCondition := mc.GetCondition(newCondition.Type) if existingCondition == nil || existingCondition.Status != newCondition.Status { - recorder.Event(mc, corev1.EventTypeNormal, reasonMemberClusterReadyToJoin, "member cluster ready to join") + recorder.Eventf(mc, nil, corev1.EventTypeNormal, reasonMemberClusterReadyToJoin, "Join", "member cluster ready to join") klog.V(2).InfoS("member cluster ready to join", "memberCluster", klog.KObj(mc)) } @@ -663,7 +676,7 @@ func markMemberClusterReadyToJoin(recorder record.EventRecorder, mc apis.Conditi } // markMemberClusterJoined is used to the update the status of the member cluster to have the joined condition. -func markMemberClusterJoined(recorder record.EventRecorder, mc apis.ConditionedObj) { +func markMemberClusterJoined(recorder events.EventRecorder, mc apis.ConditionedObj) { klog.V(4).InfoS("Mark the member cluster joined", "memberCluster", klog.KObj(mc)) newCondition := metav1.Condition{ Type: string(clusterv1beta1.ConditionTypeMemberClusterJoined), @@ -676,7 +689,7 @@ func markMemberClusterJoined(recorder record.EventRecorder, mc apis.ConditionedO // Joined status changed. existingCondition := mc.GetCondition(newCondition.Type) if existingCondition == nil || existingCondition.Status != newCondition.Status { - recorder.Event(mc, corev1.EventTypeNormal, reasonMemberClusterJoined, "member cluster joined") + recorder.Eventf(mc, nil, corev1.EventTypeNormal, reasonMemberClusterJoined, "Join", "member cluster joined") klog.V(2).InfoS("memberCluster joined", "memberCluster", klog.KObj(mc)) sharedmetrics.ReportJoinResultMetric() } @@ -685,7 +698,7 @@ func markMemberClusterJoined(recorder record.EventRecorder, mc apis.ConditionedO } // markMemberClusterLeft is used to update the status of the member cluster to have the left condition and mark member cluster as not ready to join. -func markMemberClusterLeft(recorder record.EventRecorder, mc apis.ConditionedObj) { +func markMemberClusterLeft(recorder events.EventRecorder, mc apis.ConditionedObj) { klog.V(4).InfoS("Mark the member cluster left", "memberCluster", klog.KObj(mc)) newCondition := metav1.Condition{ Type: string(clusterv1beta1.ConditionTypeMemberClusterJoined), @@ -705,7 +718,7 @@ func markMemberClusterLeft(recorder record.EventRecorder, mc apis.ConditionedObj // Joined status changed. existingCondition := mc.GetCondition(newCondition.Type) if existingCondition == nil || existingCondition.Status != newCondition.Status { - recorder.Event(mc, corev1.EventTypeNormal, reasonMemberClusterJoined, "member cluster left") + recorder.Eventf(mc, nil, corev1.EventTypeNormal, reasonMemberClusterLeft, "Leave", "member cluster left") klog.V(2).InfoS("memberCluster left", "memberCluster", klog.KObj(mc)) sharedmetrics.ReportLeaveResultMetric() } @@ -713,8 +726,8 @@ func markMemberClusterLeft(recorder record.EventRecorder, mc apis.ConditionedObj mc.SetConditions(newCondition, notReadyCondition) } -// markMemberClusterUnknown is used to update the status of the member cluster to have the left condition. -func markMemberClusterUnknown(recorder record.EventRecorder, mc apis.ConditionedObj, unknownMessage string) { +// markMemberClusterUnknown is used to update the status of the member cluster to have the joined condition set to unknown. +func markMemberClusterUnknown(recorder events.EventRecorder, mc apis.ConditionedObj, unknownMessage string) { klog.V(4).InfoS("Mark the member cluster join condition unknown", "memberCluster", klog.KObj(mc)) newCondition := metav1.Condition{ Type: string(clusterv1beta1.ConditionTypeMemberClusterJoined), @@ -727,7 +740,7 @@ func markMemberClusterUnknown(recorder record.EventRecorder, mc apis.Conditioned // Joined status changed. existingCondition := mc.GetCondition(newCondition.Type) if existingCondition == nil || existingCondition.Status != newCondition.Status { - recorder.Event(mc, corev1.EventTypeWarning, reasonMemberClusterUnknown, "member cluster join state unknown") + recorder.Eventf(mc, nil, corev1.EventTypeWarning, reasonMemberClusterUnknown, "Join", "member cluster join state unknown") klog.V(2).InfoS("memberCluster join state unknown", "memberCluster", klog.KObj(mc)) } @@ -736,7 +749,7 @@ func markMemberClusterUnknown(recorder record.EventRecorder, mc apis.Conditioned // SetupWithManager sets up the controller with the Manager. func (r *Reconciler) SetupWithManager(mgr runtime.Manager, name string) error { - r.recorder = mgr.GetEventRecorderFor("mcv1beta1") + r.recorder = mgr.GetEventRecorder("mcv1beta1") r.agents = make(map[clusterv1beta1.AgentType]bool) r.agents[clusterv1beta1.MemberAgent] = true diff --git a/pkg/controllers/membercluster/v1beta1/membercluster_controller_test.go b/pkg/controllers/membercluster/v1beta1/membercluster_controller_test.go index f393a3623..032c63f41 100644 --- a/pkg/controllers/membercluster/v1beta1/membercluster_controller_test.go +++ b/pkg/controllers/membercluster/v1beta1/membercluster_controller_test.go @@ -27,14 +27,13 @@ import ( "github.com/crossplane/crossplane-runtime/v2/pkg/test" "github.com/google/go-cmp/cmp" "github.com/google/go-cmp/cmp/cmpopts" - "github.com/stretchr/testify/assert" corev1 "k8s.io/api/core/v1" rbacv1 "k8s.io/api/rbac/v1" apierrors "k8s.io/apimachinery/pkg/api/errors" "k8s.io/apimachinery/pkg/api/resource" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/runtime/schema" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" ctrl "sigs.k8s.io/controller-runtime" "sigs.k8s.io/controller-runtime/pkg/client" @@ -71,22 +70,47 @@ func TestEnsureMemberNameLabel(t *testing.T) { wantLabels map[string]string wantErr string }{ - "label already present with correct value": { + "name and alias labels already present with correct values": { r: &Reconciler{ Client: &test.MockClient{ - MockUpdate: test.NewMockUpdateFn(fmt.Errorf("update should not be called when label is already correct")), + MockUpdate: test.NewMockUpdateFn(fmt.Errorf("update should not be called when the labels are already correct")), }, }, memberCluster: &clusterv1beta1.MemberCluster{ ObjectMeta: metav1.ObjectMeta{ Name: "mc1", Labels: map[string]string{ - placementv1beta1.MemberNameLabel: "mc1", + placementv1beta1.MemberNameLabel: "mc1", + placementv1beta1.ClusterAliasLabel: "mc1", + }, + }, + }, + wantLabels: map[string]string{ + placementv1beta1.MemberNameLabel: "mc1", + placementv1beta1.ClusterAliasLabel: "mc1", + }, + }, + // The alias is the admin's to rename: unlike the name label, a different value is left + // alone rather than reasserted, since the alias exists precisely so that a selector can + // follow a role while the cluster behind it changes. + "an alias renamed by an admin is not reverted": { + r: &Reconciler{ + Client: &test.MockClient{ + MockUpdate: test.NewMockUpdateFn(fmt.Errorf("update should not be called when the alias was deliberately renamed")), + }, + }, + memberCluster: &clusterv1beta1.MemberCluster{ + ObjectMeta: metav1.ObjectMeta{ + Name: "mc1", + Labels: map[string]string{ + placementv1beta1.MemberNameLabel: "mc1", + placementv1beta1.ClusterAliasLabel: "bravelion", }, }, }, wantLabels: map[string]string{ - placementv1beta1.MemberNameLabel: "mc1", + placementv1beta1.MemberNameLabel: "mc1", + placementv1beta1.ClusterAliasLabel: "bravelion", }, }, "no labels at all": { @@ -103,7 +127,8 @@ func TestEnsureMemberNameLabel(t *testing.T) { }, }, wantLabels: map[string]string{ - placementv1beta1.MemberNameLabel: "mc1", + placementv1beta1.MemberNameLabel: "mc1", + placementv1beta1.ClusterAliasLabel: "mc1", }, }, "labels exist but name label is missing": { @@ -123,8 +148,9 @@ func TestEnsureMemberNameLabel(t *testing.T) { }, }, wantLabels: map[string]string{ - "existing-label": "value", - placementv1beta1.MemberNameLabel: "mc1", + "existing-label": "value", + placementv1beta1.MemberNameLabel: "mc1", + placementv1beta1.ClusterAliasLabel: "mc1", }, }, "label present with wrong value": { @@ -144,7 +170,31 @@ func TestEnsureMemberNameLabel(t *testing.T) { }, }, wantLabels: map[string]string{ - placementv1beta1.MemberNameLabel: "mc1", + placementv1beta1.MemberNameLabel: "mc1", + placementv1beta1.ClusterAliasLabel: "mc1", + }, + }, + // The day-2 scenario: a member cluster labeled by the controller before the alias existed. + // Only the alias branch has anything to do, and it alone must drive the update. + "name label correct, alias absent, alias alone drives the update": { + r: &Reconciler{ + Client: &test.MockClient{ + MockUpdate: func(ctx context.Context, obj client.Object, opts ...client.UpdateOption) error { + return nil + }, + }, + }, + memberCluster: &clusterv1beta1.MemberCluster{ + ObjectMeta: metav1.ObjectMeta{ + Name: "mc1", + Labels: map[string]string{ + placementv1beta1.MemberNameLabel: "mc1", + }, + }, + }, + wantLabels: map[string]string{ + placementv1beta1.MemberNameLabel: "mc1", + placementv1beta1.ClusterAliasLabel: "mc1", }, }, "update error": { @@ -202,7 +252,7 @@ func TestReconcileEnsureMemberNameLabelError(t *testing.T) { return updateErr }, }, - recorder: record.NewFakeRecorder(10), + recorder: events.NewFakeRecorder(10), } result, err := r.Reconcile(context.Background(), ctrl.Request{ NamespacedName: client.ObjectKey{Name: "mc1"}, @@ -233,11 +283,11 @@ func TestSyncNamespace(t *testing.T) { return nil }, }, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), }, memberCluster: &clusterv1beta1.MemberCluster{ObjectMeta: metav1.ObjectMeta{Name: "mc1"}}, wantedNamespaceName: namespace1, - wantedEvent: utils.GetEventString(&clusterv1beta1.MemberCluster{ObjectMeta: metav1.ObjectMeta{Name: "mc1"}}, corev1.EventTypeNormal, eventReasonNamespaceCreated, "Namespace was created"), + wantedEvent: utils.GetEventString(corev1.EventTypeNormal, eventReasonNamespaceCreated, "Namespace was created"), wantedError: "", }, "namespace exists without label": { @@ -257,11 +307,11 @@ func TestSyncNamespace(t *testing.T) { return nil }, }, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), }, memberCluster: &clusterv1beta1.MemberCluster{ObjectMeta: metav1.ObjectMeta{Name: "mc1"}}, wantedNamespaceName: namespace1, - wantedEvent: utils.GetEventString(&clusterv1beta1.MemberCluster{ObjectMeta: metav1.ObjectMeta{Name: "mc1"}}, corev1.EventTypeNormal, eventReasonNamespacePatched, "Namespace was patched"), + wantedEvent: utils.GetEventString(corev1.EventTypeNormal, eventReasonNamespacePatched, "Namespace was patched"), wantedError: "", }, "namespace exists with label": { @@ -338,16 +388,22 @@ func TestSyncNamespace(t *testing.T) { t.Run(testName, func(t *testing.T) { got, err := tt.r.syncNamespace(context.Background(), tt.memberCluster) if tt.r.recorder != nil { - fakeRecorder := tt.r.recorder.(*record.FakeRecorder) + fakeRecorder := tt.r.recorder.(*events.FakeRecorder) event := <-fakeRecorder.Events - assert.Equal(t, tt.wantedEvent, event) + if event != tt.wantedEvent { + t.Errorf("emitted event %v, want %v", event, tt.wantedEvent) + } } if tt.wantedError == "" { - assert.Equal(t, err, nil, utils.TestCaseMsg, testName) - } else { - assert.Contains(t, err.Error(), tt.wantedError, utils.TestCaseMsg, testName) + if err != nil { + t.Errorf("syncNamespace() error = %v, want nil", err) + } + } else if err == nil || !strings.Contains(err.Error(), tt.wantedError) { + t.Errorf("syncNamespace() error = %v, want error containing %q", err, tt.wantedError) + } + if got != tt.wantedNamespaceName { + t.Errorf("syncNamespace() = %q, want %q", got, tt.wantedNamespaceName) } - assert.Equalf(t, tt.wantedNamespaceName, got, utils.TestCaseMsg, testName) }) } } @@ -355,8 +411,8 @@ func TestSyncNamespace(t *testing.T) { func TestSyncRole(t *testing.T) { expectedMemberCluster1 := clusterv1beta1.MemberCluster{ObjectMeta: metav1.ObjectMeta{Name: "mc2"}} expectedMemberCluster2 := clusterv1beta1.MemberCluster{ObjectMeta: metav1.ObjectMeta{Name: "mc3"}} - expectedEvent1 := utils.GetEventString(&expectedMemberCluster1, corev1.EventTypeNormal, eventReasonRoleUpdated, "role was updated") - expectedEvent2 := utils.GetEventString(&expectedMemberCluster2, corev1.EventTypeNormal, eventReasonRoleCreated, "role was created") + expectedEvent1 := utils.GetEventString(corev1.EventTypeNormal, eventReasonRoleUpdated, "role was updated") + expectedEvent2 := utils.GetEventString(corev1.EventTypeNormal, eventReasonRoleCreated, "role was created") tests := map[string]struct { r *Reconciler @@ -380,7 +436,7 @@ func TestSyncRole(t *testing.T) { Name: "fleet-role-mc1", Namespace: namespace1, }, - Rules: []rbacv1.PolicyRule{utils.FleetClusterRule, utils.FleetPlacementRule, utils.FleetNetworkRule, utils.EventRule}, + Rules: []rbacv1.PolicyRule{utils.FleetClusterRule, utils.FleetPlacementRule, utils.FleetNetworkRule, utils.EventRule, utils.EventsK8sIoRule}, } return nil }, @@ -408,7 +464,7 @@ func TestSyncRole(t *testing.T) { return nil }, }, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), }, memberCluster: &expectedMemberCluster1, namespaceName: namespace2, @@ -426,7 +482,7 @@ func TestSyncRole(t *testing.T) { return nil }, }, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), }, memberCluster: &expectedMemberCluster2, namespaceName: namespace3, @@ -491,16 +547,22 @@ func TestSyncRole(t *testing.T) { t.Run(testName, func(t *testing.T) { got, err := tt.r.syncRole(context.Background(), tt.memberCluster, tt.namespaceName) if tt.r.recorder != nil { - fakeRecorder := tt.r.recorder.(*record.FakeRecorder) + fakeRecorder := tt.r.recorder.(*events.FakeRecorder) event := <-fakeRecorder.Events - assert.Equal(t, tt.wantedEvent, event) + if event != tt.wantedEvent { + t.Errorf("emitted event %v, want %v", event, tt.wantedEvent) + } } if tt.wantedError == "" { - assert.Equal(t, err, nil, utils.TestCaseMsg, testName) - } else { - assert.Contains(t, err.Error(), tt.wantedError, utils.TestCaseMsg, testName) + if err != nil { + t.Errorf("syncRole() error = %v, want nil", err) + } + } else if err == nil || !strings.Contains(err.Error(), tt.wantedError) { + t.Errorf("syncRole() error = %v, want error containing %q", err, tt.wantedError) + } + if got != tt.wantedRoleName { + t.Errorf("syncRole() = %q, want %q", got, tt.wantedRoleName) } - assert.Equalf(t, tt.wantedRoleName, got, utils.TestCaseMsg, testName) }) } } @@ -540,8 +602,8 @@ func TestSyncRoleBinding(t *testing.T) { ObjectMeta: metav1.ObjectMeta{Name: "mc3"}, Spec: clusterv1beta1.MemberClusterSpec{Identity: identity}, } - expectedEvent1 := utils.GetEventString(&expectedMemberCluster1, corev1.EventTypeNormal, eventReasonRoleBindingUpdated, "role binding was updated") - expectedEvent2 := utils.GetEventString(&expectedMemberCluster2, corev1.EventTypeNormal, eventReasonRoleBindingCreated, "role binding was created") + expectedEvent1 := utils.GetEventString(corev1.EventTypeNormal, eventReasonRoleBindingUpdated, "role binding was updated") + expectedEvent2 := utils.GetEventString(corev1.EventTypeNormal, eventReasonRoleBindingCreated, "role binding was created") tests := map[string]struct { r *Reconciler @@ -644,7 +706,7 @@ func TestSyncRoleBinding(t *testing.T) { return nil }, MockUpdate: updateMock}, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), }, memberCluster: &expectedMemberCluster1, namespaceName: namespace2, @@ -659,7 +721,7 @@ func TestSyncRoleBinding(t *testing.T) { return apierrors.NewNotFound(schema.GroupResource{Group: "", Resource: "Namespace"}, "namespace") }, MockCreate: createMock}, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), }, memberCluster: &expectedMemberCluster2, namespaceName: namespace3, @@ -721,14 +783,18 @@ func TestSyncRoleBinding(t *testing.T) { t.Run(testName, func(t *testing.T) { err := tt.r.syncRoleBinding(context.Background(), tt.memberCluster, tt.namespaceName, tt.roleName) if tt.r.recorder != nil { - fakeRecorder := tt.r.recorder.(*record.FakeRecorder) + fakeRecorder := tt.r.recorder.(*events.FakeRecorder) event := <-fakeRecorder.Events - assert.Equal(t, tt.wantedEvent, event) + if event != tt.wantedEvent { + t.Errorf("emitted event %v, want %v", event, tt.wantedEvent) + } } if tt.wantedError == "" { - assert.Equal(t, err, nil, utils.TestCaseMsg, testName) - } else { - assert.Contains(t, err.Error(), tt.wantedError, utils.TestCaseMsg, testName) + if err != nil { + t.Errorf("syncRoleBinding() error = %v, want nil", err) + } + } else if err == nil || !strings.Contains(err.Error(), tt.wantedError) { + t.Errorf("syncRoleBinding() error = %v, want error containing %q", err, tt.wantedError) } }) } @@ -764,8 +830,8 @@ func TestSyncInternalMemberCluster(t *testing.T) { Spec: clusterv1beta1.MemberClusterSpec{HeartbeatPeriodSeconds: 30}, } - expectedEvent1 := utils.GetEventString(&expectedLeavingMemberCluster, corev1.EventTypeNormal, eventReasonIMCSpecUpdated, "internal member cluster spec updated") - expectedEvent2 := utils.GetEventString(&expectedMemberCluster2, corev1.EventTypeNormal, eventReasonIMCCreated, "Internal member cluster was created") + expectedEvent1 := utils.GetEventString(corev1.EventTypeNormal, eventReasonIMCSpecUpdated, "internal member cluster spec updated") + expectedEvent2 := utils.GetEventString(corev1.EventTypeNormal, eventReasonIMCCreated, "Internal member cluster was created") tests := map[string]struct { r *Reconciler @@ -780,7 +846,7 @@ func TestSyncInternalMemberCluster(t *testing.T) { r: &Reconciler{ Client: &test.MockClient{ MockUpdate: updateMock}, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), }, memberCluster: &expectedLeavingMemberCluster, namespaceName: namespace1, @@ -827,7 +893,7 @@ func TestSyncInternalMemberCluster(t *testing.T) { r: &Reconciler{ Client: &test.MockClient{ MockCreate: createMock}, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), }, memberCluster: &expectedMemberCluster2, namespaceName: "fleet-mc4", @@ -853,24 +919,30 @@ func TestSyncInternalMemberCluster(t *testing.T) { t.Run(testName, func(t *testing.T) { got, err := tt.r.syncInternalMemberCluster(context.Background(), tt.memberCluster, tt.namespaceName, tt.internalMemberCluster) if tt.r.recorder != nil { - fakeRecorder := tt.r.recorder.(*record.FakeRecorder) + fakeRecorder := tt.r.recorder.(*events.FakeRecorder) event := <-fakeRecorder.Events - assert.Equal(t, tt.wantedEvent, event) + if event != tt.wantedEvent { + t.Errorf("emitted event %v, want %v", event, tt.wantedEvent) + } } if tt.wantedInternalMemberClusterSpec != nil { - assert.Equal(t, *tt.wantedInternalMemberClusterSpec, got.Spec, utils.TestCaseMsg, testName) + if diff := cmp.Diff(got.Spec, *tt.wantedInternalMemberClusterSpec); diff != "" { + t.Errorf("syncInternalMemberCluster() spec mismatch (-got, +want):\n%s", diff) + } } if tt.wantedError == "" { - assert.Equal(t, err, nil, utils.TestCaseMsg, testName) - } else { - assert.Contains(t, err.Error(), tt.wantedError, utils.TestCaseMsg, testName) + if err != nil { + t.Errorf("syncInternalMemberCluster() error = %v, want nil", err) + } + } else if err == nil || !strings.Contains(err.Error(), tt.wantedError) { + t.Errorf("syncInternalMemberCluster() error = %v, want error containing %q", err, tt.wantedError) } }) } } func TestMarkMemberClusterJoined(t *testing.T) { - recorder := utils.NewFakeRecorder(1) + recorder := events.NewFakeRecorder(1) memberCluster := &clusterv1beta1.MemberCluster{ TypeMeta: metav1.TypeMeta{ Kind: clusterv1beta1.InternalMemberClusterKind, @@ -881,8 +953,10 @@ func TestMarkMemberClusterJoined(t *testing.T) { // check that the correct event is emitted event := <-recorder.Events - expected := utils.GetEventString(memberCluster, corev1.EventTypeNormal, reasonMemberClusterJoined, "member cluster joined") - assert.Equal(t, expected, event) + expected := utils.GetEventString(corev1.EventTypeNormal, reasonMemberClusterJoined, "member cluster joined") + if event != expected { + t.Errorf("markMemberClusterJoined() emitted event %v, want %v", event, expected) + } // Check expected conditions. expectedConditions := []metav1.Condition{ @@ -891,7 +965,9 @@ func TestMarkMemberClusterJoined(t *testing.T) { for i := range expectedConditions { actualCondition := memberCluster.GetCondition(expectedConditions[i].Type) - assert.Equal(t, "", cmp.Diff(&expectedConditions[i], actualCondition, cmpopts.IgnoreTypes(time.Time{}))) + if diff := cmp.Diff(actualCondition, &expectedConditions[i], cmpopts.IgnoreTypes(time.Time{})); diff != "" { + t.Errorf("markMemberClusterJoined() condition mismatch (-got, +want):\n%s", diff) + } } } @@ -907,7 +983,7 @@ func TestSyncInternalMemberClusterStatus(t *testing.T) { }{ "copy with Joined condition": { r: &Reconciler{ - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), agents: map[clusterv1beta1.AgentType]bool{ clusterv1beta1.MemberAgent: true, clusterv1beta1.ServiceExportImportAgent: true, @@ -1077,7 +1153,7 @@ func TestSyncInternalMemberClusterStatus(t *testing.T) { }, "copy with Left condition": { r: &Reconciler{ - recorder: utils.NewFakeRecorder(2), + recorder: events.NewFakeRecorder(2), agents: map[clusterv1beta1.AgentType]bool{ clusterv1beta1.MemberAgent: true, clusterv1beta1.ServiceExportImportAgent: true, @@ -1183,7 +1259,7 @@ func TestSyncInternalMemberClusterStatus(t *testing.T) { }, "copy with Unknown condition": { r: &Reconciler{ - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), agents: map[clusterv1beta1.AgentType]bool{ clusterv1beta1.MemberAgent: true, clusterv1beta1.ServiceExportImportAgent: true, @@ -1283,7 +1359,7 @@ func TestSyncInternalMemberClusterStatus(t *testing.T) { }, "No Agent Status": { r: &Reconciler{ - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), agents: map[clusterv1beta1.AgentType]bool{ clusterv1beta1.MemberAgent: true, }, @@ -1330,7 +1406,7 @@ func TestSyncInternalMemberClusterStatus(t *testing.T) { }, "Internal member cluster is nil": { r: &Reconciler{ - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), agents: map[clusterv1beta1.AgentType]bool{ clusterv1beta1.MemberAgent: true, }, @@ -1341,7 +1417,7 @@ func TestSyncInternalMemberClusterStatus(t *testing.T) { }, "other agent type reported in the status and should be ignored": { r: &Reconciler{ - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), agents: map[clusterv1beta1.AgentType]bool{ clusterv1beta1.MemberAgent: true, clusterv1beta1.ServiceExportImportAgent: true, @@ -1465,7 +1541,7 @@ func TestSyncInternalMemberClusterStatus(t *testing.T) { }, "less agent type reported in the status": { r: &Reconciler{ - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), agents: map[clusterv1beta1.AgentType]bool{ clusterv1beta1.MemberAgent: true, clusterv1beta1.ServiceExportImportAgent: true, @@ -1541,7 +1617,7 @@ func TestSyncInternalMemberClusterStatus(t *testing.T) { }, "condition is not reported in the status": { r: &Reconciler{ - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), agents: map[clusterv1beta1.AgentType]bool{ clusterv1beta1.MemberAgent: true, clusterv1beta1.ServiceExportImportAgent: true, @@ -1625,7 +1701,7 @@ func TestSyncInternalMemberClusterStatus(t *testing.T) { }, "agent type is not reported in the status": { r: &Reconciler{ - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), agents: map[clusterv1beta1.AgentType]bool{ clusterv1beta1.MemberAgent: true, clusterv1beta1.ServiceExportImportAgent: true, @@ -1743,7 +1819,7 @@ func TestUpdateMemberClusterStatus(t *testing.T) { count++ return nil }}, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), }, memberCluster: &clusterv1beta1.MemberCluster{}, wantedError: "", @@ -1760,7 +1836,7 @@ func TestUpdateMemberClusterStatus(t *testing.T) { } return apierrors.NewServerTimeout(schema.GroupResource{}, "", 1) }}, - recorder: utils.NewFakeRecorder(10), + recorder: events.NewFakeRecorder(10), }, memberCluster: &clusterv1beta1.MemberCluster{Spec: clusterv1beta1.MemberClusterSpec{HeartbeatPeriodSeconds: int32(5)}}, wantedError: "", @@ -1774,7 +1850,7 @@ func TestUpdateMemberClusterStatus(t *testing.T) { count++ return apierrors.NewServerTimeout(schema.GroupResource{}, "", 1) }}, - recorder: utils.NewFakeRecorder(10), + recorder: events.NewFakeRecorder(10), }, memberCluster: &clusterv1beta1.MemberCluster{}, wantedError: "The operation against could not be completed at this time, please try again.", @@ -1788,7 +1864,7 @@ func TestUpdateMemberClusterStatus(t *testing.T) { count++ return errors.New("random update error") }}, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), }, memberCluster: &clusterv1beta1.MemberCluster{}, wantedError: "random update error", @@ -1803,11 +1879,15 @@ func TestUpdateMemberClusterStatus(t *testing.T) { count = -1 err := tt.r.updateMemberClusterStatus(context.Background(), tt.memberCluster) if tt.wantedError == "" { - assert.Equal(t, err, nil, utils.TestCaseMsg, testName) - } else { - assert.Contains(t, err.Error(), tt.wantedError, utils.TestCaseMsg, testName) + if err != nil { + t.Errorf("updateMemberClusterStatus() error = %v, want nil", err) + } + } else if err == nil || !strings.Contains(err.Error(), tt.wantedError) { + t.Errorf("updateMemberClusterStatus() error = %v, want error containing %q", err, tt.wantedError) + } + if !tt.verifyNumberOfRetry() { + t.Error("updateMemberClusterStatus() retried an unexpected number of times") } - assert.Equal(t, tt.verifyNumberOfRetry(), true, utils.TestCaseMsg, testName) }) } } @@ -1827,7 +1907,7 @@ func TestHandleDelete(t *testing.T) { }{ "do nothing when the mc has no finalizer": { r: &Reconciler{Client: &test.MockClient{}, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), }, memberCluster: &clusterv1beta1.MemberCluster{}, wantResult: ctrl.Result{}, @@ -1850,7 +1930,7 @@ func TestHandleDelete(t *testing.T) { } return nil }}, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), }, memberCluster: memberClusterWithFinalizer.DeepCopy(), wantResult: ctrl.Result{}, @@ -1871,7 +1951,7 @@ func TestHandleDelete(t *testing.T) { } return nil }}, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), }, memberCluster: memberClusterWithFinalizer.DeepCopy(), wantResult: ctrl.Result{RequeueAfter: time.Second}, @@ -1892,7 +1972,7 @@ func TestHandleDelete(t *testing.T) { } return nil }}, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), }, memberCluster: memberClusterWithFinalizer.DeepCopy(), wantResult: ctrl.Result{RequeueAfter: time.Second}, @@ -1931,7 +2011,7 @@ func TestHandleDelete(t *testing.T) { return nil }, }, - recorder: utils.NewFakeRecorder(1), + recorder: events.NewFakeRecorder(1), }, memberCluster: memberClusterWithFinalizer.DeepCopy(), wantResult: ctrl.Result{Requeue: true}, diff --git a/pkg/controllers/placement/controller.go b/pkg/controllers/placement/controller.go index 42a5a3276..c051758c9 100644 --- a/pkg/controllers/placement/controller.go +++ b/pkg/controllers/placement/controller.go @@ -33,7 +33,7 @@ import ( "k8s.io/apimachinery/pkg/runtime" "k8s.io/apimachinery/pkg/types" utilerrors "k8s.io/apimachinery/pkg/util/errors" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" "k8s.io/klog/v2" ctrl "sigs.k8s.io/controller-runtime" "sigs.k8s.io/controller-runtime/pkg/client" @@ -64,7 +64,7 @@ type Reconciler struct { // It's only needed by v1beta1 APIs. UncachedReader client.Reader - Recorder record.EventRecorder + Recorder events.EventRecorder Scheme *runtime.Scheme @@ -137,7 +137,7 @@ func (r *Reconciler) handleDelete(ctx context.Context, placementObj fleetv1beta1 return ctrl.Result{}, err } klog.V(2).InfoS("Removed placement-cleanup finalizer", "placement", placementKObj) - r.Recorder.Event(placementObj, corev1.EventTypeNormal, "PlacementCleanupFinalizerRemoved", "Deleted the snapshots and removed the placement cleanup finalizer") + r.Recorder.Eventf(placementObj, nil, corev1.EventTypeNormal, "PlacementCleanupFinalizerRemoved", "RemoveFinalizer", "Deleted the snapshots and removed the placement cleanup finalizer") return ctrl.Result{}, nil } @@ -246,7 +246,7 @@ func (r *Reconciler) handleUpdate(ctx context.Context, placementObj fleetv1beta1 if !condition.IsConditionStatusTrue(oldCond, oldPlacement.GetGeneration()) && condition.IsConditionStatusTrue(newCond, placementObj.GetGeneration()) { klog.V(2).InfoS("Placement resource condition status has been changed to true", "placement", placementKObj, "generation", placementObj.GetGeneration(), "condition", conditionType) - r.Recorder.Event(placementObj, corev1.EventTypeNormal, i.EventReasonForTrue(), i.EventMessageForTrue()) + r.Recorder.Eventf(placementObj, nil, corev1.EventTypeNormal, i.EventReasonForTrue(), "UpdatePlacementStatus", i.EventMessageForTrue()) } } @@ -255,7 +255,7 @@ func (r *Reconciler) handleUpdate(ctx context.Context, placementObj fleetv1beta1 if isRolloutCompleted(placementObj) { if !isRolloutCompleted(oldPlacement) { klog.V(2).InfoS("Placement has finished the rollout process and reached the desired status", "placement", placementKObj, "generation", placementObj.GetGeneration()) - r.Recorder.Event(placementObj, corev1.EventTypeNormal, "PlacementRolloutCompleted", "Placement has finished the rollout process and reached the desired status") + r.Recorder.Eventf(placementObj, nil, corev1.EventTypeNormal, "PlacementRolloutCompleted", "UpdatePlacementStatus", "Placement has finished the rollout process and reached the desired status") } if createResourceSnapshotRes.RequeueAfter > 0 { klog.V(2).InfoS("Requeue the request to handle the new resource snapshot", "placement", placementKObj, "generation", placementObj.GetGeneration()) diff --git a/pkg/controllers/placement/controller_test.go b/pkg/controllers/placement/controller_test.go index e94b77890..a175ccd11 100644 --- a/pkg/controllers/placement/controller_test.go +++ b/pkg/controllers/placement/controller_test.go @@ -32,7 +32,7 @@ import ( metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/runtime" "k8s.io/apimachinery/pkg/types" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" "k8s.io/utils/ptr" ctrl "sigs.k8s.io/controller-runtime" "sigs.k8s.io/controller-runtime/pkg/client" @@ -740,7 +740,7 @@ func TestGetOrCreateClusterSchedulingPolicySnapshot(t *testing.T) { r := Reconciler{ Client: fakeClient, Scheme: scheme, - Recorder: record.NewFakeRecorder(10), + Recorder: events.NewFakeRecorder(10), } limit := int32(defaulter.DefaultRevisionHistoryLimitValue) if tc.revisionHistoryLimit != nil { @@ -1043,7 +1043,7 @@ func TestGetOrCreateClusterSchedulingPolicySnapshot_failure(t *testing.T) { r := Reconciler{ Client: fakeClient, Scheme: scheme, - Recorder: record.NewFakeRecorder(10), + Recorder: events.NewFakeRecorder(10), } _, err := r.getOrCreateSchedulingPolicySnapshot(ctx, crp, 1) if err == nil { // if error is nil @@ -1291,7 +1291,7 @@ func TestHandleDelete(t *testing.T) { Client: fakeClient, Scheme: scheme, UncachedReader: fakeClient, - Recorder: record.NewFakeRecorder(10), + Recorder: events.NewFakeRecorder(10), } got, err := r.handleDelete(ctx, crp) if err != nil { diff --git a/pkg/controllers/placement/placement_status_test.go b/pkg/controllers/placement/placement_status_test.go index a08fd6611..2534b887e 100644 --- a/pkg/controllers/placement/placement_status_test.go +++ b/pkg/controllers/placement/placement_status_test.go @@ -28,7 +28,7 @@ import ( "github.com/google/go-cmp/cmp/cmpopts" corev1 "k8s.io/api/core/v1" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" "k8s.io/utils/ptr" "sigs.k8s.io/controller-runtime/pkg/client" "sigs.k8s.io/controller-runtime/pkg/client/fake" @@ -5999,7 +5999,7 @@ func TestSetPlacementStatusForClusterResourcePlacement(t *testing.T) { r := Reconciler{ Client: fakeClient, Scheme: scheme, - Recorder: record.NewFakeRecorder(10), + Recorder: events.NewFakeRecorder(10), } crp.Generation = crpGeneration got, err := r.setPlacementStatus(context.Background(), crp, selectedResources, tc.latestPolicySnapshot, tc.latestResourceSnapshot) @@ -6686,7 +6686,7 @@ func TestSetResourcePlacementStatus(t *testing.T) { r := Reconciler{ Client: fakeClient, Scheme: scheme, - Recorder: record.NewFakeRecorder(10), + Recorder: events.NewFakeRecorder(10), } rp.Generation = rpGeneration got, err := r.setPlacementStatus(context.Background(), rp, selectedResources, tc.latestPolicySnapshot, tc.latestResourceSnapshot) @@ -9455,7 +9455,7 @@ func TestSetPlacementStatusPerCluster(t *testing.T) { } r := Reconciler{ - Recorder: record.NewFakeRecorder(10), + Recorder: events.NewFakeRecorder(10), } status := fleetv1beta1.PerClusterPlacementStatus{ClusterName: cluster} got := r.setPerClusterPlacementStatus(tc.placement, resourceSnapshot, "0", tc.binding, &status, tc.allConditionType) diff --git a/pkg/controllers/placement/suite_test.go b/pkg/controllers/placement/suite_test.go index 481c8683a..201480558 100644 --- a/pkg/controllers/placement/suite_test.go +++ b/pkg/controllers/placement/suite_test.go @@ -124,7 +124,7 @@ var _ = BeforeSuite(func() { Client: mgr.GetClient(), Scheme: mgr.GetScheme(), UncachedReader: mgr.GetAPIReader(), - Recorder: mgr.GetEventRecorderFor(controllerName), + Recorder: mgr.GetEventRecorder(controllerName), ResourceSelectorResolver: resourceSelectorResolver, ResourceSnapshotResolver: resourceSnapshotResolver, } diff --git a/pkg/controllers/resourcechange/resourcechange_controller.go b/pkg/controllers/resourcechange/resourcechange_controller.go index 463163f83..3fa9d9879 100644 --- a/pkg/controllers/resourcechange/resourcechange_controller.go +++ b/pkg/controllers/resourcechange/resourcechange_controller.go @@ -29,7 +29,7 @@ import ( "k8s.io/apimachinery/pkg/runtime" "k8s.io/apimachinery/pkg/runtime/schema" "k8s.io/client-go/dynamic" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" "k8s.io/klog/v2" ctrl "sigs.k8s.io/controller-runtime" "sigs.k8s.io/controller-runtime/pkg/client" @@ -59,7 +59,7 @@ type Reconciler struct { ResourcePlacementController controller.Controller // Event recorder to indicate the which placement picks up this object - Recorder record.EventRecorder + Recorder events.EventRecorder } func (r *Reconciler) Reconcile(_ context.Context, key controller.QueueKey) (ctrl.Result, error) { diff --git a/pkg/controllers/rollout/controller.go b/pkg/controllers/rollout/controller.go index 48fa0f143..00e8aaf6a 100644 --- a/pkg/controllers/rollout/controller.go +++ b/pkg/controllers/rollout/controller.go @@ -29,7 +29,7 @@ import ( metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/types" "k8s.io/apimachinery/pkg/util/intstr" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" "k8s.io/client-go/util/workqueue" "k8s.io/klog/v2" runtime "sigs.k8s.io/controller-runtime" @@ -54,7 +54,7 @@ type Reconciler struct { UncachedReader client.Reader // the max number of concurrent reconciles per controller. MaxConcurrentReconciles int - recorder record.EventRecorder + recorder events.EventRecorder // the informer contains the cache for all the resources we need. // to check the resource scope InformerManager informer.Manager @@ -691,7 +691,7 @@ func (r *Reconciler) updateBindings(ctx context.Context, bindings []toBeUpdatedB // The rollout controller watches resource snapshots and resource bindings. // It reconciles on the CRP when a new cluster resource binding is created or an existing cluster resource binding is created/updated. func (r *Reconciler) SetupWithManagerForClusterResourcePlacement(mgr runtime.Manager) error { - r.recorder = mgr.GetEventRecorderFor("cluster-resource-placement-rollout-controller") + r.recorder = mgr.GetEventRecorder("cluster-resource-placement-rollout-controller") return runtime.NewControllerManagedBy(mgr).Named("cluster-resource-placement-rollout-controller"). WithOptions(ctrl.Options{MaxConcurrentReconciles: r.MaxConcurrentReconciles}). // set the max number of concurrent reconciles Watches(&placementv1beta1.ClusterResourceSnapshot{}, resourceSnapshotObjHandlerFuncs()). @@ -736,7 +736,7 @@ func (r *Reconciler) SetupWithManagerForClusterResourcePlacement(mgr runtime.Man // The rollout controller watches resource snapshots and resource bindings. // It reconciles on the RP when a new resource binding is created or an existing resource binding is created/updated. func (r *Reconciler) SetupWithManagerForResourcePlacement(mgr runtime.Manager) error { - r.recorder = mgr.GetEventRecorderFor("resource-placement-rollout-controller") + r.recorder = mgr.GetEventRecorder("resource-placement-rollout-controller") return runtime.NewControllerManagedBy(mgr).Named("resource-placement-rollout-controller"). WithOptions(ctrl.Options{MaxConcurrentReconciles: r.MaxConcurrentReconciles}). // set the max number of concurrent reconciles Watches(&placementv1beta1.ResourceSnapshot{}, resourceSnapshotObjHandlerFuncs()). diff --git a/pkg/controllers/updaterun/controller.go b/pkg/controllers/updaterun/controller.go index 1f1bf0c17..56d50875e 100644 --- a/pkg/controllers/updaterun/controller.go +++ b/pkg/controllers/updaterun/controller.go @@ -26,7 +26,7 @@ import ( "k8s.io/apimachinery/pkg/api/meta" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/types" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" "k8s.io/client-go/util/workqueue" "k8s.io/klog/v2" runtime "sigs.k8s.io/controller-runtime" @@ -57,7 +57,7 @@ var ( // Reconciler reconciles an updateRun object. type Reconciler struct { client.Client - recorder record.EventRecorder + recorder events.EventRecorder // the informer contains the cache for all the resources we need to check the resource scope. InformerManager informer.Manager @@ -345,7 +345,7 @@ func (r *Reconciler) recordUpdateRunStatus(ctx context.Context, updateRun placem // SetupWithManagerForClusterStagedUpdateRun sets up the controller with the Manager for ClusterStagedUpdateRun resources. func (r *Reconciler) SetupWithManagerForClusterStagedUpdateRun(mgr runtime.Manager) error { - r.recorder = mgr.GetEventRecorderFor("clusterstagedupdaterun-controller") + r.recorder = mgr.GetEventRecorder("clusterstagedupdaterun-controller") return runtime.NewControllerManagedBy(mgr). Named("clusterstagedupdaterun-controller"). For(&placementv1beta1.ClusterStagedUpdateRun{}, builder.WithPredicates(predicate.GenerationChangedPredicate{})). @@ -365,7 +365,7 @@ func (r *Reconciler) SetupWithManagerForClusterStagedUpdateRun(mgr runtime.Manag // SetupWithManagerForStagedUpdateRun sets up the controller with the Manager for StagedUpdateRun resources. func (r *Reconciler) SetupWithManagerForStagedUpdateRun(mgr runtime.Manager) error { - r.recorder = mgr.GetEventRecorderFor("stagedupdaterun-controller") + r.recorder = mgr.GetEventRecorder("stagedupdaterun-controller") return runtime.NewControllerManagedBy(mgr). Named("stagedupdaterun-controller"). For(&placementv1beta1.StagedUpdateRun{}, builder.WithPredicates(predicate.GenerationChangedPredicate{})). diff --git a/pkg/controllers/workapplier/controller.go b/pkg/controllers/workapplier/controller.go index 7e91365d1..02117d572 100644 --- a/pkg/controllers/workapplier/controller.go +++ b/pkg/controllers/workapplier/controller.go @@ -30,7 +30,7 @@ import ( "k8s.io/apimachinery/pkg/runtime/schema" "k8s.io/apimachinery/pkg/types" "k8s.io/client-go/dynamic" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" "k8s.io/client-go/util/workqueue" "k8s.io/klog/v2" "k8s.io/utils/ptr" @@ -104,7 +104,7 @@ type Reconciler struct { spokeDynamicClient dynamic.Interface spokeClient client.Client restMapper meta.RESTMapper - recorder record.EventRecorder + recorder events.EventRecorder concurrentReconciles int deletionWaitTime time.Duration joined *atomic.Bool @@ -126,7 +126,7 @@ func NewReconciler( controllerName string, hubClient client.Client, workNameSpace string, spokeDynamicClient dynamic.Interface, spokeClient client.Client, restMapper meta.RESTMapper, - recorder record.EventRecorder, + recorder events.EventRecorder, concurrentReconciles int, parallelizer parallelizerutil.Parallelizer, deletionWaitTime time.Duration, diff --git a/pkg/controllers/workapplier/suite_test.go b/pkg/controllers/workapplier/suite_test.go index 4b56589a5..f94393ee3 100644 --- a/pkg/controllers/workapplier/suite_test.go +++ b/pkg/controllers/workapplier/suite_test.go @@ -319,7 +319,7 @@ var _ = BeforeSuite(func() { memberDynamicClient1, memberClient1, memberClient1.RESTMapper(), - hubMgr1.GetEventRecorderFor("work-applier"), + hubMgr1.GetEventRecorder("work-applier"), maxConcurrentReconciles, parallelizer.NewParallelizer(workerCount), 30*time.Second, @@ -370,7 +370,7 @@ var _ = BeforeSuite(func() { memberDynamicClient2, memberClient2, memberClient2.RESTMapper(), - hubMgr2.GetEventRecorderFor("work-applier-long-backoff"), + hubMgr2.GetEventRecorder("work-applier-long-backoff"), maxConcurrentReconciles, parallelizer.NewParallelizer(workerCount), 30*time.Second, @@ -409,7 +409,7 @@ var _ = BeforeSuite(func() { memberDynamicClient3, memberClient3, memberClient3.RESTMapper(), - hubMgr3.GetEventRecorderFor("work-applier"), + hubMgr3.GetEventRecorder("work-applier"), maxConcurrentReconciles, pWithDelay, 30*time.Second, @@ -446,7 +446,7 @@ var _ = BeforeSuite(func() { memberDynamicClient4, wrappedMemberClient4, memberClient4.RESTMapper(), - hubMgr4.GetEventRecorderFor("work-applier-wrapped-client"), + hubMgr4.GetEventRecorder("work-applier-wrapped-client"), maxConcurrentReconciles, parallelizer.NewParallelizer(workerCount), 30*time.Second, diff --git a/pkg/controllers/workgenerator/controller.go b/pkg/controllers/workgenerator/controller.go index 24b6d069e..67085fcc5 100644 --- a/pkg/controllers/workgenerator/controller.go +++ b/pkg/controllers/workgenerator/controller.go @@ -34,7 +34,7 @@ import ( "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" "k8s.io/apimachinery/pkg/runtime" "k8s.io/apimachinery/pkg/types" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" "k8s.io/client-go/util/retry" "k8s.io/client-go/util/workqueue" "k8s.io/klog/v2" @@ -76,7 +76,7 @@ type Reconciler struct { client.Client // the max number of concurrent reconciles per controller. MaxConcurrentReconciles int - recorder record.EventRecorder + recorder events.EventRecorder // the informer contains the cache for all the resources we need. // to check the resource scope InformerManager informer.Manager @@ -1507,7 +1507,7 @@ func extractDiffedResourcePlacementsFromWork(work *fleetv1beta1.Work) []fleetv1b // SetupWithManagerForClusterResourceBinding sets up the controller with the Manager. // It watches clusterResourceBinding events and also update/delete events for work. func (r *Reconciler) SetupWithManagerForClusterResourceBinding(mgr controllerruntime.Manager) error { - r.recorder = mgr.GetEventRecorderFor("cluster resource binding work generator") + r.recorder = mgr.GetEventRecorder("cluster-resource-binding-work-generator") return controllerruntime.NewControllerManagedBy(mgr).Named("cluster-resource-binding-work-generator"). WithOptions(ctrl.Options{MaxConcurrentReconciles: r.MaxConcurrentReconciles}). // set the max number of concurrent reconciles For(&fleetv1beta1.ClusterResourceBinding{}, builder.WithPredicates(predicate.GenerationChangedPredicate{})). @@ -1518,7 +1518,7 @@ func (r *Reconciler) SetupWithManagerForClusterResourceBinding(mgr controllerrun // SetupWithManagerForResourceBinding sets up the controller with the Manager. // It watches resourceBinding events and also update/delete events for work. func (r *Reconciler) SetupWithManagerForResourceBinding(mgr controllerruntime.Manager) error { - r.recorder = mgr.GetEventRecorderFor("resource binding work generator") + r.recorder = mgr.GetEventRecorder("resource-binding-work-generator") return controllerruntime.NewControllerManagedBy(mgr).Named("resource-binding-work-generator"). WithOptions(ctrl.Options{MaxConcurrentReconciles: r.MaxConcurrentReconciles}). // set the max number of concurrent reconciles For(&fleetv1beta1.ResourceBinding{}, builder.WithPredicates(predicate.GenerationChangedPredicate{})). diff --git a/pkg/controllers/workgenerator/controller_test.go b/pkg/controllers/workgenerator/controller_test.go index 7b4310448..1da0585c8 100644 --- a/pkg/controllers/workgenerator/controller_test.go +++ b/pkg/controllers/workgenerator/controller_test.go @@ -31,7 +31,7 @@ import ( "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" "k8s.io/apimachinery/pkg/runtime" "k8s.io/apimachinery/pkg/runtime/schema" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" "k8s.io/utils/ptr" "sigs.k8s.io/controller-runtime/pkg/client" "sigs.k8s.io/controller-runtime/pkg/client/fake" @@ -364,7 +364,7 @@ func TestUpsertWork(t *testing.T) { // Create reconciler with custom client reconciler := &Reconciler{ Client: fakeClient, - recorder: record.NewFakeRecorder(10), + recorder: events.NewFakeRecorder(10), InformerManager: &informer.FakeManager{}, } changed, _ := reconciler.upsertWork(ctx, newWork, tt.existingWork, resourceSnapshot) @@ -3602,7 +3602,7 @@ func TestUpdateBindingStatusWithRetry(t *testing.T) { // Create reconciler with custom client r := &Reconciler{ Client: conflictClient, - recorder: record.NewFakeRecorder(10), + recorder: events.NewFakeRecorder(10), InformerManager: &informer.FakeManager{}, } err := r.updateBindingStatusWithRetry(ctx, tt.resourceBinding) diff --git a/pkg/controllers/workgenerator/envelope.go b/pkg/controllers/workgenerator/envelope.go index c79383874..6e117f092 100644 --- a/pkg/controllers/workgenerator/envelope.go +++ b/pkg/controllers/workgenerator/envelope.go @@ -121,9 +121,11 @@ func (r *Reconciler) createOrUpdateEnvelopeCRWorkObj( "resourceBinding", klog.KObj(binding), "resourceSnapshot", klog.KObj(resourceSnapshot), "envelope", envelopeReader.GetEnvelopeObjRef()) - r.recorder.Eventf(binding, corev1.EventTypeWarning, "DuplicateEnvelopeWorks", - "Multiple Work objects (%v) found for envelope %v in namespace %s; delete all but the oldest to recover", - workNames, envelopeReader.GetEnvelopeObjRef(), fmt.Sprintf(utils.NamespaceNameFormat, binding.GetBindingSpec().TargetCluster)) + // events.k8s.io rejects notes longer than 1024 bytes, so cap the name list; + // the full list is on the log line above. + r.recorder.Eventf(binding, nil, corev1.EventTypeWarning, "DuplicateEnvelopeWorks", "GenerateWork", + "%d Work objects (%.512s) found for envelope %v in namespace %s; delete all but the oldest to recover", + len(workNames), strings.Join(workNames, ", "), envelopeReader.GetEnvelopeObjRef(), fmt.Sprintf(utils.NamespaceNameFormat, binding.GetBindingSpec().TargetCluster)) return nil, false, controller.NewUnexpectedBehaviorError(wrappedErr) case len(workList.Items) == 1: klog.V(2).InfoS("Found existing work object for the envelope; updating it", diff --git a/pkg/controllers/workgenerator/envelope_test.go b/pkg/controllers/workgenerator/envelope_test.go index 017909b76..be1e79503 100644 --- a/pkg/controllers/workgenerator/envelope_test.go +++ b/pkg/controllers/workgenerator/envelope_test.go @@ -32,7 +32,7 @@ import ( "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" "k8s.io/apimachinery/pkg/runtime" "k8s.io/apimachinery/pkg/runtime/schema" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" "sigs.k8s.io/controller-runtime/pkg/client" "sigs.k8s.io/controller-runtime/pkg/client/fake" @@ -470,7 +470,7 @@ func TestCreateOrUpdateEnvelopeCRWorkObj_EmptyManifestListRetained(t *testing.T) APIResources: map[schema.GroupVersionKind]bool{utils.DeploymentGVK: true}, IsClusterScopedResource: false, }, - recorder: record.NewFakeRecorder(10), + recorder: events.NewFakeRecorder(10), } roMap := map[fleetv1beta1.ResourceIdentifier][]*fleetv1beta1.ResourceOverrideSnapshot{ deploymentResourceIdentifier("app", "web"): {resourceOverrideSnapshot("delete-ro", "app", deleteOverrideRule())}, @@ -880,7 +880,7 @@ func TestCreateOrUpdateEnvelopeCRWorkObj(t *testing.T) { // Create reconciler r := &Reconciler{ Client: fakeClient, - recorder: record.NewFakeRecorder(10), + recorder: events.NewFakeRecorder(10), InformerManager: &informer.FakeManager{}, } @@ -1046,7 +1046,7 @@ func TestProcessOneSelectedResource(t *testing.T) { // Create reconciler r := &Reconciler{ Client: fakeClient, - recorder: record.NewFakeRecorder(10), + recorder: events.NewFakeRecorder(10), InformerManager: &informer.FakeManager{}, } @@ -1233,7 +1233,7 @@ func TestProcessOneSelectedResource_OverrideBehavior(t *testing.T) { }, IsClusterScopedResource: false, }, - recorder: record.NewFakeRecorder(10), + recorder: events.NewFakeRecorder(10), } activeWork := make(map[string]*fleetv1beta1.Work) gotNewWork, gotSimpleManifests, overrideFailed, err := r.processOneSelectedResource( @@ -1304,7 +1304,7 @@ func TestProcessOneSelectedResource_EnvelopeInnerOverrideFailureClassifiedAsOver }, IsClusterScopedResource: false, }, - recorder: record.NewFakeRecorder(10), + recorder: events.NewFakeRecorder(10), } roMap := map[fleetv1beta1.ResourceIdentifier][]*fleetv1beta1.ResourceOverrideSnapshot{ deploymentResourceIdentifier("app", "web"): { @@ -1469,7 +1469,7 @@ func TestCreateOrUpdateEnvelopeCRWorkObj_DuplicateWorksSurfaceWithoutMutation(t WithObjects(objs...). Build() - recorder := record.NewFakeRecorder(10) + recorder := events.NewFakeRecorder(10) r := &Reconciler{ Client: fakeClient, recorder: recorder, diff --git a/pkg/propertyprovider/azure/provider.go b/pkg/propertyprovider/azure/provider.go index 189f5d964..84da6eb97 100644 --- a/pkg/propertyprovider/azure/provider.go +++ b/pkg/propertyprovider/azure/provider.go @@ -249,7 +249,11 @@ func (p *PropertyProvider) Start(ctx context.Context, config *rest.Config) error p.region = discoveredRegion } klog.V(2).Infof("Starting with the region set to %s", *p.region) - pp := trackers.NewAKSKarpenterPricingClient(ctx, *p.region) + pp, err := trackers.NewAKSKarpenterPricingClient(ctx, *p.region) + if err != nil { + klog.ErrorS(err, "Failed to set up the pricing provider for the Azure property provider") + return err + } p.nodeTracker = trackers.NewNodeTracker(pp) default: // No node tracker has been set, and cost collection is disabled; set up a node tracker diff --git a/pkg/propertyprovider/azure/suite_test.go b/pkg/propertyprovider/azure/suite_test.go index 871234d61..3697dc4fb 100644 --- a/pkg/propertyprovider/azure/suite_test.go +++ b/pkg/propertyprovider/azure/suite_test.go @@ -112,7 +112,8 @@ var _ = BeforeSuite(func() { setUpResources() // Start an Azure property provider instance with all features on. - pp = trackers.NewAKSKarpenterPricingClient(ctx, region) + pp, err = trackers.NewAKSKarpenterPricingClient(ctx, region) + Expect(err).NotTo(HaveOccurred(), "Failed to create the AKS Karpenter pricing client") p = NewWithPricingProvider(pp, "node watcher", "pod watcher", true, true) Expect(p.Start(ctx, memberCfg)).To(Succeed()) diff --git a/pkg/propertyprovider/azure/trackers/pricing.go b/pkg/propertyprovider/azure/trackers/pricing.go index a9a5ff659..7503edefc 100644 --- a/pkg/propertyprovider/azure/trackers/pricing.go +++ b/pkg/propertyprovider/azure/trackers/pricing.go @@ -18,8 +18,10 @@ package trackers import ( "context" + "fmt" "time" + "github.com/Azure/karpenter-provider-azure/pkg/auth" "github.com/Azure/karpenter-provider-azure/pkg/providers/pricing" "github.com/Azure/karpenter-provider-azure/pkg/providers/pricing/client" ) @@ -53,7 +55,16 @@ func (k *AKSKarpenterPricingClient) LastUpdated() time.Time { // NewAKSKarpenterPricingClient returns a new AKS Karpenter pricing client, which implements // the PricingProvider interface. -func NewAKSKarpenterPricingClient(ctx context.Context, region string) *AKSKarpenterPricingClient { +func NewAKSKarpenterPricingClient(ctx context.Context, region string) (*AKSKarpenterPricingClient, error) { + // Pin the public cloud: the 1.5 pricing client had no environment and always queried the + // public retail prices endpoint, so this preserves existing behaviour. 1.14 can reject + // non-public clouds outright; wire that up with the member agent's cloud config + // (see the TODO in cmd/memberagent/main.go). + env, err := auth.EnvironmentFromName("AzurePublicCloud") + if err != nil { + return nil, fmt.Errorf("failed to resolve the Azure public cloud environment: %w", err) + } + // In the case of Azure property provider, there is no need to wait for leader election // successes; close the channel immediately to allow immediate boot-up of the pricing // client. @@ -61,6 +72,6 @@ func NewAKSKarpenterPricingClient(ctx context.Context, region string) *AKSKarpen close(ch) return &AKSKarpenterPricingClient{ - karpenterPricingClient: pricing.NewProvider(ctx, client.New(), region, ch), - } + karpenterPricingClient: pricing.NewProvider(ctx, env, client.New(env.Cloud), region, ch), + }, nil } diff --git a/pkg/propertyprovider/azure/trackers/pricing_test.go b/pkg/propertyprovider/azure/trackers/pricing_test.go new file mode 100644 index 000000000..665c153b7 --- /dev/null +++ b/pkg/propertyprovider/azure/trackers/pricing_test.go @@ -0,0 +1,55 @@ +/* +Copyright 2025 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package trackers + +import ( + "context" + "testing" +) + +// TestNewAKSKarpenterPricingClient tests the NewAKSKarpenterPricingClient function. +func TestNewAKSKarpenterPricingClient(t *testing.T) { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + pricingClient, err := NewAKSKarpenterPricingClient(ctx, "eastus") + if err != nil { + t.Fatalf("NewAKSKarpenterPricingClient(ctx, \"eastus\") = %v, want no error", err) + } + if pricingClient == nil { + t.Fatal("NewAKSKarpenterPricingClient(ctx, \"eastus\") = nil, want non-nil client") + } + + // The pricing provider ships with static pricing data, so known instance types + // resolve to a positive on-demand price even before any sync with the live API. + price, found := pricingClient.OnDemandPrice("Standard_D2s_v3") + if !found { + t.Error("OnDemandPrice(\"Standard_D2s_v3\") not found, want found") + } + if price <= 0 { + t.Errorf("OnDemandPrice(\"Standard_D2s_v3\") = %v, want positive price", price) + } + + if _, found := pricingClient.OnDemandPrice("not-a-real-instance-type"); found { + t.Error("OnDemandPrice(\"not-a-real-instance-type\") found, want not found") + } + + // LastUpdated reports the static data timestamp before any live sync completes. + if pricingClient.LastUpdated().IsZero() { + t.Error("LastUpdated() = zero time, want non-zero timestamp") + } +} diff --git a/pkg/resourcewatcher/change_detector.go b/pkg/resourcewatcher/change_detector.go index d7d2622cb..ac7307c45 100644 --- a/pkg/resourcewatcher/change_detector.go +++ b/pkg/resourcewatcher/change_detector.go @@ -154,6 +154,14 @@ func (d *ChangeDetector) discoverResources(dynamicResourceEventHandler cache.Res // dynamicResourceFilter filters out resources that we don't want to watch. func (d *ChangeDetector) dynamicResourceFilter(obj any) bool { + // A deletion the watch missed arrives as a tombstone wrapping the object's final state, not as + // the object itself. It has to be unwrapped before anything here inspects the object; filtering + // on the tombstone would silently drop every relist-detected deletion, since a tombstone is not + // a runtime object and fails the key derivation below. + if tombstone, ok := obj.(cache.DeletedFinalStateUnknown); ok { + obj = tombstone.Obj + } + key, err := controller.ClusterWideKeyFunc(obj) if err != nil { return false diff --git a/pkg/resourcewatcher/change_detector_test.go b/pkg/resourcewatcher/change_detector_test.go index cfccf63e2..78f3eeb73 100644 --- a/pkg/resourcewatcher/change_detector_test.go +++ b/pkg/resourcewatcher/change_detector_test.go @@ -99,7 +99,7 @@ func TestChangeDetector_discoverResources(t *testing.T) { for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { // Create fake discovery client - fakeClient := fake.NewSimpleClientset() + fakeClient := fake.NewClientset() fakeDiscovery, ok := fakeClient.Discovery().(*fakediscovery.FakeDiscovery) if !ok { t.Fatal("Failed to cast to FakeDiscovery") @@ -211,10 +211,22 @@ func TestChangeDetector_dynamicResourceFilter(t *testing.T) { want: false, }, { - // Tombstones from informer cache deletions are not unwrapped by ClusterWideKeyFunc, - // so the filter rejects them. The downstream delete handler unwraps tombstones separately. - name: "tombstone object is filtered out", + // A relist-detected deletion arrives as a tombstone wrapping the object's final state. + // The filter must judge the wrapped object, not the tombstone: rejecting tombstones + // wholesale would silently drop every such deletion before the delete handler -- which + // is what unwraps them for use -- ever saw it. + name: "tombstone wrapping a watched object passes the filter", obj: cache.DeletedFinalStateUnknown{Key: "default/cm", Obj: unstructuredConfigMap("default", "cm")}, + want: true, + }, + { + name: "tombstone wrapping an object in a skipped namespace is filtered out", + obj: cache.DeletedFinalStateUnknown{Key: "kube-system/cm", Obj: unstructuredConfigMap("kube-system", "cm")}, + want: false, + }, + { + name: "tombstone wrapping garbage is filtered out", + obj: cache.DeletedFinalStateUnknown{Key: "default/cm", Obj: "not-a-runtime-object"}, want: false, }, { diff --git a/pkg/resourcewatcher/informer_populator_test.go b/pkg/resourcewatcher/informer_populator_test.go index 92a162760..47c592c55 100644 --- a/pkg/resourcewatcher/informer_populator_test.go +++ b/pkg/resourcewatcher/informer_populator_test.go @@ -107,7 +107,7 @@ func TestInformerPopulator_discoverAndCreateInformers(t *testing.T) { for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { // Create fake discovery client - fakeClient := fake.NewSimpleClientset() + fakeClient := fake.NewClientset() fakeDiscovery, ok := fakeClient.Discovery().(*fakediscovery.FakeDiscovery) if !ok { t.Fatal("Failed to cast to FakeDiscovery") @@ -166,7 +166,7 @@ func TestInformerPopulator_discoverAndCreateInformers(t *testing.T) { func TestInformerPopulator_Start(t *testing.T) { // Create fake discovery client with some resources - fakeClient := fake.NewSimpleClientset() + fakeClient := fake.NewClientset() fakeDiscovery, ok := fakeClient.Discovery().(*fakediscovery.FakeDiscovery) if !ok { t.Fatal("Failed to cast to FakeDiscovery") @@ -230,7 +230,7 @@ func TestInformerPopulator_Integration(t *testing.T) { // This test verifies the integration between InformerPopulator and the informer manager // Create fake discovery with multiple resource types - fakeClient := fake.NewSimpleClientset() + fakeClient := fake.NewClientset() fakeDiscovery, ok := fakeClient.Discovery().(*fakediscovery.FakeDiscovery) if !ok { t.Fatal("Failed to cast to FakeDiscovery") @@ -297,7 +297,7 @@ func TestInformerPopulator_Integration(t *testing.T) { func TestInformerPopulator_PeriodicDiscovery(t *testing.T) { // This test verifies that the populator continues to discover resources periodically - fakeClient := fake.NewSimpleClientset() + fakeClient := fake.NewClientset() fakeDiscovery, ok := fakeClient.Discovery().(*fakediscovery.FakeDiscovery) if !ok { t.Fatal("Failed to cast to FakeDiscovery") diff --git a/pkg/scheduler/framework/framework.go b/pkg/scheduler/framework/framework.go index 58484b44e..2c263e876 100644 --- a/pkg/scheduler/framework/framework.go +++ b/pkg/scheduler/framework/framework.go @@ -32,7 +32,7 @@ import ( "k8s.io/apimachinery/pkg/api/meta" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/types" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" "k8s.io/client-go/util/retry" "k8s.io/klog/v2" ctrl "sigs.k8s.io/controller-runtime" @@ -94,7 +94,7 @@ type Handle interface { // UncachedReader returns an uncached read-only client, which allows direct (uncached) access to the API server. UncachedReader() client.Reader // EventRecorder returns an event recorder. - EventRecorder() record.EventRecorder + EventRecorder() events.EventRecorder // ClusterEligibilityChecker returns the cluster eligibility checker associated with the scheduler. ClusterEligibilityChecker() *clustereligibilitychecker.ClusterEligibilityChecker } @@ -124,7 +124,7 @@ type framework struct { // manager is the controller manager in use by the scheduler framework. manager ctrl.Manager // eventRecorder is the event recorder in use by the scheduler framework. - eventRecorder record.EventRecorder + eventRecorder events.EventRecorder // parallelizer is a utility which helps run tasks in parallel. parallelizer parallelizer.Parallelizer @@ -215,7 +215,7 @@ func NewFramework(profile *Profile, manager ctrl.Manager, opts ...Option) Framew client: manager.GetClient(), uncachedReader: manager.GetAPIReader(), manager: manager, - eventRecorder: manager.GetEventRecorderFor(fmt.Sprintf(eventRecorderNameTemplate, profile.Name())), + eventRecorder: manager.GetEventRecorder(fmt.Sprintf(eventRecorderNameTemplate, profile.Name())), parallelizer: parallelizer.NewParallelizer(options.numOfWorkers), maxUnselectedClusterDecisionCount: options.maxUnselectedClusterDecisionCount, clusterEligibilityChecker: options.clusterEligibilityChecker, @@ -243,7 +243,7 @@ func (f *framework) UncachedReader() client.Reader { } // EventRecorder returns the event recorder in use by the scheduler framework. -func (f *framework) EventRecorder() record.EventRecorder { +func (f *framework) EventRecorder() events.EventRecorder { return f.eventRecorder } @@ -546,8 +546,10 @@ func (f *framework) runAllPluginsForPickAllPlacementType( if err != nil { klog.ErrorS(err, "Failed to run filter plugins", "policySnapshot", policyRef) // Emit an event to inform the user about the scheduling error. - f.eventRecorder.Event(policy, corev1.EventTypeWarning, SchedulingErrorReason, - fmt.Sprintf("Failed to run filter plugins: %v", err)) + if f.eventRecorder != nil { + f.eventRecorder.Eventf(policy, nil, corev1.EventTypeWarning, SchedulingErrorReason, "RunFilterPlugins", + fmt.Sprintf("Failed to run filter plugins: %v", err)) + } // Check if the error has a retry policy configured. // If the error (or any error in its chain) implements ErrorWithRetryPolicy and indicates // it's retryable, return it as-is so the scheduler can requeue. @@ -1176,8 +1178,10 @@ func (f *framework) runAllPluginsForPickNPlacementType( if err != nil { klog.ErrorS(err, "Failed to run filter plugins", "policySnapshot", policyRef) // Emit an event to inform the user about the scheduling error. - f.eventRecorder.Event(policy, corev1.EventTypeWarning, SchedulingErrorReason, - fmt.Sprintf("Failed to run filter plugins: %v", err)) + if f.eventRecorder != nil { + f.eventRecorder.Eventf(policy, nil, corev1.EventTypeWarning, SchedulingErrorReason, "RunFilterPlugins", + fmt.Sprintf("Failed to run filter plugins: %v", err)) + } // Check if the error has a retry policy configured. // If the error (or any error in its chain) implements ErrorWithRetryPolicy and indicates // it's retryable, return it as-is so the scheduler can requeue. diff --git a/pkg/scheduler/framework/framework_test.go b/pkg/scheduler/framework/framework_test.go index a8cfd6fd0..5c29a5127 100644 --- a/pkg/scheduler/framework/framework_test.go +++ b/pkg/scheduler/framework/framework_test.go @@ -36,7 +36,6 @@ import ( "k8s.io/apimachinery/pkg/runtime/schema" "k8s.io/apimachinery/pkg/types" "k8s.io/client-go/kubernetes/scheme" - "k8s.io/client-go/tools/record" "sigs.k8s.io/controller-runtime/pkg/client" "sigs.k8s.io/controller-runtime/pkg/client/fake" @@ -1326,7 +1325,7 @@ func TestRunAllPluginsForPickAllPlacementType(t *testing.T) { f := &framework{ profile: profile, parallelizer: parallelizer.NewParallelizer(parallelizer.DefaultNumOfWorkers), - eventRecorder: record.NewFakeRecorder(10), + eventRecorder: nil, } ctx := context.Background() @@ -6257,7 +6256,7 @@ func TestRunAllPluginsForPickNPlacementType(t *testing.T) { f := &framework{ profile: profile, parallelizer: parallelizer.NewParallelizer(parallelizer.DefaultNumOfWorkers), - eventRecorder: record.NewFakeRecorder(10), + eventRecorder: nil, } ctx := context.Background() diff --git a/pkg/scheduler/framework/plugins/clustereligibility/plugin_test.go b/pkg/scheduler/framework/plugins/clustereligibility/plugin_test.go index 1eb0c3aeb..f14fca271 100644 --- a/pkg/scheduler/framework/plugins/clustereligibility/plugin_test.go +++ b/pkg/scheduler/framework/plugins/clustereligibility/plugin_test.go @@ -24,7 +24,7 @@ import ( "github.com/google/go-cmp/cmp" "github.com/google/go-cmp/cmp/cmpopts" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" ctrl "sigs.k8s.io/controller-runtime" "sigs.k8s.io/controller-runtime/pkg/client" @@ -56,7 +56,7 @@ var ( func (mh *MockHandle) Client() client.Client { return nil } func (mh *MockHandle) Manager() ctrl.Manager { return nil } func (mh *MockHandle) UncachedReader() client.Reader { return nil } -func (mh *MockHandle) EventRecorder() record.EventRecorder { return nil } +func (mh *MockHandle) EventRecorder() events.EventRecorder { return nil } func (mh *MockHandle) ClusterEligibilityChecker() *clustereligibilitychecker.ClusterEligibilityChecker { return mh.clusterEligibilityChecker } diff --git a/pkg/scheduler/scheduler.go b/pkg/scheduler/scheduler.go index 7dfa3a943..8cc0658d7 100644 --- a/pkg/scheduler/scheduler.go +++ b/pkg/scheduler/scheduler.go @@ -27,7 +27,7 @@ import ( apiErrors "k8s.io/apimachinery/pkg/api/errors" "k8s.io/apimachinery/pkg/types" utilruntime "k8s.io/apimachinery/pkg/util/runtime" - "k8s.io/client-go/tools/record" + "k8s.io/client-go/tools/events" "k8s.io/klog/v2" ctrl "sigs.k8s.io/controller-runtime" "sigs.k8s.io/controller-runtime/pkg/client" @@ -73,7 +73,7 @@ type Scheduler struct { workerNumber int // eventRecorder is the event recorder in use by the scheduler. - eventRecorder record.EventRecorder + eventRecorder events.EventRecorder } // NewScheduler creates a scheduler. @@ -92,7 +92,7 @@ func NewScheduler( uncachedReader: manager.GetAPIReader(), manager: manager, workerNumber: workerNumber, - eventRecorder: manager.GetEventRecorderFor(name), + eventRecorder: manager.GetEventRecorder(name), } } diff --git a/pkg/utils/common.go b/pkg/utils/common.go index f16aa7125..9d57b42d6 100644 --- a/pkg/utils/common.go +++ b/pkg/utils/common.go @@ -118,11 +118,22 @@ var ( APIGroups: []string{placementv1beta1.GroupVersion.Group}, Resources: []string{"*"}, } + // EventRule grants access to core/v1 Events. The Fleet controllers have + // moved to the events.k8s.io recorder, but this rule is still required: + // the fleet-networking agents that share this role emit core/v1 Events. EventRule = rbacv1.PolicyRule{ Verbs: []string{"get", "list", "update", "patch", "watch", "create"}, APIGroups: []string{""}, Resources: []string{"events"}, } + // EventsK8sIoRule grants the access needed by the events.k8s.io event + // recorder the controllers use. The recorder's sink only ever creates or + // patches Event objects, so no read access is granted here. + EventsK8sIoRule = rbacv1.PolicyRule{ + Verbs: []string{"create", "patch"}, + APIGroups: []string{"events.k8s.io"}, + Resources: []string{"events"}, + } FleetNetworkRule = rbacv1.PolicyRule{ Verbs: []string{"*"}, APIGroups: []string{NetworkingGroupName}, diff --git a/pkg/utils/controller/resource_selector_resolver_test.go b/pkg/utils/controller/resource_selector_resolver_test.go index 6c197f6ec..9ffd31208 100644 --- a/pkg/utils/controller/resource_selector_resolver_test.go +++ b/pkg/utils/controller/resource_selector_resolver_test.go @@ -42,10 +42,10 @@ import ( testinformer "go.goms.io/fleet/test/utils/informer" ) -func makeIPFamilyPolicyTypePointer(policyType corev1.IPFamilyPolicyType) *corev1.IPFamilyPolicyType { +func makeIPFamilyPolicyTypePointer(policyType corev1.IPFamilyPolicy) *corev1.IPFamilyPolicy { return &policyType } -func makeServiceInternalTrafficPolicyPointer(policyType corev1.ServiceInternalTrafficPolicyType) *corev1.ServiceInternalTrafficPolicyType { +func makeServiceInternalTrafficPolicyPointer(policyType corev1.ServiceInternalTrafficPolicy) *corev1.ServiceInternalTrafficPolicy { return &policyType } @@ -172,7 +172,7 @@ func TestGenerateResourceContent(t *testing.T) { LoadBalancerIP: "192.168.1.3", LoadBalancerSourceRanges: []string{"192.168.1.1"}, ExternalName: "svc-spec-externalName", - ExternalTrafficPolicy: corev1.ServiceExternalTrafficPolicyType("svc-spec-externalTrafficPolicy"), + ExternalTrafficPolicy: corev1.ServiceExternalTrafficPolicy("svc-spec-externalTrafficPolicy"), PublishNotReadyAddresses: false, SessionAffinityConfig: &corev1.SessionAffinityConfig{ClientIP: &corev1.ClientIPConfig{TimeoutSeconds: ptr.To(int32(60))}}, IPFamilies: []corev1.IPFamily{ @@ -229,7 +229,7 @@ func TestGenerateResourceContent(t *testing.T) { LoadBalancerIP: "192.168.1.3", LoadBalancerSourceRanges: []string{"192.168.1.1"}, ExternalName: "svc-spec-externalName", - ExternalTrafficPolicy: corev1.ServiceExternalTrafficPolicyType("svc-spec-externalTrafficPolicy"), + ExternalTrafficPolicy: corev1.ServiceExternalTrafficPolicy("svc-spec-externalTrafficPolicy"), PublishNotReadyAddresses: false, SessionAffinityConfig: &corev1.SessionAffinityConfig{ClientIP: &corev1.ClientIPConfig{TimeoutSeconds: ptr.To(int32(60))}}, IPFamilies: []corev1.IPFamily{ diff --git a/pkg/utils/informer/informermanager.go b/pkg/utils/informer/informermanager.go index 07aed02fb..b7d151910 100644 --- a/pkg/utils/informer/informermanager.go +++ b/pkg/utils/informer/informermanager.go @@ -30,6 +30,14 @@ import ( ctrlcache "sigs.k8s.io/controller-runtime/pkg/cache" ) +// Note (chenyu1): many methods in this utility, such as IsInformerSynced and Lister, will implicitly create an informer +// for the queried resource if one does not exist already. This might have side effects as such informers +// will not start until the manager's Start() method is called, provided that such resources have support +// for LIST/WATCH ops. Normally this is fine as the resource watcher is configured to periodically register +// all applicable resources in the informer manager, but the gaps between the synchronization might lead to +// unexpected behaviors (hopefully temporary). For newer code that needs to integrate with the informer manager, +// consider calling IsInformerSet first to check if an informer has been set up, before calling other methods. + // InformerManager manages dynamic shared informer for all resources, include Kubernetes resource and // custom resources defined by CustomResourceDefinition. type Manager interface { @@ -42,6 +50,9 @@ type Manager interface { // IsInformerSynced checks if the resource's informer is synced. IsInformerSynced(resource schema.GroupVersionResource) bool + // IsInformerSet returns if an informer has been set up for the given resource. + IsInformerSet(gvk schema.GroupVersionKind) bool + // Start will run all informers, the informers will keep running until the channel closed. // It is intended to be called after create new informer(s), and it's safe to call multi times. Start() @@ -153,6 +164,14 @@ func (s *informerManagerImpl) IsInformerSynced(resource schema.GroupVersionResou return s.informerFactory.ForResource(resource).Informer().HasSynced() } +func (s *informerManagerImpl) IsInformerSet(gvk schema.GroupVersionKind) bool { + s.resourcesLock.RLock() + defer s.resourcesLock.RUnlock() + + _, ok := s.apiResources[gvk] + return ok +} + func (s *informerManagerImpl) Lister(resource schema.GroupVersionResource) cache.GenericLister { return s.informerFactory.ForResource(resource).Lister() } diff --git a/pkg/utils/test_util.go b/pkg/utils/test_util.go index a665d0bb2..94fb76f48 100644 --- a/pkg/utils/test_util.go +++ b/pkg/utils/test_util.go @@ -34,7 +34,6 @@ import ( "k8s.io/apimachinery/pkg/runtime/serializer" "k8s.io/apimachinery/pkg/util/yaml" "k8s.io/client-go/kubernetes/scheme" - "k8s.io/client-go/tools/record" ) var ( @@ -50,18 +49,9 @@ const ( TestCaseMsg string = "\nTest case: %s" ) -// NewFakeRecorder makes a new fake event recorder that prints the object. -func NewFakeRecorder(bufferSize int) *record.FakeRecorder { - recorder := record.NewFakeRecorder(bufferSize) - recorder.IncludeObject = true - return recorder -} - -// GetEventString get the exact string literal of the event created by the fake event library. -func GetEventString(object runtime.Object, eventtype, reason, messageFmt string, args ...interface{}) string { - return fmt.Sprintf(eventtype+" "+reason+" "+messageFmt, args...) + - fmt.Sprintf(" involvedObject{kind=%s,apiVersion=%s}", - object.GetObjectKind().GroupVersionKind().Kind, object.GetObjectKind().GroupVersionKind().GroupVersion()) +// GetEventString gets the exact string literal of the event created by the fake event library. +func GetEventString(eventtype, reason, messageFmt string, args ...any) string { + return fmt.Sprintf(eventtype+" "+reason+" "+messageFmt, args...) } // GetObjectFromRawExtension returns an object decoded from the raw byte array. @@ -114,7 +104,7 @@ type NotFoundMatcher struct { } // Match matches the api error. -func (matcher NotFoundMatcher) Match(actual interface{}) (success bool, err error) { +func (matcher NotFoundMatcher) Match(actual any) (success bool, err error) { if actual == nil { return false, nil } @@ -123,12 +113,12 @@ func (matcher NotFoundMatcher) Match(actual interface{}) (success bool, err erro } // FailureMessage builds an error message. -func (matcher NotFoundMatcher) FailureMessage(actual interface{}) (message string) { +func (matcher NotFoundMatcher) FailureMessage(actual any) (message string) { return format.Message(actual, "to be not found") } // NegatedFailureMessage builds an error message. -func (matcher NotFoundMatcher) NegatedFailureMessage(actual interface{}) (message string) { +func (matcher NotFoundMatcher) NegatedFailureMessage(actual any) (message string) { return format.Message(actual, "to be found") } @@ -137,7 +127,7 @@ type AlreadyExistMatcher struct { } // Match matches error. -func (matcher AlreadyExistMatcher) Match(actual interface{}) (success bool, err error) { +func (matcher AlreadyExistMatcher) Match(actual any) (success bool, err error) { if actual == nil { return false, nil } @@ -146,12 +136,12 @@ func (matcher AlreadyExistMatcher) Match(actual interface{}) (success bool, err } // FailureMessage builds an error message. -func (matcher AlreadyExistMatcher) FailureMessage(actual interface{}) (message string) { +func (matcher AlreadyExistMatcher) FailureMessage(actual any) (message string) { return format.Message(actual, "to be already exist") } // NegatedFailureMessage builds an error message. -func (matcher AlreadyExistMatcher) NegatedFailureMessage(actual interface{}) (message string) { +func (matcher AlreadyExistMatcher) NegatedFailureMessage(actual any) (message string) { return format.Message(actual, "not to be already exist") } diff --git a/pkg/utils/test_util_test.go b/pkg/utils/test_util_test.go new file mode 100644 index 000000000..009be9405 --- /dev/null +++ b/pkg/utils/test_util_test.go @@ -0,0 +1,31 @@ +/* +Copyright 2025 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package utils + +import ( + "testing" + + corev1 "k8s.io/api/core/v1" +) + +func TestGetEventString(t *testing.T) { + got := GetEventString(corev1.EventTypeNormal, "SomeReason", "something %s happened", "good") + want := "Normal SomeReason something good happened" + if got != want { + t.Errorf("GetEventString() = %q, want %q", got, want) + } +} diff --git a/pkg/v1/controllers/workgenerator/cleanup.go b/pkg/v1/controllers/workgenerator/cleanup.go new file mode 100644 index 000000000..b96dce543 --- /dev/null +++ b/pkg/v1/controllers/workgenerator/cleanup.go @@ -0,0 +1,84 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package workgenerator + +import ( + "context" + "fmt" + + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/klog/v2" + "sigs.k8s.io/controller-runtime/pkg/controller/controllerutil" + + placementv1alpha1 "go.goms.io/fleet/apis/kubefleet.dev/placement/v1alpha1" + "go.goms.io/fleet/pkg/utils" + "go.goms.io/fleet/pkg/utils/errors" +) + +func (r *Reconciler) addPlacementBindingCleanupFinalizer(ctx context.Context, placementBinding placementv1alpha1.PlacementBindingAccessor) error { + if controllerutil.ContainsFinalizer(placementBinding, workGeneratorCleanupFinalizer) { + return nil + } + controllerutil.AddFinalizer(placementBinding, workGeneratorCleanupFinalizer) + if err := r.hubClient.Update(ctx, placementBinding); err != nil { + return errors.NewAPIServerError(err, "failed to add cleanup finalizer to placement binding", false) + } + return nil +} + +// cleanupWorks deletes the primary Work object owned by a placement binding in the reserved namespace of the +// target cluster; all the other Work objects are cleaned up via owner-reference cascade deletion. +func (r *Reconciler) cleanupWorks(ctx context.Context, placementBinding placementv1alpha1.PlacementBindingAccessor) error { + if !controllerutil.ContainsFinalizer(placementBinding, workGeneratorCleanupFinalizer) { + // The cleanup finalizer has been dropped; no cleanup is needed. + return nil + } + + derivedFromSourceFormatter := &placementResourceSnapshotDerivedFromSourceFormatter{ + snapshotSubIdx: "0", + } + workName := uniqueNameForWorkDerivedFromPlacementResourceSnapshot(placementBinding, true, derivedFromSourceFormatter) + workForPrimaryResSnapshot := &placementv1alpha1.Work{ + ObjectMeta: metav1.ObjectMeta{ + Namespace: fmt.Sprintf(utils.NamespaceNameFormat, placementBinding.GetSpec().ClusterName), + Name: workName, + }, + } + if err := r.hubClient.Delete(ctx, workForPrimaryResSnapshot); err != nil && !apierrors.IsNotFound(err) { + return errors.NewAPIServerError(err, "failed to delete work object for primary placement resource snapshot", false, + "work", klog.KObj(workForPrimaryResSnapshot)) + } + // This work object is set as the owner of all other work objects created for this placement binding; + // no further cleanup is needed. + + // If the binding has been suspended, unset its last processed resource snapshot name. + if placementBinding.GetSpec().Suspended && placementBinding.GetDeletionTimestamp().IsZero() { + placementBinding.GetStatus().LastProcessedResourceSnapshotName = nil + + if err := r.hubClient.Status().Update(ctx, placementBinding); err != nil { + return errors.NewAPIServerError(err, "failed to update placement binding status to reset last processed resource snapshot name", false) + } + } + + // Remove the cleanup finalizer from the placement binding. + controllerutil.RemoveFinalizer(placementBinding, workGeneratorCleanupFinalizer) + if err := r.hubClient.Update(ctx, placementBinding); err != nil { + return errors.NewAPIServerError(err, "failed to remove cleanup finalizer from placement binding", false) + } + return nil +} diff --git a/pkg/v1/controllers/workgenerator/controller.go b/pkg/v1/controllers/workgenerator/controller.go new file mode 100644 index 000000000..be37106e2 --- /dev/null +++ b/pkg/v1/controllers/workgenerator/controller.go @@ -0,0 +1,272 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +// Package workgenerator contains the controller logic for reconciling placement binding objects. +package workgenerator + +import ( + "context" + "time" + + "k8s.io/client-go/util/workqueue" + "k8s.io/klog/v2" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/builder" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/controller" + "sigs.k8s.io/controller-runtime/pkg/event" + "sigs.k8s.io/controller-runtime/pkg/handler" + "sigs.k8s.io/controller-runtime/pkg/predicate" + "sigs.k8s.io/controller-runtime/pkg/reconcile" + + "k8s.io/apimachinery/pkg/api/equality" + apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/types" + + placementv1alpha1 "go.goms.io/fleet/apis/kubefleet.dev/placement/v1alpha1" + "go.goms.io/fleet/pkg/utils/errors" + parallelizerutil "go.goms.io/fleet/pkg/utils/parallelizer" +) + +const ( + controllerName = "work-generator" + + workGeneratorCleanupFinalizer = "placement.kubefleet.dev/work-generator-cleanup" +) + +type Reconciler struct { + hubClient client.Client + + parallelizer parallelizerutil.Parallelizer +} + +func New(hubClient client.Client, workerCnt int) *Reconciler { + parallelizer := parallelizerutil.NewParallelizer(workerCnt) + + return &Reconciler{ + hubClient: hubClient, + parallelizer: parallelizer, + } +} + +// TO-DO (chenyu1): switch to field-based indexes for better performance when listing objects. + +func (r *Reconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl.Result, error) { + startTime := time.Now() + klog.V(2).InfoS("Reconciliation starts", "placementBinding", req.NamespacedName, "controller", controllerName) + defer func() { + latency := time.Since(startTime).Milliseconds() + klog.V(2).InfoS("Reconciliation ends", "placementBinding", req.NamespacedName, "controller", controllerName, "latency", latency) + }() + + // Retrieve the PlacementBinding object. + placementBinding, err := r.retrievePlacementBinding(ctx, req.NamespacedName) + if err != nil { + if apierrors.IsNotFound(err) { + klog.V(2).InfoS("placement binding is not found", "namespacedName", req.NamespacedName, "controller", controllerName) + return ctrl.Result{}, nil + } + klog.ErrorS(err, "", "namespacedName", req.NamespacedName, "controller", controllerName) + return ctrl.Result{}, errors.Wraps(err, "", "namespacedName", req.NamespacedName, "controller", controllerName) + } + + placementBindingSpec := placementBinding.GetSpec() + // Clean up the work objects for the placement binding if it has been marked for deletion or if it has been + // suspended. + if placementBinding.GetDeletionTimestamp() != nil || placementBindingSpec.Suspended { + if err := r.cleanupWorks(ctx, placementBinding); err != nil { + wrappedErr := errors.Wraps(err, "failed to clean up work objects for placement binding", + "placementBinding", klog.KObj(placementBinding), "controller", controllerName) + klog.ErrorS(wrappedErr, "failed to clean up work objects for placement binding", + errors.Args(wrappedErr)...) + return ctrl.Result{}, wrappedErr + } + return ctrl.Result{}, nil + } + // Add the cleanup finalizer if it is not already present. + if err := r.addPlacementBindingCleanupFinalizer(ctx, placementBinding); err != nil { + wrappedErr := errors.Wraps(err, "failed to add cleanup finalizer to placement binding", + "placementBinding", klog.KObj(placementBinding), "controller", controllerName) + klog.ErrorS(wrappedErr, "failed to add cleanup finalizer to placement binding", errors.Args(wrappedErr)...) + return ctrl.Result{}, wrappedErr + } + + // Do a sanity check; verify if a target cluster and a (primary) placement resource snapshot have been assigned. + if len(placementBindingSpec.ClusterName) == 0 || len(placementBindingSpec.ResourceSnapshotName) == 0 { + wrappedErr := errors.NewUnexpectedError(nil, "the placement binding does not have a target cluster or a placement resource snapshot assigned", + "placementBinding", klog.KObj(placementBinding), "controller", controllerName) + klog.ErrorS(wrappedErr, "failed to process placement binding", errors.Args(wrappedErr)...) + return ctrl.Result{}, wrappedErr + } + + // Retrieve the Work objects owned by the placement binding. + works, err := r.listWorksByOwnerBinding(ctx, placementBindingSpec.ClusterName, placementBinding.GetNamespace(), placementBinding.GetName()) + if err != nil { + wrappedErr := errors.Wraps(err, "", "placementBinding", klog.KObj(placementBinding), + "targetCluster", placementBindingSpec.ClusterName, "controller", controllerName) + klog.ErrorS(wrappedErr, "failed to list work objects owned by binding", errors.Args(wrappedErr)...) + return ctrl.Result{}, wrappedErr + } + + // Check if the Work objects are consistent with the assigned primary and secondary placement resource snapshots. + // If so, no need to update the spec of the Work objects; just sync the status back to the placement binding + // instead. + // + // Note (chenyu1): this check is intended as a shortcut to avoid constant re-generation and validation of + // work objects (which can be expensive when there are a large number of manifests to place); once the controller + // signals that it has completed processing a placement binding given a specific configuration (a specific + // set of placement resource snapshots) and generated all the needed work objects, the control loop will skip + // to status reporting. In general we do not try to guard against byzantine faults here, especially + // considering that work objects are KubeFleet internal API objects that reside in reserved namespaces; if a + // non-KubeFleet agent decides to tamper with work objects, the system is not guaranteed to auto-recover. + // The changes, however, will be overwritten upon rollouts. + upToDate, err := areWorksUpToDate(placementBinding, works) // codespell:ignore + if err != nil { + wrappedErr := errors.Wraps(err, "failed to check if work objects are up-to-date", + "placementBinding", klog.KObj(placementBinding), "targetCluster", placementBindingSpec.ClusterName, + "controller", controllerName) + klog.ErrorS(wrappedErr, "failed to check if work objects are up-to-date", errors.Args(wrappedErr)...) + return ctrl.Result{}, wrappedErr + } + if upToDate { // codespell:ignore + if err := r.refreshPlacementBindingStatus(ctx, placementBinding, works); err != nil { + wrappedErr := errors.Wraps(err, "failed to refresh placement binding status", + "placementBinding", klog.KObj(placementBinding), "targetCluster", placementBindingSpec.ClusterName, + "controller", controllerName) + klog.ErrorS(wrappedErr, "failed to refresh placement binding status", errors.Args(wrappedErr)...) + return ctrl.Result{}, wrappedErr + } + return ctrl.Result{}, nil + } + + // The Work objects are absent or not up-to-date. Retrieve the placement resource snapshots and create/update + // the Work objects accordingly. + + // Retrieve the assigned primary and secondary placement resource snapshots referenced by the placement binding. + placementResourceSnapshots, err := r.retrievePrimaryAndSecondaryPlacementResourceSnapshots(ctx, placementBinding) + if err != nil { + wrappedErr := errors.Wraps(err, "failed to retrieve placement resource snapshots", + "placementBinding", klog.KObj(placementBinding), "targetCluster", placementBindingSpec.ClusterName, + "controller", controllerName) + klog.ErrorS(wrappedErr, "failed to retrieve placement resource snapshots referenced by binding", + errors.Args(wrappedErr)...) + return ctrl.Result{}, wrappedErr + } + + // Create or update the work objects. + createdOrUpdatedWorks, writtenToStorage, err := r.refreshWorks(ctx, placementBinding, placementResourceSnapshots, works) + if err != nil { + wrappedErr := errors.Wraps(err, "failed to refresh work objects", + "placementBinding", klog.KObj(placementBinding), "targetCluster", placementBindingSpec.ClusterName, + "controller", controllerName) + klog.ErrorS(wrappedErr, "failed to refresh work objects for placement binding", + errors.Args(wrappedErr)...) + return ctrl.Result{}, wrappedErr + } + + // Report the processing progress via placement binding status. + if err := r.reportPlacementBindingProcessingProgress(ctx, placementBinding, placementResourceSnapshots[0], createdOrUpdatedWorks); err != nil { + wrappedErr := errors.Wraps(err, "failed to report placement binding processing progress", + "placementBinding", klog.KObj(placementBinding), "targetCluster", placementBindingSpec.ClusterName, + "controller", controllerName) + klog.ErrorS(wrappedErr, "failed to report placement binding processing progress", + errors.Args(wrappedErr)...) + return ctrl.Result{}, wrappedErr + } + + // The work objects have been refreshed. Normally the controller needs only to wait for the work objects + // to be processed by the KubeFleet member agent, then refresh the placement binding status upon receiving + // create/update events from the work objects, and there is no need to requeue manually. However, there exists + // a corner case in which a rollout attempt does not involve any change in the work objects; in this case there + // will not be any change events from the work objects and the work generator needs to requeue manually to + // have the placement binding status refreshed. + if !writtenToStorage { + // The work objects have not been created or updated; requeue manually. + return ctrl.Result{RequeueAfter: 1 * time.Second}, nil + } + // The work objects have been created or updated; wait for change events from the work objects to refresh + // the placement binding status. + return ctrl.Result{}, nil +} + +func (r *Reconciler) SetupWithManager(mgr ctrl.Manager, maxConcurrentReconciles int) error { + // enqueueOwnerBindingForWork resolves the owner placement binding from a work object's metadata and enqueues it + // for reconciliation. eventType is used for logging only. + enqueueOwnerBindingForWork := func(work client.Object, eventType string, q workqueue.TypedRateLimitingInterface[reconcile.Request]) { + if work == nil { + wrappedErr := errors.NewUnexpectedError(nil, "received a nil work object", "eventType", eventType, "controller", controllerName) + klog.ErrorS(wrappedErr, "received a nil work object", errors.Args(wrappedErr)...) + return + } + labels := work.GetLabels() + annotations := work.GetAnnotations() + ownerBindingNSName, nsNameFound := labels[placementv1alpha1.WorkOwnerNamespaceLabelKey] + ownerBindingName, bindingNameFound := annotations[placementv1alpha1.WorkOwnedByPlacementBindingAnnotationKey] + if !nsNameFound || !bindingNameFound { + err := errors.NewUnexpectedError(nil, "work object is missing required labels or annotations", + "work", klog.KObj(work), "eventType", eventType, "controller", controllerName) + klog.ErrorS(err, "work object is missing required labels or annotations", errors.Args(err)...) + return + } + ownerBinding := types.NamespacedName{Namespace: ownerBindingNSName, Name: ownerBindingName} + klog.V(2).InfoS("Enqueue the owner placement binding for reconciliation", + "work", klog.KObj(work), "eventType", eventType, "placementBinding", ownerBinding) + q.Add(reconcile.Request{NamespacedName: ownerBinding}) + } + + workObjHandlerFuncs := handler.Funcs{ + // The controller needs to watch for work object create events as the client-side cache might + // lag under heavy load, i.e., it might learn about a work object only after its status has been updated. + CreateFunc: func(_ context.Context, e event.TypedCreateEvent[client.Object], q workqueue.TypedRateLimitingInterface[reconcile.Request]) { + enqueueOwnerBindingForWork(e.Object, "create", q) + }, + UpdateFunc: func(_ context.Context, e event.TypedUpdateEvent[client.Object], q workqueue.TypedRateLimitingInterface[reconcile.Request]) { + if e.ObjectOld == nil || e.ObjectNew == nil { + wrappedErr := errors.NewUnexpectedError(nil, "received nil work objects in update event", "controller", controllerName) + klog.ErrorS(wrappedErr, "received nil work objects in update event", errors.Args(wrappedErr)...) + return + } + + oldWork, canCastOldWork := e.ObjectOld.(*placementv1alpha1.Work) + newWork, canCastNewWork := e.ObjectNew.(*placementv1alpha1.Work) + if !canCastOldWork || !canCastNewWork { + wrappedErr := errors.NewUnexpectedError(nil, "failed to cast work objects in update event", "controller", controllerName) + klog.ErrorS(wrappedErr, "failed to cast work objects in update event", errors.Args(wrappedErr)...) + return + } + + // Only enqueue when the status has changed, so that status can be synced back to the owner binding. + if !equality.Semantic.DeepEqual(oldWork.Status, newWork.Status) { + enqueueOwnerBindingForWork(e.ObjectNew, "update", q) + } + }, + DeleteFunc: func(_ context.Context, e event.TypedDeleteEvent[client.Object], q workqueue.TypedRateLimitingInterface[reconcile.Request]) { + enqueueOwnerBindingForWork(e.Object, "delete", q) + }, + } + + return ctrl.NewControllerManagedBy(mgr). + Named(controllerName). + WithOptions(controller.Options{MaxConcurrentReconciles: maxConcurrentReconciles}). + // The controller watches placement binding objects (both namespace-scoped and cluster-scoped) for spec + // changes (generation predicate). + Watches(&placementv1alpha1.PlacementBinding{}, &handler.EnqueueRequestForObject{}, builder.WithPredicates(predicate.GenerationChangedPredicate{})). + Watches(&placementv1alpha1.ClusterPlacementBinding{}, &handler.EnqueueRequestForObject{}, builder.WithPredicates(predicate.GenerationChangedPredicate{})). + // The controller watches work objects for status changes, so that status can be synced back to their + // owner placement bindings. + Watches(&placementv1alpha1.Work{}, workObjHandlerFuncs). + Complete(r) +} diff --git a/pkg/v1/controllers/workgenerator/derivedfrom.go b/pkg/v1/controllers/workgenerator/derivedfrom.go new file mode 100644 index 000000000..48d13efb2 --- /dev/null +++ b/pkg/v1/controllers/workgenerator/derivedfrom.go @@ -0,0 +1,56 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package workgenerator + +// Verify that all formatter implements the derivedFromSourceFormatter interface. +var _ derivedFromSourceFormatter = &placementResourceSnapshotDerivedFromSourceFormatter{} + +// derivedFromSourceFormatter is an interface that helps format the ID of a source that derives a work object for +// various use cases, primarily for preparing unique names for work objects. +type derivedFromSourceFormatter interface { + // StrictDNSLabel returns a string that is a valid DNS label (max. 63 chars, all lowercase, alphanumeric characters + // and hyphens, and must start and end with an alphanumeric character). + // + // This value is used as a sub-component of the unique name for a work object derived from a source. It is + // for informational purposes only and does not need to be unique across all work objects that are created/updated + // for the same placement binding. + StrictDNSLabel() string + // SourceType returns a string that identifies the type of the source that derives a work object, e.g., + // `placement-resource-snapshot` for placement resource snapshots. + SourceType() string + // SourceID returns a string that uniquely identifies the source (of the same type) that derives a + // work object. + SourceID() string +} + +// placementResourceSnapshotDerivedFromSourceFormatter is a formatter for placement resource snapshots that implements +// the derivedFromSourceFormatter interface. +type placementResourceSnapshotDerivedFromSourceFormatter struct { + snapshotSubIdx string +} + +func (f *placementResourceSnapshotDerivedFromSourceFormatter) SourceID() string { + return f.snapshotSubIdx +} + +func (f *placementResourceSnapshotDerivedFromSourceFormatter) SourceType() string { + return "placement-resource-snapshot" +} + +func (f *placementResourceSnapshotDerivedFromSourceFormatter) StrictDNSLabel() string { + return f.snapshotSubIdx +} diff --git a/pkg/v1/controllers/workgenerator/retrieval.go b/pkg/v1/controllers/workgenerator/retrieval.go new file mode 100644 index 000000000..cb0a10b57 --- /dev/null +++ b/pkg/v1/controllers/workgenerator/retrieval.go @@ -0,0 +1,184 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package workgenerator + +import ( + "context" + "fmt" + "sort" + "strconv" + + "k8s.io/apimachinery/pkg/types" + "k8s.io/klog/v2" + "sigs.k8s.io/controller-runtime/pkg/client" + + placementv1alpha1 "go.goms.io/fleet/apis/kubefleet.dev/placement/v1alpha1" + "go.goms.io/fleet/pkg/utils" + "go.goms.io/fleet/pkg/utils/errors" +) + +func (r *Reconciler) retrievePlacementBinding(ctx context.Context, namespacedName types.NamespacedName) (placementv1alpha1.PlacementBindingAccessor, error) { + var placementBinding placementv1alpha1.PlacementBindingAccessor + if namespacedName.Namespace == "" { + // The placement binding is cluster-scoped. + placementBinding = &placementv1alpha1.ClusterPlacementBinding{} + } else { + // The placement binding is namespace-scoped. + placementBinding = &placementv1alpha1.PlacementBinding{} + } + + if err := r.hubClient.Get(ctx, namespacedName, placementBinding); err != nil { + return nil, errors.NewAPIServerError(err, "failed to retrieve placement binding", true) + } + return placementBinding, nil +} + +// listWorksByOwnerBinding lists the Work objects owned by a placement binding within a Fleet member cluster reserved +// namespace. +func (r *Reconciler) listWorksByOwnerBinding(ctx context.Context, clusterName, ownerBindingNSName, ownerBindingName string) ([]placementv1alpha1.Work, error) { + memberClusterNamespace := fmt.Sprintf(utils.NamespaceNameFormat, clusterName) + + workList := &placementv1alpha1.WorkList{} + listOptions := []client.ListOption{ + client.InNamespace(memberClusterNamespace), + client.MatchingLabels{ + placementv1alpha1.WorkOwnedByPlacementBindingLabelKey: workOwnerLabelValue(ownerBindingName), + placementv1alpha1.WorkOwnerNamespaceLabelKey: ownerBindingNSName, + }, + } + if err := r.hubClient.List(ctx, workList, listOptions...); err != nil { + return nil, errors.NewAPIServerError(err, "failed to list work objects", true) + } + return workList.Items, nil +} + +// retrievePrimaryAndSecondaryPlacementResourceSnapshots retrieves the primary placement resource snapshot referenced +// by the placement binding, along with any secondary snapshots that share the same index. The returned snapshots are +// sorted in ascending order of their sub-indices (the primary, sub-index 0, comes first). +func (r *Reconciler) retrievePrimaryAndSecondaryPlacementResourceSnapshots( + ctx context.Context, + placementBinding placementv1alpha1.PlacementBindingAccessor, +) ([]placementv1alpha1.PlacementResourceSnapshotAccessor, error) { + namespace := placementBinding.GetNamespace() + primarySnapshotName := placementBinding.GetSpec().ResourceSnapshotName + + // Retrieve the primary placement resource snapshot referenced by the binding. + var primarySnapshot placementv1alpha1.PlacementResourceSnapshotAccessor + if namespace == "" { + // The placement binding is cluster-scoped. + primarySnapshot = &placementv1alpha1.ClusterPlacementResourceSnapshot{} + } else { + // The placement binding is namespace-scoped. + primarySnapshot = &placementv1alpha1.PlacementResourceSnapshot{} + } + if err := r.hubClient.Get(ctx, types.NamespacedName{Namespace: namespace, Name: primarySnapshotName}, primarySnapshot); err != nil { + return nil, errors.NewAPIServerError(err, "failed to retrieve the primary placement resource snapshot", true, + "primaryPlacementResourceSnapshotName", primarySnapshotName) + } + + // Determine how many snapshots share the same index via the count label on the primary snapshot. + countStr := primarySnapshot.GetLabels()[placementv1alpha1.SubIndexedPlacementResourceSnapshotCountLabelKey] + count, err := strconv.Atoi(countStr) + if err != nil || count < 1 { + return nil, errors.NewUnexpectedError(err, "invalid sub-indexed placement resource snapshot count label on the primary placement resource snapshot", + "primaryPlacementResourceSnapshot", klog.KObj(primarySnapshot), "countVal", countStr) + } + if count == 1 { + // The primary snapshot is the only snapshot associated with this index. + return []placementv1alpha1.PlacementResourceSnapshotAccessor{primarySnapshot}, nil + } + + // There are secondary snapshots; list all snapshots that share the same owner and index. + ownedBy := primarySnapshot.GetLabels()[placementv1alpha1.PlacementResourceSnapshotOwnedByLabelKey] + index := primarySnapshot.GetLabels()[placementv1alpha1.PlacementResourceSnapshotIndexLabelKey] + if ownedBy == "" || index == "" { + return nil, errors.NewUnexpectedError(nil, "the primary placement resource snapshot is missing required labels", + "primaryPlacementResourceSnapshot", klog.KObj(primarySnapshot)) + } + labelMatchers := client.MatchingLabels{ + placementv1alpha1.PlacementResourceSnapshotOwnedByLabelKey: ownedBy, + placementv1alpha1.PlacementResourceSnapshotIndexLabelKey: index, + } + + var snapshots []placementv1alpha1.PlacementResourceSnapshotAccessor + if namespace == "" { + snapshotList := &placementv1alpha1.ClusterPlacementResourceSnapshotList{} + if err := r.hubClient.List(ctx, snapshotList, labelMatchers); err != nil { + return nil, errors.NewAPIServerError(err, "failed to list cluster placement resource snapshots", true) + } + snapshots = make([]placementv1alpha1.PlacementResourceSnapshotAccessor, len(snapshotList.Items)) + for i := range snapshotList.Items { + snapshots[i] = &snapshotList.Items[i] + } + } else { + snapshotList := &placementv1alpha1.PlacementResourceSnapshotList{} + if err := r.hubClient.List(ctx, snapshotList, client.InNamespace(namespace), labelMatchers); err != nil { + return nil, errors.NewAPIServerError(err, "failed to list placement resource snapshots", true) + } + snapshots = make([]placementv1alpha1.PlacementResourceSnapshotAccessor, len(snapshotList.Items)) + for i := range snapshotList.Items { + snapshots[i] = &snapshotList.Items[i] + } + } + + // Sort the snapshots by their sub-indices in ascending order. + var sortErrs []error + sort.Slice(snapshots, func(i, j int) bool { + subIdxI, iErr := strconv.Atoi(snapshots[i].GetLabels()[placementv1alpha1.PlacementResourceSnapshotSubIndexLabelKey]) + subIdxJ, jErr := strconv.Atoi(snapshots[j].GetLabels()[placementv1alpha1.PlacementResourceSnapshotSubIndexLabelKey]) + if iErr != nil { + sortErrs = append(sortErrs, fmt.Errorf("failed to convert sub-index label to integer: %w (placementResourceSnapshot: %s)", iErr, snapshots[i].GetName())) + return false + } + if jErr != nil { + sortErrs = append(sortErrs, fmt.Errorf("failed to convert sub-index label to integer: %w (placementResourceSnapshot: %s)", jErr, snapshots[j].GetName())) + return false + } + return subIdxI < subIdxJ + }) + if len(sortErrs) > 0 { + return nil, errors.NewUnexpectedError(nil, "failed to sort placement resource snapshots by sub-index", "errs", sortErrs) + } + + // Do some sanity checks; verify that all snapshots dictated by the count label are present and they have + // the same snapshotted resource hash. + + if len(snapshots) < count { + // Normally this branch will never run, as the placement resource snapshot manager creates secondary + // snapshots first, then the primary snapshot with the count label. + return nil, errors.NewUnexpectedError(nil, "there are fewer placement resource snapshots than the count label indicates", + "primaryPlacementResourceSnapshot", klog.KObj(primarySnapshot), "expectedCount", count, "actualCount", len(snapshots)) + } + + primarySnapshottedResHash := primarySnapshot.GetAnnotations()[placementv1alpha1.PlacementResourceSnapshotContentsHashAnnotationKey] + for i := range snapshots[:count] { + snapshottedResHash := snapshots[i].GetAnnotations()[placementv1alpha1.PlacementResourceSnapshotContentsHashAnnotationKey] + if snapshottedResHash != primarySnapshottedResHash { + // Normally this branch will never run, as the placement resource snapshot manager uses ordered creation + // to make sure that hashes are consistent across all snapshots with the same index. + return nil, errors.NewUnexpectedError(nil, "the contents hash of a placement resource snapshot does not match the primary snapshot", + "primaryPlacementResourceSnapshot", klog.KObj(primarySnapshot), + "hashMismatchedPlacementResourceSnapshot", klog.KObj(snapshots[i]), + "hashOnPrimaryPlacementResourceSnapshot", primarySnapshottedResHash, + "mismatchedHash", snapshottedResHash) + } + } + + // Any snapshots beyond the count are orphans from an overwritten resource change; return only the ones + // dictated by the count label, which are guaranteed to be consistent. + return snapshots[:count], nil +} diff --git a/pkg/v1/controllers/workgenerator/status.go b/pkg/v1/controllers/workgenerator/status.go new file mode 100644 index 000000000..dc5b89147 --- /dev/null +++ b/pkg/v1/controllers/workgenerator/status.go @@ -0,0 +1,272 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package workgenerator + +import ( + "context" + + "k8s.io/apimachinery/pkg/api/equality" + "k8s.io/apimachinery/pkg/api/meta" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/klog/v2" + "k8s.io/utils/ptr" + + placementv1alpha1 "go.goms.io/fleet/apis/kubefleet.dev/placement/v1alpha1" + "go.goms.io/fleet/pkg/utils/condition" + "go.goms.io/fleet/pkg/utils/errors" +) + +func (r *Reconciler) refreshPlacementBindingStatus( + ctx context.Context, + placementBinding placementv1alpha1.PlacementBindingAccessor, + works []placementv1alpha1.Work, +) (err error) { + oldStatus := placementBinding.GetStatus().DeepCopy() + + if notReady := refreshPlacementBindingSyncCond(placementBinding, works); notReady { + klog.V(2).InfoS("the synchronized condition is not yet ready to be refreshed; skipping status update for now") + return nil + } + if notReady := refreshPlacementBindingAvailableCond(placementBinding, works); notReady { + klog.V(2).InfoS("the available condition is not yet ready to be refreshed; skipping status update for now") + return nil + } + + total, synced, available, failed := countResourcesInWorksByProcessingResults(works) + placementBinding.GetStatus().SelectedResources = ptr.To(total) + placementBinding.GetStatus().SynchronizedResources = ptr.To(synced) + placementBinding.GetStatus().AvailableResources = ptr.To(available) + if len(failed) > 50 { + klog.V(2).InfoS("Too many failed resources to report in placement binding status; truncating the list to 50", + "placementBinding", klog.KObj(placementBinding), "totalFailedResources", len(failed)) + failed = failed[:50] + } + placementBinding.GetStatus().FailedResources = failed + + // Skip the update if the status has not changed. + if equality.Semantic.DeepEqual(oldStatus, placementBinding.GetStatus()) { + klog.V(2).InfoS("No need to update placement binding status as it has not changed", + "placementBinding", klog.KObj(placementBinding), + "selectedResources", total, "synchronizedResources", synced, "availableResources", available, + "failedResources", len(failed)) + return nil + } + + if err := r.hubClient.Status().Update(ctx, placementBinding); err != nil { + return errors.NewAPIServerError(err, "failed to update placement binding status", true) + } + klog.V(2).InfoS("Updated placement binding status", + "placementBinding", klog.KObj(placementBinding), + "selectedResources", total, "synchronizedResources", synced, "availableResources", available, + "failedResources", len(failed)) + return nil +} + +func (r *Reconciler) reportPlacementBindingProcessingProgress( + ctx context.Context, + placementBinding placementv1alpha1.PlacementBindingAccessor, + primaryPlacementResourceSnapshot placementv1alpha1.PlacementResourceSnapshotAccessor, + worksToCreateOrUpdate []*placementv1alpha1.Work, +) error { + placementBindingStatus := placementBinding.GetStatus() + // Set the last processed placement resource snapshot name on the placement binding status. + placementBindingStatus.LastProcessedResourceSnapshotName = ptr.To(primaryPlacementResourceSnapshot.GetName()) + + // Set a false Synchronized condition with the WaitingForSynchronization reason on the placement binding. + meta.SetStatusCondition(&placementBindingStatus.Conditions, metav1.Condition{ + Type: placementv1alpha1.PlacementBindingCondTypeSynchronized, + Status: metav1.ConditionFalse, + ObservedGeneration: placementBinding.GetGeneration(), + Reason: placementv1alpha1.PlacementBindingSynchronizedCondReasonWaitingForSynchronization, + Message: "Waiting for the resources to be synchronized to the target cluster", + }) + // Set an unknown Available condition with the WaitingForAvailabilityCheck reason on the placement binding. + meta.SetStatusCondition(&placementBindingStatus.Conditions, metav1.Condition{ + Type: placementv1alpha1.PlacementBindingCondTypeAvailable, + Status: metav1.ConditionUnknown, + ObservedGeneration: placementBinding.GetGeneration(), + Reason: placementv1alpha1.PlacementBindingAvailableCondReasonWaitingForAvailabilityCheck, + Message: "Waiting for the resources to be checked for availability in the target cluster", + }) + + // Count the number of manifests in all created/updated work objects. + total := int32(0) + for idx := range worksToCreateOrUpdate { + total += int32(len(worksToCreateOrUpdate[idx].Spec.Manifests)) //nolint:gosec // safe: there is no risk of overflowing due to API-level restrictions. + } + placementBindingStatus.SelectedResources = ptr.To(total) + + // Clear the other counters and failed resources as their previous values no longer apply. + placementBindingStatus.SynchronizedResources = nil + placementBindingStatus.AvailableResources = nil + placementBindingStatus.FailedResources = nil + + if err := r.hubClient.Status().Update(ctx, placementBinding); err != nil { + return errors.NewAPIServerError(err, "failed to update placement binding status", true) + } + klog.V(2).InfoS("Reported placement binding processing progress", + "placementBinding", klog.KObj(placementBinding), "selectedResources", total) + return nil +} + +func refreshPlacementBindingSyncCond(placementBinding placementv1alpha1.PlacementBindingAccessor, works []placementv1alpha1.Work) (notReady bool) { + // The binding is synchronized only if every work has been applied and its applied condition is up-to-date. + synchronized := true + for idx := range works { + work := &works[idx] + appliedCond := meta.FindStatusCondition(work.Status.Conditions, placementv1alpha1.WorkCondTypeApplied) + if appliedCond == nil || appliedCond.ObservedGeneration != work.Generation { + // The work object has not been applied or its applied condition is outdated. Instead of refreshing + // the placement binding status conditions on stale data, return a transient error and wait for + // the status to be updated. + return true + } + if !condition.IsConditionStatusTrue(appliedCond, work.GetGeneration()) { + synchronized = false + break + } + } + + var syncCond metav1.Condition + if synchronized { + syncCond = metav1.Condition{ + Type: placementv1alpha1.PlacementBindingCondTypeSynchronized, + Status: metav1.ConditionTrue, + ObservedGeneration: placementBinding.GetGeneration(), + Reason: placementv1alpha1.PlacementBindingSynchronizedCondReasonAllResourcesSynchronized, + Message: "All resources have been synchronized to the target cluster", + } + } else { + syncCond = metav1.Condition{ + Type: placementv1alpha1.PlacementBindingCondTypeSynchronized, + Status: metav1.ConditionFalse, + ObservedGeneration: placementBinding.GetGeneration(), + Reason: placementv1alpha1.PlacementBindingSynchronizedCondReasonFailedToSynchronizeSomeResources, + Message: "Some resources might be out of sync in the target cluster", + } + } + meta.SetStatusCondition(&placementBinding.GetStatus().Conditions, syncCond) + return false +} + +func refreshPlacementBindingAvailableCond(placementBinding placementv1alpha1.PlacementBindingAccessor, works []placementv1alpha1.Work) (notReady bool) { + // The binding is available only if every work is available and its available condition is up-to-date. + available := true + for idx := range works { + work := &works[idx] + availableCond := meta.FindStatusCondition(work.Status.Conditions, placementv1alpha1.WorkCondTypeAvailable) + if availableCond == nil || availableCond.ObservedGeneration != work.Generation { + // The work object has not been marked as available or its available condition is outdated. + // Instead of refreshing the placement binding status conditions on stale data, return a transient error + // and wait for the status to be updated. + return true + } + if !condition.IsConditionStatusTrue(availableCond, work.GetGeneration()) { + available = false + break + } + } + + var availableCond metav1.Condition + if available { + availableCond = metav1.Condition{ + Type: placementv1alpha1.PlacementBindingCondTypeAvailable, + Status: metav1.ConditionTrue, + ObservedGeneration: placementBinding.GetGeneration(), + Reason: placementv1alpha1.PlacementBindingAvailableCondReasonAllResourcesAvailable, + Message: "All resources are available in the target cluster", + } + } else { + availableCond = metav1.Condition{ + Type: placementv1alpha1.PlacementBindingCondTypeAvailable, + Status: metav1.ConditionFalse, + ObservedGeneration: placementBinding.GetGeneration(), + Reason: placementv1alpha1.PlacementBindingAvailableCondReasonSomeResourcesUnavailable, + Message: "Some resources might be unavailable in the target cluster", + } + } + meta.SetStatusCondition(&placementBinding.GetStatus().Conditions, availableCond) + return false +} + +func countResourcesInWorksByProcessingResults(works []placementv1alpha1.Work) ( + total, synced, available int32, + failed []placementv1alpha1.FailedResource, +) { + for i := range works { + work := &works[i] + total += int32(len(work.Spec.Manifests)) //nolint:gosec // safe: there is no risk of overflowing due to API-level restrictions. + for j := range work.Status.Manifests { + manifest := &work.Status.Manifests[j] + + appliedCond := meta.FindStatusCondition(manifest.Conditions, placementv1alpha1.ManifestCondTypeApplied) + // Note that the checks below do not take into account the condition's observed generation; this is + // because for manifest conditions KubeFleet uses the generation of the actual manifest object + // being applied, not the generation of the work object. + switch { + case appliedCond == nil: + // The Applied condition has not been set yet; the manifest has not been processed. + continue + case appliedCond.Status != metav1.ConditionTrue: + // The manifest has failed to be applied. + failed = append(failed, failedResourceFromManifestStatus(manifest, appliedCond)) + continue + default: + // The manifest has been applied. + synced++ + } + + availableCond := meta.FindStatusCondition(manifest.Conditions, placementv1alpha1.ManifestCondTypeAvailable) + switch { + case availableCond == nil: + // The Available condition has not been set yet; the manifest has not been processed. + continue + case availableCond.Status != metav1.ConditionTrue: + // The manifest is not yet available. We consider an applied manifest in a failed state if + // it fails the availability check; see the work applier for the rules. + failed = append(failed, failedResourceFromManifestStatus(manifest, availableCond)) + continue + default: + // The manifest is available. + available++ + } + } + } + return total, synced, available, failed +} + +// failedResourceFromManifestStatus builds a FailedResource from a per-manifest status and the condition that +// is not true (nil if the condition is absent). +func failedResourceFromManifestStatus(manifest *placementv1alpha1.PerManifestStatus, falseCond *metav1.Condition) placementv1alpha1.FailedResource { + failedResource := placementv1alpha1.FailedResource{ + ObjectRef: placementv1alpha1.ObjectReference{ + Namespace: manifest.Identifier.Namespace, + Name: manifest.Identifier.Name, + APIGroup: manifest.Identifier.APIGroup, + APIVersion: manifest.Identifier.APIVersion, + Kind: manifest.Identifier.Kind, + }, + DiffDetails: manifest.DiffDetails, + } + if falseCond != nil { + // Note that per KubeFleet API semantics, the observed generation set in the copied condition is the generation + // of the actual manifest object being applied in the member cluster, not the generation of the work object + // nor the placement binding object. + failedResource.Conditions = []metav1.Condition{*falseCond} + } + return failedResource +} diff --git a/pkg/v1/controllers/workgenerator/uniquename.go b/pkg/v1/controllers/workgenerator/uniquename.go new file mode 100644 index 000000000..e97448b7d --- /dev/null +++ b/pkg/v1/controllers/workgenerator/uniquename.go @@ -0,0 +1,133 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package workgenerator + +import ( + "crypto/sha256" + "fmt" + "strings" + + placementv1alpha1 "go.goms.io/fleet/apis/kubefleet.dev/placement/v1alpha1" +) + +const ( + nameLenLimit = 251 + hashSegLen = 12 +) + +const ( + // The name format for work objects when they are derived from placement resource snapshots. + // Typically, these objects are named using the format: + // + // `[PLACEMENT-POLICY-NAMESPACED-NAME]-work-[HASH]`, if the work is derived from the primary + // placement resource snapshot, or + // `[PLACEMENT-POLICY-NAMESPACED-NAME]-work-[DERIVED-FROM-SOURCE-LABEL]-[HASH]`, if the work is + // derived from other sources (e.g., a secondary placement resource snapshot). + // + // where + // + // * `[PLACEMENT-POLICY-NAMESPACED-NAME]` is the namespace and name of the placement policy that owns the + // placement resource snapshot (and indirectly owns the work objects via placement binding), in the + // format `[NAMESPACE]-[NAME]` (if the placement policy is cluster-scoped, the namespaced name segment is + // simply the placement policy name); + // * `[DERIVED-FROM-SOURCE-LABEL]` is a label that helps identify the source where the work object is derived + // from; it is not guaranteed to be unique and is added for informational purposes only. + // * `[HASH]` is the first few characters of the hash of the value + // `[PLACEMENT-POLICY-NAMESPACE]/[PLACEMENT-POLICY-NAME]-work` or + // `[PLACEMENT-POLICY-NAMESPACE]/[PLACEMENT-POLICY-NAME]-work-[DERIVED-FROM-SOURCE-TYPE]/[DERIVED-FROM-SOURCE-ID]` + // respectively, where `[DERIVED-FROM-SOURCE-TYPE]` is the type of the source where the work object is + // derived from (e.g., `placement-resource-snapshot` for placement resource snapshots), and + // `[DERIVED-FROM-SOURCE-ID]` is an identifier of the source object. Together the segment uniquely identifies + // the source where the work object is derived from (among all the work objects that are created/updated + // for the placement binding). + // + // The slash is used here instead of a dash to avoid collisions between different namespace/name combinations, + // e.g., to make sure that a placement policy named `red` in namespace `team-a` and a placement policy named + // `a-red` in namespace `team` do not produce the same hash. + // + // If the name becomes too long (> 251 characters), KubeFleet will truncate the placement policy namespaced name + // segment and the derived from source marker segment as appropriate. + workDerivedFromPrimarySnapshotSourceNameFmt = "%s-work-%s" + workDerivedFromOtherSourcesNameFmt = "%s-work-%s-%s" +) + +// uniqueNameForWorkDerivedFromPlacementResourceSnapshot generates a unique name for a work object derived from a +// placement resource snapshot, given the owner placement binding and the snapshot sub-index (0 = primary). +func uniqueNameForWorkDerivedFromPlacementResourceSnapshot( + placementBinding placementv1alpha1.PlacementBindingAccessor, + isFromPrimarySnapshot bool, + derivedFromSrcFormatter derivedFromSourceFormatter, +) string { + namespace := placementBinding.GetNamespace() + policyName := placementBinding.GetSpec().PlacementPolicyName + + // The namespaced name of the owner placement policy, in the format `[NAMESPACE]-[NAME]`; for cluster-scoped + // placement policies, it is simply the placement policy name. + namespacedName := policyName + if namespace != "" { + namespacedName = fmt.Sprintf("%s-%s", namespace, policyName) + } + + // The hash is computed over the namespace and name (separated by a slash) plus the derived from source marker, + // so that different namespace/name combinations never collide, and so that a hash suffix is always present. + hashInput := fmt.Sprintf("%s/%s-work", namespace, policyName) + if !isFromPrimarySnapshot { + hashInput = fmt.Sprintf("%s/%s-work-%s/%s", namespace, policyName, derivedFromSrcFormatter.SourceType(), derivedFromSrcFormatter.SourceID()) + } + hash := fmt.Sprintf("%x", sha256.Sum256([]byte(hashInput)))[:hashSegLen] + + // Remove all dots from the namespaced name segment so that truncation cannot leave a trailing dot, + // which would produce an invalid DNS subdomain label. + namespacedName = strings.ReplaceAll(namespacedName, ".", "") + + if isFromPrimarySnapshot { + // The work is derived from the primary placement resource snapshot; the name omits the source marker segment. + name := fmt.Sprintf(workDerivedFromPrimarySnapshotSourceNameFmt, namespacedName, hash) + if len(name) <= nameLenLimit { + return name + } + + // The name is too long; truncate the namespaced name segment. The hash suffix always disambiguates. + reservedLen := len(fmt.Sprintf(workDerivedFromPrimarySnapshotSourceNameFmt, "", hash)) + availableLen := nameLenLimit - reservedLen + if len(namespacedName) > availableLen { + namespacedName = namespacedName[:availableLen] + } + return fmt.Sprintf(workDerivedFromPrimarySnapshotSourceNameFmt, namespacedName, hash) + } + + // The work is derived from another source (e.g., a secondary placement resource snapshot); the name carries + // the source marker segment. + derivedFromSrcLabel := derivedFromSrcFormatter.StrictDNSLabel() + name := fmt.Sprintf(workDerivedFromOtherSourcesNameFmt, namespacedName, derivedFromSrcLabel, hash) + if len(name) <= nameLenLimit { + return name + } + + // The name is too long; truncate the namespaced name and source marker segments, splitting the available + // space evenly between them. The hash suffix always disambiguates. + reservedLen := len(fmt.Sprintf(workDerivedFromOtherSourcesNameFmt, "", "", hash)) + availableLen := nameLenLimit - reservedLen + availablePerSeg := availableLen / 2 + if len(namespacedName) > availablePerSeg { + namespacedName = namespacedName[:availablePerSeg] + } + if len(derivedFromSrcLabel) > availablePerSeg { + derivedFromSrcLabel = derivedFromSrcLabel[:availablePerSeg] + } + return fmt.Sprintf(workDerivedFromOtherSourcesNameFmt, namespacedName, derivedFromSrcLabel, hash) +} diff --git a/pkg/v1/controllers/workgenerator/works.go b/pkg/v1/controllers/workgenerator/works.go new file mode 100644 index 000000000..fe9455b9f --- /dev/null +++ b/pkg/v1/controllers/workgenerator/works.go @@ -0,0 +1,404 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package workgenerator + +import ( + "context" + "crypto/sha256" + "fmt" + "strconv" + "strings" + "sync/atomic" + + "k8s.io/apimachinery/pkg/api/equality" + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime/schema" + "k8s.io/apimachinery/pkg/util/sets" + "k8s.io/klog/v2" + "sigs.k8s.io/controller-runtime/pkg/controller/controllerutil" + + placementv1alpha1 "go.goms.io/fleet/apis/kubefleet.dev/placement/v1alpha1" + "go.goms.io/fleet/pkg/utils" + "go.goms.io/fleet/pkg/utils/errors" + "go.goms.io/fleet/pkg/utils/parallelizer" +) + +const ( + workOwnerLabelValueLengthLimit = 61 + workOwnerLabelValueHashLength = 12 +) + +var ( + workGVK = schema.GroupVersionKind{ + Group: placementv1alpha1.GroupVersion.Group, + Version: placementv1alpha1.GroupVersion.Version, + Kind: "Work", + } +) + +// areWorksUpToDate checks if the work objects for a placement binding are up-to-date, i.e., if all the work objects +// needed given the current placement binding spec have been created/updated. If so, the work generation can skip +// the work object create/update ops and skip to refreshing the placement binding status. +func areWorksUpToDate(placementBinding placementv1alpha1.PlacementBindingAccessor, works []placementv1alpha1.Work) (bool, error) { + // Check if the last processed placement resource snapshot name in the placement binding status matches the + // primary placement resource snapshot name in the placement binding spec. If so, it ensures that all the needed + // work objects have been created/updated for the placement binding. + lastProcessedSnapshotName := "" + if placementBinding.GetStatus().LastProcessedResourceSnapshotName != nil { + lastProcessedSnapshotName = *placementBinding.GetStatus().LastProcessedResourceSnapshotName + } + primarySnapshotName := placementBinding.GetSpec().ResourceSnapshotName + + if lastProcessedSnapshotName != primarySnapshotName { + return false, nil + } + + // Do some sanity checks, just to make sure that the cache is up to date. + if len(works) == 0 { + // No work objects exist for the placement binding. The cache might not have caught up yet. + return false, errors.NewTransientError(nil, "no work objects are found; the cache might be stale") + } + + // Check if all the work objects have been linked to the expected placement resource snapshot (in the spec), + // and verify that the linked work count recorded on the primary work matches the number of listed works. + linkedWorkCount := -1 + for idx := range works { + work := &works[idx] + annotations := work.GetAnnotations() + if linked := annotations[placementv1alpha1.WorkLinkedToPrimaryPlacementResourceSnapshotAnnotationKey]; linked != primarySnapshotName { + // The work object is linked to a different placement resource snapshot than the one in the placement binding spec. + // This might happen if the cache is stale. + return false, errors.NewTransientError(nil, "found a work object that is not linked to the expected primary placement resource snapshot", + "work", klog.KObj(work), "linkedPlacementResourceSnapshotName", linked, "expectedPlacementResourceSnapshotName", primarySnapshotName) + } + + linkedWorkCountStr, found := annotations[placementv1alpha1.LinkedWorkCountAnnotationKey] + if !found { + continue + } + if linkedWorkCount != -1 { + // At any time there should be exactly one primary placement resource snapshot that has the + // linked work count annotation. + return false, errors.NewUnexpectedError(nil, "multiple primary placement resource snapshots have the linked work count annotation", + "work", klog.KObj(work), "linkedWorkCount", linkedWorkCountStr) + } + var err error + linkedWorkCount, err = strconv.Atoi(linkedWorkCountStr) + if err != nil || linkedWorkCount < 1 { + return false, errors.NewUnexpectedError(err, "invalid linked work count annotation on work object", + "work", klog.KObj(work), "linkedWorkCount", linkedWorkCountStr) + } + } + if linkedWorkCount != len(works) { + return false, errors.NewTransientError(nil, "the number of work objects is not as expected", + "expectedWorkCount", linkedWorkCount, "actualWorkCount", len(works)) + } + + // Check if the sync strategy of the placement binding still matches that on the work objects. + syncStrategy := placementBinding.GetSpec().SyncStrategy + for idx := range works { + work := &works[idx] + if !equality.Semantic.DeepEqual(work.Spec.SyncStrategy, syncStrategy) { + return false, nil + } + } + + return true, nil +} + +func (r *Reconciler) refreshWorks(ctx context.Context, + placementBinding placementv1alpha1.PlacementBindingAccessor, + sortedPlacementResourceSnapshots []placementv1alpha1.PlacementResourceSnapshotAccessor, + works []placementv1alpha1.Work, +) ([]*placementv1alpha1.Work, bool, error) { + writtenToStorage := false + worksToDelete := []*placementv1alpha1.Work{} + + // Build an index of work objects by their names. + existingWorksByName := make(map[string]*placementv1alpha1.Work, len(works)) + for idx := range works { + work := &works[idx] + existingWorksByName[work.GetName()] = work + } + + seenWorkNames := sets.Set[string]{} + // First, build a work object for the primary placement resource snapshot. This is considered to be the + // primary work object for the placement binding. + // + // This work object serves as the owner of all other work objects created for this placement binding. KubeFleet + // leverages this setup to ensure that if a placement binding is deleted, all the work objects created for it + // will be cleaned up automatically by K8s' built-in GC process. + // + // We cannot set the placement binding itself as the owner of the work objects as they might reside in + // different namespaces, and cross-namespace ownership is not allowed in K8s. The list-then-delete loop has + // limitations as well, as stale cache might leave some work objects behind. + primaryPlacementResourceSnapshot := sortedPlacementResourceSnapshots[0] + primaryWorkToCreateOrUpdate, err := buildWorkObjectFor(primaryPlacementResourceSnapshot, placementBinding, primaryPlacementResourceSnapshot.GetName()) + if err != nil { + return nil, false, errors.Wraps(err, "failed to build work object for primary placement resource snapshot", + "primaryPlacementResourceSnapshot", klog.KObj(primaryPlacementResourceSnapshot)) + } + seenWorkNames.Insert(primaryWorkToCreateOrUpdate.GetName()) + + // Then build work objects for any secondary placement resource snapshots. These work objects are considered to + // be secondary work objects for the placement binding. + var additionalWorksToCreateOrUpdate []*placementv1alpha1.Work + for idx := 1; idx < len(sortedPlacementResourceSnapshots); idx++ { + snapshot := sortedPlacementResourceSnapshots[idx] + work, err := buildWorkObjectFor(snapshot, placementBinding, primaryPlacementResourceSnapshot.GetName()) + if err != nil { + return nil, false, errors.Wraps(err, "failed to build work object for placement resource snapshot", + "placementResourceSnapshot", klog.KObj(snapshot)) + } + if seenWorkNames.Has(work.GetName()) { + return nil, false, errors.NewUnexpectedError(nil, "duplicate work object built for placement resource snapshot", + "work", klog.KObj(work), "placementResourceSnapshot", klog.KObj(snapshot)) + } + additionalWorksToCreateOrUpdate = append(additionalWorksToCreateOrUpdate, work) + seenWorkNames.Insert(work.GetName()) + } + + // Add the linked work object count annotation on the primary work object. The count is the total number of + // work objects created for the placement binding, including the primary work object itself. + primaryWorkToCreateOrUpdate.GetAnnotations()[placementv1alpha1.LinkedWorkCountAnnotationKey] = fmt.Sprintf("%d", len(additionalWorksToCreateOrUpdate)+1) + + // Check for dangling work objects (those that are no longer linked with any source) and add them to the + // deletion list. + for _, work := range existingWorksByName { + if seenWorkNames.Has(work.GetName()) { + continue + } + klog.V(2).InfoS("A work object is no longer needed; mark it for deletion", "work", klog.KObj(work)) + worksToDelete = append(worksToDelete, work) + } + + // Issue the delete ops in parallel. The control loop deletes the dangling work objects first to avoid + // potential conflicts (e.g., creating the same object twice). This is a best-effort attempt as we cannot + // create/update/delete work objects in a transactional manner. + if err := r.deleteWorkObjects(ctx, worksToDelete, placementBinding); err != nil { + return nil, false, errors.Wraps(err, "failed to delete dangling work objects") + } + + // Create the primary work object first. This is needed as the controller needs its object UID to set + // owner references on the secondary work objects. + createdOrUpdatedWorks, primaryWorkObjWrittenToStorage, err := r.createOrUpdateWorkObjects(ctx, + []*placementv1alpha1.Work{primaryWorkToCreateOrUpdate}, placementBinding) + if err != nil { + return nil, false, errors.Wraps(err, "failed to create or update work object for primary placement resource snapshot", + "primaryPlacementResourceSnapshot", klog.KObj(primaryPlacementResourceSnapshot)) + } + ownerWorkObjRef := metav1.NewControllerRef(createdOrUpdatedWorks[0], workGVK) + writtenToStorage = primaryWorkObjWrittenToStorage + + // Set the owner reference on all secondary work objects. + for idx := range additionalWorksToCreateOrUpdate { + work := additionalWorksToCreateOrUpdate[idx] + work.SetOwnerReferences([]metav1.OwnerReference{*ownerWorkObjRef}) + } + + // Issue the create or update ops for the secondary work objects in parallel. + additionalCreatedOrUpdatedWorks, additionalCreatedOrUpdated, err := r.createOrUpdateWorkObjects(ctx, additionalWorksToCreateOrUpdate, placementBinding) + if err != nil { + return nil, false, errors.Wraps(err, "failed to create or update additional work objects for secondary placement resource snapshots") + } + createdOrUpdatedWorks = append(createdOrUpdatedWorks, additionalCreatedOrUpdatedWorks...) + if !writtenToStorage { + writtenToStorage = additionalCreatedOrUpdated + } + + return createdOrUpdatedWorks, writtenToStorage, nil +} + +func buildWorkObjectFor( + placementResourceSnapshot placementv1alpha1.PlacementResourceSnapshotAccessor, + placementBinding placementv1alpha1.PlacementBindingAccessor, + primaryPlacementResourceSnapshotName string, +) (*placementv1alpha1.Work, error) { + snapshotSubIdx := placementResourceSnapshot.GetLabels()[placementv1alpha1.PlacementResourceSnapshotSubIndexLabelKey] + if len(snapshotSubIdx) == 0 { + return nil, errors.NewUnexpectedError(nil, "no sub-index label found on the placement resource snapshot") + } + derivedFromSnapshotSrcFormatter := &placementResourceSnapshotDerivedFromSourceFormatter{ + snapshotSubIdx: snapshotSubIdx, + } + placementBindingSpec := placementBinding.GetSpec() + placementResourceSnapshotSpec := placementResourceSnapshot.GetSpec() + + workName := uniqueNameForWorkDerivedFromPlacementResourceSnapshot(placementBinding, snapshotSubIdx == "0", derivedFromSnapshotSrcFormatter) + + work := &placementv1alpha1.Work{ + ObjectMeta: metav1.ObjectMeta{ + Namespace: fmt.Sprintf(utils.NamespaceNameFormat, placementBinding.GetSpec().ClusterName), + Name: workName, + }, + } + updateWorkObjectMetadataAndSpec( + work, + placementBinding.GetNamespace(), + placementBindingSpec.PlacementPolicyName, + placementBinding.GetName(), + primaryPlacementResourceSnapshotName, + derivedFromSnapshotSrcFormatter, + placementResourceSnapshotSpec.Resources, + placementBindingSpec.SyncStrategy.DeepCopy(), + ) + return work, nil +} + +func updateWorkObjectMetadataAndSpec( + work *placementv1alpha1.Work, + ownerNSName, ownerPlacementPolicyName, ownerPlacementBindingName string, + primaryPlacementResourceSnapshotName string, + derivedFromSrcFormatter derivedFromSourceFormatter, + resources []placementv1alpha1.SnapshottedResource, + syncStrategy *placementv1alpha1.SyncStrategy, +) { + // Set annotations on the work object. + annotations := work.GetAnnotations() + if annotations == nil { + annotations = make(map[string]string) + } + // Set the linked to primary placement resource snapshot annotation on the work object. + annotations[placementv1alpha1.WorkLinkedToPrimaryPlacementResourceSnapshotAnnotationKey] = primaryPlacementResourceSnapshotName + + // Set the derived from source annotation on the work object. + // + // For work objects derived from placement resource snapshots, the annotation is set with the value + // `placement-resource-snapshot/[SUB-INDEX]`, where `[SUB-INDEX]` is the sub-index of the placement + // resource snapshot that the work object is derived from. + // + // Sub-indices are used here instead of indices to avoid any fluctuations caused by the progression + // of placement resource snapshots over rollouts. + annotations[placementv1alpha1.WorkDerivedFromSourceAnnotationKey] = fmt.Sprintf("%s/%s", + derivedFromSrcFormatter.SourceType(), derivedFromSrcFormatter.SourceID()) + annotations[placementv1alpha1.WorkOwnedByPlacementPolicyAnnotationKey] = ownerPlacementPolicyName + annotations[placementv1alpha1.WorkOwnedByPlacementBindingAnnotationKey] = ownerPlacementBindingName + work.SetAnnotations(annotations) + + // Set the owner labels on the work object. + labels := work.GetLabels() + if labels == nil { + labels = make(map[string]string) + } + labels[placementv1alpha1.WorkOwnerNamespaceLabelKey] = ownerNSName + labels[placementv1alpha1.WorkOwnedByPlacementPolicyLabelKey] = workOwnerLabelValue(ownerPlacementPolicyName) + labels[placementv1alpha1.WorkOwnedByPlacementBindingLabelKey] = workOwnerLabelValue(ownerPlacementBindingName) + work.SetLabels(labels) + + // Set the snapshotted resources on the work object. + manifests := make([]placementv1alpha1.Manifest, len(resources)) + for i := range resources { + manifests[i] = placementv1alpha1.Manifest{RawExtension: resources[i].Manifest} + } + work.Spec.Manifests = manifests + + // Set the sync strategy on the work object. + work.Spec.SyncStrategy = syncStrategy +} + +func (r *Reconciler) createOrUpdateWorkObjects( + ctx context.Context, + worksToCreateOrUpdate []*placementv1alpha1.Work, + placementBinding placementv1alpha1.PlacementBindingAccessor, +) ([]*placementv1alpha1.Work, bool, error) { + childCtx, childCancel := context.WithCancel(ctx) + defer childCancel() + + createdOrUpdatedWorks := make([]*placementv1alpha1.Work, len(worksToCreateOrUpdate)) + errFlag := parallelizer.NewErrorFlag() + createdOrUpdated := atomic.Bool{} + r.parallelizer.ParallelizeUntil(childCtx, len(worksToCreateOrUpdate), func(idx int) { + work := worksToCreateOrUpdate[idx] + + createdOrUpdatedWork := &placementv1alpha1.Work{ + ObjectMeta: metav1.ObjectMeta{ + Namespace: work.GetNamespace(), + Name: work.GetName(), + }, + } + resOp, err := controllerutil.CreateOrUpdate(childCtx, r.hubClient, createdOrUpdatedWork, func() error { + // Work objects are considered to be fully internal KubeFleet resources; for this reason + // here the control loop chooses to overwrite the spec, labels, annotations, and owner references of + // the work object with the latest values instead of attempting to do a merge. + createdOrUpdatedWork.Spec = work.Spec + createdOrUpdatedWork.SetLabels(work.GetLabels()) + createdOrUpdatedWork.SetAnnotations(work.GetAnnotations()) + createdOrUpdatedWork.SetOwnerReferences(work.GetOwnerReferences()) + return nil + }) + if err != nil { + wrappedErr := errors.Wraps(err, "failed to create or update work object", + "work", klog.KObj(work), "resOp", resOp) + errFlag.Raise(wrappedErr) + childCancel() + return + } + + createdOrUpdatedWorks[idx] = createdOrUpdatedWork + if resOp != controllerutil.OperationResultNone { + // The work object has been created or updated. + createdOrUpdated.CompareAndSwap(false, true) + } + klog.V(2).InfoS("Successfully created or updated work object", + "work", klog.KObj(createdOrUpdatedWork), "resOp", resOp, + "placementBinding", klog.KObj(placementBinding)) + }, "createOrUpdateWorkObjects") + if err := errFlag.Lower(); err != nil { + return nil, false, err + } + return createdOrUpdatedWorks, createdOrUpdated.Load(), nil +} + +func (r *Reconciler) deleteWorkObjects( + ctx context.Context, + worksToDelete []*placementv1alpha1.Work, + placementBinding placementv1alpha1.PlacementBindingAccessor, +) error { + childCtx, childCancel := context.WithCancel(ctx) + defer childCancel() + + errFlag := parallelizer.NewErrorFlag() + r.parallelizer.ParallelizeUntil(childCtx, len(worksToDelete), func(idx int) { + work := worksToDelete[idx] + + if err := r.hubClient.Delete(childCtx, work); err != nil && !apierrors.IsNotFound(err) { + wrappedErr := errors.Wraps(err, "failed to delete work object", "work", klog.KObj(work)) + errFlag.Raise(wrappedErr) + childCancel() + return + } + klog.V(2).InfoS("Successfully deleted work object", + "work", klog.KObj(work), + "placementBinding", klog.KObj(placementBinding)) + }, "deleteWorkObjects") + return errFlag.Lower() +} + +func workOwnerLabelValue(ownerName string) string { + if len(ownerName) <= workOwnerLabelValueLengthLimit && !strings.Contains(ownerName, ".") { + return ownerName + } + + hash := fmt.Sprintf("%x", sha256.Sum256([]byte(ownerName)))[:workOwnerLabelValueHashLength] + name := strings.ReplaceAll(ownerName, ".", "") + prefixLength := workOwnerLabelValueLengthLimit - workOwnerLabelValueHashLength - 1 + if len(name) > prefixLength { + name = name[:prefixLength] + } + return fmt.Sprintf("%s-%s", name, hash) +} diff --git a/pkg/v1/managers/placementresourcesnapshot/manager.go b/pkg/v1/managers/placementresourcesnapshot/manager.go new file mode 100644 index 000000000..a81b8a71e --- /dev/null +++ b/pkg/v1/managers/placementresourcesnapshot/manager.go @@ -0,0 +1,108 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package placementresourcesnapshot + +import ( + "fmt" + "hash/fnv" + "sync" + + "k8s.io/apimachinery/pkg/api/meta" + "k8s.io/client-go/dynamic" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + + placementv1alpha1 "go.goms.io/fleet/apis/kubefleet.dev/placement/v1alpha1" + errors "go.goms.io/fleet/pkg/utils/errors" + "go.goms.io/fleet/pkg/utils/informer" +) + +const ( + managerName = "placementresourcesnapshot" +) + +const ( + // The format of the key used to find the mutex for a placement policy in the mutex array. + // + // Note that slashes are used to avoid unexpected collisions. + placementPolicyKeyFmt = "%s/%s" + + minSlotCnt = 256 +) + +type Manager struct { + hubClient client.Client + hubUncachedReader client.Reader + hubDynamicClient dynamic.Interface + hubDynamicInformerManager informer.Manager + + restMapper meta.RESTMapper + + mus []sync.Mutex + muSlotCnt uint32 +} + +// New returns a new Manager. +func New(mgr ctrl.Manager, + hubDynamicClient dynamic.Interface, + hubDynamicInformerManager informer.Manager, + restMapper meta.RESTMapper, + muSlotCnt int32, +) (*Manager, error) { + if muSlotCnt < minSlotCnt { + return nil, errors.NewUserError(nil, "mu slot size must be greater than or equal to the minimum limit", + "manager", managerName, "limit", minSlotCnt, "actual", muSlotCnt) + } + + return &Manager{ + hubClient: mgr.GetClient(), + hubUncachedReader: mgr.GetAPIReader(), + hubDynamicClient: hubDynamicClient, + hubDynamicInformerManager: hubDynamicInformerManager, + restMapper: restMapper, + mus: make([]sync.Mutex, muSlotCnt), + muSlotCnt: uint32(muSlotCnt), + }, nil +} + +// acquireLock acquires a mutex for a given placement policy. +// +// The placement resource snapshot manager features a slot-based locking mechanism to ensure that KubeFleet always +// snapshots resources for one placement policy at a time. A fixed number of slots are assigned when the manager +// is initialized. There might be a small chance where two placement policies need to contend for the same slot. +// +// Slots are used to avoid GC complications. +func (m *Manager) acquireLock(placementPolicy placementv1alpha1.PlacementPolicyAccessor) { + placementPolicyKey := fmt.Sprintf(placementPolicyKeyFmt, placementPolicy.GetNamespace(), placementPolicy.GetName()) + + hasher := fnv.New32a() + hasher.Write([]byte(placementPolicyKey)) + + slot := int(hasher.Sum32() % m.muSlotCnt) + m.mus[slot].Lock() +} + +// releaseLock releases the mutex for a given placement policy. +func (m *Manager) releaseLock(placementPolicy placementv1alpha1.PlacementPolicyAccessor) { + placementPolicyKey := fmt.Sprintf(placementPolicyKeyFmt, placementPolicy.GetNamespace(), placementPolicy.GetName()) + + hasher := fnv.New32a() + hasher.Write([]byte(placementPolicyKey)) + + slot := int(hasher.Sum32() % m.muSlotCnt) + m.mus[slot].Unlock() +} diff --git a/pkg/v1/managers/placementresourcesnapshot/ops.go b/pkg/v1/managers/placementresourcesnapshot/ops.go new file mode 100644 index 000000000..7e70e490c --- /dev/null +++ b/pkg/v1/managers/placementresourcesnapshot/ops.go @@ -0,0 +1,519 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package placementresourcesnapshot + +import ( + "context" + "fmt" + "sort" + "strconv" + + "k8s.io/klog/v2" + "sigs.k8s.io/controller-runtime/pkg/client" + + placementv1alpha1 "go.goms.io/fleet/apis/kubefleet.dev/placement/v1alpha1" + errors "go.goms.io/fleet/pkg/utils/errors" + "go.goms.io/fleet/pkg/v1/utils/fieldindexers" +) + +// SnapshotResourcesIfNoSnapshotExists creates a new placement resource snapshot only if no snapshots exist. +func (m *Manager) SnapshotResourcesIfNoSnapshotExists(ctx context.Context, + placementPolicy placementv1alpha1.PlacementPolicyAccessor, +) ([]placementv1alpha1.PlacementResourceSnapshotAccessor, bool, error) { + return m.snapshotResources(ctx, placementPolicy, true) +} + +// SnapshotResourcesIfStale creates a new placement resource snapshot if the current latest snapshot has become stale. +func (m *Manager) SnapshotResourcesIfStale(ctx context.Context, + placementPolicy placementv1alpha1.PlacementPolicyAccessor, +) ([]placementv1alpha1.PlacementResourceSnapshotAccessor, bool, error) { + return m.snapshotResources(ctx, placementPolicy, false) +} + +// snapshotResources checks for the latest placement resource snapshot(s) associated +// with a placement policy; it will then: +// +// a) create new placement resource snapshot(s) if none exists, or +// b) return the latest placement resource snapshot(s) if they exist and are up-to-date; +// c) create new placement resource snapshot(s) if the latest ones exist but have become stale. +// +// Set the createOnlyWhenMissing flag to true if one only needs to create a new snapshot when none exists. +func (m *Manager) snapshotResources(ctx context.Context, + placementPolicy placementv1alpha1.PlacementPolicyAccessor, + createOnlyWhenMissing bool, +) ([]placementv1alpha1.PlacementResourceSnapshotAccessor, bool, error) { + // Do a sanity check. + if placementPolicy == nil { + return nil, false, errors.NewUnexpectedError(nil, "placement policy accessor is nil", "manager", managerName) + } + + // Acquire the mutex for the placement policy. + m.acquireLock(placementPolicy) + defer m.releaseLock(placementPolicy) + + // Retrieve the latest placement resource snapshot(s) associated with the placement policy. + snapshots, err := m.retrieveLatestSnapshot(ctx, placementPolicy) + if err != nil { + return nil, false, errors.Wraps(err, "failed to retrieve the latest placement resource snapshot(s)") + } + + // Retrieve the currently selected resources and their hash based on the placement policy. + currentResources, currentHash, err := m.retrieveAndHashSelectedResources(ctx, placementPolicy) + if err != nil { + return nil, false, errors.Wraps(err, "failed to retrieve and hash the selected resources") + } + + var latestPrimarySnapshot placementv1alpha1.PlacementResourceSnapshotAccessor + var isUpToDate bool + if len(snapshots) > 0 { + // A latest placement resource snapshot exists; check if it is up-to-date. + latestPrimarySnapshot = snapshots[0] + + isUpToDate, err = m.isSnapshotUpToDate(ctx, placementPolicy, latestPrimarySnapshot, currentHash) + if err != nil { + return nil, false, errors.Wraps(err, "failed to check if the latest placement resource snapshot is up-to-date", + "primaryPlacementResourceSnapshot", klog.KObj(latestPrimarySnapshot)) + } + } + + switch { + case createOnlyWhenMissing && len(snapshots) > 0: + // A placement resource snapshot already exists, and the requester dictates that a new snapshot can only be created if none exists. + // Return the retrieved snapshots and their freshness state. + return snapshots, isUpToDate, nil + case len(snapshots) == 0: + // No placement resource snapshot exists; create one. + createdSnapshots, err := m.createResourceSnapshotAnyway(ctx, placementPolicy, nil, currentResources, currentHash) + if err != nil { + return nil, false, errors.Wraps(err, "failed to create a placement resource snapshot", "manager", managerName) + } + return createdSnapshots, true, nil + case isUpToDate: + // The latest placement resource snapshot is up-to-date; return it as it is. + return snapshots, isUpToDate, nil + default: + // The latest placement resource snapshot exists but has become stale; create a new one. + createdSnapshots, err := m.createResourceSnapshotAnyway(ctx, placementPolicy, latestPrimarySnapshot, currentResources, currentHash) + if err != nil { + return nil, false, errors.Wraps(err, "failed to create a placement resource snapshot", "manager", managerName) + } + return createdSnapshots, true, nil + } +} + +// retrieveLatestSnapshot retrieves the latest placement resource snapshot(s) associated with a placement policy. +// +// If there are multiple placement resource snapshots with the same index, they will be returned in the ascending order +// of their sub-indices. +// +// Note that this method assumes that the corresponding mutex for the placement policy has been acquired before +// calling this method. +func (m *Manager) retrieveLatestSnapshot(ctx context.Context, placementPolicy placementv1alpha1.PlacementPolicyAccessor) ( + placementResourceSnapshot []placementv1alpha1.PlacementResourceSnapshotAccessor, err error) { + // Retrieve the primary resource snapshots associated with the placement policy. + var snapshots []placementv1alpha1.PlacementResourceSnapshotAccessor + fieldMatchers := client.MatchingFields{ + fieldindexers.PlacementResourceSnapshotOwnedByAndSubIndexedCustomFieldName: fmt.Sprintf(fieldindexers.PlacementResourceSnapshotOwnedByAndSubIndexedCustomFieldValFmt, placementPolicyOwnerLabelVal(placementPolicy), "0"), + } + if placementPolicy.GetNamespace() == "" { + // The placement policy is cluster-scoped; list cluster placement resource snapshots. + placementResourceSnapshotList := &placementv1alpha1.ClusterPlacementResourceSnapshotList{} + if err := m.hubClient.List(ctx, placementResourceSnapshotList, fieldMatchers); err != nil { + return nil, errors.NewAPIServerError(err, "failed to list cluster placement resource snapshots", true) + } + snapshots = make([]placementv1alpha1.PlacementResourceSnapshotAccessor, len(placementResourceSnapshotList.Items)) + for i := range placementResourceSnapshotList.Items { + snapshots[i] = &placementResourceSnapshotList.Items[i] + } + } else { + // The placement policy is namespace-scoped; list placement resource snapshots in the same namespace. + placementResourceSnapshotList := &placementv1alpha1.PlacementResourceSnapshotList{} + if err := m.hubClient.List(ctx, placementResourceSnapshotList, + client.InNamespace(placementPolicy.GetNamespace()), fieldMatchers); err != nil { + return nil, errors.NewAPIServerError(err, "failed to list placement resource snapshots", true) + } + snapshots = make([]placementv1alpha1.PlacementResourceSnapshotAccessor, len(placementResourceSnapshotList.Items)) + for i := range placementResourceSnapshotList.Items { + snapshots[i] = &placementResourceSnapshotList.Items[i] + } + } + + if len(snapshots) == 0 { + // No placement resource snapshot exists for the placement policy. + return nil, nil + } + + // Sort the primary snapshots by their indices. + var sortErrs []error + sort.Slice(snapshots, func(i, j int) bool { + indexIStr := snapshots[i].GetLabels()[placementv1alpha1.PlacementResourceSnapshotIndexLabelKey] + indexJStr := snapshots[j].GetLabels()[placementv1alpha1.PlacementResourceSnapshotIndexLabelKey] + indexI, iErr := strconv.Atoi(indexIStr) + indexJ, jErr := strconv.Atoi(indexJStr) + if iErr != nil { + sortErrs = append(sortErrs, fmt.Errorf("failed to convert index label to integer: %w (placementResourceSnapshot: %v)", + iErr, klog.KObj(snapshots[i]))) + return false + } + if jErr != nil { + sortErrs = append(sortErrs, fmt.Errorf("failed to convert index label to integer: %w (placementResourceSnapshot: %v)", + jErr, klog.KObj(snapshots[j]))) + return false + } + return indexI < indexJ + }) + if len(sortErrs) > 0 { + return nil, errors.NewUnexpectedError(nil, "failed to sort primary placement resource snapshots", "errs", sortErrs) + } + + latestPrimarySnapshot := snapshots[len(snapshots)-1] + // Check if there are snapshots with the same index. + subIndexedSnapshotCntStr := latestPrimarySnapshot.GetLabels()[placementv1alpha1.SubIndexedPlacementResourceSnapshotCountLabelKey] + if subIndexedSnapshotCntStr == "1" { + // The primary placement resource snapshot is the only snapshot with the latest index; return it. + return []placementv1alpha1.PlacementResourceSnapshotAccessor{latestPrimarySnapshot}, nil + } + subIndexedSnapshotCnt, err := strconv.Atoi(subIndexedSnapshotCntStr) + if err != nil { + return nil, errors.NewUnexpectedError(err, "failed to convert sub-indexed placement resource snapshot count label to integer", + "placementResourceSnapshot", klog.KObj(latestPrimarySnapshot)) + } + + if subIndexedSnapshotCnt < 1 { + // Do a sanity check. + return nil, errors.NewUnexpectedError(nil, "sub-indexed placement resource snapshot count label is less than 1", + "placementResourceSnapshot", klog.KObj(latestPrimarySnapshot), "subIndexedSnapshotCount", subIndexedSnapshotCnt) + } + + // There are sub-indexed placement resource snapshots with the same index; retrieve them. + latestIndex := latestPrimarySnapshot.GetLabels()[placementv1alpha1.PlacementResourceSnapshotIndexLabelKey] + fieldMatchers = client.MatchingFields{ + fieldindexers.PlacementResourceSnapshotOwnedByAndIndexedCustomFieldName: fmt.Sprintf(fieldindexers.PlacementResourceSnapshotOwnedByAndIndexedCustomFieldValFmt, placementPolicyOwnerLabelVal(placementPolicy), latestIndex), + } + var subIndexedSnapshots []placementv1alpha1.PlacementResourceSnapshotAccessor + if placementPolicy.GetNamespace() == "" { + // The placement policy is cluster-scoped; list cluster placement resource snapshots. + placementResourceSnapshotList := &placementv1alpha1.ClusterPlacementResourceSnapshotList{} + if err := m.hubClient.List(ctx, placementResourceSnapshotList, fieldMatchers); err != nil { + return nil, errors.NewAPIServerError(err, "failed to list cluster placement resource snapshots", true) + } + subIndexedSnapshots = make([]placementv1alpha1.PlacementResourceSnapshotAccessor, len(placementResourceSnapshotList.Items)) + for i := range placementResourceSnapshotList.Items { + subIndexedSnapshots[i] = &placementResourceSnapshotList.Items[i] + } + } else { + // The placement policy is namespace-scoped; list placement resource snapshots in the same namespace. + placementResourceSnapshotList := &placementv1alpha1.PlacementResourceSnapshotList{} + if err := m.hubClient.List(ctx, placementResourceSnapshotList, + client.InNamespace(placementPolicy.GetNamespace()), fieldMatchers); err != nil { + return nil, errors.NewAPIServerError(err, "failed to list placement resource snapshots", true) + } + subIndexedSnapshots = make([]placementv1alpha1.PlacementResourceSnapshotAccessor, len(placementResourceSnapshotList.Items)) + for i := range placementResourceSnapshotList.Items { + subIndexedSnapshots[i] = &placementResourceSnapshotList.Items[i] + } + } + // Sort the sub-indexed snapshots by their sub-indices. + sortErrs = nil + sort.Slice(subIndexedSnapshots, func(i, j int) bool { + subIndexIStr := subIndexedSnapshots[i].GetLabels()[placementv1alpha1.PlacementResourceSnapshotSubIndexLabelKey] + subIndexJStr := subIndexedSnapshots[j].GetLabels()[placementv1alpha1.PlacementResourceSnapshotSubIndexLabelKey] + subIndexI, iErr := strconv.Atoi(subIndexIStr) + subIndexJ, jErr := strconv.Atoi(subIndexJStr) + if iErr != nil { + sortErrs = append(sortErrs, fmt.Errorf("failed to convert sub-index label to integer: %w (placementResourceSnapshot: %v)", + iErr, klog.KObj(subIndexedSnapshots[i]))) + return false + } + if jErr != nil { + sortErrs = append(sortErrs, fmt.Errorf("failed to convert sub-index label to integer: %w (placementResourceSnapshot: %v)", + jErr, klog.KObj(subIndexedSnapshots[j]))) + return false + } + return subIndexI < subIndexJ + }) + if len(sortErrs) > 0 { + return nil, errors.NewUnexpectedError(nil, "failed to sort sub-indexed placement resource snapshots", "errs", sortErrs) + } + + // Verify that there are enough sub-indexed placement resource snapshots as dictated by the count label. + if len(subIndexedSnapshots) < subIndexedSnapshotCnt { + // Normally this would never happen, as the manager creates secondary placement resource snapshots first + // before creating the primary placement resource snapshot with the count label. + return nil, errors.NewUnexpectedError(nil, "there are fewer sub-indexed placement resource snapshots than the count label indicates", + "expectedCount", subIndexedSnapshotCnt, "actualCount", len(subIndexedSnapshots)) + } + + // As there is no way to create multiple placement resource snapshots with the same index in a transactional + // manner, there exists a corner case where the manager, when going through several snapshot creation passes, + // created more placement resource snapshots than the count label indicates. The extra snapshots are orphans from + // resource changes that have been overwritten. + // + // This is not registered as an error, and here the manager returns only the number of placement resource snapshots + // dictated by the count label, which is guaranteed to be consistent. The orphaned snapshots will eventually + // be cleaned up. + if len(subIndexedSnapshots) > subIndexedSnapshotCnt { + // There are more snapshots than expected; log a warning and only return the ones dictated by the count. + klog.Warningf("found more sub-indexed placement resource snapshots (%d) than the count label indicates (%d) for placement policy %v; only returning the first %d", + len(subIndexedSnapshots), subIndexedSnapshotCnt, klog.KObj(placementPolicy), subIndexedSnapshotCnt) + } + + return subIndexedSnapshots[:subIndexedSnapshotCnt], nil +} + +// isSnapshotUpToDate checks if the given placement resource snapshot is up-to-date, i.e., the snapshot is +// consistent with the current state of the resources as selected by the placement policy. +// +// Note that this method assumes that the corresponding mutex for the placement policy has been acquired before +// calling this method. +func (m *Manager) isSnapshotUpToDate( + ctx context.Context, + placementPolicy placementv1alpha1.PlacementPolicyAccessor, + primaryPlacementResourceSnapshot placementv1alpha1.PlacementResourceSnapshotAccessor, + currentHash string, +) (bool, error) { + // Get the contents hash annotation from the given primary placement resource snapshot. + snapshotHash := primaryPlacementResourceSnapshot.GetAnnotations()[placementv1alpha1.PlacementResourceSnapshotContentsHashAnnotationKey] + + if snapshotHash != currentHash { + // The hashes do not match; the placement resource snapshot is not up-to-date. + // + // Note that due to the check being carried out using a cached client, false negatives are possible, i.e., + // a newer snapshot with matching hash might have been created, yet the cache has not been updated yet. + // However, this is considered OK as any attempt to create a new snapshot based on the false negative + // will lead to a failure (`AlreadyExists` error). Eventually the cache will catch up, and consistency + // will be restored. + return false, nil + } + + // The hashes do match. + // + // Note that this check is being carried out using a cached client, and false positives can occur in the + // situation where the user does an A -> B -> A type of resource change; in this scenario the false positive + // might lead to side effects, e.g., empty rollouts and inconsistent status reporting. Here KubeFleet does a + // quorum read to verify that the snapshot is indeed up-to-date. + + // Compute the index of the snapshot that would be created next. + currentIdxStr := primaryPlacementResourceSnapshot.GetLabels()[placementv1alpha1.PlacementResourceSnapshotIndexLabelKey] + currentIdx, err := strconv.Atoi(currentIdxStr) + if err != nil { + return false, errors.NewUnexpectedError(err, "failed to convert index label to integer") + } + nextIdx := currentIdx + 1 + + found, err := m.primaryPlacementResourceSnapshotExistsAtIdx(ctx, placementPolicy, nextIdx) + if err != nil { + return false, err + } + if found { + // A primary placement resource snapshot already exists at the given index; the currently observed primary + // resource snapshot is not up-to-date. + return false, errors.NewTransientError(nil, "a newer snapshot already exists (found via quorum reads); the client cache might be stale", + "primaryPlacementResourceSnapshotName", uniqueNameForPrimaryPlacementResourceSnapshot(placementPolicy.GetName(), nextIdx), + "snapshotIndex", nextIdx) + } + + // No newer snapshot exists at the next index. + return true, nil +} + +// createResourceSnapshotAnyway creates a new placement resource snapshot for the given placement policy. +// +// Snapshot creation spans multiple objects (secondaries then the primary) and is not transactional; it relies on +// the mutex plus the hub controller manager's leader election for serialization. Because the listing/cleanup steps +// read from a cached client, a stale cache can lead to `AlreadyExists` errors on create, or to a quorum read finding +// a primary snapshot that the cache has yet to observe. These are expected and surfaced to the caller so that it +// requeues; each retry re-runs the orphan cleanup from a clean slate, and the operation converges once the cache +// catches up. +// +// Note that this method assumes that the corresponding mutex for the placement policy has been acquired before +// calling this method. +func (m *Manager) createResourceSnapshotAnyway( + ctx context.Context, + placementPolicy placementv1alpha1.PlacementPolicyAccessor, + latestPrimaryPlacementResourceSnapshot placementv1alpha1.PlacementResourceSnapshotAccessor, + currentResources []placementv1alpha1.SnapshottedResource, + currentHash string, +) ([]placementv1alpha1.PlacementResourceSnapshotAccessor, error) { + // Compute the index of the snapshot that would be created next. + nextSnapshotIdx := 0 + if latestPrimaryPlacementResourceSnapshot != nil { + lastSeenSnapshotIdxStr := latestPrimaryPlacementResourceSnapshot.GetLabels()[placementv1alpha1.PlacementResourceSnapshotIndexLabelKey] + lastSeenSnapshotIdx, err := strconv.Atoi(lastSeenSnapshotIdxStr) + if err != nil { + return nil, errors.NewUnexpectedError(err, + "failed to convert last seen primary placement resource snapshot index label to integer") + } + nextSnapshotIdx = lastSeenSnapshotIdx + 1 + } + + // Clean up orphaned secondary placement resource snapshots (if any). + // + // Due to the inability to create multiple placement resource snapshots with the same index in a + // transactional manner, it is possible that the manager has already created a few secondary placement + // resource snapshot in a previous pass. In this case, the manager should delete the existing snapshots + // (its content might be outdated, and the object spec is immutable anyway) before creating new ones. + acted, err := m.cleanUpOrphanedSecondarySnapshots(ctx, placementPolicy, nextSnapshotIdx) + if err != nil { + return nil, errors.Wraps(err, "failed to clean up orphaned secondary placement resource snapshots") + } + if acted { + // Ask the caller to requeue when there are orphaned secondary placement resource snapshots to be cleaned up. + // This helps avoid oscillation issues where a not fully deleted snapshot blocks later creation. + return nil, errors.NewTransientError(nil, "cleaned up orphaned secondary placement resource snapshots; requeue before creating new snapshots", "snapshotIndex", nextSnapshotIdx) + } + + // Split the resources into size-controlled groups. Each group corresponds to a placement resource snapshot + // that will be created. + resGroups, err := splitResourcesIntoSizeControlledGroups(currentResources) + if err != nil { + return nil, errors.Wraps(err, "failed to split the selected resources into size-controlled groups") + } + + // createdSnapshots holds the created snapshots for the new index, keyed by their sub-indices. + createdSnapshots := make([]placementv1alpha1.PlacementResourceSnapshotAccessor, len(resGroups)) + + // Note (chenyu1): evaluate if parallelization is needed here. In most cases the number of secondary + // placement resource snapshots is small, so the overhead of parallelization might not be worth it. + if len(resGroups) > 1 { + // Create the secondary placement resource snapshots first. Start with the last resource group and work + // backwards, so that the primary snapshot (which carries the count label) is created last. + for subIdx := len(resGroups) - 1; subIdx >= 1; subIdx-- { + secondaryName := uniqueNameForSecondaryPlacementResourceSnapshot(placementPolicy.GetName(), nextSnapshotIdx, subIdx) + secondarySnapshot, err := secondaryPlacementResourceSnapshot( + placementPolicy.GetNamespace(), secondaryName, placementPolicy, nextSnapshotIdx, subIdx, resGroups[subIdx], currentHash, m.hubClient.Scheme()) + if err != nil { + return nil, errors.Wraps(err, "failed to build a secondary placement resource snapshot", + "secondaryPlacementResourceSnapshotName", secondaryName, + "snapshotIndex", nextSnapshotIdx, "snapshotSubIndex", subIdx) + } + + if err := m.hubClient.Create(ctx, secondarySnapshot); err != nil { + return nil, errors.NewAPIServerError(err, "failed to create a secondary placement resource snapshot", false, + "secondaryPlacementResourceSnapshot", klog.KObj(secondarySnapshot), + "snapshotIndex", nextSnapshotIdx, "snapshotSubIndex", subIdx) + } + + createdSnapshots[subIdx] = secondarySnapshot + } + } + + // Create the primary placement resource snapshot last, with the count label. + primaryName := uniqueNameForPrimaryPlacementResourceSnapshot(placementPolicy.GetName(), nextSnapshotIdx) + primarySnapshot, err := primaryPlacementResourceSnapshot( + placementPolicy.GetNamespace(), primaryName, placementPolicy, nextSnapshotIdx, resGroups[0], currentHash, len(resGroups), m.hubClient.Scheme()) + if err != nil { + return nil, errors.Wraps(err, "failed to build the primary placement resource snapshot", + "primaryPlacementResourceSnapshotName", primaryName, "snapshotIndex", nextSnapshotIdx) + } + + if err := m.hubClient.Create(ctx, primarySnapshot); err != nil { + // Note that if the primary placement resource snapshot already exists, no deletion will be attempted. The + // caller must retry and create the next placement resource snapshot with a new index. + return nil, errors.NewAPIServerError(err, "failed to create the primary placement resource snapshot", false, + "primaryPlacementResourceSnapshot", klog.KObj(primarySnapshot), "snapshotIndex", nextSnapshotIdx) + } + + createdSnapshots[0] = primarySnapshot + return createdSnapshots, nil +} + +// cleanUpOrphanedSecondarySnapshots deletes all secondary placement resource snapshots at the given index. +// +// Note that this method assumes that the corresponding mutex for the placement policy has been acquired before +// calling this method. +func (m *Manager) cleanUpOrphanedSecondarySnapshots( + ctx context.Context, + placementPolicy placementv1alpha1.PlacementPolicyAccessor, + nextSnapshotIdx int, +) (bool, error) { + // List all placement resource snapshots at the given index. + fieldMatchers := client.MatchingFields{ + fieldindexers.PlacementResourceSnapshotOwnedByAndIndexedCustomFieldName: fmt.Sprintf(fieldindexers.PlacementResourceSnapshotOwnedByAndIndexedCustomFieldValFmt, placementPolicyOwnerLabelVal(placementPolicy), strconv.Itoa(nextSnapshotIdx)), + } + + var snapshots []placementv1alpha1.PlacementResourceSnapshotAccessor + if placementPolicy.GetNamespace() == "" { + // The placement policy is cluster-scoped; list cluster placement resource snapshots. + placementResourceSnapshotList := &placementv1alpha1.ClusterPlacementResourceSnapshotList{} + if err := m.hubClient.List(ctx, placementResourceSnapshotList, fieldMatchers); err != nil { + return false, errors.NewAPIServerError(err, "failed to list cluster placement resource snapshots", true) + } + snapshots = make([]placementv1alpha1.PlacementResourceSnapshotAccessor, len(placementResourceSnapshotList.Items)) + for i := range placementResourceSnapshotList.Items { + snapshots[i] = &placementResourceSnapshotList.Items[i] + } + } else { + // The placement policy is namespace-scoped; list placement resource snapshots in the same namespace. + placementResourceSnapshotList := &placementv1alpha1.PlacementResourceSnapshotList{} + if err := m.hubClient.List(ctx, placementResourceSnapshotList, + client.InNamespace(placementPolicy.GetNamespace()), fieldMatchers); err != nil { + return false, errors.NewAPIServerError(err, "failed to list placement resource snapshots", true) + } + snapshots = make([]placementv1alpha1.PlacementResourceSnapshotAccessor, len(placementResourceSnapshotList.Items)) + for i := range placementResourceSnapshotList.Items { + snapshots[i] = &placementResourceSnapshotList.Items[i] + } + } + + if len(snapshots) == 0 { + // No placement resource snapshots are found at the index. + return false, nil + } + + // Do a sanity check; verify that there is no primary placement resource snapshot at the given index. + for idx := range snapshots { + snapshot := snapshots[idx] + subIdxStr := snapshot.GetLabels()[placementv1alpha1.PlacementResourceSnapshotSubIndexLabelKey] + if subIdxStr == "0" { + // This normally should never occur. + return false, errors.NewUnexpectedError(nil, + "found a primary placement resource snapshot at the given index while cleaning up orphaned secondary snapshots", + "primaryPlacementResourceSnapshot", klog.KObj(snapshot)) + } + } + + // There exists a corner case, where, due to the staleness of cache, a primary placement resource snapshot has been created + // at the given (next) index yet has not been registered in the cache. Do a quorum read to confirm this. + found, err := m.primaryPlacementResourceSnapshotExistsAtIdx(ctx, placementPolicy, nextSnapshotIdx) + if err != nil { + return false, errors.Wraps(err, "failed to perform a quorum read for the primary placement resource snapshot at the given index", + "snapshotIdx", nextSnapshotIdx) + } + if found { + // A primary placement resource snapshot already exists at the given index; the secondary snapshots found + // here are not orphans. Report this as an error; the caller should requeue and wait for the cache to catch up. + return false, errors.NewTransientError(nil, "a primary placement resource snapshot already exists at the given index (found via quorum read); the client cache might be stale", + "primaryPlacementResourceSnapshotName", uniqueNameForPrimaryPlacementResourceSnapshot(placementPolicy.GetName(), nextSnapshotIdx), + "snapshotIndex", nextSnapshotIdx) + } + + // Delete all the secondary placement resource snapshots at the given index. + for idx := range snapshots { + snapshot := snapshots[idx] + if !snapshot.GetDeletionTimestamp().IsZero() { + // The secondary placement resource snapshot has been marked for deletion; wait for it to complete. + continue + } + + if err := m.hubClient.Delete(ctx, snapshot); err != nil { + return false, errors.NewAPIServerError(err, "failed to delete an orphaned secondary placement resource snapshot", + false, "secondaryPlacementResourceSnapshot", klog.KObj(snapshot)) + } + } + return true, nil +} diff --git a/pkg/v1/managers/placementresourcesnapshot/quorumread.go b/pkg/v1/managers/placementresourcesnapshot/quorumread.go new file mode 100644 index 000000000..5d8206637 --- /dev/null +++ b/pkg/v1/managers/placementresourcesnapshot/quorumread.go @@ -0,0 +1,56 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package placementresourcesnapshot + +import ( + "context" + + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/types" + + placementv1alpha1 "go.goms.io/fleet/apis/kubefleet.dev/placement/v1alpha1" + errors "go.goms.io/fleet/pkg/utils/errors" +) + +func (m *Manager) primaryPlacementResourceSnapshotExistsAtIdx( + ctx context.Context, + placementPolicy placementv1alpha1.PlacementPolicyAccessor, + idx int) (bool, error) { + name := uniqueNameForPrimaryPlacementResourceSnapshot(placementPolicy.GetName(), idx) + namespace := placementPolicy.GetNamespace() + + kind := placementv1alpha1.ClusterPlacementResourceSnapshotKind + if namespace != "" { + kind = placementv1alpha1.PlacementResourceSnapshotKind + } + + // A metadata-only read; the snapshot spec may be large and is not needed here. + metadata := metav1.PartialObjectMetadata{} + metadata.SetGroupVersionKind(placementv1alpha1.GroupVersion.WithKind(kind)) + + // Read from the API server directly, bypassing the (possibly stale) cache. + if err := m.hubUncachedReader.Get(ctx, types.NamespacedName{Name: name, Namespace: namespace}, &metadata); err != nil { + if apierrors.IsNotFound(err) { + return false, nil + } + return false, errors.NewAPIServerError(err, + "failed to get the partial object metadata of the primary placement resource snapshot", false, + "primaryPlacementResourceSnapshotName", name, "snapshotIndex", idx) + } + return true, nil +} diff --git a/pkg/v1/managers/placementresourcesnapshot/resources.go b/pkg/v1/managers/placementresourcesnapshot/resources.go new file mode 100644 index 000000000..1a0bd1fc6 --- /dev/null +++ b/pkg/v1/managers/placementresourcesnapshot/resources.go @@ -0,0 +1,425 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package placementresourcesnapshot + +import ( + "context" + "fmt" + "sort" + + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/meta" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/apimachinery/pkg/runtime/schema" + "k8s.io/apimachinery/pkg/util/sets" + "k8s.io/klog/v2" + "k8s.io/kubectl/pkg/util/deployment" + + placementv1alpha1 "go.goms.io/fleet/apis/kubefleet.dev/placement/v1alpha1" + errors "go.goms.io/fleet/pkg/utils/errors" + hasher "go.goms.io/fleet/pkg/utils/resource" +) + +const ( + // etcd has a 1.5 MiB limit for objects by default, and Kubernetes clients might + // reject request entities too large (~2/~3 MiB, depending on the protocol in use). + // + // With these factors considered, we set the maximum size of all resource data in a single placement + // resource snapshot to be ~1.2 MiB, or ~1.26 MB, which should be safe in most cases. Note that the padding + // space is not just reserved for safety reasons, but also to accommodate the additional fields in + // the placement resource snapshot object, such as metadata and labels. + maxPerSnapshotResourceDataSizeBytes = 1258291 // 1.2 MiB, or ~1.26 MB. + maxPerSnapshotResourceCnt = 50 +) + +const ( + // The format in use to generate a unique identifier for each selected resource. + // + // The format is `[API-GROUP]/[KIND]/[NAMESPACE]/[NAME]`, where + // `[API-GROUP]` is the API group of the resource, `[KIND]` is the kind of the resource, `[NAMESPACE]` is the namespace of the resource, + // and `[NAME]` is the name of the resource. + // + // The API version value is omitted from the unique identifier as for the same resource, only one version is picked as the Kubernetes + // storage version. It would not make sense to select the same resources twice using different API versions in the same placement policy. + // When this happens, KubeFleet will pick the resource as selected by the resource selector that appears first. + // + // Also note that for cluster-scoped resources, the `[NAMESPACE]` segment will be empty. + resourceUniqueIdStrFmt = "%s/%s/%s/%s" +) + +func (m *Manager) retrieveAndHashSelectedResources( + ctx context.Context, + placementPolicyAccessor placementv1alpha1.PlacementPolicyAccessor, +) ( + resources []placementv1alpha1.SnapshottedResource, + hash string, + err error, +) { + placementPolicySpec := placementPolicyAccessor.GetSpec() + resources = make([]placementv1alpha1.SnapshottedResource, 0, len(placementPolicySpec.ResourceSelectors)) + seen := sets.Set[string]{} + + if len(placementPolicySpec.ResourceSelectors) == 0 { + // KubeFleet does not consider the absence of resource selectors to be an error; however, this special + // case should be handled by the caller (i.e., the placement resource snapshot manager should not be + // called at all if there are no resource selectors), hence the unexpected error returned here. + return nil, "", errors.NewUnexpectedError(nil, "no resource selectors are present") + } + + for idx := range placementPolicySpec.ResourceSelectors { + selector := placementPolicySpec.ResourceSelectors[idx] + + resourcesAsSelected := make([]*placementv1alpha1.SnapshottedResource, 0, 5) + switch { + case len(selector.Name) != 0: + // Retrieve the resource by name. + res, err := m.retrieveResourceByName(ctx, placementPolicyAccessor, selector) + if err != nil { + return nil, "", errors.Wraps(err, "failed to retrieve a selected resource (name-based selector)", + "resourceSelectorIndex", idx) + } + resourcesAsSelected = append(resourcesAsSelected, &res) + case selector.LabelSelector != nil: + // Retrieve the resources by label selector. + resList, err := m.retrieveResourcesByLabelSelector(ctx, placementPolicyAccessor, selector) + if err != nil { + return nil, "", errors.Wraps(err, "failed to retrieve selected resources (label selector-based selector)", + "resourceSelectorIndex", idx) + } + for ridx := range resList { + resourcesAsSelected = append(resourcesAsSelected, &resList[ridx]) + } + default: + return nil, "", errors.NewUserError(nil, "invalid resource selector: neither name nor label selector is specified", + "resourceSelectorIndex", idx) + } + + // Make sure that no resources are selected more than once. Duplicates are skipped. + for ridx := range resourcesAsSelected { + resource := resourcesAsSelected[ridx] + resourceId := resourceUniqueId(resource) + if !seen.Has(resourceId) { + // The resource has not been seen before; add it to the list of selected resources. + seen.Insert(resourceId) + resources = append(resources, *resource) + } else { + // The resource has already been seen; skip it and log a message. + klog.V(2).InfoS("Found duplicate selected resource; skipping it", "resourceId", resourceId, "resourceSelectorIndex", idx) + } + } + } + + // Sort the selected resources to ensure deterministic outcomes. + sort.Slice(resources, func(i, j int) bool { + return resourceUniqueId(&resources[i]) < resourceUniqueId(&resources[j]) + }) + + hash, err = hasher.HashOf(resources) + if err != nil { + return nil, "", errors.Wraps(err, "failed to compute the hash of the selected resources") + } + return resources, hash, nil +} + +func (m *Manager) retrieveResourceByName( + ctx context.Context, + placementPolicyAccessor placementv1alpha1.PlacementPolicyAccessor, + resourceSelector placementv1alpha1.ResourceSelector, +) (placementv1alpha1.SnapshottedResource, error) { + gvk, gvr, namespace, err := m.lookUpGVKGVRAndNamespace(resourceSelector, placementPolicyAccessor.GetNamespace()) + if err != nil { + return placementv1alpha1.SnapshottedResource{}, errors.Wraps(err, "failed to look up GVK, GVR and namespace", + "resourceSelector", resourceSelector) + } + + var resource *unstructured.Unstructured + // Before retrieving the resource via cache, verify if the informer has been synced. + if m.hubDynamicInformerManager.IsInformerSet(gvk) && m.hubDynamicInformerManager.IsInformerSynced(gvr) { + // An informer for the selected resource has been set up and synced; proceed to retrieve the resource from the cache. + var obj runtime.Object + if namespace == "" { + obj, err = m.hubDynamicInformerManager.Lister(gvr).Get(resourceSelector.Name) + } else { + obj, err = m.hubDynamicInformerManager.Lister(gvr).ByNamespace(namespace).Get(resourceSelector.Name) + } + if err != nil { + return placementv1alpha1.SnapshottedResource{}, errors.NewAPIServerError(err, "failed to get selected resource", true, + "gvr", gvr, "namespace", namespace, "name", resourceSelector.Name) + } + var ok bool + resource, ok = obj.(*unstructured.Unstructured) + if !ok { + return placementv1alpha1.SnapshottedResource{}, errors.NewUnexpectedError(nil, "failed to convert the retrieved resource to unstructured", + "gvr", gvr, "namespace", namespace, "name", resourceSelector.Name) + } + } else { + // No informer is set up for the selected resource, or the informer has not been synced yet. + // + // As a fallback, retrieve the resource directly from the API server. + klog.V(2).InfoS("Informer for the selected resource is not set up or not synced; retrieving the resource directly from the API server", + "gvr", gvr) + if namespace == "" { + resource, err = m.hubDynamicClient.Resource(gvr).Get(ctx, resourceSelector.Name, metav1.GetOptions{}) + } else { + resource, err = m.hubDynamicClient.Resource(gvr).Namespace(namespace).Get(ctx, resourceSelector.Name, metav1.GetOptions{}) + } + if err != nil { + return placementv1alpha1.SnapshottedResource{}, errors.NewAPIServerError(err, "failed to get selected resource directly from the API server", false, + "gvr", gvr, "namespace", namespace, "name", resourceSelector.Name) + } + } + + snapshottedResource, err := snapshotResource(resource) + if err != nil { + return placementv1alpha1.SnapshottedResource{}, + errors.Wraps(err, "failed to snapshot selected resource", + "gvr", gvr, "namespace", namespace, "name", resourceSelector.Name) + } + return snapshottedResource, nil +} + +func (m *Manager) retrieveResourcesByLabelSelector( + ctx context.Context, + placementPolicyAccessor placementv1alpha1.PlacementPolicyAccessor, + resourceSelector placementv1alpha1.ResourceSelector, +) ([]placementv1alpha1.SnapshottedResource, error) { + gvk, gvr, namespace, err := m.lookUpGVKGVRAndNamespace(resourceSelector, placementPolicyAccessor.GetNamespace()) + if err != nil { + return nil, errors.Wraps(err, "failed to look up GVK, GVR and namespace", "resourceSelector", resourceSelector) + } + + // Convert the label selector into a selector string. + selector, err := metav1.LabelSelectorAsSelector(resourceSelector.LabelSelector) + if err != nil { + return nil, errors.NewUserError(err, "invalid label selector", "gvk", gvk, "labelSelector", resourceSelector.LabelSelector) + } + + var resources []*unstructured.Unstructured + if m.hubDynamicInformerManager.IsInformerSet(gvk) && m.hubDynamicInformerManager.IsInformerSynced(gvr) { + // An informer for the selected resources has been set up and synced; proceed to retrieve the resources from the cache. + var objList []runtime.Object + if namespace == "" { + objList, err = m.hubDynamicInformerManager.Lister(gvr).List(selector) + } else { + objList, err = m.hubDynamicInformerManager.Lister(gvr).ByNamespace(namespace).List(selector) + } + if err != nil { + return nil, errors.NewAPIServerError(err, "failed to list the selected resources", true, + "gvr", gvr, "namespace", namespace, "labelSelector", selector.String()) + } + + for idx := range objList { + obj := objList[idx] + resource, ok := obj.(*unstructured.Unstructured) + if !ok { + return nil, errors.NewUnexpectedError(nil, "failed to convert the retrieved resource to unstructured", + "gvr", gvr, "namespace", namespace) + } + resources = append(resources, resource) + } + } else { + // No informer is set up for the selected resources, or the informer has not been synced yet. + // + // As a fallback, retrieve the resources directly from the API server. + klog.V(2).InfoS("Informer for the selected resources is not set up or not synced; retrieving the resources directly from the API server", + "gvr", gvr) + var resourceList *unstructured.UnstructuredList + if namespace == "" { + resourceList, err = m.hubDynamicClient.Resource(gvr).List(ctx, metav1.ListOptions{ + LabelSelector: selector.String(), + }) + } else { + resourceList, err = m.hubDynamicClient.Resource(gvr).Namespace(namespace).List(ctx, metav1.ListOptions{ + LabelSelector: selector.String(), + }) + } + if err != nil { + return nil, errors.NewAPIServerError(err, "failed to list the selected resources", false, + "gvr", gvr, "namespace", namespace, "labelSelector", selector.String()) + } + + for idx := range resourceList.Items { + resource := &resourceList.Items[idx] + resources = append(resources, resource) + } + } + + snapshottedResources := make([]placementv1alpha1.SnapshottedResource, len(resources)) + for idx := range resources { + resource := resources[idx] + snapshottedResources[idx], err = snapshotResource(resource) + if err != nil { + return nil, errors.Wraps(err, "failed to snapshot selected resource", + "gvr", gvr, "namespace", namespace, "name", resource.GetName()) + } + } + return snapshottedResources, nil +} + +func (m *Manager) lookUpGVKGVRAndNamespace(resourceSelector placementv1alpha1.ResourceSelector, placementPolicyNSName string) ( + schema.GroupVersionKind, schema.GroupVersionResource, string, error) { + gvk := schema.GroupVersionKind{ + Group: resourceSelector.APIGroup, + Version: resourceSelector.APIVersion, + Kind: resourceSelector.Kind, + } + + // Convert the GVK to a GVR using the REST mapper. + mapping, err := m.restMapper.RESTMapping(gvk.GroupKind(), gvk.Version) + if err != nil { + return schema.GroupVersionKind{}, schema.GroupVersionResource{}, "", errors.NewUnexpectedError(err, "failed to map GVK to GVR", "gvk", gvk) + } + gvr := mapping.Resource + scope := mapping.Scope.Name() + + // Determine the namespace of the selected resources. + // + // If the placement policy is namespace-scoped, the selected resources are assumed to be from the same namespace as the placement policy; + // if the placement policy is cluster-scoped, the namespace is taken from the resource selector. + namespace := resourceSelector.Namespace + if placementPolicyNSName != "" { + namespace = placementPolicyNSName + } + + // Check if the resolved scope matches with the placement policy, i.e., PlacementPolicy can only select namespaced resources, while + // ClusterPlacementPolicy can select any resource. + if scope == meta.RESTScopeNameRoot && placementPolicyNSName != "" { + return schema.GroupVersionKind{}, schema.GroupVersionResource{}, "", + errors.NewUserError(nil, "cluster-scoped resource cannot be selected by a placement policy; use cluster placement policy instead", + "resourceSelector", resourceSelector) + } + // Check if the resolved scope matches with the resource selector, i.e., when selecting a cluster-scoped resource, no namespace can + // be specified. + if scope == meta.RESTScopeNameRoot && namespace != "" { + return schema.GroupVersionKind{}, schema.GroupVersionResource{}, "", + errors.NewUserError(nil, "namespace must not be specified for cluster-scoped resources", "resourceSelector", resourceSelector) + } + + return gvk, gvr, namespace, nil +} + +// snapshotResource removes fields that are not needed in a snapshot from an unstructured resource and converts it +// into a SnapshottedResource. +func snapshotResource(resource *unstructured.Unstructured) (placementv1alpha1.SnapshottedResource, error) { + // Create a deep copy of the resource. + resourceCopy := resource.DeepCopy() + + // Remove certain labels and annotations. + if annotations := resourceCopy.GetAnnotations(); annotations != nil { + // Remove the last applied configuration set by kubectl. + delete(annotations, corev1.LastAppliedConfigAnnotation) + + // Remove the revision annotation set by deployment controller. + delete(annotations, deployment.RevisionAnnotation) + + if len(annotations) == 0 { + resourceCopy.SetAnnotations(nil) + } else { + resourceCopy.SetAnnotations(annotations) + } + } + + // Remove certain system-managed fields. + resourceCopy.SetOwnerReferences(nil) + resourceCopy.SetManagedFields(nil) + + // Remove the read-only fields. + resourceCopy.SetCreationTimestamp(metav1.Time{}) + resourceCopy.SetDeletionTimestamp(nil) + resourceCopy.SetDeletionGracePeriodSeconds(nil) + resourceCopy.SetGeneration(0) + resourceCopy.SetResourceVersion("") + resourceCopy.SetSelfLink("") + resourceCopy.SetUID("") + + // Remove the status field. + unstructured.RemoveNestedField(resourceCopy.Object, "status") + + resourceCopyRawData, err := resourceCopy.MarshalJSON() + if err != nil { + return placementv1alpha1.SnapshottedResource{}, errors.NewUnexpectedError(err, "failed to marshal the resource copy to JSON", + "resource", klog.KObj(resourceCopy)) + } + + gvk := resource.GroupVersionKind() + + // Note that for regular Kubernetes resources, the additional information field is always left empty. + return placementv1alpha1.SnapshottedResource{ + Identifier: placementv1alpha1.ObjectReference{ + Namespace: resourceCopy.GetNamespace(), + Name: resourceCopy.GetName(), + APIGroup: gvk.Group, + APIVersion: gvk.Version, + Kind: gvk.Kind, + }, + Manifest: runtime.RawExtension{Raw: resourceCopyRawData}, + }, nil +} + +func resourceUniqueId(resource *placementv1alpha1.SnapshottedResource) string { + return fmt.Sprintf(resourceUniqueIdStrFmt, + resource.Identifier.APIGroup, + resource.Identifier.Kind, + resource.Identifier.Namespace, + resource.Identifier.Name) +} + +func splitResourcesIntoSizeControlledGroups(resources []placementv1alpha1.SnapshottedResource) ([][]placementv1alpha1.SnapshottedResource, error) { + if len(resources) == 0 { + // Return one single empty group. + return [][]placementv1alpha1.SnapshottedResource{{}}, nil + } + + var groups [][]placementv1alpha1.SnapshottedResource + // Pre-allocate with a reasonably guessed initial capacity. + currentGroup := make([]placementv1alpha1.SnapshottedResource, 0, 10) + currentSize := 0 + + for i := range resources { + resource := resources[i] + resourceSize := len(resource.Manifest.Raw) + for _, info := range resource.AdditionalInfo { + resourceSize += len(info) + } + + if resourceSize > maxPerSnapshotResourceDataSizeBytes { + // A single resource exceeds the per-snapshot size limit; it can never fit into any group. + return nil, errors.NewUserError(nil, "a single selected resource is too large to fit in a placement resource snapshot", + "resource", resource.Identifier, + "resourceSizeBytes", resourceSize, "maxPerSnapshotResourceDataSizeBytes", maxPerSnapshotResourceDataSizeBytes) + } + + // Start a new group if adding this resource would exceed either the size or the count limit. + if len(currentGroup) > 0 && + (currentSize+resourceSize > maxPerSnapshotResourceDataSizeBytes || len(currentGroup) >= maxPerSnapshotResourceCnt) { + groups = append(groups, currentGroup) + currentGroup = nil + currentSize = 0 + } + + currentGroup = append(currentGroup, resource) + currentSize += resourceSize + } + + if len(currentGroup) > 0 { + groups = append(groups, currentGroup) + } + + return groups, nil +} diff --git a/pkg/v1/managers/placementresourcesnapshot/snapshots.go b/pkg/v1/managers/placementresourcesnapshot/snapshots.go new file mode 100644 index 000000000..20338c769 --- /dev/null +++ b/pkg/v1/managers/placementresourcesnapshot/snapshots.go @@ -0,0 +1,154 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package placementresourcesnapshot + +import ( + "crypto/sha256" + "fmt" + "strconv" + + placementv1alpha1 "go.goms.io/fleet/apis/kubefleet.dev/placement/v1alpha1" + errors "go.goms.io/fleet/pkg/utils/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/klog/v2" + "sigs.k8s.io/controller-runtime/pkg/controller/controllerutil" +) + +const ( + // A limit of 61 is set here; Kubernetes labels accept values of up to 63 characters, and + // KubeFleet reserves 2 more characters as a buffer. + placementPolicyOwnerLabelValLenLimit = 61 + placementPolicyOwnerLabelHashLen = 12 +) + +func primaryPlacementResourceSnapshot( + namespace, name string, + ownerPlacementPolicy placementv1alpha1.PlacementPolicyAccessor, + idx int, + resources []placementv1alpha1.SnapshottedResource, + resourceHash string, + snapshotCount int, + scheme *runtime.Scheme, +) (placementv1alpha1.PlacementResourceSnapshotAccessor, error) { + labels := map[string]string{ + placementv1alpha1.PlacementResourceSnapshotOwnedByLabelKey: placementPolicyOwnerLabelVal(ownerPlacementPolicy), + placementv1alpha1.PlacementResourceSnapshotIndexLabelKey: strconv.Itoa(idx), + placementv1alpha1.PlacementResourceSnapshotSubIndexLabelKey: "0", + placementv1alpha1.SubIndexedPlacementResourceSnapshotCountLabelKey: strconv.Itoa(snapshotCount), + } + annotations := map[string]string{ + placementv1alpha1.PlacementResourceSnapshotContentsHashAnnotationKey: resourceHash, + } + + var primarySnapshot placementv1alpha1.PlacementResourceSnapshotAccessor + if namespace == "" { + primarySnapshot = &placementv1alpha1.ClusterPlacementResourceSnapshot{ + ObjectMeta: metav1.ObjectMeta{ + Name: name, + Labels: labels, + Annotations: annotations, + }, + Spec: placementv1alpha1.PlacementResourceSnapshotSpec{ + Resources: resources, + }, + } + } else { + primarySnapshot = &placementv1alpha1.PlacementResourceSnapshot{ + ObjectMeta: metav1.ObjectMeta{ + Name: name, + Namespace: namespace, + Labels: labels, + Annotations: annotations, + }, + Spec: placementv1alpha1.PlacementResourceSnapshotSpec{ + Resources: resources, + }, + } + } + + if err := controllerutil.SetControllerReference(ownerPlacementPolicy, primarySnapshot, scheme); err != nil { + return nil, errors.NewUnexpectedError(err, "failed to set controller reference on the primary placement resource snapshot", + "primaryPlacementResourceSnapshot", klog.KObj(primarySnapshot)) + } + return primarySnapshot, nil +} + +func secondaryPlacementResourceSnapshot( + namespace, name string, + ownerPlacementPolicy placementv1alpha1.PlacementPolicyAccessor, + idx, subIdx int, + resources []placementv1alpha1.SnapshottedResource, + resourceHash string, + scheme *runtime.Scheme, +) (placementv1alpha1.PlacementResourceSnapshotAccessor, error) { + labels := map[string]string{ + placementv1alpha1.PlacementResourceSnapshotOwnedByLabelKey: placementPolicyOwnerLabelVal(ownerPlacementPolicy), + placementv1alpha1.PlacementResourceSnapshotIndexLabelKey: strconv.Itoa(idx), + placementv1alpha1.PlacementResourceSnapshotSubIndexLabelKey: strconv.Itoa(subIdx), + } + annotations := map[string]string{ + placementv1alpha1.PlacementResourceSnapshotContentsHashAnnotationKey: resourceHash, + } + + var secondarySnapshot placementv1alpha1.PlacementResourceSnapshotAccessor + if namespace == "" { + secondarySnapshot = &placementv1alpha1.ClusterPlacementResourceSnapshot{ + ObjectMeta: metav1.ObjectMeta{ + Name: name, + Labels: labels, + Annotations: annotations, + }, + Spec: placementv1alpha1.PlacementResourceSnapshotSpec{ + Resources: resources, + }, + } + } else { + secondarySnapshot = &placementv1alpha1.PlacementResourceSnapshot{ + ObjectMeta: metav1.ObjectMeta{ + Name: name, + Namespace: namespace, + Labels: labels, + Annotations: annotations, + }, + Spec: placementv1alpha1.PlacementResourceSnapshotSpec{ + Resources: resources, + }, + } + } + + if err := controllerutil.SetControllerReference(ownerPlacementPolicy, secondarySnapshot, scheme); err != nil { + return nil, errors.NewUnexpectedError(err, "failed to set controller reference on a secondary placement resource snapshot", + "secondaryPlacementResourceSnapshot", klog.KObj(secondarySnapshot)) + } + return secondarySnapshot, nil +} + +// placementPolicyOwnerLabelVal returns the value for the PlacementResourceSnapshotOwnedByLabelKey label. +// +// If the placement policy's name does not exceed the length limit, the label value is simply the name itself. +// Otherwise, the name is truncated and appended with a hash suffix to ensure uniqueness. +func placementPolicyOwnerLabelVal(placementPolicy placementv1alpha1.PlacementPolicyAccessor) string { + name := placementPolicy.GetName() + if len(name) <= placementPolicyOwnerLabelValLenLimit { + return name + } + + hash := fmt.Sprintf("%x", sha256.Sum256([]byte(name)))[:placementPolicyOwnerLabelHashLen] + prefixLen := placementPolicyOwnerLabelValLenLimit - placementPolicyOwnerLabelHashLen - 1 + return fmt.Sprintf("%s-%s", name[:prefixLen], hash) +} diff --git a/pkg/v1/managers/placementresourcesnapshot/uniquename.go b/pkg/v1/managers/placementresourcesnapshot/uniquename.go new file mode 100644 index 000000000..ecd752b68 --- /dev/null +++ b/pkg/v1/managers/placementresourcesnapshot/uniquename.go @@ -0,0 +1,150 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package placementresourcesnapshot + +import ( + "crypto/sha256" + "fmt" + "strconv" + "strings" +) + +const ( + nameLenLimit = 251 + hashSegLen = 12 +) + +const ( + // The name format for primary placement resource snapshots. Typically, these snapshots are named using the + // format: + // + // `[PLACEMENT-POLICY-NAME]-resource-snapshot-[SNAPSHOT-INDEX]`, + // + // where `[PLACEMENT-POLICY-NAME]` is the name of the owner placement policy, and + // `[SNAPSHOT-INDEX]` is the monotonically increasing index of the snapshot. + // + // If the name becomes too long (> 251 characters) or contains dots, KubeFleet will drop the dots and truncate the + // placement policy name segment and the snapshot index segment as appropriate and add a hash suffix, i.e., + // + // `[PLACEMENT-POLICY-NAME-TRUNCATED]-resource-snapshot-[SNAPSHOT-INDEX-TRUNCATED]-[HASH]`, + // + // where `[HASH]` is the first few characters of the hash of the value + // `[PLACEMENT-POLICY-NAME]-resource-snapshot-[SNAPSHOT-INDEX]`. + PrimaryPlacementResourceSnapshotNameFmt = "%s-resource-snapshot-%d" + PrimaryPlacementResourceSnapshotNameWithHashFmt = "%s-resource-snapshot-%s-%s" + + // The name format for secondary placement resource snapshots. Typically, these snapshots are named using the + // format: + // + // `[PLACEMENT-POLICY-NAME]-resource-snapshot-[SNAPSHOT-INDEX]-[SNAPSHOT-SUB-INDEX]`, + // + // where `[PLACEMENT-POLICY-NAME]` is the name of the owner placement policy, + // `[SNAPSHOT-INDEX]` is the monotonically increasing index of the snapshot, and + // `[SNAPSHOT-SUB-INDEX]` is the monotonically increasing sub-index of the snapshot. + // + // If the name becomes too long (> 251 characters) or contains dots, KubeFleet will drop the dots and truncate the + // placement policy name segment and the `[SNAPSHOT-INDEX]-[SNAPSHOT-SUB-INDEX]` segment as appropriate and add + // a hash suffix, i.e., + // + // `[PLACEMENT-POLICY-NAME-TRUNCATED]-resource-snapshot-[INDEX-TRUNCATED]-[HASH]`, + // + // where `[HASH]` is the first few characters of the hash of the value + // `[PLACEMENT-POLICY-NAME]-resource-snapshot-[SNAPSHOT-INDEX]-[SNAPSHOT-SUB-INDEX]`. + SecondaryPlacementResourceSnapshotNameFmt = "%s-resource-snapshot-%d-%d" + SecondaryPlacementResourceSnapshotNameWithHashFmt = "%s-resource-snapshot-%s-%s" +) + +// uniqueNameForPrimaryPlacementResourceSnapshot generates a unique name for a primary placement resource snapshot. +func uniqueNameForPrimaryPlacementResourceSnapshot(placementPolicyName string, idx int) string { + name := fmt.Sprintf(PrimaryPlacementResourceSnapshotNameFmt, placementPolicyName, idx) + if len(name) <= nameLenLimit && !strings.Contains(name, ".") { + return name + } + + // The name is too long or contains dots; sanitize and truncate the placement policy name segment and append a hash suffix. + // The hash is computed over the full (untruncated) name. + // + // Note that here only the first few (12) characters are kept. This does lead to increased risk of name + // collisions, but the chances still remain extremely low. If such a collision does occur, manual intervention + // is needed for resolution. + hash := fmt.Sprintf("%x", sha256.Sum256([]byte(name)))[:hashSegLen] + + // Compute how many characters are left for the two variable segments (the placement policy name and the + // snapshot index); the index segment gets the space it needs, and the placement policy name segment takes + // whatever remains. + // + // reservedLen accounts for the static decoration and the hash suffix only. + // + // The offset 1 is the length of placeholder index (0). + reservedLen := len(fmt.Sprintf(PrimaryPlacementResourceSnapshotNameWithHashFmt, "", "0", hash)) - 1 + availableLen := nameLenLimit - reservedLen + availableLenForIdxSeg := 10 // The maximum number of digits for an int32 value. + availableLenForNameSeg := availableLen - availableLenForIdxSeg + + // Remove all dots from the placement policy name segment so that truncation cannot leave a trailing dot, + // which would produce an invalid DNS subdomain label. + truncatedPlacementPolicyName := strings.ReplaceAll(placementPolicyName, ".", "") + if len(truncatedPlacementPolicyName) > availableLenForNameSeg { + truncatedPlacementPolicyName = truncatedPlacementPolicyName[:availableLenForNameSeg] + } + + truncatedIdxStr := strconv.Itoa(idx) + if len(truncatedIdxStr) > availableLenForIdxSeg { + truncatedIdxStr = truncatedIdxStr[:availableLenForIdxSeg] + } + + return fmt.Sprintf(PrimaryPlacementResourceSnapshotNameWithHashFmt, truncatedPlacementPolicyName, truncatedIdxStr, hash) +} + +func uniqueNameForSecondaryPlacementResourceSnapshot(placementPolicyName string, idx int, subIdx int) string { + name := fmt.Sprintf(SecondaryPlacementResourceSnapshotNameFmt, placementPolicyName, idx, subIdx) + if len(name) <= nameLenLimit && !strings.Contains(name, ".") { + return name + } + + // The name is too long or contains dots; sanitize and truncate the placement policy name segment and append a hash suffix. + // The hash is computed over the full (untruncated) name. + // + // Note that here only the first few (12) characters are kept. This does lead to increased risk of name + // collisions, but the chances still remain extremely low. If such a collision does occur, manual intervention + // is needed for resolution. + hash := fmt.Sprintf("%x", sha256.Sum256([]byte(name)))[:hashSegLen] + + // Compute how many characters are left for the two variable segments (the placement policy name and the + // combined snapshot index/sub-index segment); the index segment gets the space it needs, and the placement + // policy name segment takes whatever remains. + // + // reservedLen accounts for the static decoration and the hash suffix only. + reservedLen := len(fmt.Sprintf(SecondaryPlacementResourceSnapshotNameWithHashFmt, "", "", hash)) + availableLen := nameLenLimit - reservedLen + availableLenForIdxSeg := 12 // The maximum number of characters for an int32 value, plus the room for a dash and a single-digit sub-index. + availableLenForNameSeg := availableLen - availableLenForIdxSeg + + // Remove all dots from the placement policy name segment so that truncation cannot leave a trailing dot, + // which would produce an invalid DNS subdomain label. + truncatedPlacementPolicyName := strings.ReplaceAll(placementPolicyName, ".", "") + if len(truncatedPlacementPolicyName) > availableLenForNameSeg { + truncatedPlacementPolicyName = truncatedPlacementPolicyName[:availableLenForNameSeg] + } + + truncatedIdxStr := fmt.Sprintf("%d-%d", idx, subIdx) + if len(truncatedIdxStr) > availableLenForIdxSeg { + truncatedIdxStr = truncatedIdxStr[:availableLenForIdxSeg] + } + + return fmt.Sprintf(SecondaryPlacementResourceSnapshotNameWithHashFmt, truncatedPlacementPolicyName, truncatedIdxStr, hash) +} diff --git a/pkg/v1/utils/bindingmanager/manager.go b/pkg/v1/utils/bindingmanager/manager.go new file mode 100644 index 000000000..a42879248 --- /dev/null +++ b/pkg/v1/utils/bindingmanager/manager.go @@ -0,0 +1,200 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +// Package bindingmanager provides utilities for managing the binding manager role for placement policies. +package bindingmanager + +import ( + "context" + "reflect" + + "k8s.io/klog/v2" + "sigs.k8s.io/controller-runtime/pkg/client" + + placementv1alpha1 "go.goms.io/fleet/apis/kubefleet.dev/placement/v1alpha1" + "go.goms.io/fleet/pkg/utils/errors" +) + +// ClaimRoleAs adds an object reference (under the management of a controller) as the binding manager for +// a placement policy. +// +// This function returns (true, nil), if the binding manager role has been successfully claimed; +// (false, nil) if another source currently holds the binding manager role; and (false, err) if an error occurs +// during the attempt. +func ClaimRoleAs( + ctx context.Context, + hubClient client.Client, + placementPolicy placementv1alpha1.PlacementPolicyAccessor, + controllerName string, + objectRef placementv1alpha1.ObjectReference, +) (bool, error) { + if placementPolicy == nil || reflect.ValueOf(placementPolicy).IsNil() { + return false, errors.NewUnexpectedError(nil, "the placement policy is nil") + } + if len(controllerName) == 0 { + return false, errors.NewUnexpectedError(nil, "no controller name is provided") + } + // The name, API version, and kind fields are required by the API definition. + if len(objectRef.Name) == 0 || len(objectRef.APIVersion) == 0 || len(objectRef.Kind) == 0 { + return false, errors.NewUnexpectedError(nil, "the object reference is incomplete") + } + + bindingManager := placementPolicy.GetStatus().BindingManager + bindingManagerCopy := bindingManager.DeepCopy() + if bindingManager == nil { + bindingManager = &placementv1alpha1.BindingManager{ + ControllerName: controllerName, + ObjectRefs: []placementv1alpha1.ObjectReference{ + objectRef, + }, + } + placementPolicy.GetStatus().BindingManager = bindingManager + if err := hubClient.Status().Update(ctx, placementPolicy); err != nil { + // Reset the binding manager to its previous state. + placementPolicy.GetStatus().BindingManager = bindingManagerCopy + return false, errors.NewAPIServerError(err, "failed to update placement policy status", false) + } + klog.V(2).InfoS("Successfully claimed the binding manager role", "placementPolicy", klog.KObj(placementPolicy), "controllerName", controllerName, "objectRef", objectRef) + return true, nil + } + + if bindingManager.ControllerName != controllerName { + klog.V(2).InfoS("The binding manager role has already been claimed by another controller; should retry later", + "placementPolicy", klog.KObj(placementPolicy), + "currentControllerName", bindingManager.ControllerName, + "applicantControllerName", controllerName, "applicantObjectRef", objectRef) + return false, nil + } + + found := false + for idx := range bindingManager.ObjectRefs { + if bindingManager.ObjectRefs[idx] == objectRef { + found = true + break + } + } + if found { + // The given object reference is already present in the binding manager claim. Verify if the view is still up-to-date. + if err := dryRunPatch(ctx, hubClient, placementPolicy); err != nil { + return false, errors.Wraps(err, "failed to verify state freshness (the given object reference is already present in the binding manager claim)") + } + klog.V(2).InfoS("The given object reference is already present in the binding manager claim; no further action is needed", + "placementPolicy", klog.KObj(placementPolicy), "controllerName", controllerName, "objectRef", objectRef) + return true, nil + } + + bindingManager.ObjectRefs = append(bindingManager.ObjectRefs, objectRef) + if err := hubClient.Status().Update(ctx, placementPolicy); err != nil { + // Reset the binding manager to its previous state. + placementPolicy.GetStatus().BindingManager = bindingManagerCopy + return false, errors.NewAPIServerError(err, "failed to update placement policy status", false) + } + klog.V(2).InfoS("Successfully added the object reference to the binding manager claim", + "placementPolicy", klog.KObj(placementPolicy), "controllerName", controllerName, "objectRef", objectRef) + return true, nil +} + +// RelinquishRoleFor releases the given object reference (under the management of a controller) from the +// binding manager role for a placement policy. +// +// If the passed-in object reference is the last entry in the binding manager claim, the claim is dropped altogether. +// +// An error will be returned if the current binding manager view is not up-to-date. If the given object +// reference is not currently holding the binding manager role, the function will return with no error. +func RelinquishRoleFor( + ctx context.Context, + hubClient client.Client, + placementPolicy placementv1alpha1.PlacementPolicyAccessor, + controllerName string, + objectRef placementv1alpha1.ObjectReference, +) error { + if placementPolicy == nil || reflect.ValueOf(placementPolicy).IsNil() { + return errors.NewUnexpectedError(nil, "the placement policy is nil") + } + if len(controllerName) == 0 { + return errors.NewUnexpectedError(nil, "no controller name is provided") + } + if len(objectRef.Name) == 0 || len(objectRef.APIVersion) == 0 || len(objectRef.Kind) == 0 { + return errors.NewUnexpectedError(nil, "the object reference is incomplete") + } + + bindingManager := placementPolicy.GetStatus().BindingManager + bindingManagerCopy := bindingManager.DeepCopy() + if bindingManager == nil || bindingManager.ControllerName != controllerName { + // The given object (or controller) no longer holds the binding manager role based on the current state. + // However, the current state might be stale; do a dry-run patch to verify its freshness. + if err := dryRunPatch(ctx, hubClient, placementPolicy); err != nil { + return errors.Wraps(err, "failed to verify state freshness (no binding manager role is held by the given controller)") + } + // The current state is up to date; no further action is needed. + klog.V(2).InfoS("No binding manager role is held by the given controller; relinquishing is not needed", + "placementPolicy", klog.KObj(placementPolicy), "controllerName", controllerName) + return nil + } + + found := false + updatedObjectRefs := make([]placementv1alpha1.ObjectReference, 0, len(bindingManager.ObjectRefs)) + for idx := range bindingManager.ObjectRefs { + if bindingManager.ObjectRefs[idx] == objectRef { + found = true + continue + } + updatedObjectRefs = append(updatedObjectRefs, bindingManager.ObjectRefs[idx]) + } + if !found { + // The given object (or controller) no longer holds the binding manager role based on the current state. However, the + // current state might be stale; do a dry-run patch to verify its freshness. + if err := dryRunPatch(ctx, hubClient, placementPolicy); err != nil { + return errors.Wraps(err, "failed to verify state freshness (the given object reference is not found in the binding manager claim)") + } + // The current state is up to date; no further action is needed. + klog.V(2).InfoS("The given object reference is not found in the binding manager claim; relinquishing is not needed", + "placementPolicy", klog.KObj(placementPolicy), "controllerName", controllerName, "objectRef", objectRef) + return nil + } + + // Remove the given object reference from the binding manager claim. If the list of object references becomes empty, + // remove the binding manager claim altogether. + bindingManager.ObjectRefs = updatedObjectRefs + if len(bindingManager.ObjectRefs) == 0 { + placementPolicy.GetStatus().BindingManager = nil + } + + if err := hubClient.Status().Update(ctx, placementPolicy); err != nil { + // Reset the binding manager to its previous state. + placementPolicy.GetStatus().BindingManager = bindingManagerCopy + return errors.NewAPIServerError(err, "failed to update placement policy status", false) + } + klog.V(2).InfoS("Relinquished the binding manager role from the object reference", "placementPolicy", klog.KObj(placementPolicy), "controllerName", controllerName, "objectRef", objectRef) + return nil +} + +// dryRunPatch verifies that the caller's view of the placement policy is still current by issuing a no-op status +// patch that carries the object's resource version; a stale view yields a conflict error. +func dryRunPatch(ctx context.Context, hubClient client.Client, placementPolicy placementv1alpha1.PlacementPolicyAccessor) error { + placementToPatch := placementPolicy.DeepCopyObject().(placementv1alpha1.PlacementPolicyAccessor) + if err := hubClient.Status().Patch( + ctx, + placementToPatch, + client.MergeFromWithOptions(placementPolicy, client.MergeFromWithOptimisticLock{}), + client.DryRunAll, + ); err != nil { + wrappedErr := errors.NewAPIServerError(err, + "failed to complete the dry-run: the current state might be stale, or an unexpected API server error has occurred", false) + return wrappedErr + } + return nil +} diff --git a/pkg/v1/utils/bindingmanager/manager_test.go b/pkg/v1/utils/bindingmanager/manager_test.go new file mode 100644 index 000000000..a07be3a36 --- /dev/null +++ b/pkg/v1/utils/bindingmanager/manager_test.go @@ -0,0 +1,676 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package bindingmanager + +import ( + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + "github.com/google/go-cmp/cmp" + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/types" + + placementv1alpha1 "go.goms.io/fleet/apis/kubefleet.dev/placement/v1alpha1" +) + +var _ = Describe("Claiming as the binding manager (ClusterPlacementPolicy)", func() { + Context("when the placement policy is nil", func() { + It("should return an error without claiming the binding manager role", func() { + claimed, err := ClaimRoleAs(ctx, hubClient, nil, "test-controller", placementv1alpha1.ObjectReference{}) + Expect(err).To(HaveOccurred()) + Expect(claimed).To(BeFalse()) + }) + + It("should return an error when a typed nil pointer is given", func() { + var policy *placementv1alpha1.ClusterPlacementPolicy + claimed, err := ClaimRoleAs(ctx, hubClient, policy, "test-controller", placementv1alpha1.ObjectReference{}) + Expect(err).To(HaveOccurred()) + Expect(claimed).To(BeFalse()) + }) + }) + + Context("when no controller name is provided", func() { + It("should return an error without claiming the binding manager role", func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{ + Name: "test-cluster-placement-policy", + }, + } + claimed, err := ClaimRoleAs(ctx, hubClient, policy, "", placementv1alpha1.ObjectReference{}) + Expect(err).To(HaveOccurred()) + Expect(claimed).To(BeFalse()) + }) + }) + + DescribeTable("when the object reference is incomplete", + func(objectRef placementv1alpha1.ObjectReference) { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Name: "test-cluster-placement-policy"}, + } + claimed, err := ClaimRoleAs(ctx, hubClient, policy, "test-controller", objectRef) + Expect(err).To(HaveOccurred()) + Expect(claimed).To(BeFalse()) + }, + Entry("no name", placementv1alpha1.ObjectReference{ + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + }), + Entry("no API version", placementv1alpha1.ObjectReference{ + Name: "test-object", + Kind: "DummyOwner", + }), + Entry("no kind", placementv1alpha1.ObjectReference{ + Name: "test-object", + APIVersion: placementv1alpha1.GroupVersion.Version, + }), + ) + + Context("when the binding manager role has not been claimed yet", Ordered, func() { + const ( + controllerName = "test-controller" + policyName = "fresh-claim" + ) + + objectRef := placementv1alpha1.ObjectReference{ + Name: "test-object", + APIGroup: placementv1alpha1.GroupVersion.Group, + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + } + + BeforeAll(func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Name: policyName}, + Spec: placementv1alpha1.PlacementPolicySpec{ + ResourceSelectors: []placementv1alpha1.ResourceSelector{ + {APIVersion: "v1", Kind: "Namespace", Name: "test-namespace"}, + }, + }, + } + Expect(hubClient.Create(ctx, policy)).To(Succeed()) + }) + + AfterAll(func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Name: policyName}, + } + Expect(hubClient.Delete(ctx, policy)).To(Succeed()) + }) + + It("should claim the role and record the object reference", func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, policy)).To(Succeed()) + + claimed, err := ClaimRoleAs(ctx, hubClient, policy, controllerName, objectRef) + Expect(err).ToNot(HaveOccurred()) + Expect(claimed).To(BeTrue()) + + updated := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, updated)).To(Succeed()) + want := &placementv1alpha1.BindingManager{ + ControllerName: controllerName, + ObjectRefs: []placementv1alpha1.ObjectReference{objectRef}, + } + Expect(cmp.Diff(updated.Status.BindingManager, want)).To(BeEmpty()) + }) + }) + + Context("when the binding manager role has been claimed by another controller", Ordered, func() { + const ( + controllerName = "test-controller" + otherControllerName = "other-controller" + policyName = "claimed-by-another-controller" + ) + + objectRef := placementv1alpha1.ObjectReference{ + Name: "test-object", + APIGroup: placementv1alpha1.GroupVersion.Group, + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + } + otherObjectRef := placementv1alpha1.ObjectReference{ + Name: "other-object", + APIGroup: placementv1alpha1.GroupVersion.Group, + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + } + wantBindingManager := &placementv1alpha1.BindingManager{ + ControllerName: otherControllerName, + ObjectRefs: []placementv1alpha1.ObjectReference{otherObjectRef}, + } + + BeforeAll(func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Name: policyName}, + Spec: placementv1alpha1.PlacementPolicySpec{ + ResourceSelectors: []placementv1alpha1.ResourceSelector{ + {APIVersion: "v1", Kind: "Namespace", Name: "test-namespace"}, + }, + }, + } + Expect(hubClient.Create(ctx, policy)).To(Succeed()) + policy.Status.BindingManager = wantBindingManager.DeepCopy() + Expect(hubClient.Status().Update(ctx, policy)).To(Succeed()) + }) + + AfterAll(func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Name: policyName}, + } + Expect(hubClient.Delete(ctx, policy)).To(Succeed()) + }) + + It("should not claim the role and should leave the existing claim untouched", func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, policy)).To(Succeed()) + + claimed, err := ClaimRoleAs(ctx, hubClient, policy, controllerName, objectRef) + Expect(err).ToNot(HaveOccurred()) + Expect(claimed).To(BeFalse()) + + updated := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, updated)).To(Succeed()) + Expect(cmp.Diff(updated.Status.BindingManager, wantBindingManager)).To(BeEmpty()) + }) + }) + + Context("when the binding manager role has already been claimed by the same controller", Ordered, func() { + const ( + controllerName = "test-controller" + policyName = "claimed-by-same-controller" + ) + + existingRef := placementv1alpha1.ObjectReference{ + Name: "existing-object", + APIGroup: placementv1alpha1.GroupVersion.Group, + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + } + newRef := placementv1alpha1.ObjectReference{ + Name: "new-object", + APIGroup: placementv1alpha1.GroupVersion.Group, + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + } + + BeforeAll(func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Name: policyName}, + Spec: placementv1alpha1.PlacementPolicySpec{ + ResourceSelectors: []placementv1alpha1.ResourceSelector{ + {APIVersion: "v1", Kind: "Namespace", Name: "test-namespace"}, + }, + }, + } + Expect(hubClient.Create(ctx, policy)).To(Succeed()) + policy.Status.BindingManager = &placementv1alpha1.BindingManager{ + ControllerName: controllerName, + ObjectRefs: []placementv1alpha1.ObjectReference{existingRef}, + } + Expect(hubClient.Status().Update(ctx, policy)).To(Succeed()) + }) + + AfterAll(func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Name: policyName}, + } + Expect(hubClient.Delete(ctx, policy)).To(Succeed()) + }) + + It("should be a no-op when the object reference is already present", func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, policy)).To(Succeed()) + + claimed, err := ClaimRoleAs(ctx, hubClient, policy, controllerName, existingRef) + Expect(err).ToNot(HaveOccurred()) + Expect(claimed).To(BeTrue()) + + updated := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, updated)).To(Succeed()) + want := &placementv1alpha1.BindingManager{ + ControllerName: controllerName, + ObjectRefs: []placementv1alpha1.ObjectReference{existingRef}, + } + Expect(cmp.Diff(updated.Status.BindingManager, want)).To(BeEmpty()) + }) + + It("should append a new object reference to the existing claim", func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, policy)).To(Succeed()) + + claimed, err := ClaimRoleAs(ctx, hubClient, policy, controllerName, newRef) + Expect(err).ToNot(HaveOccurred()) + Expect(claimed).To(BeTrue()) + + updated := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, updated)).To(Succeed()) + want := &placementv1alpha1.BindingManager{ + ControllerName: controllerName, + ObjectRefs: []placementv1alpha1.ObjectReference{existingRef, newRef}, + } + Expect(cmp.Diff(updated.Status.BindingManager, want)).To(BeEmpty()) + }) + }) + + Context("when the view of the placement policy is stale (dry-run verification)", Ordered, func() { + const ( + controllerName = "test-controller" + otherControllerName = "other-controller" + policyName = "stale-view-on-claim" + ) + + objectRef := placementv1alpha1.ObjectReference{ + Name: "existing-object", + APIGroup: placementv1alpha1.GroupVersion.Group, + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + } + // The claim written out of band, behind the back of the stale copy. + wantBindingManager := &placementv1alpha1.BindingManager{ + ControllerName: otherControllerName, + ObjectRefs: []placementv1alpha1.ObjectReference{ + { + Name: "other-object", + APIGroup: placementv1alpha1.GroupVersion.Group, + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + }, + }, + } + + // A copy of the policy that is read before the out-of-band claim below. + var stalePolicy *placementv1alpha1.ClusterPlacementPolicy + + BeforeAll(func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Name: policyName}, + Spec: placementv1alpha1.PlacementPolicySpec{ + ResourceSelectors: []placementv1alpha1.ResourceSelector{ + {APIVersion: "v1", Kind: "Namespace", Name: "test-namespace"}, + }, + }, + } + Expect(hubClient.Create(ctx, policy)).To(Succeed()) + // The stale copy must carry a matching claim; otherwise the dry-run branches are never reached. + policy.Status.BindingManager = &placementv1alpha1.BindingManager{ + ControllerName: controllerName, + ObjectRefs: []placementv1alpha1.ObjectReference{objectRef}, + } + Expect(hubClient.Status().Update(ctx, policy)).To(Succeed()) + + stalePolicy = &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, stalePolicy)).To(Succeed()) + + latestPolicy := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, latestPolicy)).To(Succeed()) + latestPolicy.Status.BindingManager = wantBindingManager.DeepCopy() + Expect(hubClient.Status().Update(ctx, latestPolicy)).To(Succeed()) + }) + + AfterAll(func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Name: policyName}, + } + Expect(hubClient.Delete(ctx, policy)).To(Succeed()) + }) + + It("should report a conflict when the object reference is already present", func() { + claimed, err := ClaimRoleAs(ctx, hubClient, stalePolicy, controllerName, objectRef) + Expect(apierrors.IsConflict(err)).To(BeTrue(), "ClaimRoleAs() = %v, want a conflict error", err) + Expect(claimed).To(BeFalse()) + }) + + It("should leave the persisted binding manager claim untouched", func() { + updated := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, updated)).To(Succeed()) + Expect(cmp.Diff(updated.Status.BindingManager, wantBindingManager)).To(BeEmpty()) + }) + }) +}) + +var _ = Describe("Claiming as the binding manager (PlacementPolicy)", func() { + Context("when the binding manager role has not been claimed yet", Ordered, func() { + const ( + controllerName = "test-controller" + policyName = "ns-fresh-claim" + ) + + objectRef := placementv1alpha1.ObjectReference{ + Namespace: playgroundNamespace, + Name: "test-object", + APIGroup: placementv1alpha1.GroupVersion.Group, + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + } + + BeforeAll(func() { + policy := &placementv1alpha1.PlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Namespace: playgroundNamespace, Name: policyName}, + Spec: placementv1alpha1.PlacementPolicySpec{ + ResourceSelectors: []placementv1alpha1.ResourceSelector{ + {APIVersion: "v1", Kind: "ConfigMap", Name: "test-config-map"}, + }, + }, + } + Expect(hubClient.Create(ctx, policy)).To(Succeed()) + }) + + AfterAll(func() { + policy := &placementv1alpha1.PlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Namespace: playgroundNamespace, Name: policyName}, + } + Expect(hubClient.Delete(ctx, policy)).To(Succeed()) + }) + + It("should claim the role and record the object reference", func() { + policy := &placementv1alpha1.PlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Namespace: playgroundNamespace, Name: policyName}, policy)).To(Succeed()) + + claimed, err := ClaimRoleAs(ctx, hubClient, policy, controllerName, objectRef) + Expect(err).ToNot(HaveOccurred()) + Expect(claimed).To(BeTrue()) + + updated := &placementv1alpha1.PlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Namespace: playgroundNamespace, Name: policyName}, updated)).To(Succeed()) + want := &placementv1alpha1.BindingManager{ + ControllerName: controllerName, + ObjectRefs: []placementv1alpha1.ObjectReference{objectRef}, + } + Expect(cmp.Diff(updated.Status.BindingManager, want)).To(BeEmpty()) + }) + }) +}) + +var _ = Describe("Relinquishing the binding manager role (ClusterPlacementPolicy)", func() { + Context("when the placement policy is nil", func() { + It("should return an error", func() { + err := RelinquishRoleFor(ctx, hubClient, nil, "test-controller", placementv1alpha1.ObjectReference{}) + Expect(err).To(HaveOccurred()) + }) + + It("should return an error when a typed nil pointer is given", func() { + var policy *placementv1alpha1.ClusterPlacementPolicy + err := RelinquishRoleFor(ctx, hubClient, policy, "test-controller", placementv1alpha1.ObjectReference{}) + Expect(err).To(HaveOccurred()) + }) + }) + + Context("when no controller name is provided", func() { + It("should return an error", func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Name: "relinquish-no-controller-name"}, + } + err := RelinquishRoleFor(ctx, hubClient, policy, "", placementv1alpha1.ObjectReference{}) + Expect(err).To(HaveOccurred()) + }) + }) + + DescribeTable("when the object reference is incomplete", + func(objectRef placementv1alpha1.ObjectReference) { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Name: "relinquish-incomplete-object-ref"}, + } + err := RelinquishRoleFor(ctx, hubClient, policy, "test-controller", objectRef) + Expect(err).To(HaveOccurred()) + }, + Entry("no name", placementv1alpha1.ObjectReference{ + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + }), + Entry("no API version", placementv1alpha1.ObjectReference{ + Name: "test-object", + Kind: "DummyOwner", + }), + Entry("no kind", placementv1alpha1.ObjectReference{ + Name: "test-object", + APIVersion: placementv1alpha1.GroupVersion.Version, + }), + ) + + Context("when relinquishing an object reference", Ordered, func() { + const ( + controllerName = "test-controller" + policyName = "relinquish-object-ref" + ) + + refA := placementv1alpha1.ObjectReference{ + Name: "object-a", + APIGroup: placementv1alpha1.GroupVersion.Group, + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + } + refB := placementv1alpha1.ObjectReference{ + Name: "object-b", + APIGroup: placementv1alpha1.GroupVersion.Group, + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + } + + BeforeAll(func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Name: policyName}, + Spec: placementv1alpha1.PlacementPolicySpec{ + ResourceSelectors: []placementv1alpha1.ResourceSelector{ + {APIVersion: "v1", Kind: "Namespace", Name: "test-namespace"}, + }, + }, + } + Expect(hubClient.Create(ctx, policy)).To(Succeed()) + policy.Status.BindingManager = &placementv1alpha1.BindingManager{ + ControllerName: controllerName, + ObjectRefs: []placementv1alpha1.ObjectReference{refA, refB}, + } + Expect(hubClient.Status().Update(ctx, policy)).To(Succeed()) + }) + + AfterAll(func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Name: policyName}, + } + Expect(hubClient.Delete(ctx, policy)).To(Succeed()) + }) + + It("should remove the object reference while keeping the remaining ones", func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, policy)).To(Succeed()) + + err := RelinquishRoleFor(ctx, hubClient, policy, controllerName, refA) + Expect(err).ToNot(HaveOccurred()) + + updated := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, updated)).To(Succeed()) + want := &placementv1alpha1.BindingManager{ + ControllerName: controllerName, + ObjectRefs: []placementv1alpha1.ObjectReference{refB}, + } + Expect(cmp.Diff(updated.Status.BindingManager, want)).To(BeEmpty()) + }) + + It("should remove the binding manager claim when the last object reference is relinquished", func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, policy)).To(Succeed()) + + err := RelinquishRoleFor(ctx, hubClient, policy, controllerName, refB) + Expect(err).ToNot(HaveOccurred()) + + updated := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, updated)).To(Succeed()) + Expect(updated.Status.BindingManager).To(BeNil()) + }) + }) + + Context("when the view of the placement policy is stale (dry-run verification)", Ordered, func() { + const ( + controllerName = "test-controller" + otherControllerName = "other-controller" + policyName = "stale-view-on-relinquish" + ) + + objectRef := placementv1alpha1.ObjectReference{ + Name: "existing-object", + APIGroup: placementv1alpha1.GroupVersion.Group, + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + } + unknownRef := placementv1alpha1.ObjectReference{ + Name: "unknown-object", + APIGroup: placementv1alpha1.GroupVersion.Group, + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + } + // The claim written out of band, behind the back of the stale copy. + wantBindingManager := &placementv1alpha1.BindingManager{ + ControllerName: otherControllerName, + ObjectRefs: []placementv1alpha1.ObjectReference{ + { + Name: "other-object", + APIGroup: placementv1alpha1.GroupVersion.Group, + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + }, + }, + } + + // A copy of the policy that is read before the out-of-band claim below. + var stalePolicy *placementv1alpha1.ClusterPlacementPolicy + + BeforeAll(func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Name: policyName}, + Spec: placementv1alpha1.PlacementPolicySpec{ + ResourceSelectors: []placementv1alpha1.ResourceSelector{ + {APIVersion: "v1", Kind: "Namespace", Name: "test-namespace"}, + }, + }, + } + Expect(hubClient.Create(ctx, policy)).To(Succeed()) + policy.Status.BindingManager = &placementv1alpha1.BindingManager{ + ControllerName: controllerName, + ObjectRefs: []placementv1alpha1.ObjectReference{objectRef}, + } + Expect(hubClient.Status().Update(ctx, policy)).To(Succeed()) + + stalePolicy = &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, stalePolicy)).To(Succeed()) + + latestPolicy := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, latestPolicy)).To(Succeed()) + latestPolicy.Status.BindingManager = wantBindingManager.DeepCopy() + Expect(hubClient.Status().Update(ctx, latestPolicy)).To(Succeed()) + }) + + AfterAll(func() { + policy := &placementv1alpha1.ClusterPlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Name: policyName}, + } + Expect(hubClient.Delete(ctx, policy)).To(Succeed()) + }) + + It("should report a conflict when the given controller does not hold the role", func() { + err := RelinquishRoleFor(ctx, hubClient, stalePolicy, "unknown-controller", objectRef) + Expect(apierrors.IsConflict(err)).To(BeTrue(), "RelinquishRoleFor() = %v, want a conflict error", err) + }) + + It("should report a conflict when the object reference is not found", func() { + err := RelinquishRoleFor(ctx, hubClient, stalePolicy, controllerName, unknownRef) + Expect(apierrors.IsConflict(err)).To(BeTrue(), "RelinquishRoleFor() = %v, want a conflict error", err) + }) + + It("should leave the persisted binding manager claim untouched", func() { + updated := &placementv1alpha1.ClusterPlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Name: policyName}, updated)).To(Succeed()) + Expect(cmp.Diff(updated.Status.BindingManager, wantBindingManager)).To(BeEmpty()) + }) + }) +}) + +var _ = Describe("Relinquishing the binding manager role (PlacementPolicy)", func() { + Context("when relinquishing an object reference", Ordered, func() { + const ( + controllerName = "test-controller" + policyName = "ns-relinquish-object-ref" + ) + + refA := placementv1alpha1.ObjectReference{ + Namespace: playgroundNamespace, + Name: "object-a", + APIGroup: placementv1alpha1.GroupVersion.Group, + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + } + refB := placementv1alpha1.ObjectReference{ + Namespace: playgroundNamespace, + Name: "object-b", + APIGroup: placementv1alpha1.GroupVersion.Group, + APIVersion: placementv1alpha1.GroupVersion.Version, + Kind: "DummyOwner", + } + + BeforeAll(func() { + policy := &placementv1alpha1.PlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Namespace: playgroundNamespace, Name: policyName}, + Spec: placementv1alpha1.PlacementPolicySpec{ + ResourceSelectors: []placementv1alpha1.ResourceSelector{ + {APIVersion: "v1", Kind: "ConfigMap", Name: "test-config-map"}, + }, + }, + } + Expect(hubClient.Create(ctx, policy)).To(Succeed()) + policy.Status.BindingManager = &placementv1alpha1.BindingManager{ + ControllerName: controllerName, + ObjectRefs: []placementv1alpha1.ObjectReference{refA, refB}, + } + Expect(hubClient.Status().Update(ctx, policy)).To(Succeed()) + }) + + AfterAll(func() { + policy := &placementv1alpha1.PlacementPolicy{ + ObjectMeta: metav1.ObjectMeta{Namespace: playgroundNamespace, Name: policyName}, + } + Expect(hubClient.Delete(ctx, policy)).To(Succeed()) + }) + + It("should remove the object reference while keeping the remaining ones", func() { + policy := &placementv1alpha1.PlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Namespace: playgroundNamespace, Name: policyName}, policy)).To(Succeed()) + + err := RelinquishRoleFor(ctx, hubClient, policy, controllerName, refA) + Expect(err).ToNot(HaveOccurred()) + + updated := &placementv1alpha1.PlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Namespace: playgroundNamespace, Name: policyName}, updated)).To(Succeed()) + want := &placementv1alpha1.BindingManager{ + ControllerName: controllerName, + ObjectRefs: []placementv1alpha1.ObjectReference{refB}, + } + Expect(cmp.Diff(updated.Status.BindingManager, want)).To(BeEmpty()) + }) + + It("should remove the binding manager claim when the last object reference is relinquished", func() { + policy := &placementv1alpha1.PlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Namespace: playgroundNamespace, Name: policyName}, policy)).To(Succeed()) + + err := RelinquishRoleFor(ctx, hubClient, policy, controllerName, refB) + Expect(err).ToNot(HaveOccurred()) + + updated := &placementv1alpha1.PlacementPolicy{} + Expect(hubClient.Get(ctx, types.NamespacedName{Namespace: playgroundNamespace, Name: policyName}, updated)).To(Succeed()) + Expect(updated.Status.BindingManager).To(BeNil()) + }) + }) +}) diff --git a/pkg/v1/utils/bindingmanager/suite_test.go b/pkg/v1/utils/bindingmanager/suite_test.go new file mode 100644 index 000000000..c39757d69 --- /dev/null +++ b/pkg/v1/utils/bindingmanager/suite_test.go @@ -0,0 +1,102 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package bindingmanager + +import ( + "context" + "flag" + "path/filepath" + "testing" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/client-go/kubernetes/scheme" + "k8s.io/client-go/rest" + "k8s.io/klog/v2" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/envtest" + "sigs.k8s.io/controller-runtime/pkg/log/zap" + + placementv1alpha1 "go.goms.io/fleet/apis/kubefleet.dev/placement/v1alpha1" +) + +// Note (chenyu1): this package uses envtest-based environment for testing purposes as some of the ops in the logic, +// specifically the dry-run ops, require interaction with a real API server. + +var ( + hubCfg *rest.Config + hubEnv *envtest.Environment + hubClient client.Client + + ctx context.Context + cancel context.CancelFunc + + playgroundNamespace = "playground" +) + +func TestAPIs(t *testing.T) { + RegisterFailHandler(Fail) + + RunSpecs(t, "Binding Manager Integration Test Suite") +} + +var _ = BeforeSuite(func() { + ctx, cancel = context.WithCancel(context.TODO()) + + By("Setup klog") + fs := flag.NewFlagSet("klog", flag.ContinueOnError) + klog.InitFlags(fs) + Expect(fs.Parse([]string{"--v", "5", "-add_dir_header", "true"})).Should(Succeed()) + + logger := zap.New(zap.WriteTo(GinkgoWriter), zap.UseDevMode(true)) + klog.SetLogger(logger) + ctrl.SetLogger(logger) + + By("Bootstrapping test environment") + hubEnv = &envtest.Environment{ + CRDDirectoryPaths: []string{filepath.Join("../../../../", "config", "crd", "bases")}, + } + + var err error + hubCfg, err = hubEnv.Start() + Expect(err).ToNot(HaveOccurred()) + Expect(hubCfg).ToNot(BeNil()) + + By("Setting up the scheme") + Expect(placementv1alpha1.AddToScheme(scheme.Scheme)).To(Succeed()) + + By("Building the K8s client") + hubClient, err = client.New(hubCfg, client.Options{Scheme: scheme.Scheme}) + Expect(err).ToNot(HaveOccurred()) + Expect(hubClient).ToNot(BeNil()) + + By("Creating the test namespace") + Expect(hubClient.Create(ctx, &corev1.Namespace{ + ObjectMeta: metav1.ObjectMeta{Name: playgroundNamespace}, + })).To(Succeed()) +}) + +var _ = AfterSuite(func() { + defer klog.Flush() + + cancel() + By("Tearing down the test environment") + Expect(hubEnv.Stop()).To(Succeed()) +}) diff --git a/pkg/v1/utils/fieldindexers/hub.go b/pkg/v1/utils/fieldindexers/hub.go new file mode 100644 index 000000000..2e651cddb --- /dev/null +++ b/pkg/v1/utils/fieldindexers/hub.go @@ -0,0 +1,149 @@ +/* +Copyright 2026 The KubeFleet Authors. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +*/ + +package fieldindexers + +import ( + "context" + "fmt" + + "k8s.io/klog/v2" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + + placementv1alpha1 "go.goms.io/fleet/apis/kubefleet.dev/placement/v1alpha1" + "go.goms.io/fleet/pkg/utils/errors" +) + +const ( + // The field-based indexes set up for KubeFleet API objects. + // + // Important: many KubeFleet components run under the assumption that proper custom fields + // have been added and indexed in the cache when running. Failure to complete such prior setup **before + // the manager starts** will result in unexpected behaviors. Make sure that all applicable components + // are properly set up using the client provided by the hub controller manager, and `SetupWithManager` is + // called before the manager starts. + + // PlacementResourceSnapshotOwnedByAndSubIndexedCustomFieldName is the name of the custom field that indexes + // placement resource snapshots by their owner placement policies and their sub-indices. + // + // This is added to help the placement resource snapshot manager retrieve all primary placement resource + // snapshots (i.e., those with a sub-index of 0) associated with a placement policy. + PlacementResourceSnapshotOwnedByAndSubIndexedCustomFieldName = "ownedByWithSubIndex" + + // PlacementResourceSnapshotOwnedByAndIndexedCustomFieldName is the name of the custom field that indexes + // placement resource snapshots by their owner placement policies and their indices. + // + // This is added to help the placement resource snapshot manager retrieve all placement resource snapshots of a + // specific index associated with a placement policy. + PlacementResourceSnapshotOwnedByAndIndexedCustomFieldName = "ownedByWithIndex" +) + +const ( + // The format of the custom field values for the field-based indexes defined above. + + // PlacementResourceSnapshotOwnedByAndSubIndexedCustomFieldValFmt is used to format the value for the custom field, + // `PlacementResourceSnapshotOwnedByAndSubIndexedCustomFieldName`. + // + // Note that slashes are used to avoid unexpected collisions. + PlacementResourceSnapshotOwnedByAndSubIndexedCustomFieldValFmt = "%s/%s" + + // PlacementResourceSnapshotOwnedByAndIndexedCustomFieldValFmt is used to format the value for the custom field, + // `PlacementResourceSnapshotOwnedByAndIndexedCustomFieldName`. + // + // Note that slashes are used to avoid unexpected collisions. + PlacementResourceSnapshotOwnedByAndIndexedCustomFieldValFmt = "%s/%s" +) + +type fieldValueExtractor func(obj client.Object) ([]string, error) + +func indexCompositeField(ctx context.Context, + fieldIdxer client.FieldIndexer, + obj client.Object, + fieldName string, fieldValueExt fieldValueExtractor) error { + if err := fieldIdxer.IndexField(ctx, obj, fieldName, func(rawObj client.Object) []string { + fieldVals, extErr := fieldValueExt(rawObj) + if extErr != nil { + wrappedErr := errors.NewUnexpectedError(extErr, "failed to extract field value", "object", klog.KObj(rawObj)) + klog.ErrorS(wrappedErr, "failed to index field", errors.Args(wrappedErr)...) + return nil + } + return fieldVals + }); err != nil { + wrappedErr := errors.NewUnexpectedError(err, "", "fieldName", fieldName, "object", klog.KObj(obj)) + klog.ErrorS(wrappedErr, "failed to index field", errors.Args(wrappedErr)...) + return wrappedErr + } + return nil +} + +var ( + placementResourceSnapshotOwnedByAndSubIdxedFieldExtractor fieldValueExtractor = func(obj client.Object) ([]string, error) { + ownedBy := obj.GetLabels()[placementv1alpha1.PlacementResourceSnapshotOwnedByLabelKey] + subIndex := obj.GetLabels()[placementv1alpha1.PlacementResourceSnapshotSubIndexLabelKey] + if ownedBy == "" || subIndex == "" { + wrappedErr := errors.NewUnexpectedError(nil, "placement resource snapshot is missing required labels") + return nil, wrappedErr + } + return []string{fmt.Sprintf(PlacementResourceSnapshotOwnedByAndSubIndexedCustomFieldValFmt, ownedBy, subIndex)}, nil + } + + placementResourceSnapshotOwnedByAndIdxedFieldExtractor fieldValueExtractor = func(obj client.Object) ([]string, error) { + ownedBy := obj.GetLabels()[placementv1alpha1.PlacementResourceSnapshotOwnedByLabelKey] + index := obj.GetLabels()[placementv1alpha1.PlacementResourceSnapshotIndexLabelKey] + if ownedBy == "" || index == "" { + wrappedErr := errors.NewUnexpectedError(nil, "placement resource snapshot is missing required labels") + return nil, wrappedErr + } + return []string{fmt.Sprintf(PlacementResourceSnapshotOwnedByAndIndexedCustomFieldValFmt, ownedBy, index)}, nil + } +) + +// SetupWithHubControllerManager sets up the indices that controllers from the KubeFleet hub agent need to run properly. +// It must be called before the manager starts. +func SetupWithHubControllerManager(ctx context.Context, mgr ctrl.Manager) error { + fieldIdxer := mgr.GetFieldIndexer() + + if err := indexCompositeField(ctx, fieldIdxer, + &placementv1alpha1.PlacementResourceSnapshot{}, + PlacementResourceSnapshotOwnedByAndSubIndexedCustomFieldName, placementResourceSnapshotOwnedByAndSubIdxedFieldExtractor, + ); err != nil { + return errors.Wraps(err, "failed to set up placement resource snapshot owner and sub-index field index") + } + + if err := indexCompositeField(ctx, fieldIdxer, + &placementv1alpha1.PlacementResourceSnapshot{}, + PlacementResourceSnapshotOwnedByAndIndexedCustomFieldName, placementResourceSnapshotOwnedByAndIdxedFieldExtractor, + ); err != nil { + return errors.Wraps(err, "failed to set up placement resource snapshot owner and index field index") + } + + if err := indexCompositeField(ctx, fieldIdxer, + &placementv1alpha1.ClusterPlacementResourceSnapshot{}, + PlacementResourceSnapshotOwnedByAndSubIndexedCustomFieldName, placementResourceSnapshotOwnedByAndSubIdxedFieldExtractor, + ); err != nil { + return errors.Wraps(err, "failed to set up cluster placement resource snapshot owner and sub-index field index") + } + + if err := indexCompositeField(ctx, fieldIdxer, + &placementv1alpha1.ClusterPlacementResourceSnapshot{}, + PlacementResourceSnapshotOwnedByAndIndexedCustomFieldName, placementResourceSnapshotOwnedByAndIdxedFieldExtractor, + ); err != nil { + return errors.Wraps(err, "failed to set up cluster placement resource snapshot owner and index field index") + } + + return nil +} diff --git a/pkg/webhook/clusterresourceplacement/v1beta1_clusterresourceplacement_mutating_webhook_test.go b/pkg/webhook/clusterresourceplacement/v1beta1_clusterresourceplacement_mutating_webhook_test.go index 4f6f1e363..a988a23af 100644 --- a/pkg/webhook/clusterresourceplacement/v1beta1_clusterresourceplacement_mutating_webhook_test.go +++ b/pkg/webhook/clusterresourceplacement/v1beta1_clusterresourceplacement_mutating_webhook_test.go @@ -19,13 +19,13 @@ package clusterresourceplacement import ( "context" "encoding/json" + "strconv" "testing" "gomodules.xyz/jsonpatch/v2" "github.com/google/go-cmp/cmp" "github.com/google/go-cmp/cmp/cmpopts" - "github.com/stretchr/testify/assert" admissionv1 "k8s.io/api/admission/v1" authenticationv1 "k8s.io/api/authentication/v1" corev1 "k8s.io/api/core/v1" @@ -349,7 +349,9 @@ func TestMutatingHandle(t *testing.T) { crpUpdateAllFieldsNewBytes, _ := json.Marshal(crpUpdateAllFieldsNew) scheme := runtime.NewScheme() - assert.Nil(t, placementv1beta1.AddToScheme(scheme)) + if err := placementv1beta1.AddToScheme(scheme); err != nil { + t.Fatalf("AddToScheme() = %v, want nil", err) + } decoder := admission.NewDecoder(scheme) mutator := &clusterResourcePlacementMutator{decoder: decoder} @@ -378,7 +380,7 @@ func TestMutatingHandle(t *testing.T) { { Operation: "add", Path: "/spec/revisionHistoryLimit", - Value: float64(defaulter.DefaultRevisionHistoryLimitValue), + Value: json.Number(strconv.Itoa(defaulter.DefaultRevisionHistoryLimitValue)), }, }, AdmissionResponse: admissionv1.AdmissionResponse{ @@ -450,7 +452,7 @@ func TestMutatingHandle(t *testing.T) { Value: map[string]any{ "maxSurge": defaulter.DefaultMaxSurgeValue, "maxUnavailable": defaulter.DefaultMaxUnavailableValue, - "unavailablePeriodSeconds": float64(defaulter.DefaultUnavailablePeriodSeconds), + "unavailablePeriodSeconds": json.Number(strconv.Itoa(defaulter.DefaultUnavailablePeriodSeconds)), }, }, { @@ -529,7 +531,7 @@ func TestMutatingHandle(t *testing.T) { Value: map[string]any{ "maxSurge": defaulter.DefaultMaxSurgeValue, "maxUnavailable": defaulter.DefaultMaxUnavailableValue, - "unavailablePeriodSeconds": float64(defaulter.DefaultUnavailablePeriodSeconds), + "unavailablePeriodSeconds": json.Number(strconv.Itoa(defaulter.DefaultUnavailablePeriodSeconds)), }, }, }, @@ -688,7 +690,7 @@ func TestMutatingHandle(t *testing.T) { Value: map[string]any{ "maxSurge": defaulter.DefaultMaxSurgeValue, "maxUnavailable": defaulter.DefaultMaxUnavailableValue, - "unavailablePeriodSeconds": float64(defaulter.DefaultUnavailablePeriodSeconds), + "unavailablePeriodSeconds": json.Number(strconv.Itoa(defaulter.DefaultUnavailablePeriodSeconds)), }, }, { @@ -704,7 +706,7 @@ func TestMutatingHandle(t *testing.T) { { Operation: "add", Path: "/spec/revisionHistoryLimit", - Value: float64(defaulter.DefaultRevisionHistoryLimitValue), + Value: json.Number(strconv.Itoa(defaulter.DefaultRevisionHistoryLimitValue)), }, }, }, diff --git a/pkg/webhook/membercluster/membercluster_validating_webhook.go b/pkg/webhook/membercluster/membercluster_validating_webhook.go index 589ea3851..c6ed59fd5 100644 --- a/pkg/webhook/membercluster/membercluster_validating_webhook.go +++ b/pkg/webhook/membercluster/membercluster_validating_webhook.go @@ -4,6 +4,7 @@ import ( "context" "fmt" "net/http" + "strings" admissionv1 "k8s.io/api/admission/v1" "k8s.io/apimachinery/pkg/types" @@ -14,6 +15,7 @@ import ( "sigs.k8s.io/controller-runtime/pkg/webhook/admission" clusterv1beta1 "go.goms.io/fleet/apis/cluster/v1beta1" + placementv1beta1 "go.goms.io/fleet/apis/placement/v1beta1" "go.goms.io/fleet/pkg/utils" "go.goms.io/fleet/pkg/utils/validator" @@ -87,5 +89,63 @@ func (v *memberClusterValidator) Handle(ctx context.Context, req admission.Reque klog.V(2).ErrorS(err, "Member cluster has invalid fields, request is denied", "operation", req.Operation, "memberCluster", mcObjectName) return admission.Denied(err.Error()) } - return admission.Allowed("Member cluster has valid fields") + + response := admission.Allowed("Member cluster has valid fields") + if warning := v.clusterAliasCollisionWarning(ctx, &mc); warning != "" { + response = response.WithWarnings(warning) + } + return response +} + +// clusterAliasCollisionWarning returns a warning message if another member cluster already carries +// the alias this one is being labelled with, or the empty string otherwise. +// +// The alias selects a cluster by a name of the admin's choosing, so it is meant to identify one +// cluster; two clusters sharing an alias makes an alias-based selector match both. It is only a +// warning, never a denial: labelling a replacement cluster with the outgoing one's alias before +// removing it from the outgoing one is exactly the handoff the alias exists to allow, and that +// handoff passes through a state where two clusters share the alias. For the same reason a failure +// to list the member clusters does not block the request -- an advisory check must not stand +// between an admin and the cluster they are registering. +func (v *memberClusterValidator) clusterAliasCollisionWarning(ctx context.Context, mc *clusterv1beta1.MemberCluster) string { + alias, ok := mc.Labels[placementv1beta1.ClusterAliasLabel] + if !ok || alias == "" { + return "" + } + + // The list is served from the manager's cache, so it costs no API call, and the label selector + // is applied before the cache copies anything: only the clusters that actually hold the alias + // are copied out, rather than the whole inventory on every member cluster admission. + memberClusterList := &clusterv1beta1.MemberClusterList{} + if err := v.client.List(ctx, memberClusterList, client.MatchingLabels{placementv1beta1.ClusterAliasLabel: alias}); err != nil { + klog.V(2).ErrorS(err, "Failed to list member clusters for the alias uniqueness check; admitting without a warning", "memberCluster", klog.KObj(mc)) + return "" + } + + // The message names at most a few holders: it is read on a terminal, and the actionable half is + // the alias value and that someone else holds it, not an exhaustive roll call. Only those few + // names are kept, while the count runs over all of them, so that a fleet where every cluster + // carries the alias is reported accurately without collecting a name per cluster. + const maxNamedHolders = 3 + named := make([]string, 0, maxNamedHolders+1) + total := 0 + for i := range memberClusterList.Items { + other := &memberClusterList.Items[i] + if other.Name == mc.Name { + // The cluster under admission holds the alias by definition; only another holder is + // a collision. + continue + } + total++ + if len(named) < maxNamedHolders { + named = append(named, other.Name) + } + } + if total == 0 { + return "" + } + if total > maxNamedHolders { + named = append(named, fmt.Sprintf("and %d more", total-maxNamedHolders)) + } + return fmt.Sprintf("cluster alias %q is already used by %s; an alias-based cluster selector will match more than one cluster while this is the case", alias, strings.Join(named, ", ")) } diff --git a/pkg/webhook/membercluster/membercluster_validating_webhook_test.go b/pkg/webhook/membercluster/membercluster_validating_webhook_test.go index 3cec42679..c7d9537b0 100644 --- a/pkg/webhook/membercluster/membercluster_validating_webhook_test.go +++ b/pkg/webhook/membercluster/membercluster_validating_webhook_test.go @@ -13,12 +13,14 @@ import ( "k8s.io/apimachinery/pkg/types" clusterv1beta1 "go.goms.io/fleet/apis/cluster/v1beta1" + placementv1beta1 "go.goms.io/fleet/apis/placement/v1beta1" "go.goms.io/fleet/pkg/utils" fleetnetworkingv1alpha1 "go.goms.io/fleet-networking/api/v1alpha1" "sigs.k8s.io/controller-runtime/pkg/client" "sigs.k8s.io/controller-runtime/pkg/client/fake" + "sigs.k8s.io/controller-runtime/pkg/client/interceptor" "sigs.k8s.io/controller-runtime/pkg/webhook/admission" ) @@ -139,3 +141,173 @@ func newInternalServiceExport(clusterID, namespace string) *fleetnetworkingv1alp }, } } + +func buildCreateRequestFromObject(t *testing.T, mc *clusterv1beta1.MemberCluster) admission.Request { + t.Helper() + + raw, err := json.Marshal(mc) + if err != nil { + t.Fatalf("failed to marshal member cluster: %v", err) + } + return admission.Request{ + AdmissionRequest: admissionv1.AdmissionRequest{ + Operation: admissionv1.Create, + Name: mc.Name, + Object: runtime.RawExtension{Raw: raw}, + }, + } +} + +func memberClusterWithAlias(name, alias string) *clusterv1beta1.MemberCluster { + mc := &clusterv1beta1.MemberCluster{ObjectMeta: metav1.ObjectMeta{Name: name}} + if alias != "" { + mc.Labels = map[string]string{placementv1beta1.ClusterAliasLabel: alias} + } + return mc +} + +// TestHandleClusterAliasCollision covers the alias uniqueness warning: a member cluster whose alias +// is already held by another is admitted with a warning, never denied, and the cases that must stay +// silent (no alias, a unique alias, and the alias's own holder) produce none. +func TestHandleClusterAliasCollision(t *testing.T) { + t.Parallel() + + testCases := map[string]struct { + incoming *clusterv1beta1.MemberCluster + wantWarning bool + }{ + "no alias label is silent": { + incoming: memberClusterWithAlias("cluster-two", ""), + wantWarning: false, + }, + "a unique alias is silent": { + incoming: memberClusterWithAlias("cluster-two", "api-primary"), + wantWarning: false, + }, + "an alias already held by another cluster warns": { + incoming: memberClusterWithAlias("cluster-two", "web-primary"), + wantWarning: true, + }, + "the alias's own holder does not warn about itself": { + incoming: memberClusterWithAlias("cluster-one", "web-primary"), + wantWarning: false, + }, + } + + for name, tc := range testCases { + tc := tc + t.Run(name, func(t *testing.T) { + t.Parallel() + + // A fresh object per subtest: the fake client stamps a resourceVersion onto the objects + // it is built with, so a shared pointer would be written concurrently under -race. + existing := memberClusterWithAlias("cluster-one", "web-primary") + validator := newMemberClusterValidatorForTest(t, false, existing) + resp := validator.Handle(context.Background(), buildCreateRequestFromObject(t, tc.incoming)) + + if !resp.Allowed { + t.Fatalf("Handle() = denied, want allowed regardless of alias collision: %+v", resp.Result) + } + if gotWarning := len(resp.Warnings) > 0; gotWarning != tc.wantWarning { + t.Errorf("Handle() produced a warning = %v (%v), want %v", gotWarning, resp.Warnings, tc.wantWarning) + } + }) + } +} + +// TestClusterAliasCollisionWarningTruncatesHolders covers the many-holders path: the message names +// at most three holders and summarizes the rest, with the surplus counted over every holder rather +// than over the names kept. +func TestClusterAliasCollisionWarningTruncatesHolders(t *testing.T) { + t.Parallel() + + testCases := []struct { + name string + holders int + want string + }{ + {name: "one holder is named on its own", holders: 1, want: "holder-0"}, + {name: "holders up to the cap are all named", holders: 3, want: "holder-0, holder-1, holder-2"}, + {name: "one holder past the cap is summarized", holders: 4, want: "holder-0, holder-1, holder-2, and 1 more"}, + {name: "many holders are counted in full", holders: 6, want: "holder-0, holder-1, holder-2, and 3 more"}, + } + + for _, tc := range testCases { + t.Run(tc.name, func(t *testing.T) { + t.Parallel() + + scheme := runtime.NewScheme() + if err := clusterv1beta1.AddToScheme(scheme); err != nil { + t.Fatalf("failed to add member cluster scheme: %v", err) + } + seed := make([]client.Object, 0, tc.holders+2) + for i := 0; i < tc.holders; i++ { + seed = append(seed, memberClusterWithAlias(fmt.Sprintf("holder-%d", i), "web-primary")) + } + // Decoys: the count must come from the alias, not from the fleet size. These fail the + // selector, so dropping it from the List would show up here as an inflated count. + seed = append(seed, memberClusterWithAlias("other-alias", "db-primary"), memberClusterWithAlias("no-alias", "")) + c := fake.NewClientBuilder().WithScheme(scheme).WithObjects(seed...).Build() + v := &memberClusterValidator{client: c, decoder: admission.NewDecoder(scheme)} + + got := v.clusterAliasCollisionWarning(context.Background(), memberClusterWithAlias("newcomer", "web-primary")) + // The holder list is pinned up to the semicolon that ends it, so that a case naming + // fewer holders than the cap also asserts that no surplus summary was appended. + if want := fmt.Sprintf("used by %s;", tc.want); !strings.Contains(got, want) { + t.Errorf("clusterAliasCollisionWarning() = %q, want it to name the holders as %q", got, want) + } + }) + } +} + +// TestClusterAliasCollisionWarningEmptyAlias covers the guard on the alias value: a member cluster +// carrying the alias label with an explicit empty value is not an alias at all, and must not be +// matched against every other cluster whose alias is likewise empty. +func TestClusterAliasCollisionWarningEmptyAlias(t *testing.T) { + t.Parallel() + + scheme := runtime.NewScheme() + if err := clusterv1beta1.AddToScheme(scheme); err != nil { + t.Fatalf("failed to add member cluster scheme: %v", err) + } + emptyAliasCluster := func(name string) *clusterv1beta1.MemberCluster { + return &clusterv1beta1.MemberCluster{ + ObjectMeta: metav1.ObjectMeta{ + Name: name, + Labels: map[string]string{placementv1beta1.ClusterAliasLabel: ""}, + }, + } + } + c := fake.NewClientBuilder().WithScheme(scheme).WithObjects(emptyAliasCluster("other")).Build() + v := &memberClusterValidator{client: c, decoder: admission.NewDecoder(scheme)} + + if got := v.clusterAliasCollisionWarning(context.Background(), emptyAliasCluster("newcomer")); got != "" { + t.Errorf("clusterAliasCollisionWarning() = %q, want no warning for an empty alias value", got) + } +} + +// TestClusterAliasCollisionWarningListError covers the fail-open path: a member cluster list that +// errors must admit the request without a warning rather than block it, since the check is advisory. +func TestClusterAliasCollisionWarningListError(t *testing.T) { + t.Parallel() + + scheme := runtime.NewScheme() + if err := clusterv1beta1.AddToScheme(scheme); err != nil { + t.Fatalf("failed to add member cluster scheme: %v", err) + } + // A real collision exists in the store, so a successful list WOULD warn; only the injected list + // error can produce the empty result this asserts, which is what makes it a fail-open test + // rather than a trivially-no-collisions one. + failingClient := fake.NewClientBuilder().WithScheme(scheme). + WithObjects(memberClusterWithAlias("cluster-one", "web-primary")). + WithInterceptorFuncs(interceptor.Funcs{ + List: func(context.Context, client.WithWatch, client.ObjectList, ...client.ListOption) error { + return fmt.Errorf("the member cluster list is unwell") + }, + }).Build() + v := &memberClusterValidator{client: failingClient, decoder: admission.NewDecoder(scheme)} + + if got := v.clusterAliasCollisionWarning(context.Background(), memberClusterWithAlias("cluster-two", "web-primary")); got != "" { + t.Errorf("clusterAliasCollisionWarning() = %q, want empty on a list error (fail open)", got) + } +} diff --git a/pkg/webhook/validation/uservalidation.go b/pkg/webhook/validation/uservalidation.go index 49b12c303..3002570b9 100644 --- a/pkg/webhook/validation/uservalidation.go +++ b/pkg/webhook/validation/uservalidation.go @@ -6,6 +6,7 @@ import ( "encoding/json" "fmt" "reflect" + "slices" "strings" authenticationv1 "k8s.io/api/authentication/v1" @@ -13,7 +14,6 @@ import ( metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/types" "k8s.io/klog/v2" - "k8s.io/utils/strings/slices" clusterinventory "sigs.k8s.io/cluster-inventory-api/apis/v1alpha1" "sigs.k8s.io/controller-runtime/pkg/client" "sigs.k8s.io/controller-runtime/pkg/webhook/admission" @@ -94,8 +94,9 @@ func ValidateFleetMemberClusterUpdate(currentMC, oldMC clusterv1beta1.MemberClus } isLabelUpdated := isMapFieldUpdated(currentMC.GetLabels(), oldMC.GetLabels()) - if isLabelUpdated && !isUserInGroup(userInfo, mastersGroup) && shouldDenyLabelModification(currentMC.GetLabels(), oldMC.GetLabels(), denyModifyMemberClusterLabels) { - // allow any user to modify kubernetes-fleet.io/* labels, but restricts other label modifications given denyModifyMemberClusterLabels is true. + if isLabelUpdated && !isUserInGroup(userInfo, mastersGroup) && shouldDenyLabelModification(currentMC.GetLabels(), oldMC.GetLabels(), denyModifyMemberClusterLabels, isUserAuthenticatedServiceAccount(userInfo)) { + // allow any user to modify kubernetes-fleet.io/* labels and service accounts to modify kubefleet.dev/* labels, + // but restricts other label modifications given denyModifyMemberClusterLabels is true. klog.V(2).InfoS(DeniedModifyMemberClusterLabels, "user", userInfo.Username, "groups", userInfo.Groups, "operation", req.Operation, "GVK", req.RequestKind, "subResource", req.SubResource, "namespacedName", namespacedName) return admission.Denied(DeniedModifyMemberClusterLabels) } @@ -160,22 +161,29 @@ func isUserInGroup(userInfo authenticationv1.UserInfo, groupName string) bool { return slices.Contains(userInfo.Groups, groupName) } -// shouldDenyLabelModification returns true if any labels (besides kubernetes-fleet.io/* labels) are being modified and denyModifyMemberClusterLabels is true. -func shouldDenyLabelModification(currentLabels, oldLabels map[string]string, denyModifyMemberClusterLabels bool) bool { +// shouldDenyLabelModification returns true if any labels besides the ones fleet reserves are being +// modified and denyModifyMemberClusterLabels is true. kubernetes-fleet.io/* labels are exempt for +// every user; kubefleet.dev/* labels only for service accounts, so that the hub agent (which is not +// in system:masters) can seed the cluster alias while a plain user under the guard cannot move an +// alias, which is a scheduling label, from one cluster to another. +func shouldDenyLabelModification(currentLabels, oldLabels map[string]string, denyModifyMemberClusterLabels, isServiceAccount bool) bool { if !denyModifyMemberClusterLabels { return false } + exempt := func(k string) bool { + return strings.HasPrefix(k, placementv1beta1.FleetPrefix) || (isServiceAccount && strings.HasPrefix(k, placementv1beta1.KubeFleetPrefix)) + } for k, v := range currentLabels { oldV, exists := oldLabels[k] if !exists || oldV != v { - if !strings.HasPrefix(k, placementv1beta1.FleetPrefix) { + if !exempt(k) { return true } } } for k := range oldLabels { if _, exists := currentLabels[k]; !exists { - if !strings.HasPrefix(k, placementv1beta1.FleetPrefix) { + if !exempt(k) { return true } } diff --git a/pkg/webhook/validation/uservalidation_test.go b/pkg/webhook/validation/uservalidation_test.go index aedb57cf7..b9a912710 100644 --- a/pkg/webhook/validation/uservalidation_test.go +++ b/pkg/webhook/validation/uservalidation_test.go @@ -396,6 +396,74 @@ func TestValidateFleetMemberClusterUpdate(t *testing.T) { wantResponse: admission.Allowed(fmt.Sprintf(ResourceAllowedFormat, "nonSystemMastersUser", utils.GenerateGroupString([]string{"someGroup"}), admissionv1.Update, &utils.MCMetaGVK, "", types.NamespacedName{Name: "test-mc"})), }, + // The hub agent seeds kubefleet.dev/cluster-alias and is not in system:masters, so the + // kubefleet.dev/ prefix must pass this guard for service accounts. + "allow label creation by service accounts for kubefleet.dev/* labels": { + denyModifyMemberClusterLabels: true, + oldMC: &clusterv1beta1.MemberCluster{ + ObjectMeta: metav1.ObjectMeta{ + Name: "test-mc", + Annotations: map[string]string{ + "fleet.azure.com/cluster-resource-id": "test-cluster-resource-id", + }, + }, + }, + newMC: &clusterv1beta1.MemberCluster{ + ObjectMeta: metav1.ObjectMeta{ + Name: "test-mc", + Labels: map[string]string{"kubefleet.dev/cluster-alias": "test-mc"}, + Annotations: map[string]string{ + "fleet.azure.com/cluster-resource-id": "test-cluster-resource-id", + }, + }, + }, + req: admission.Request{ + AdmissionRequest: admissionv1.AdmissionRequest{ + Name: "test-mc", + UserInfo: authenticationv1.UserInfo{ + Username: "system:serviceaccount:fleet-system:hub-agent-sa", + Groups: []string{"system:serviceaccounts"}, + }, + RequestKind: &utils.MCMetaGVK, + Operation: admissionv1.Update, + }, + }, + wantResponse: admission.Allowed(fmt.Sprintf(ResourceAllowedFormat, "system:serviceaccount:fleet-system:hub-agent-sa", utils.GenerateGroupString([]string{"system:serviceaccounts"}), + admissionv1.Update, &utils.MCMetaGVK, "", types.NamespacedName{Name: "test-mc"})), + }, + "deny label modification by non-system:masters user for kubefleet.dev/* labels": { + denyModifyMemberClusterLabels: true, + oldMC: &clusterv1beta1.MemberCluster{ + ObjectMeta: metav1.ObjectMeta{ + Name: "test-mc", + Labels: map[string]string{"kubefleet.dev/cluster-alias": "test-mc"}, + Annotations: map[string]string{ + "fleet.azure.com/cluster-resource-id": "test-cluster-resource-id", + }, + }, + }, + newMC: &clusterv1beta1.MemberCluster{ + ObjectMeta: metav1.ObjectMeta{ + Name: "test-mc", + Labels: map[string]string{"kubefleet.dev/cluster-alias": "prod-primary"}, + Annotations: map[string]string{ + "fleet.azure.com/cluster-resource-id": "test-cluster-resource-id", + }, + }, + }, + req: admission.Request{ + AdmissionRequest: admissionv1.AdmissionRequest{ + Name: "test-mc", + UserInfo: authenticationv1.UserInfo{ + Username: "nonSystemMastersUser", + Groups: []string{"someGroup"}, + }, + RequestKind: &utils.MCMetaGVK, + Operation: admissionv1.Update, + }, + }, + wantResponse: admission.Denied(DeniedModifyMemberClusterLabels), + }, "allow label deletion by any user for kubernetes-fleet.io/* labels": { denyModifyMemberClusterLabels: true, oldMC: &clusterv1beta1.MemberCluster{ diff --git a/test/apis/placement/v1beta1/api_validation_integration_test.go b/test/apis/placement/v1beta1/api_validation_integration_test.go index 04df99f65..9e8c1f143 100644 --- a/test/apis/placement/v1beta1/api_validation_integration_test.go +++ b/test/apis/placement/v1beta1/api_validation_integration_test.go @@ -30,6 +30,7 @@ import ( metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" "k8s.io/apimachinery/pkg/util/intstr" + "k8s.io/utils/ptr" "sigs.k8s.io/controller-runtime/pkg/client" placementv1beta1 "go.goms.io/fleet/apis/placement/v1beta1" @@ -564,6 +565,112 @@ var _ = Describe("Test placement v1beta1 API validation", func() { Expect(errors.As(err, &statusErr)).To(BeTrue(), "The returned error is not a StatusError") Expect(statusErr.Status().Message).Should(ContainSubstring("operator must be Exists when key is empty")) }) + + // The rolling update bounds are an int-or-string whose pattern constrains the string form + // only; the CEL rules are what keep the integer form non-negative, so these cases exercise + // the integer form specifically, with the string entries pinning that the pattern still + // owns its side. + DescribeTable("the rolling update bounds of a ClusterResourcePlacement", + func(mutate func(*placementv1beta1.RollingUpdateConfig), wantMessage string) { + crpName := fmt.Sprintf(crpNameTemplate, GinkgoParallelProcess()) + rollingUpdate := &placementv1beta1.RollingUpdateConfig{} + mutate(rollingUpdate) + crp := &placementv1beta1.ClusterResourcePlacement{ + ObjectMeta: metav1.ObjectMeta{ + Name: crpName, + }, + Spec: placementv1beta1.PlacementSpec{ + ResourceSelectors: []placementv1beta1.ResourceSelectorTerm{ + { + Group: "", + Version: "v1", + Kind: "Namespace", + Name: nonExistentNSName, + }, + }, + Strategy: placementv1beta1.RolloutStrategy{ + Type: placementv1beta1.RollingUpdateRolloutStrategyType, + RollingUpdate: rollingUpdate, + }, + }, + } + + err := hubClient.Create(ctx, crp) + if wantMessage == "" { + Expect(err).To(Succeed(), "Expected the CRP to be accepted") + return + } + Expect(err).To(HaveOccurred(), "Expected error when creating CRP with an out-of-range rolling update bound") + var statusErr *k8sErrors.StatusError + Expect(errors.As(err, &statusErr)).To(BeTrue(), "The returned error is not a StatusError") + Expect(statusErr.Status().Message).Should(ContainSubstring(wantMessage)) + }, + Entry("maxUnavailable 0 as an integer is accepted", func(c *placementv1beta1.RollingUpdateConfig) { + c.MaxUnavailable = ptr.To(intstr.FromInt32(0)) + }, ""), + Entry("maxUnavailable 5 as an integer is accepted", func(c *placementv1beta1.RollingUpdateConfig) { + c.MaxUnavailable = ptr.To(intstr.FromInt32(5)) + }, ""), + Entry("maxSurge 0 as an integer is accepted", func(c *placementv1beta1.RollingUpdateConfig) { + c.MaxSurge = ptr.To(intstr.FromInt32(0)) + }, ""), + Entry("maxUnavailable 25% is accepted", func(c *placementv1beta1.RollingUpdateConfig) { + c.MaxUnavailable = ptr.To(intstr.FromString("25%")) + }, ""), + Entry("maxUnavailable 100% is accepted", func(c *placementv1beta1.RollingUpdateConfig) { + c.MaxUnavailable = ptr.To(intstr.FromString("100%")) + }, ""), + Entry("maxUnavailable -1 as an integer is rejected", func(c *placementv1beta1.RollingUpdateConfig) { + c.MaxUnavailable = ptr.To(intstr.FromInt32(-1)) + }, "maxUnavailable must be a non-negative integer or a percentage"), + Entry("maxSurge -1 as an integer is rejected", func(c *placementv1beta1.RollingUpdateConfig) { + c.MaxSurge = ptr.To(intstr.FromInt32(-1)) + }, "maxSurge must be a non-negative integer or a percentage"), + Entry("maxUnavailable -1 as a string is rejected by the pattern", func(c *placementv1beta1.RollingUpdateConfig) { + c.MaxUnavailable = ptr.To(intstr.FromString("-1")) + }, "spec.strategy.rollingUpdate.maxUnavailable in body should match"), + Entry("maxUnavailable 101% is rejected by the pattern", func(c *placementv1beta1.RollingUpdateConfig) { + c.MaxUnavailable = ptr.To(intstr.FromString("101%")) + }, "spec.strategy.rollingUpdate.maxUnavailable in body should match"), + // The digit branch of the pattern is bounded to nine digits, all of which fit an int32 + // comfortably; an unbounded digit string used to pass validation only to fail in + // whatever later consumed it. + Entry("maxUnavailable with nine digits is accepted", func(c *placementv1beta1.RollingUpdateConfig) { + c.MaxUnavailable = ptr.To(intstr.FromString("999999999")) + }, ""), + Entry("maxUnavailable with ten digits is rejected by the pattern", func(c *placementv1beta1.RollingUpdateConfig) { + c.MaxUnavailable = ptr.To(intstr.FromString("9999999999")) + }, "spec.strategy.rollingUpdate.maxUnavailable in body should match"), + ) + + It("does not re-litigate the rolling update bounds on an unrelated update", func() { + crpName := fmt.Sprintf(crpNameTemplate, GinkgoParallelProcess()) + crp := &placementv1beta1.ClusterResourcePlacement{ + ObjectMeta: metav1.ObjectMeta{ + Name: crpName, + }, + Spec: placementv1beta1.PlacementSpec{ + ResourceSelectors: []placementv1beta1.ResourceSelectorTerm{ + { + Group: "", + Version: "v1", + Kind: "Namespace", + Name: nonExistentNSName, + }, + }, + Strategy: placementv1beta1.RolloutStrategy{ + Type: placementv1beta1.RollingUpdateRolloutStrategyType, + RollingUpdate: &placementv1beta1.RollingUpdateConfig{ + MaxUnavailable: ptr.To(intstr.FromString("25%")), + }, + }, + }, + } + Expect(hubClient.Create(ctx, crp)).To(Succeed()) + + crp.Spec.RevisionHistoryLimit = ptr.To(int32(5)) + Expect(hubClient.Update(ctx, crp)).To(Succeed(), "Expected an update leaving the bounds untouched to pass their validation") + }) }) Context("Test ClusterResourcePlacement API validation - invalid update cases", func() { @@ -1826,6 +1933,79 @@ var _ = Describe("Test placement v1beta1 API validation", func() { }) }) + Context("Test ResourcePlacement rolling update bounds", func() { + rpNamespace := "default" + + AfterEach(func() { + rpName := fmt.Sprintf(rpNameTemplate, GinkgoParallelProcess()) + Eventually(func() error { + rp := &placementv1beta1.ResourcePlacement{ + ObjectMeta: metav1.ObjectMeta{ + Name: rpName, + Namespace: rpNamespace, + }, + } + if err := hubClient.Delete(ctx, rp); err != nil && !k8sErrors.IsNotFound(err) { + return fmt.Errorf("failed to delete RP: %w", err) + } + if err := hubClient.Get(ctx, client.ObjectKey{Name: rpName, Namespace: rpNamespace}, &placementv1beta1.ResourcePlacement{}); !k8sErrors.IsNotFound(err) { + return fmt.Errorf("RP still exists after deletion attempt (error: %w)", err) + } + return nil + }, eventuallyDuration, eventuallyInterval).Should(Succeed()) + }) + + // ResourcePlacement shares RollingUpdateConfig with ClusterResourcePlacement, but unlike + // the cluster-scoped placement it has no validating webhook behind it, so the CRD schema + // is the only line of defense here. + DescribeTable("the rolling update bounds of a ResourcePlacement", + func(mutate func(*placementv1beta1.RollingUpdateConfig), wantMessage string) { + rpName := fmt.Sprintf(rpNameTemplate, GinkgoParallelProcess()) + rollingUpdate := &placementv1beta1.RollingUpdateConfig{} + mutate(rollingUpdate) + rp := &placementv1beta1.ResourcePlacement{ + ObjectMeta: metav1.ObjectMeta{ + Name: rpName, + Namespace: rpNamespace, + }, + Spec: placementv1beta1.PlacementSpec{ + ResourceSelectors: []placementv1beta1.ResourceSelectorTerm{ + { + Group: "", + Version: "v1", + Kind: "ConfigMap", + Name: "app", + }, + }, + Strategy: placementv1beta1.RolloutStrategy{ + Type: placementv1beta1.RollingUpdateRolloutStrategyType, + RollingUpdate: rollingUpdate, + }, + }, + } + + err := hubClient.Create(ctx, rp) + if wantMessage == "" { + Expect(err).To(Succeed(), "Expected the RP to be accepted") + return + } + Expect(err).To(HaveOccurred(), "Expected error when creating RP with an out-of-range rolling update bound") + var statusErr *k8sErrors.StatusError + Expect(errors.As(err, &statusErr)).To(BeTrue(), "The returned error is not a StatusError") + Expect(statusErr.Status().Message).Should(ContainSubstring(wantMessage)) + }, + Entry("maxUnavailable 0 as an integer is accepted", func(c *placementv1beta1.RollingUpdateConfig) { + c.MaxUnavailable = ptr.To(intstr.FromInt32(0)) + }, ""), + Entry("maxUnavailable -1 as an integer is rejected", func(c *placementv1beta1.RollingUpdateConfig) { + c.MaxUnavailable = ptr.To(intstr.FromInt32(-1)) + }, "maxUnavailable must be a non-negative integer or a percentage"), + Entry("maxSurge -1 as an integer is rejected", func(c *placementv1beta1.RollingUpdateConfig) { + c.MaxSurge = ptr.To(intstr.FromInt32(-1)) + }, "maxSurge must be a non-negative integer or a percentage"), + ) + }) + Context("Test ResourcePlacement API validation - invalid update cases", func() { var rp placementv1beta1.ResourcePlacement rpName := fmt.Sprintf(rpNameTemplate, GinkgoParallelProcess()) diff --git a/test/apis/v1alpha1/zz_generated.deepcopy.go b/test/apis/v1alpha1/zz_generated.deepcopy.go index 143bdee7b..081bec913 100644 --- a/test/apis/v1alpha1/zz_generated.deepcopy.go +++ b/test/apis/v1alpha1/zz_generated.deepcopy.go @@ -21,7 +21,7 @@ limitations under the License. package v1alpha1 import ( - v1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1" runtime "k8s.io/apimachinery/pkg/runtime" ) diff --git a/test/e2e/api_progression_test.go b/test/e2e/api_progression_test.go index 8e87f8e6e..ae6dbf319 100644 --- a/test/e2e/api_progression_test.go +++ b/test/e2e/api_progression_test.go @@ -18,6 +18,7 @@ package e2e import ( "fmt" + "time" "github.com/google/go-cmp/cmp" "github.com/google/go-cmp/cmp/cmpopts" @@ -34,6 +35,9 @@ import ( placementv1beta1 "go.goms.io/fleet/apis/placement/v1beta1" "go.goms.io/fleet/pkg/controllers/workapplier" "go.goms.io/fleet/pkg/utils" + "go.goms.io/fleet/pkg/utils/condition" + "go.goms.io/fleet/test/e2e/framework" + testutilseviction "go.goms.io/fleet/test/utils/eviction" ) var ( @@ -49,8 +53,19 @@ var ( ignorePlacementStatusDiffedPlacementsTimestampFieldsV1, cmpopts.EquateEmpty(), } + + placementStatusCmpOptionsOnCreateV1 = append( + cmp.Options{ + ignorePlacementStatusObservedResourceIndexFieldV1, + ignorePerClusterPlacementStatusObservedResourceIndexFieldV1, + }, + placementStatusCmpOptionsV1..., + ) ) +// The helpers below are v1 API counterparts of the shared (v1beta1) E2E utilities; they read and +// write exclusively through the v1 API so that the API progression specs never fall back to v1beta1. + func ensureCRPRemovalV1(crpName string) { Eventually(func() error { crp := &placementv1.ClusterResourcePlacement{ @@ -69,6 +84,235 @@ func ensureCRPRemovalV1(crpName string) { }, eventuallyDuration, eventuallyInterval).Should(Succeed(), "Failed to wait for CRP deletion") } +func retrievePlacementV1(placementKey types.NamespacedName) (placementv1.PlacementObj, error) { + var placement placementv1.PlacementObj + if placementKey.Namespace == "" { + placement = &placementv1.ClusterResourcePlacement{} + } else { + placement = &placementv1.ResourcePlacement{} + } + if err := hubClient.Get(ctx, placementKey, placement); err != nil { + return nil, err + } + return placement, nil +} + +func placementRemovedActualV1(placementKey types.NamespacedName) func() error { + return func() error { + if _, err := retrievePlacementV1(placementKey); !errors.IsNotFound(err) { + return fmt.Errorf("placement %s still exists or an unexpected error occurred: %w", placementKey, err) + } + return nil + } +} + +func allFinalizersExceptForCustomDeletionBlockerRemovedFromPlacementActualV1(placementKey types.NamespacedName) func() error { + return func() error { + placement, err := retrievePlacementV1(placementKey) + if err != nil { + if errors.IsNotFound(err) { + return nil + } + return err + } + + wantFinalizers := []string{customDeletionBlockerFinalizer} + if diff := cmp.Diff(placement.GetFinalizers(), wantFinalizers); diff != "" { + return fmt.Errorf("placement finalizers diff (-got, +want): %s", diff) + } + return nil + } +} + +func crpEvictionRemovedActualV1(crpEvictionName string) func() error { + return func() error { + if err := hubClient.Get(ctx, types.NamespacedName{Name: crpEvictionName}, &placementv1.ClusterResourcePlacementEviction{}); !errors.IsNotFound(err) { + return fmt.Errorf("CRP eviction still exists or an unexpected error occurred: %w", err) + } + return nil + } +} + +func crpDisruptionBudgetRemovedActualV1(crpDisruptionBudgetName string) func() error { + return func() error { + if err := hubClient.Get(ctx, types.NamespacedName{Name: crpDisruptionBudgetName}, &placementv1.ClusterResourcePlacementDisruptionBudget{}); !errors.IsNotFound(err) { + return fmt.Errorf("CRP disruption budget still exists or an unexpected error occurred: %w", err) + } + return nil + } +} + +func cleanupPlacementV1(placementKey types.NamespacedName) { + Eventually(func() error { + placement, err := retrievePlacementV1(placementKey) + if errors.IsNotFound(err) { + return nil + } + if err != nil { + return err + } + + // Delete the placement (again, if applicable); this helps the After All node to run + // successfully even if the steps above fail early. + if err := hubClient.Delete(ctx, placement); err != nil { + return err + } + + placement.SetFinalizers([]string{}) + return hubClient.Update(ctx, placement) + }, eventuallyDuration, eventuallyInterval).Should(Succeed(), "Failed to delete placement %s", placementKey) + + Eventually(placementRemovedActualV1(placementKey), workloadEventuallyDuration, eventuallyInterval).Should(Succeed(), "Failed to remove placement %s", placementKey) + + // Wait for the Work objects to be deleted as well; leftover Work objects (which are kept + // around by a finalizer until all applied resources are gone) may lead to resource overlaps + // and flakiness in subsequent specs. + By("Check if work is deleted") + workName := fmt.Sprintf("%s-work", placementKey.Name) + if placementKey.Namespace != "" { + workName = fmt.Sprintf("%s.%s", placementKey.Namespace, workName) + } + Eventually(func() error { + for idx := range allMemberClusterNames { + workNS := fmt.Sprintf(utils.NamespaceNameFormat, allMemberClusterNames[idx]) + if err := hubClient.Get(ctx, types.NamespacedName{Name: workName, Namespace: workNS}, &placementv1.Work{}); !errors.IsNotFound(err) { + return fmt.Errorf("work object %s/%s still exists or an unexpected error occurred: %w", workNS, workName, err) + } + } + return nil + }, workloadEventuallyDuration, eventuallyInterval).Should(Succeed(), "Failed to remove work objects derived from placement %s", placementKey) +} + +func ensureCRPEvictionDeletedV1(crpEvictionName string) { + crpe := &placementv1.ClusterResourcePlacementEviction{ + ObjectMeta: metav1.ObjectMeta{ + Name: crpEvictionName, + }, + } + Expect(hubClient.Delete(ctx, crpe)).Should(SatisfyAny(Succeed(), utils.NotFoundMatcher{}), "Failed to delete CRP eviction") + Eventually(crpEvictionRemovedActualV1(crpEvictionName), eventuallyDuration, eventuallyInterval).Should(Succeed(), "CRP eviction still exists") +} + +func ensureCRPDisruptionBudgetDeletedV1(crpDisruptionBudgetName string) { + crpdb := &placementv1.ClusterResourcePlacementDisruptionBudget{ + ObjectMeta: metav1.ObjectMeta{ + Name: crpDisruptionBudgetName, + }, + } + Expect(hubClient.Delete(ctx, crpdb)).Should(SatisfyAny(Succeed(), utils.NotFoundMatcher{}), "Failed to delete CRP disruption budget") + Eventually(crpDisruptionBudgetRemovedActualV1(crpDisruptionBudgetName), eventuallyDuration, eventuallyInterval).Should(Succeed(), "CRP disruption budget still exists") +} + +func ensureCRPAndRelatedResourcesDeletedV1(crpName string, memberClusters []*framework.Cluster) { + crp := &placementv1.ClusterResourcePlacement{ + ObjectMeta: metav1.ObjectMeta{ + Name: crpName, + }, + } + Expect(hubClient.Delete(ctx, crp)).Should(SatisfyAny(Succeed(), utils.NotFoundMatcher{}), "Failed to delete CRP") + + // Verify that all resources placed have been removed from the specified member clusters. + for idx := range memberClusters { + memberCluster := memberClusters[idx] + + workResourcesRemovedActual := workNamespaceRemovedFromClusterActual(memberCluster) + Eventually(workResourcesRemovedActual, workloadEventuallyDuration, time.Second*5).Should(Succeed(), "Failed to remove work resources from member cluster %s", memberCluster.ClusterName) + } + + // Verify that related finalizers have been removed from the CRP. + finalizerRemovedActual := allFinalizersExceptForCustomDeletionBlockerRemovedFromPlacementActualV1(types.NamespacedName{Name: crpName}) + Eventually(finalizerRemovedActual, workloadEventuallyDuration, eventuallyInterval).Should(Succeed(), "Failed to remove controller finalizers from CRP") + + // Remove the custom deletion blocker finalizer from the CRP. + cleanupPlacementV1(types.NamespacedName{Name: crpName}) + + // Delete the created resources. + cleanupWorkResources() +} + +func workResourceIdentifiersV1() []placementv1.ResourceIdentifier { + workNamespaceName := fmt.Sprintf(workNamespaceNameTemplate, GinkgoParallelProcess()) + appConfigMapName := fmt.Sprintf(appConfigMapNameTemplate, GinkgoParallelProcess()) + + return []placementv1.ResourceIdentifier{ + { + Kind: "Namespace", + Name: workNamespaceName, + Version: "v1", + }, + { + Kind: "ConfigMap", + Name: appConfigMapName, + Version: "v1", + Namespace: workNamespaceName, + }, + } +} + +func crpStatusUpdatedActualV1(wantSelectedResourceIdentifiers []placementv1.ResourceIdentifier, wantSelectedClusters, wantUnselectedClusters []string, wantObservedResourceIndex string) func() error { + crpKey := types.NamespacedName{Name: fmt.Sprintf(crpNameTemplate, GinkgoParallelProcess())} + return func() error { + placement, err := retrievePlacementV1(crpKey) + if err != nil { + return fmt.Errorf("failed to get placement %s: %w", crpKey, err) + } + + wantStatus := buildWantPlacementStatusV1(crpKey, placement.GetGeneration(), wantSelectedResourceIdentifiers, wantSelectedClusters, wantUnselectedClusters, wantObservedResourceIndex) + cmpOptions := placementStatusCmpOptionsV1 + if wantObservedResourceIndex == "0" { + // The placement has just been created; the observed resource index might not have been + // populated yet. + cmpOptions = placementStatusCmpOptionsOnCreateV1 + } + if diff := cmp.Diff(placement.GetPlacementStatus(), wantStatus, cmpOptions...); diff != "" { + return fmt.Errorf("placement status diff (-got, +want): %s for placement %v", diff, crpKey) + } + return nil + } +} + +func buildWantPlacementStatusV1( + placementKey types.NamespacedName, + placementGeneration int64, + wantSelectedResourceIdentifiers []placementv1.ResourceIdentifier, + wantSelectedClusters, wantUnselectedClusters []string, + wantObservedResourceIndex string, +) *placementv1.PlacementStatus { + wantPerClusterPlacementStatuses := []placementv1.PerClusterPlacementStatus{} + for _, name := range wantSelectedClusters { + wantPerClusterPlacementStatuses = append(wantPerClusterPlacementStatuses, placementv1.PerClusterPlacementStatus{ + ClusterName: name, + ObservedResourceIndex: wantObservedResourceIndex, + Conditions: perClusterRolloutCompletedConditions(placementGeneration, true, false), + }) + } + for i := 0; i < len(wantUnselectedClusters); i++ { + wantPerClusterPlacementStatuses = append(wantPerClusterPlacementStatuses, placementv1.PerClusterPlacementStatus{ + Conditions: perClusterScheduleFailedConditions(placementGeneration), + }) + } + + var wantPlacementConditions []metav1.Condition + switch { + case len(wantSelectedClusters) > 0 && len(wantUnselectedClusters) > 0: + wantPlacementConditions = placementSchedulePartiallyFailedConditions(placementKey, placementGeneration) + case len(wantSelectedClusters) > 0: + wantPlacementConditions = placementRolloutCompletedConditions(placementKey, placementGeneration, false) + case len(wantUnselectedClusters) > 0: + // The remaining resource conditions are not set if there is no cluster to select. + wantPlacementConditions = placementScheduleFailedConditions(placementKey, placementGeneration) + default: + wantPlacementConditions = placementScheduledConditions(placementKey, placementGeneration) + } + + return &placementv1.PlacementStatus{ + Conditions: wantPlacementConditions, + PerClusterPlacementStatuses: wantPerClusterPlacementStatuses, + SelectedResources: wantSelectedResourceIdentifiers, + ObservedResourceIndex: wantObservedResourceIndex, + } +} + // Test specs in this file help verify the progression from one API version to another (e.g., v1beta1 to v1); // the logic is more focuses on API compatibility and is less focused on behavioral correctness for simplicity reasons. @@ -484,3 +728,183 @@ var _ = Describe("takeover, drift detection, and reportDiff mode (v1beta1 to v1) }) }) }) + +var _ = Describe("eviction and disruption budget", func() { + Context("eviction of a PickAll CRP protected by a disruption budget (read and write in v1)", Ordered, func() { + crpName := fmt.Sprintf(crpNameTemplate, GinkgoParallelProcess()) + crpEvictionName := fmt.Sprintf(crpEvictionNameTemplate, GinkgoParallelProcess()) + + BeforeAll(func() { + createWorkResources() + + crp := &placementv1.ClusterResourcePlacement{ + ObjectMeta: metav1.ObjectMeta{ + Name: crpName, + }, + Spec: placementv1.PlacementSpec{ + Policy: &placementv1.PlacementPolicy{ + PlacementType: placementv1.PickAllPlacementType, + }, + ResourceSelectors: []placementv1.ResourceSelectorTerm{ + { + Group: "", + Version: "v1", + Kind: "Namespace", + Name: fmt.Sprintf(workNamespaceNameTemplate, GinkgoParallelProcess()), + }, + }, + }, + } + Expect(hubClient.Create(ctx, crp)).To(Succeed(), "Failed to create CRP %s", crpName) + }) + + AfterAll(func() { + ensureCRPEvictionDeletedV1(crpEvictionName) + ensureCRPDisruptionBudgetDeletedV1(crpName) + ensureCRPAndRelatedResourcesDeletedV1(crpName, allMemberClusters) + }) + + It("should place resources on all available member clusters", func() { + crpStatusUpdatedActual := crpStatusUpdatedActualV1(workResourceIdentifiersV1(), allMemberClusterNames, nil, "0") + Eventually(crpStatusUpdatedActual, eventuallyDuration, eventuallyInterval).Should(Succeed(), "Failed to update CRP status as expected") + }) + + It("should create a disruption budget that protects all placements", func() { + crpdb := &placementv1.ClusterResourcePlacementDisruptionBudget{ + ObjectMeta: metav1.ObjectMeta{ + Name: crpName, + }, + Spec: placementv1.PlacementDisruptionBudgetSpec{ + MinAvailable: ptr.To(intstr.FromInt32(int32(len(allMemberClusterNames)))), + }, + } + Expect(hubClient.Create(ctx, crpdb)).To(Succeed(), "Failed to create CRP disruption budget %s", crpName) + }) + + It("should create an eviction targeting a bound cluster", func() { + crpe := &placementv1.ClusterResourcePlacementEviction{ + ObjectMeta: metav1.ObjectMeta{ + Name: crpEvictionName, + }, + Spec: placementv1.PlacementEvictionSpec{ + PlacementName: crpName, + ClusterName: memberCluster1EastProdName, + }, + } + Expect(hubClient.Create(ctx, crpe)).To(Succeed(), "Failed to create CRP eviction %s", crpEvictionName) + }) + + It("should deny the disruption", func() { + crpEvictionStatusUpdatedActual := testutilseviction.StatusUpdatedActual( + ctx, hubClient, crpEvictionName, + &testutilseviction.IsValidEviction{IsValid: true, Msg: condition.EvictionValidMessage}, + &testutilseviction.IsExecutedEviction{ + IsExecuted: false, + Msg: fmt.Sprintf( + condition.EvictionBlockedPDBSpecifiedMessageFmt, + len(allMemberClusterNames), + len(allMemberClusterNames), + ), + }, + ) + Eventually(crpEvictionStatusUpdatedActual, eventuallyDuration, eventuallyInterval).Should(Succeed(), "Failed to deny CRP eviction as expected") + }) + }) + + Context("eviction of a PickN CRP protected by a disruption budget (read and write in v1)", Ordered, Serial, func() { + crpName := fmt.Sprintf(crpNameTemplate, GinkgoParallelProcess()) + crpEvictionName := fmt.Sprintf(crpEvictionNameTemplate, GinkgoParallelProcess()) + taintClusterNames := []string{memberCluster1EastProdName} + noTaintClusterNames := []string{memberCluster2EastCanaryName, memberCluster3WestProdName} + + BeforeAll(func() { + createWorkResources() + + crp := &placementv1.ClusterResourcePlacement{ + ObjectMeta: metav1.ObjectMeta{ + Name: crpName, + }, + Spec: placementv1.PlacementSpec{ + Policy: &placementv1.PlacementPolicy{ + PlacementType: placementv1.PickNPlacementType, + NumberOfClusters: ptr.To(int32(len(allMemberClusterNames))), + }, + ResourceSelectors: []placementv1.ResourceSelectorTerm{ + { + Group: "", + Version: "v1", + Kind: "Namespace", + Name: fmt.Sprintf(workNamespaceNameTemplate, GinkgoParallelProcess()), + }, + }, + }, + } + Expect(hubClient.Create(ctx, crp)).To(Succeed(), "Failed to create CRP %s", crpName) + }) + + AfterAll(func() { + removeTaintsFromMemberClusters(taintClusterNames) + ensureCRPEvictionDeletedV1(crpEvictionName) + ensureCRPDisruptionBudgetDeletedV1(crpName) + ensureCRPAndRelatedResourcesDeletedV1(crpName, allMemberClusters) + }) + + It("should place resources on all available member clusters", func() { + crpStatusUpdatedActual := crpStatusUpdatedActualV1(workResourceIdentifiersV1(), allMemberClusterNames, nil, "0") + Eventually(crpStatusUpdatedActual, eventuallyDuration, eventuallyInterval).Should(Succeed(), "Failed to update CRP status as expected") + }) + + It("should create a disruption budget that allows one unavailable placement", func() { + crpdb := &placementv1.ClusterResourcePlacementDisruptionBudget{ + ObjectMeta: metav1.ObjectMeta{ + Name: crpName, + }, + Spec: placementv1.PlacementDisruptionBudgetSpec{ + MaxUnavailable: ptr.To(intstr.FromInt32(1)), + }, + } + Expect(hubClient.Create(ctx, crpdb)).To(Succeed(), "Failed to create CRP disruption budget %s", crpName) + }) + + It("should taint the target cluster to prevent it from being picked again", func() { + addTaintsToMemberClusters(taintClusterNames, buildTaints(taintClusterNames)) + }) + + It("should create an eviction targeting a bound cluster", func() { + crpe := &placementv1.ClusterResourcePlacementEviction{ + ObjectMeta: metav1.ObjectMeta{ + Name: crpEvictionName, + }, + Spec: placementv1.PlacementEvictionSpec{ + PlacementName: crpName, + ClusterName: memberCluster1EastProdName, + }, + } + Expect(hubClient.Create(ctx, crpe)).To(Succeed(), "Failed to create CRP eviction %s", crpEvictionName) + }) + + It("should allow the disruption", func() { + crpEvictionStatusUpdatedActual := testutilseviction.StatusUpdatedActual( + ctx, hubClient, crpEvictionName, + &testutilseviction.IsValidEviction{IsValid: true, Msg: condition.EvictionValidMessage}, + &testutilseviction.IsExecutedEviction{ + IsExecuted: true, + Msg: fmt.Sprintf( + condition.EvictionAllowedPDBSpecifiedMessageFmt, + len(allMemberClusterNames), + len(allMemberClusterNames), + ), + }, + ) + Eventually(crpEvictionStatusUpdatedActual, eventuallyDuration, eventuallyInterval).Should(Succeed(), "Failed to allow CRP eviction as expected") + }) + + It("should complete the disruption", func() { + workResourcesRemovedActual := workNamespaceRemovedFromClusterActual(memberCluster1EastProd) + Eventually(workResourcesRemovedActual, workloadEventuallyDuration, eventuallyInterval).Should(Succeed(), "Failed to remove work resources from evicted member cluster") + + crpStatusUpdatedActual := crpStatusUpdatedActualV1(workResourceIdentifiersV1(), noTaintClusterNames, taintClusterNames, "0") + Eventually(crpStatusUpdatedActual, eventuallyDuration, eventuallyInterval).Should(Succeed(), "Failed to update CRP status after eviction") + }) + }) +}) diff --git a/test/e2e/setup_test.go b/test/e2e/setup_test.go index 4d3b1307c..986819d9b 100644 --- a/test/e2e/setup_test.go +++ b/test/e2e/setup_test.go @@ -234,13 +234,15 @@ var ( // disappear from the status of the MemberCluster object. c.Type == string(clusterv1beta1.ConditionTypeClusterPropertyProviderStarted) }) - ignoreTimeTypeFields = cmpopts.IgnoreTypes(time.Time{}, metav1.Time{}) - ignorePlacementStatusDriftedPlacementsTimestampFields = cmpopts.IgnoreFields(placementv1beta1.DriftedResourcePlacement{}, "ObservationTime", "FirstDriftedObservedTime") - ignorePlacementStatusDriftedPlacementsTimestampFieldsV1 = cmpopts.IgnoreFields(placementv1.DriftedResourcePlacement{}, "ObservationTime", "FirstDriftedObservedTime") - ignorePlacementStatusDiffedPlacementsTimestampFields = cmpopts.IgnoreFields(placementv1beta1.DiffedResourcePlacement{}, "ObservationTime", "FirstDiffedObservedTime") - ignorePlacementStatusDiffedPlacementsTimestampFieldsV1 = cmpopts.IgnoreFields(placementv1.DiffedResourcePlacement{}, "ObservationTime", "FirstDiffedObservedTime") - ignorePerClusterPlacementStatusObservedResourceIndexField = cmpopts.IgnoreFields(placementv1beta1.PerClusterPlacementStatus{}, "ObservedResourceIndex") - ignorePlacementStatusObservedResourceIndexField = cmpopts.IgnoreFields(placementv1beta1.PlacementStatus{}, "ObservedResourceIndex") + ignoreTimeTypeFields = cmpopts.IgnoreTypes(time.Time{}, metav1.Time{}) + ignorePlacementStatusDriftedPlacementsTimestampFields = cmpopts.IgnoreFields(placementv1beta1.DriftedResourcePlacement{}, "ObservationTime", "FirstDriftedObservedTime") + ignorePlacementStatusDriftedPlacementsTimestampFieldsV1 = cmpopts.IgnoreFields(placementv1.DriftedResourcePlacement{}, "ObservationTime", "FirstDriftedObservedTime") + ignorePlacementStatusDiffedPlacementsTimestampFields = cmpopts.IgnoreFields(placementv1beta1.DiffedResourcePlacement{}, "ObservationTime", "FirstDiffedObservedTime") + ignorePlacementStatusDiffedPlacementsTimestampFieldsV1 = cmpopts.IgnoreFields(placementv1.DiffedResourcePlacement{}, "ObservationTime", "FirstDiffedObservedTime") + ignorePerClusterPlacementStatusObservedResourceIndexField = cmpopts.IgnoreFields(placementv1beta1.PerClusterPlacementStatus{}, "ObservedResourceIndex") + ignorePerClusterPlacementStatusObservedResourceIndexFieldV1 = cmpopts.IgnoreFields(placementv1.PerClusterPlacementStatus{}, "ObservedResourceIndex") + ignorePlacementStatusObservedResourceIndexField = cmpopts.IgnoreFields(placementv1beta1.PlacementStatus{}, "ObservedResourceIndex") + ignorePlacementStatusObservedResourceIndexFieldV1 = cmpopts.IgnoreFields(placementv1.PlacementStatus{}, "ObservedResourceIndex") placementStatusCmpOptions = cmp.Options{ cmpopts.SortSlices(lessFuncCondition), @@ -382,35 +384,26 @@ func beforeSuiteForAllProcesses() { sysMastersClient = hubCluster.SystemMastersClient Expect(sysMastersClient).NotTo(BeNil(), "Failed to initialize impersonate client for accessing Kubernetes cluster") - var pricingProvider1 trackers.PricingProvider - if isAzurePropertyProviderEnabled { - pricingProvider1 = trackers.NewAKSKarpenterPricingClient(ctx, memberCluster1AKSRegion) - } + pricingProvider1 := newPricingProvider(ctx, memberCluster1AKSRegion) memberCluster1EastProd = framework.NewCluster(memberCluster1EastProdName, memberCluster1EastProdSAName, scheme, pricingProvider1) Expect(memberCluster1EastProd).NotTo(BeNil(), "Failed to initialize cluster object") framework.GetClusterClient(memberCluster1EastProd) memberCluster1EastProdClient = memberCluster1EastProd.KubeClient Expect(memberCluster1EastProdClient).NotTo(BeNil(), "Failed to initialize client for accessing Kubernetes cluster") - var pricingProvider2 trackers.PricingProvider - if isAzurePropertyProviderEnabled { - pricingProvider2 = trackers.NewAKSKarpenterPricingClient(ctx, memberCluster2AKSRegion) - } + pricingProvider2 := newPricingProvider(ctx, memberCluster2AKSRegion) memberCluster2EastCanary = framework.NewCluster(memberCluster2EastCanaryName, memberCluster2EastCanarySAName, scheme, pricingProvider2) Expect(memberCluster2EastCanary).NotTo(BeNil(), "Failed to initialize cluster object") framework.GetClusterClient(memberCluster2EastCanary) memberCluster2EastCanaryClient = memberCluster2EastCanary.KubeClient Expect(memberCluster2EastCanaryClient).NotTo(BeNil(), "Failed to initialize client for accessing Kubernetes cluster") - var pricingProvider3 trackers.PricingProvider - if isAzurePropertyProviderEnabled { - pricingProvider3 = trackers.NewAKSKarpenterPricingClient(ctx, memberCluster3AKSRegion) - } + pricingProvider3 := newPricingProvider(ctx, memberCluster3AKSRegion) memberCluster3WestProd = framework.NewCluster(memberCluster3WestProdName, memberCluster3WestProdSAName, scheme, pricingProvider3) Expect(memberCluster3WestProd).NotTo(BeNil(), "Failed to initialize cluster object") framework.GetClusterClient(memberCluster3WestProd) memberCluster3WestProdClient = memberCluster3WestProd.KubeClient - Expect(memberCluster3WestProdClient).NotTo(BeNil(), "Failed to initialize client for accessing kubernetes cluster") + Expect(memberCluster3WestProdClient).NotTo(BeNil(), "Failed to initialize client for accessing Kubernetes cluster") allMemberClusters = []*framework.Cluster{memberCluster1EastProd, memberCluster2EastCanary, memberCluster3WestProd} once.Do(func() { @@ -429,6 +422,20 @@ func beforeSuiteForAllProcesses() { checkVAPAndBindingExistence(hubCluster) } +// newPricingProvider returns an AKS Karpenter pricing client for the given region, or +// nil when the Azure property provider is disabled. Returning an untyped nil matters: +// the node tracker decides whether to collect cost properties by comparing the provider +// against nil. +func newPricingProvider(ctx context.Context, region string) trackers.PricingProvider { + if !isAzurePropertyProviderEnabled { + return nil + } + + pp, err := trackers.NewAKSKarpenterPricingClient(ctx, region) + Expect(err).NotTo(HaveOccurred(), "Failed to create the AKS Karpenter pricing client for region %s", region) + return pp +} + func maxDuration(a, b time.Duration) time.Duration { if a > b { return a diff --git a/test/upgrade/before/scenarios_test.go b/test/upgrade/before/scenarios_test.go index f8c9e8a3c..9d770359e 100644 --- a/test/upgrade/before/scenarios_test.go +++ b/test/upgrade/before/scenarios_test.go @@ -26,6 +26,8 @@ import ( corev1 "k8s.io/api/core/v1" "k8s.io/apimachinery/pkg/api/meta" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime" "k8s.io/apimachinery/pkg/types" "k8s.io/apimachinery/pkg/util/intstr" "k8s.io/utils/ptr" @@ -626,7 +628,10 @@ var _ = Describe("CRP stuck in the rollout process (blocked by apply op failure) configMap.Data["custom"] = "foo" // Unset this field as required by the server. configMap.ObjectMeta.ManagedFields = nil - Expect(memberCluster.KubeClient.Patch(ctx, configMap, client.Apply, &client.PatchOptions{FieldManager: "handover", Force: ptr.To(true)})).To(Succeed(), "Failed to update config map %s", appConfigMapName) + unstructuredMap, err := runtime.DefaultUnstructuredConverter.ToUnstructured(configMap) + Expect(err).To(BeNil(), "Failed to convert config map %s to unstructured", appConfigMapName) + applyConfig := client.ApplyConfigurationFromUnstructured(&unstructured.Unstructured{Object: unstructuredMap}) + Expect(memberCluster.KubeClient.Apply(ctx, applyConfig, &client.ApplyOptions{FieldManager: "handover", Force: ptr.To(true)})).To(Succeed(), "Failed to update config map %s", appConfigMapName) } }) diff --git a/test/utils/informer/manager.go b/test/utils/informer/manager.go index d96fa1b91..91eb17513 100644 --- a/test/utils/informer/manager.go +++ b/test/utils/informer/manager.go @@ -192,3 +192,8 @@ func (m *FakeManager) AddEventHandlerToInformer(_ schema.GroupVersionResource, _ func (m *FakeManager) CreateInformerForResource(_ informer.APIResourceMeta) { // No-op for testing } + +func (m *FakeManager) IsInformerSet(_ schema.GroupVersionKind) bool { + // For testing, we can assume that the informer is always set for the given resource. + return true +}