diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index d348a395..40f94a08 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -95,10 +95,15 @@ jobs: run: | just ci::run just check::test -c backend just ci::run just check::test -c frontend - - name: Seed + # Brings the real stack up (Postgres, Keycloak, backend), fills it, and + # then actually asks it something. The unit suites run on SQLite behind a + # mock keyfunc over bufconn, so the Postgres schema, the Keycloak token + # exchange and the gRPC port are only ever exercised here. + - name: Seed and smoke test run: | just ci::run just deploy::up just ci::run just db::seed + just ci::run just check::smoke images: runs-on: ubuntu-24.04 diff --git a/CLAUDE.md b/CLAUDE.md index c0015c21..db64857b 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -113,6 +113,7 @@ just helm::publish # push the chart if Chart.yaml's version is unpu just cluster::backup # pg_dumpall to ~/hackagon-backups just cluster::wipe # backup, then drop + recreate the app DB; dev restarts empty just cluster::reseed # backup, wipe, then seed; refuses if components/backend differs from origin/main +just cluster::smoke # run check::smoke against dev through a port-forward; needs a seeded dev ``` Backend listens on **:3000**, frontend on **:8081**. Dev users (Keycloak diff --git a/components/backend/test/smoke/smoke_suite_test.go b/components/backend/test/smoke/smoke_suite_test.go new file mode 100644 index 00000000..8dcf075b --- /dev/null +++ b/components/backend/test/smoke/smoke_suite_test.go @@ -0,0 +1,142 @@ +//go:build test && integration + +// Package smoke_test drives a *running* Hackagon stack over the network. +// +// Everything under `internal/**` is tested against in-memory SQLite, a mock +// keyfunc and a bufconn listener — no Postgres, no Keycloak, no sockets. That +// is the right trade for testing behaviour, and it is why those suites are +// where behaviour belongs. It also means three things production depends on are +// never exercised: the Postgres schema the migration actually produced, the +// token exchange with Keycloak, and the gRPC server on a real port. +// +// This suite exercises exactly those three and little else. It is deliberately +// thin: "is the deployed thing alive, does auth work end to end, does Postgres +// round-trip, and does a refusal refuse cleanly". Logic assertions belong in +// `internal/service` where they run in a second without a stack. +package smoke_test + +import ( + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "net/url" + "os" + "strings" + "testing" + "time" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + "google.golang.org/grpc/metadata" +) + +func TestSmoke(t *testing.T) { + RegisterFailHandler(Fail) + RunSpecs(t, "Smoke Suite") +} + +// Where the suite points. The defaults are the local process-compose stack +// (`just deploy::up`), so it runs unconfigured on a developer machine and in CI +// alike; the env vars are there to aim it at a deployed environment instead. +var ( + backendAddr = envOr("HACKAGON_SMOKE_BACKEND_ADDR", "localhost:3000") + keycloakURL = envOr("HACKAGON_SMOKE_KEYCLOAK_URL", "http://localhost:8180") + realm = envOr("HACKAGON_SMOKE_REALM", "hackagon") + clientID = envOr("HACKAGON_SMOKE_CLIENT_ID", "hackagon-backend") + + // The dev-fixture password every seeded Keycloak user shares, documented in + // CLAUDE.md. Overridable so the suite can be pointed somewhere its fixtures + // differ — it is not a secret in any environment this suite should run + // against, and pointing this at production is not a supported use. + fixturePassword = envOr("HACKAGON_SMOKE_PASSWORD", "aliceandbob") +) + +// specTimeout bounds every spec. A stack that is down should fail the suite in +// seconds with a legible message rather than hang until the CI job's 30-minute +// ceiling, which is the difference between a useful red build and one people +// learn to cancel. +const specTimeout = 30 * time.Second + +func envOr(key, fallback string) string { + if v := os.Getenv(key); v != "" { + return v + } + + return fallback +} + +// tokens caches one access token per username. Specs run serially, and re-asking +// Keycloak for the same token on every call would add latency and failure +// surface to assertions that are not about Keycloak. +var tokens = map[string]string{} + +// accessToken performs the OIDC password grant against Keycloak. +// +// This is the half `cmd/seed` structurally cannot cover: the seed signs its own +// tokens with a key the backend is configured to trust, so a realm with a +// renamed client, a missing user or direct-access-grants switched off still +// seeds perfectly green. Here the token has to come from Keycloak itself, and +// the backend has to accept it after fetching JWKS over the network. +func accessToken(ctx context.Context, username string) string { + GinkgoHelper() + + if cached, ok := tokens[username]; ok { + return cached + } + + endpoint := fmt.Sprintf( + "%s/realms/%s/protocol/openid-connect/token", keycloakURL, realm, + ) + + form := url.Values{ + "client_id": {clientID}, + "username": {username}, + "password": {fixturePassword}, + "grant_type": {"password"}, + "scope": {"openid profile"}, + } + + // The request carries the spec's deadline. http.DefaultClient has no + // timeout of its own, so without a context a Keycloak that accepted the + // connection and then never answered would hang this goroutine long after + // the spec that owns it has been reported. + req, err := http.NewRequestWithContext( + ctx, http.MethodPost, endpoint, strings.NewReader(form.Encode()), + ) + Expect(err).NotTo(HaveOccurred()) + req.Header.Set("Content-Type", "application/x-www-form-urlencoded") + + resp, err := http.DefaultClient.Do(req) + Expect(err).NotTo(HaveOccurred(), + "Keycloak unreachable at %s — is the stack up? (just deploy::up)", keycloakURL) + defer func() { _ = resp.Body.Close() }() + + body, err := io.ReadAll(resp.Body) + Expect(err).NotTo(HaveOccurred()) + Expect(resp.StatusCode).To(Equal(http.StatusOK), + "Keycloak refused a token for %q: %s", username, body) + + var payload struct { + AccessToken string `json:"access_token"` + } + Expect(json.Unmarshal(body, &payload)).To(Succeed()) + Expect(payload.AccessToken).NotTo(BeEmpty(), + "Keycloak answered 200 for %q but the response carried no access_token", username) + + tokens[username] = payload.AccessToken + + return payload.AccessToken +} + +// as derives an outgoing context authenticated as `username`, keeping the +// spec's deadline. Deriving from the spec context rather than Background is +// what makes specTimeout actually bound the RPC. +func as(ctx context.Context, username string) context.Context { + GinkgoHelper() + + return metadata.AppendToOutgoingContext( + ctx, "authorization", "Bearer "+accessToken(ctx, username), + ) +} diff --git a/components/backend/test/smoke/smoke_test.go b/components/backend/test/smoke/smoke_test.go new file mode 100644 index 00000000..49b95299 --- /dev/null +++ b/components/backend/test/smoke/smoke_test.go @@ -0,0 +1,189 @@ +//go:build test && integration + +package smoke_test + +import ( + "fmt" + "time" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + "google.golang.org/grpc" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/credentials/insecure" + healthgrpc "google.golang.org/grpc/health/grpc_health_v1" + "google.golang.org/grpc/status" + + hackathonSvc "github.com/swissdatasciencecenter/hackagon/components/backend/internal/proto/hackathon" + hackMsgs "github.com/swissdatasciencecenter/hackagon/components/backend/internal/proto/hackathon/messages/hackathon_svc" + trackMsgs "github.com/swissdatasciencecenter/hackagon/components/backend/internal/proto/hackathon/messages/track_svc" + userSvc "github.com/swissdatasciencecenter/hackagon/components/backend/internal/proto/user" + userEnts "github.com/swissdatasciencecenter/hackagon/components/backend/internal/proto/user/entities" + userMsgs "github.com/swissdatasciencecenter/hackagon/components/backend/internal/proto/user/messages/user_svc" +) + +// The seeded fixture this suite leans on. Kept to the three hackathon names and +// the two people with Keycloak accounts, because every further detail is one +// more way for a seed change to redden a build that is meant to be about the +// deployment rather than about the fixture. +const ( + sentinelHackathon = "AI Innovation Challenge 2026" + climateHackathon = "Climate Tech Hackathon 2026" + sprintHackathon = "Internal Product Sprint" + + // alice organizes the sentinel hackathon; bob merely takes part in it. + organizer = "alice" + participant = "bob" +) + +var _ = Describe("A running Hackagon stack", Ordered, func() { + var ( + health healthgrpc.HealthClient + users userSvc.UserServiceClient + hackathons hackathonSvc.HackathonServiceClient + tracks hackathonSvc.TrackServiceClient + ) + + BeforeAll(func() { + conn, err := grpc.NewClient( + backendAddr, + grpc.WithTransportCredentials(insecure.NewCredentials()), + ) + Expect(err).NotTo(HaveOccurred()) + DeferCleanup(func() { _ = conn.Close() }) + + health = healthgrpc.NewHealthClient(conn) + users = userSvc.NewUserServiceClient(conn) + hackathons = hackathonSvc.NewHackathonServiceClient(conn) + tracks = hackathonSvc.NewTrackServiceClient(conn) + }) + + // findHackathon resolves a seeded hackathon to its id as `username` sees it. + // Going through List rather than hardcoding an id keeps the suite working + // against any freshly seeded database, where the ids are new every time. + findHackathon := func(ctx SpecContext, username, name string) string { + GinkgoHelper() + + resp, err := hackathons.List(as(ctx, username), &hackMsgs.ListRequest{}) + Expect(err).NotTo(HaveOccurred()) + + for _, h := range resp.GetHackathons() { + if h.GetName() == name { + return h.GetId() + } + } + + Fail(fmt.Sprintf("%q is not visible to %q", name, username)) + + return "" + } + + It("serves gRPC on its port", func(ctx SpecContext) { + resp, err := health.Check(ctx, &healthgrpc.HealthCheckRequest{}) + Expect(err).NotTo(HaveOccurred(), + "no gRPC health response from %s — is the stack up? (just deploy::up)", backendAddr) + Expect(resp.GetStatus()).To(Equal(healthgrpc.HealthCheckResponse_SERVING)) + }, SpecTimeout(specTimeout)) + + // The end-to-end auth assertion, and the reason this suite exists at all. + // A token minted by Keycloak has to survive the backend fetching JWKS, + // checking the issuer and resolving `sub` against a row seeded by a + // completely separate process. The role is what proves that last step: a + // user conjured from token claims alone would come back with none. + It("accepts a token Keycloak issued, for the user the seed created", func(ctx SpecContext) { + resp, err := users.WhoAmI(as(ctx, organizer), &userMsgs.WhoAmIRequest{}) + Expect(err).NotTo(HaveOccurred(), + "the backend rejected a genuine Keycloak token for %q", organizer) + + Expect(resp.GetUser().GetUsername()).To(Equal(organizer)) + Expect(resp.GetUser().GetRoles()).To( + ContainElement(userEnts.GlobalRole_GLOBAL_ROLE_HACKATHON_ORGANIZER), + "%q resolved to a user without the seeded organizer role, "+ + "so the token's subject did not match the seeded row", organizer) + }, SpecTimeout(specTimeout)) + + It("reads the seeded fixture back out of Postgres", func(ctx SpecContext) { + resp, err := hackathons.List(as(ctx, organizer), &hackMsgs.ListRequest{}) + Expect(err).NotTo(HaveOccurred()) + + names := []string{} + for _, h := range resp.GetHackathons() { + names = append(names, h.GetName()) + } + + Expect(names).To(ContainElements( + sentinelHackathon, climateHackathon, sprintHackathon, + ), "seeded hackathons missing — did `just db::seed` run against this database?") + }, SpecTimeout(specTimeout)) + + // The write path on the real driver. Everything under `internal/**` writes + // to SQLite, so a column, default or constraint that only the Postgres + // migration produces has nowhere else to go wrong in front of us. + // + // A track, and not a hackathon, because a track is the smallest thing the + // API can both create and delete. Nothing deletes a hackathon, so writing + // one would leave a row behind on every run — and a suite that silts up a + // developer's database is one people stop running locally, which is where it + // is most useful. Deleting also buys a third assertion for free: the row is + // really gone afterwards. + It("writes to Postgres, reads the row back, and deletes it", func(ctx SpecContext) { + hackathonID := findHackathon(ctx, organizer, sentinelHackathon) + name := fmt.Sprintf("smoke test %s", time.Now().UTC().Format(time.RFC3339)) + + created, err := tracks.Create(as(ctx, organizer), &trackMsgs.CreateRequest{ + HackathonId: hackathonID, + Name: name, + Description: "Created and removed again by the smoke suite.", + }) + Expect(err).NotTo(HaveOccurred()) + + trackID := created.GetTrackId() + Expect(trackID).NotTo(BeEmpty()) + + // Registered before the assertions below, so that a failing read-back + // still takes the row with it on the way out. + DeferCleanup(func(ctx SpecContext) { + _, _ = tracks.Delete(as(ctx, organizer), &trackMsgs.DeleteRequest{TrackId: trackID}) + }) + + got, err := tracks.Get(as(ctx, organizer), &trackMsgs.GetRequest{TrackId: trackID}) + Expect(err).NotTo(HaveOccurred(), "the track just created could not be read back") + Expect(got.GetTrack().GetName()).To(Equal(name)) + Expect(got.GetTrack().GetHackathonId()).To(Equal(hackathonID)) + + _, err = tracks.Delete(as(ctx, organizer), &trackMsgs.DeleteRequest{TrackId: trackID}) + Expect(err).NotTo(HaveOccurred()) + + _, err = tracks.Get(as(ctx, organizer), &trackMsgs.GetRequest{TrackId: trackID}) + Expect(status.Code(err)).To(Equal(codes.NotFound), + "the track was deleted but is still readable") + }, SpecTimeout(specTimeout)) + + // Both refusals below assert the *code*, not merely that the call failed. + // PERMISSION_DENIED means the system refused on purpose and the frontend + // renders a clean 403; INTERNAL means something fell over on the way to + // refusing and the user gets a 500. Only one of those is correct, and a test + // that accepts any error cannot tell them apart. + It("refuses an admin-only call from an ordinary participant", func(ctx SpecContext) { + _, err := users.AddRole(as(ctx, participant), &userMsgs.AddRoleRequest{ + UserId: "00000000-0000-0000-0000-000000000000", + Role: userEnts.GlobalRole_GLOBAL_ROLE_HACKATHON_ORGANIZER, + }) + + Expect(status.Code(err)).To(Equal(codes.PermissionDenied), + "granting a global role as %q should be refused, not fail some other way", participant) + }, SpecTimeout(specTimeout)) + + It("refuses to let a participant edit the hackathon they joined", func(ctx SpecContext) { + id := findHackathon(ctx, participant, sentinelHackathon) + + newName := "renamed by a participant" + _, err := hackathons.Edit(as(ctx, participant), &hackMsgs.EditRequest{ + HackathonId: id, + Name: &newName, + }) + + Expect(status.Code(err)).To(Equal(codes.PermissionDenied), + "%q is a participant in %q, not an owner of it", participant, sentinelHackathon) + }, SpecTimeout(specTimeout)) +}) diff --git a/tools/just/check.just b/tools/just/check.just index cf765582..2377189d 100644 --- a/tools/just/check.just +++ b/tools/just/check.just @@ -15,6 +15,9 @@ help: echo -e " ${cyan}just check::test${reset} -c " echo -e " ${dim}e.g. just check::test -c frontend${reset}" echo "" + echo -e " ${cyan}just check::smoke${reset}" + echo -e " ${dim}Smoke-test a running stack (needs 'just deploy::up' + 'just db::seed')${reset}" + echo "" echo -e " ${cyan}just check::format${reset} -c " echo -e " ${dim}e.g. just check::format -c backend${reset}" echo "" @@ -51,6 +54,18 @@ test *args: rm -rf .output/*/coverage/data just quitsh test "$@" +# Smoke-test a running stack. Needs `just deploy::up` and `just db::seed` first. +# Usage: just check::smoke +[group('checks')] +smoke *args: + #!/usr/bin/env bash + set -eu + cd "{{root_dir}}/components/backend" + # `-count=1` because Go's test cache keys on source and inputs, and knows + # nothing about the stack these specs talk to. A cached PASS from a run + # against a different deployment is worse than no result at all. + go test -tags 'test integration' -count=1 -v ./test/smoke/ "$@" + # Lint a component. Usage: just check::lint -c backend [group('checks')] lint *args: diff --git a/tools/just/cluster.just b/tools/just/cluster.just index fe082d98..94f3d2a4 100644 --- a/tools/just/cluster.just +++ b/tools/just/cluster.just @@ -6,6 +6,8 @@ root_dir := `git rev-parse --show-toplevel` # Names the chart gives a release called `hackagon` (see helm-chart/templates). namespace := "hackagon" backend_deploy := "hackagon-backend" +backend_svc := "hackagon-backend" +backend_port := "3000" backend_config := "hackagon-backend-config" backend_db_secret := "hackagon-backend-db" postgres_sts := "hackagon-postgresql" @@ -20,6 +22,7 @@ dev_domain := "hackagon-dev.dscompute.ch" backup_dir := env("HOME") + "/hackagon-backups" local_pg_port := "15432" +local_backend_port := "13000" # Runs inside the postgres pod: finds the superuser password the bitnami image # was started with (env var or mounted file) so it never leaves the pod. @@ -59,6 +62,10 @@ help: echo -e " ${cyan}just cluster::reseed${reset} " echo -e " ${dim}Back up, wipe, then run cmd/seed (backend code must match origin/main).${reset}" echo "" + echo -e " ${cyan}just cluster::smoke${reset} " + echo -e " ${dim}Run the smoke suite (check::smoke) against dev. Read-only apart from a${reset}" + echo -e " ${dim}track it creates and deletes; needs a seeded dev (cluster::reseed).${reset}" + echo "" echo -e " ${dim}e.g. just cluster::wipe sck-sit-dev${reset}" echo "" @@ -96,8 +103,10 @@ wipe context: [ "${replicas:-0}" -gt 0 ] || replicas=1 restore_backend() { - echo "==> Scaling backend back to $replicas (it recreates the schema on startup)" - "${kc[@]}" scale "deploy/{{backend_deploy}}" --replicas="$replicas" + echo "==> Bringing the backend back (it recreates the schema on startup)" + # An EXIT trap keeps the recipe's exit status unless it exits itself, + # so a backend that does not come back would otherwise report success. + just -f "{{source_file()}}" _backend-up "{{context}}" "$replicas" || exit 1 } # The backend must be off while its database is replaced; bring it back # however this ends, including when stopping it or the wipe fails halfway. @@ -123,8 +132,10 @@ reseed context: restore_backend() { # Starting fresh is also what loads the casbin rows the seed wrote. - echo "==> Scaling backend back to $replicas" - "${kc[@]}" scale "deploy/{{backend_deploy}}" --replicas="$replicas" + echo "==> Bringing the backend back" + # An EXIT trap keeps the recipe's exit status unless it exits itself, + # so a backend that does not come back would otherwise report success. + just -f "{{source_file()}}" _backend-up "{{context}}" "$replicas" || exit 1 } # However this ends: close the tunnel, delete the temp dir, and bring the # backend back if it was taken down. @@ -182,6 +193,60 @@ reseed context: "$tmp/seed" --config-dir "$tmp/" echo "✓ Seeded" +# Run the smoke suite (check::smoke) against dev, through a port-forward to the +# backend. Needs a seeded dev: the specs log in as the fixture's alice and bob. +[group('cluster')] +smoke context: + #!/usr/bin/env bash + set -euo pipefail + just -f "{{source_file()}}" _guard "{{context}}" + + kc=(kubectl --context "{{context}}" -n "{{namespace}}") + pf_pid="" + stop_tunnel() { [ -n "$pf_pid" ] && kill "$pf_pid" 2>/dev/null || true; } + trap stop_tunnel EXIT + + echo "==> Waiting up to 3 minutes for the backend to be ready" + "${kc[@]}" rollout status "deploy/{{backend_deploy}}" --timeout=180s + + # Ask Keycloak for tokens at the issuer the backend itself trusts, so a + # token can never be refused for coming from the wrong address. + issuer="$("${kc[@]}" get cm "{{backend_config}}" -o jsonpath='{.data.config\.yaml}' \ + | sed -n 's/^ *issuerurl: *"\{0,1\}\([^"]*\)"\{0,1\} *$/\1/p')" + case "$issuer" in + http*://*/realms/*) ;; + *) echo "✗ No oidc issuerurl in {{backend_config}} (found: '${issuer}')." >&2; exit 1 ;; + esac + keycloak_url="${issuer%/realms/*}" + realm="${issuer##*/realms/}" + + echo "==> Port-forwarding backend to localhost:{{local_backend_port}}" + "${kc[@]}" port-forward "svc/{{backend_svc}}" "{{local_backend_port}}:{{backend_port}}" >/dev/null & + pf_pid=$! + + # Wait up to 10s for the tunnel to open. + for _ in $(seq 20); do + nc -z 127.0.0.1 "{{local_backend_port}}" 2>/dev/null && break + sleep 0.5 + done + nc -z 127.0.0.1 "{{local_backend_port}}" 2>/dev/null \ + || { echo "✗ The port-forward to the backend did not open within 10s." >&2; exit 1; } + + echo "==> Running the smoke suite (Keycloak $keycloak_url, realm $realm)" + cd "{{root_dir}}" + HACKAGON_SMOKE_BACKEND_ADDR="127.0.0.1:{{local_backend_port}}" \ + HACKAGON_SMOKE_KEYCLOAK_URL="$keycloak_url" \ + HACKAGON_SMOKE_REALM="$realm" \ + just check::smoke || { + echo "✗ Smoke suite failed; the first [FAILED] line above says which check broke." >&2 + echo " Usual causes on dev:" >&2 + echo " - dev is not seeded: just cluster::reseed {{context}}" >&2 + echo " - alice/bob missing in the realm, or another password: set HACKAGON_SMOKE_PASSWORD" >&2 + echo " - client hackagon-backend has 'Direct access grants' switched off" >&2 + exit 1 + } + echo "✓ Smoke suite passed against {{context}}" + # Scale the backend to 0 and wait until its pods are gone. Scaling it back up is # the caller's job, in an EXIT trap set before calling, so it happens even when # this or a later step fails halfway. @@ -207,6 +272,24 @@ _backend-down context: [ -z "$(backend_pods)" ] \ || { echo "✗ Backend pods still running after 2 minutes." >&2; exit 1; } +# Scale the backend to `replicas` and wait until it is serving again, so a +# recipe only reports success once dev is actually back. +[private] +_backend-up context replicas: + #!/usr/bin/env bash + set -euo pipefail + + # Guard the context, so a recipe pointed at the prod context by mistake stops before touching anything. + just -f "{{source_file()}}" _guard "{{context}}" + kc=(kubectl --context "{{context}}" -n "{{namespace}}") + + echo "==> Scaling backend to {{replicas}}" + "${kc[@]}" scale "deploy/{{backend_deploy}}" --replicas="{{replicas}}" + + echo "==> Waiting up to 3 minutes for the backend to be ready" + "${kc[@]}" rollout status "deploy/{{backend_deploy}}" --timeout=180s \ + || { echo "✗ Backend not ready; see: kubectl logs deploy/{{backend_deploy}}" >&2; exit 1; } + # Drop and recreate the app database and run _backend-down first, # so the backend does not try to connect to it while it is being replaced. [private]