From 6ce21550727aeb57213a4b0c4818ff54bf420c62 Mon Sep 17 00:00:00 2001 From: unit-adm <314923187+mumtaz6@users.noreply.github.com> Date: Sun, 4 Oct 2026 11:29:33 +0530 Subject: [PATCH 1/3] Put a node that answers back in the ring A leader that shuts down takes itself out of the ring on its way out. The next leader inherits that ring, and never saw the node fail, so when the node came back and answered its pings, nothing changed from its point of view: it never set the ring again, and the node stayed out for good. A rolling restart that restarted the leader could leave it there. The leader now puts back a node that answers its ping, isn't leaving and isn't in the ring. Only one that answers: a leader that left and is gone hasn't failed yet in the new leader's count, and putting it back would send requests to it until it did. Co-Authored-By: Claude Opus 5.5 (1M context) --- server/e2e/cluster_test.go | 26 ++++++++++++++++++++++++++ server/e2e/health_test.go | 21 +++++++++++++++++++++ server/internal/cluster_leader.go | 21 +++++++++++++++++++++ 3 files changed, 68 insertions(+) diff --git a/server/e2e/cluster_test.go b/server/e2e/cluster_test.go index 132c5ae..ffcf642 100644 --- a/server/e2e/cluster_test.go +++ b/server/e2e/cluster_test.go @@ -2590,3 +2590,29 @@ func TestClusterRestartedOwnerKeepsSubscriptions(t *testing.T) { } c.assertAlive(t, c.nodes, "after the restart") } + +// TestClusterGracefulRestartRejoins restarts the leader gracefully, then a +// follower: each is back in the ring, and ready, soon after. A leader that +// shuts down takes itself out of the ring; the next leader inherited that +// ring, and never saw the node fail, so it never set the ring again when +// the node came back. +func TestClusterGracefulRestartRejoins(t *testing.T) { + c := startCluster(t, names...) + leader, err := c.waitLeader(c.nodes, 10*time.Second) + if err != nil { + t.Fatal(err) + } + for _, name := range []string{leader, followerOf(leader)} { + n := c.node(name) + n.shutdown() + if err := n.start(); err != nil { + t.Fatal(err) + } + for _, m := range c.nodes { + waitReadyz(t, m, 30*time.Second) + } + if _, err := c.waitLeader(c.nodes, 10*time.Second); err != nil { + t.Fatalf("after %s's restart: %v", name, err) + } + } +} diff --git a/server/e2e/health_test.go b/server/e2e/health_test.go index 3366bf4..2bfaed4 100644 --- a/server/e2e/health_test.go +++ b/server/e2e/health_test.go @@ -131,3 +131,24 @@ func TestHealthCluster(t *testing.T) { } } } + +// waitReadyz waits until n's /_readyz says it's ready, and returns what it +// said before, each answer once. +func waitReadyz(t *testing.T, n *clusterNode, timeout time.Duration) []string { + t.Helper() + var notReady []string + deadline := time.Now().Add(timeout) + for { + code, body := monitorGet(t, n.server, "/_readyz") + if code == http.StatusOK { + return notReady + } + if len(notReady) == 0 || notReady[len(notReady)-1] != body { + notReady = append(notReady, body) + } + if time.Now().After(deadline) { + t.Fatalf("%s not ready after %s: %v\nlogs:\n%s", n.name, timeout, notReady, n.logs.String()) + } + time.Sleep(20 * time.Millisecond) + } +} diff --git a/server/internal/cluster_leader.go b/server/internal/cluster_leader.go index d87bc12..b636253 100644 --- a/server/internal/cluster_leader.go +++ b/server/internal/cluster_leader.go @@ -206,6 +206,8 @@ func (c *Cluster) sendPings() { // The ring version the followers saw the cluster route by: the lowest, // as a switch only ever goes up. routed := 0 + // The nodes that answered this ping, and aren't leaving. + answered := make(map[string]bool) for _, node := range c.nodes { var pong ClusterPong // A node that stalls without its connection failing does not answer: @@ -220,6 +222,7 @@ func (c *Cluster) sendPings() { RingVersion: c.getRingVersion()}, &pong, c.fo.heartBeat) if err == nil { node.setCapabilities(pong.NodeCapabilities) + answered[node.name] = !pong.Leaving if v := pong.RingVersion; v != 0 && (routed == 0 || v < routed) { routed = v } @@ -276,6 +279,24 @@ func (c *Cluster) sendPings() { c.adoptRingVersion(v) } + // The ring may lack a node that answers, which this leader never saw + // fail: a leader that left (shutting down) took itself out of the ring, + // and the next leader, inheriting that ring, sees the node answer when + // it is back, which is no change of its own. So a node that answers and + // isn't in the ring is put back. Only one that answers: one that left + // and is gone hasn't failed yet here, and is taken out as it fails. + if !rehash { + inRing := map[string]bool{} + for _, name := range c.getRingNodes() { + inRing[name] = true + } + for _, node := range live { + if answered[node.name] && !inRing[node.name] { + rehash = true + } + } + } + if rehash { var activeNodes []string for _, node := range live { From 58b9ea5d68a186d75be4b3f69dfd8e558d623c7b Mon Sep 17 00:00:00 2001 From: unit-adm <314923187+mumtaz6@users.noreply.github.com> Date: Sun, 4 Oct 2026 11:29:51 +0530 Subject: [PATCH 2/3] Backups off the cluster, reconciliation after a restore, and the runbooks docs/backup-restore.md: - A checkpoint describes itself (checkpoint.json: node, run, ring and engine versions, key ids, the copy's counts), and a node won't start at one into a running cluster unless its peers were restored from the same run, or it is started with -restored. - The backup run (server/cmd/backup, deploy/kubernetes/backups.yaml) checkpoints every node under one run id, keeps the manifest in each checkpoint, and has each node upload its checkpoint, archived, compressed and encrypted with age, to S3 with Object Lock, locked and tagged by tier (deploy/aws: bucket, lifecycle, put-only writer, reader and escrow policies). - -restored reconciles each topic with its other holders by message digests and counts, idempotently, before the node takes clients. - The security journal: each security change a node makes is fsynced on the node, uploaded every 10 s, and replayed by -journal on a restore, so nothing revoked after the run comes back. - The keyring is escrowed in Secrets Manager (backup escrow-keyring). - The weekly restore test (backup verify, restore-test.yaml) opens each node's checkpoint of the newest run on a scratch server, checks it and a canary written before the run, and pushes its success; backup-alerts.yaml alerts on it. - Runbooks A to D, a restore Job per node, backup runs and fetch -node. The new cluster calls check their sender over TLS, as the others do. A checkpoint counts memdb's records as it copies them: memdb's Size counts a key's versions. storedOn, an e2e helper, reads until the relay is quiet: it stopped at n messages, and missed some when a topic held more. Co-Authored-By: Claude Opus 5.5 (1M context) --- deploy/aws/README.md | 105 +++++ deploy/aws/escrow-policy.json | 15 + deploy/aws/lifecycle.json | 15 + deploy/aws/reader-policy.json | 27 ++ deploy/aws/writer-policy.json | 16 + deploy/kubernetes/backup-alerts.yaml | 68 +++ deploy/kubernetes/backups.yaml | 57 +++ deploy/kubernetes/restore-node-job.yaml | 60 +++ deploy/kubernetes/restore-test.yaml | 80 ++++ docs/README.md | 3 + docs/backup-restore.md | 198 +++++++++ go.mod | 23 + go.sum | 50 +++ server/cmd/backup/main.go | 209 +++++++++ server/cmd/backup/tools.go | 315 ++++++++++++++ server/cmd/backup/verify.go | 359 ++++++++++++++++ server/e2e/backup_run_test.go | 160 +++++++ server/e2e/offsite_test.go | 268 ++++++++++++ server/e2e/reconcile_test.go | 263 ++++++++++++ server/e2e/restore_cluster_test.go | 322 ++++++++++++++ server/e2e/restore_test_run_test.go | 107 +++++ server/e2e/runbook_test.go | 126 ++++++ server/internal/backup/archive.go | 145 +++++++ server/internal/backup/backup.go | 119 ++++++ server/internal/backup/backup_test.go | 120 ++++++ server/internal/backup/escrow.go | 150 +++++++ server/internal/backup/s3.go | 98 +++++ server/internal/checkpoint.go | 152 ++++++- server/internal/cluster.go | 33 +- server/internal/cluster_caps.go | 2 +- server/internal/cluster_health.go | 3 + server/internal/cluster_reconcile.go | 425 +++++++++++++++++++ server/internal/cluster_tls.go | 28 ++ server/internal/cluster_tls_test.go | 26 +- server/internal/db/adapter.go | 14 +- server/internal/db/checkpoint_info.go | 105 +++++ server/internal/db/unitdb/checkpoint.go | 46 +- server/internal/db/unitdb/checkpoint_test.go | 36 +- server/internal/health.go | 3 + server/internal/metrics.go | 1 + server/internal/offsite.go | 411 ++++++++++++++++++ server/internal/offsite_test.go | 98 +++++ server/internal/restore.go | 137 ++++++ server/internal/revocation.go | 19 +- server/internal/store/canary.go | 56 +++ server/internal/store/canary_test.go | 20 + server/internal/store/checkpoint_test.go | 7 +- server/internal/store/security.go | 31 +- server/main.go | 24 ++ 49 files changed, 5113 insertions(+), 42 deletions(-) create mode 100644 deploy/aws/README.md create mode 100644 deploy/aws/escrow-policy.json create mode 100644 deploy/aws/lifecycle.json create mode 100644 deploy/aws/reader-policy.json create mode 100644 deploy/aws/writer-policy.json create mode 100644 deploy/kubernetes/backup-alerts.yaml create mode 100644 deploy/kubernetes/backups.yaml create mode 100644 deploy/kubernetes/restore-node-job.yaml create mode 100644 deploy/kubernetes/restore-test.yaml create mode 100644 docs/backup-restore.md create mode 100644 server/cmd/backup/main.go create mode 100644 server/cmd/backup/tools.go create mode 100644 server/cmd/backup/verify.go create mode 100644 server/e2e/backup_run_test.go create mode 100644 server/e2e/offsite_test.go create mode 100644 server/e2e/reconcile_test.go create mode 100644 server/e2e/restore_cluster_test.go create mode 100644 server/e2e/restore_test_run_test.go create mode 100644 server/e2e/runbook_test.go create mode 100644 server/internal/backup/archive.go create mode 100644 server/internal/backup/backup.go create mode 100644 server/internal/backup/backup_test.go create mode 100644 server/internal/backup/escrow.go create mode 100644 server/internal/backup/s3.go create mode 100644 server/internal/cluster_reconcile.go create mode 100644 server/internal/db/checkpoint_info.go create mode 100644 server/internal/offsite.go create mode 100644 server/internal/offsite_test.go create mode 100644 server/internal/restore.go create mode 100644 server/internal/store/canary.go create mode 100644 server/internal/store/canary_test.go diff --git a/deploy/aws/README.md b/deploy/aws/README.md new file mode 100644 index 0000000..e23826b --- /dev/null +++ b/deploy/aws/README.md @@ -0,0 +1,105 @@ +# unitdb's backups on AWS + +What a cluster's off-site backups need on AWS (docs/backup-restore.md), +made once per cluster by an AWS admin: a bucket with Object Lock, a +put-only identity for the cluster (an access key in a Kubernetes Secret +here; any of the AWS SDK's credential sources will do), and the backup key +and the escrowed keyring in Secrets Manager, readable only by the weekly +restore test's identity and named admins. A cluster per AWS account keeps +staging and production apart. + +Below, `CLUSTER` is the cluster's name (`staging`, `production`): its +prefix in the bucket and in the secrets' names. `BUCKET` is the bucket, +`REGION` its region, `ACCOUNT` the account id. Replace them in the JSON +files here before using them. + +## 1. The bucket: Object Lock, and a lifecycle by tier + +Object Lock can only be turned on when the bucket is made. There is no +default retention: each object is locked in compliance mode by the node +that uploads it, for as long as its tier is kept, so that daily runs can +expire after a week while monthly ones are kept 13 months (server/internal/backup: +daily 8 days, weekly 29, monthly and the journal 396). Compliance mode: not +even the account's root user deletes a locked version before its date. + +```sh +aws s3api create-bucket --bucket BUCKET --region REGION \ + --create-bucket-configuration LocationConstraint=REGION \ + --object-lock-enabled-for-bucket +aws s3api put-public-access-block --bucket BUCKET \ + --public-access-block-configuration BlockPublicAcls=true,IgnorePublicAcls=true,BlockPublicPolicy=true,RestrictPublicBuckets=true +aws s3api put-bucket-encryption --bucket BUCKET \ + --server-side-encryption-configuration '{"Rules":[{"ApplyServerSideEncryptionByDefault":{"SSEAlgorithm":"AES256"}}]}' +aws s3api put-bucket-lifecycle-configuration --bucket BUCKET --lifecycle-configuration file://lifecycle.json +``` + +The lifecycle rules (`lifecycle.json`) expire each object by its `tier` +tag once its lock has passed; the old versions go a day later. + +## 2. The cluster's identity: put only + +An IAM user, `unitdb-backup-writer-CLUSTER`, with `writer-policy.json` +only: it puts objects, with their lock and tag, under `CLUSTER/`, and can +neither read, list nor delete them. A compromised cluster can't read or +erase its backups. + +```sh +aws iam create-user --user-name unitdb-backup-writer-CLUSTER +aws iam put-user-policy --user-name unitdb-backup-writer-CLUSTER \ + --policy-name unitdb-backup-put-only --policy-document file://writer-policy.json +aws iam create-access-key --user-name unitdb-backup-writer-CLUSTER +kubectl -n unitdb create secret generic unitdb-backup-writer \ + --from-literal=AWS_ACCESS_KEY_ID= \ + --from-literal=AWS_SECRET_ACCESS_KEY= +``` + +Check the policy before the first run, with the IAM policy simulator or a +test upload: the nodes' uploads use exactly `s3:PutObject` (multipart +included), `s3:PutObjectRetention`, `s3:PutObjectTagging` and, when an +upload fails, `s3:AbortMultipartUpload`. The e2e tests run against MinIO as +its root user, so they don't check the policy. + +Rotate the key by making a second one, updating the Secret, restarting the +nodes, and deleting the first. + +## 3. The backup key + +Archives are encrypted with [age](https://age-encryption.org) to the backup +key's public half; the private half is only in Secrets Manager. + +```sh +age-keygen -o backup-key.txt # prints the public key: age1... +aws secretsmanager create-secret --region REGION \ + --name unitdb/CLUSTER/backup-key --secret-string file://backup-key.txt +shred -u backup-key.txt +``` + +The public half goes into the nodes' environment as `BACKUP_AGE_RECIPIENT` +(docs/backup-restore.md lists the nodes' settings). + +## 4. The keyring's escrow + +A backup restores only with the keyring of its time. Put the keyring in +escrow before every change of the cluster's `UNITDB_KEYRING`, with an +admin's identity (`escrow-policy.json`): + +```sh +BACKUP_CLUSTER=CLUSTER backup escrow-keyring -keyring keyring.json +``` + +It keeps every key it ever held: a key gone from the keyring stays as a +`read` key, and a key id with another key is refused. Only then update +the cluster's Secret. + +## 5. Readers + +`reader-policy.json` reads the bucket under `CLUSTER/` and the two secrets: +for the weekly restore test's identity and the named admins' roles, +nothing else. A restore (docs/backup-restore.md): + +```sh +export BACKUP_S3_BUCKET=BUCKET BACKUP_CLUSTER=CLUSTER +backup fetch -run -out restore/run # each node's checkpoint +backup fetch-journal -since -out restore/journal +# then each node: -db_path=restore/run/ -restored -journal=restore/journal +``` diff --git a/deploy/aws/escrow-policy.json b/deploy/aws/escrow-policy.json new file mode 100644 index 0000000..bb15f13 --- /dev/null +++ b/deploy/aws/escrow-policy.json @@ -0,0 +1,15 @@ +{ + "Version": "2012-10-17", + "Statement": [ + { + "Sid": "UnitdbKeyringEscrow", + "Effect": "Allow", + "Action": [ + "secretsmanager:GetSecretValue", + "secretsmanager:PutSecretValue", + "secretsmanager:CreateSecret" + ], + "Resource": "arn:aws:secretsmanager:REGION:ACCOUNT:secret:unitdb/CLUSTER/keyring-*" + } + ] +} diff --git a/deploy/aws/lifecycle.json b/deploy/aws/lifecycle.json new file mode 100644 index 0000000..d48a2cf --- /dev/null +++ b/deploy/aws/lifecycle.json @@ -0,0 +1,15 @@ +{ + "Rules": [ + {"ID": "daily", "Status": "Enabled", "Filter": {"Tag": {"Key": "tier", "Value": "daily"}}, + "Expiration": {"Days": 8}, "NoncurrentVersionExpiration": {"NoncurrentDays": 1}}, + {"ID": "weekly", "Status": "Enabled", "Filter": {"Tag": {"Key": "tier", "Value": "weekly"}}, + "Expiration": {"Days": 29}, "NoncurrentVersionExpiration": {"NoncurrentDays": 1}}, + {"ID": "monthly", "Status": "Enabled", "Filter": {"Tag": {"Key": "tier", "Value": "monthly"}}, + "Expiration": {"Days": 396}, "NoncurrentVersionExpiration": {"NoncurrentDays": 1}}, + {"ID": "journal", "Status": "Enabled", "Filter": {"Tag": {"Key": "tier", "Value": "journal"}}, + "Expiration": {"Days": 396}, "NoncurrentVersionExpiration": {"NoncurrentDays": 1}}, + {"ID": "delete-markers", "Status": "Enabled", "Filter": {}, + "Expiration": {"ExpiredObjectDeleteMarker": true}, + "AbortIncompleteMultipartUpload": {"DaysAfterInitiation": 2}} + ] +} diff --git a/deploy/aws/reader-policy.json b/deploy/aws/reader-policy.json new file mode 100644 index 0000000..892cade --- /dev/null +++ b/deploy/aws/reader-policy.json @@ -0,0 +1,27 @@ +{ + "Version": "2012-10-17", + "Statement": [ + { + "Sid": "UnitdbBackupRead", + "Effect": "Allow", + "Action": ["s3:GetObject", "s3:GetObjectVersion"], + "Resource": "arn:aws:s3:::BUCKET/CLUSTER/*" + }, + { + "Sid": "UnitdbBackupList", + "Effect": "Allow", + "Action": "s3:ListBucket", + "Resource": "arn:aws:s3:::BUCKET", + "Condition": {"StringLike": {"s3:prefix": ["CLUSTER/*"]}} + }, + { + "Sid": "UnitdbBackupKeys", + "Effect": "Allow", + "Action": "secretsmanager:GetSecretValue", + "Resource": [ + "arn:aws:secretsmanager:REGION:ACCOUNT:secret:unitdb/CLUSTER/backup-key-*", + "arn:aws:secretsmanager:REGION:ACCOUNT:secret:unitdb/CLUSTER/keyring-*" + ] + } + ] +} diff --git a/deploy/aws/writer-policy.json b/deploy/aws/writer-policy.json new file mode 100644 index 0000000..d0bc8c6 --- /dev/null +++ b/deploy/aws/writer-policy.json @@ -0,0 +1,16 @@ +{ + "Version": "2012-10-17", + "Statement": [ + { + "Sid": "UnitdbBackupPutOnly", + "Effect": "Allow", + "Action": [ + "s3:PutObject", + "s3:PutObjectRetention", + "s3:PutObjectTagging", + "s3:AbortMultipartUpload" + ], + "Resource": "arn:aws:s3:::BUCKET/CLUSTER/*" + } + ] +} diff --git a/deploy/kubernetes/backup-alerts.yaml b/deploy/kubernetes/backup-alerts.yaml new file mode 100644 index 0000000..db04733 --- /dev/null +++ b/deploy/kubernetes/backup-alerts.yaml @@ -0,0 +1,68 @@ +# Alerts on unitdb's backups (docs/backup-restore.md), for the Prometheus +# Operator. The metrics: each node's /_metrics on its monitor +# port (checkpoints, uploads, the security journal), the Pushgateway (the +# weekly restore test), and kube-state-metrics (the backup job). +apiVersion: monitoring.coreos.com/v1 +kind: PrometheusRule +metadata: + name: unitdb-backups + namespace: unitdb +spec: + groups: + - name: unitdb-backups + rules: + # A node's last checkpoint is more than a day old: the daily run + # (backups.yaml) didn't reach it. + - alert: UnitdbCheckpointStale + expr: time() - min by (pod) (unitdb_checkpoint_last_success_timestamp_seconds) > 26 * 3600 + for: 10m + labels: + severity: critical + annotations: + summary: "{{ $labels.pod }}: no checkpoint for over 26 h" + - alert: UnitdbCheckpointFailing + expr: increase(unitdb_checkpoint_failures_total[2h]) > 0 + labels: + severity: warning + annotations: + summary: "{{ $labels.pod }}: a checkpoint failed (see its logs, context checkpoint)" + # A node's last upload to the bucket is more than a day old. + - alert: UnitdbBackupUploadStale + expr: time() - min by (pod) (unitdb_backup_upload_last_success_timestamp_seconds) > 26 * 3600 + for: 10m + labels: + severity: critical + annotations: + summary: "{{ $labels.pod }}: no backup uploaded for over 26 h" + - alert: UnitdbBackupJobFailed + expr: kube_job_status_failed{namespace="unitdb", job_name=~"unitdb-backup-.*"} > 0 + labels: + severity: critical + annotations: + summary: "the backup run {{ $labels.job_name }} failed: a node's checkpoint or upload (see the Job's output, its manifest)" + # Security changes not in the bucket: a whole-cluster restore now + # would bring back what they revoked. + - alert: UnitdbSecurityJournalLagging + expr: max by (pod) (unitdb_security_journal_lag_seconds) > 300 + for: 5m + labels: + severity: critical + annotations: + summary: "{{ $labels.pod }}: security changes over 5 min old aren't in the bucket" + - alert: UnitdbSecurityJournalNotRecording + expr: increase(unitdb_security_journal_record_errors_total[10m]) > 0 + labels: + severity: critical + annotations: + summary: "{{ $labels.pod }}: a security change couldn't be written to the journal" + # The weekly restore test hasn't passed for over 8 days, or never + # has: this fires from deployment until its first pass. + - alert: UnitdbRestoreTestStale + expr: | + (time() - max by (cluster) (unitdb_backup_restore_verified_timestamp_seconds) > 8 * 86400) + or absent(unitdb_backup_restore_verified_timestamp_seconds) + for: 1h + labels: + severity: critical + annotations: + summary: "no backup has been proven to restore for over 8 days (restore-test.yaml)" diff --git a/deploy/kubernetes/backups.yaml b/deploy/kubernetes/backups.yaml new file mode 100644 index 0000000..3be809f --- /dev/null +++ b/deploy/kubernetes/backups.yaml @@ -0,0 +1,57 @@ +# unitdb's daily backup run (docs/backup-restore.md): at 21:30 UTC, the +# backup command (server/cmd/backup, in the unitdb image) has every +# node take a checkpoint of one run, one node after another, each with up to +# three tries, into checkpoints/ckpt-- on its volume, and keeps +# the run's manifest.json (each node's checkpoint.json, or why it failed) in +# every node's checkpoint. Then each node uploads its checkpoint, encrypted, +# and the manifest, to the bucket (-upload; deploy/aws/README.md). A node +# that failed either fails the job: a restore needs every node of one run. +# A volume snapshot of the nodes' claims, if you take one, goes after it: +# only the checkpoints in it restore, not the live store beside them. +# +# The job's exit status is its metric: alert on a failed Job, or on +# kube_cronjob_status_last_successful_time older than 26 h. +# +# Restoring: docs/backup-restore.md. One node lost: an empty claim, not a +# checkpoint. The whole cluster: every node at its checkpoint of one run. +apiVersion: batch/v1 +kind: CronJob +metadata: + name: unitdb-backup + namespace: unitdb +spec: + schedule: "30 21 * * *" + concurrencyPolicy: Forbid + successfulJobsHistoryLimit: 3 + failedJobsHistoryLimit: 3 + jobTemplate: + spec: + # The command retries each node itself; a second Job would start a new + # run. + backoffLimit: 0 + template: + spec: + restartPolicy: Never + # No Service links in the environment: a Service named web would set + # WEB_PORT=tcp://..., which unitdb reads as its own port. + enableServiceLinks: false + containers: + - name: backup + image: unitdb:latest + command: ["/backup"] + args: + - -upload + - -nodes + - http://unitdb-0.unitdb.unitdb.svc.cluster.local:7374,http://unitdb-1.unitdb.unitdb.svc.cluster.local:7374,http://unitdb-2.unitdb.unitdb.svc.cluster.local:7374 + env: + - name: CHECKPOINT_TOKEN + valueFrom: + secretKeyRef: + name: unitdb + key: CHECKPOINT_TOKEN + resources: + requests: + cpu: 10m + memory: 16Mi + limits: + memory: 64Mi diff --git a/deploy/kubernetes/restore-node-job.yaml b/deploy/kubernetes/restore-node-job.yaml new file mode 100644 index 0000000..7257c03 --- /dev/null +++ b/deploy/kubernetes/restore-node-job.yaml @@ -0,0 +1,60 @@ +# One node's part of a whole-cluster restore (docs/backup-restore.md, +# runbook B): on the node's new, empty claim, its checkpoint of the run, +# decrypted, where the node's store is (/var/udb/data/unitdb), and the +# security journal since the run (/var/udb/data/restore-journal). Then the +# node starts with -restored and -journal (runbook B, step 6). +# +# Per node: replace NODE (unitdb-0, unitdb-1, unitdb-2) and RUN (the run's +# id, from `backup` runs listed in the bucket), and apply: +# sed -e s/NODE/unitdb-0/g -e s/RUN/20261004T213000Z/g restore-node-job.yaml | kubectl apply -f - +# +# It runs as the user unitdb runs as: the files it writes are the node's +# store. If unitdb runs as another user than root, give the Job the same +# securityContext. +# +# It reads the bucket and the backup key, so it needs the reader's identity +# in the unitdb namespace for the restore only: the unitdb-backup-reader +# Secret, made at runbook B's step 3 and deleted at its end. +apiVersion: batch/v1 +kind: Job +metadata: + name: unitdb-restore-NODE + namespace: unitdb +spec: + backoffLimit: 0 + template: + spec: + restartPolicy: Never + enableServiceLinks: false + initContainers: + - name: checkpoint + image: unitdb:latest + command: ["/backup"] + args: ["fetch", "-run=RUN", "-node=NODE", "-out=/var/udb/data/unitdb"] + envFrom: + - secretRef: + name: unitdb-backup-reader + env: &restoreEnv + - name: BACKUP_S3_BUCKET + value: unitdb-backups-production + - name: BACKUP_S3_REGION + value: + - name: BACKUP_CLUSTER + value: production + volumeMounts: &restoreMounts + - name: data + mountPath: /var/udb/data + containers: + - name: journal + image: unitdb:latest + command: ["/backup"] + args: ["fetch-journal", "-since=RUN", "-out=/var/udb/data/restore-journal"] + envFrom: + - secretRef: + name: unitdb-backup-reader + env: *restoreEnv + volumeMounts: *restoreMounts + volumes: + - name: data + persistentVolumeClaim: + claimName: data-NODE diff --git a/deploy/kubernetes/restore-test.yaml b/deploy/kubernetes/restore-test.yaml new file mode 100644 index 0000000..c122b96 --- /dev/null +++ b/deploy/kubernetes/restore-test.yaml @@ -0,0 +1,80 @@ +# unitdb's weekly restore test (docs/backup-restore.md): the backup +# command's verify downloads the newest complete run from the bucket, +# decrypts it, and opens each node's checkpoint on a scratch server in this +# pod (no cluster, no clients), with the escrowed keyring: it must open and +# be ready, hold what the manifest says, and hold the run's canary. It +# pushes unitdb_backup_restore_verified_timestamp_seconds to the Pushgateway +# only if every node passes; backup-alerts.yaml fires when that is more than +# 8 days old. +# +# Its own namespace and identity: the reader's (deploy/aws/README.md, step +# 5), the only one besides named admins that may read the bucket, the +# backup key and the escrowed keyring. Nothing in the unitdb namespace can. +# +# Before applying: +# kubectl create namespace unitdb-restore-test +# kubectl -n unitdb-restore-test create secret generic unitdb-backup-reader \ +# --from-literal=AWS_ACCESS_KEY_ID= \ +# --from-literal=AWS_SECRET_ACCESS_KEY= +apiVersion: batch/v1 +kind: CronJob +metadata: + name: unitdb-restore-test + namespace: unitdb-restore-test +spec: + schedule: "30 23 * * 6" # Saturdays 23:30 UTC, after that day's run + concurrencyPolicy: Forbid + successfulJobsHistoryLimit: 3 + failedJobsHistoryLimit: 3 + jobTemplate: + spec: + backoffLimit: 0 + activeDeadlineSeconds: 21600 + template: + spec: + restartPolicy: Never + enableServiceLinks: false + containers: + - name: verify + image: unitdb:latest + command: ["/backup"] + args: + - verify + - -run=latest + - -server=/unitdb + - -work=/scratch + - -pushgateway=http://pushgateway.monitoring.svc.cluster.local:9091 + env: + - name: BACKUP_S3_BUCKET + value: unitdb-backups-production + - name: BACKUP_S3_REGION + value: + - name: BACKUP_CLUSTER + value: production + - name: AWS_ACCESS_KEY_ID + valueFrom: + secretKeyRef: + name: unitdb-backup-reader + key: AWS_ACCESS_KEY_ID + - name: AWS_SECRET_ACCESS_KEY + valueFrom: + secretKeyRef: + name: unitdb-backup-reader + key: AWS_SECRET_ACCESS_KEY + # One scratch server at a time; its memory store is the + # nodes' (mem_size, 500 MB). + resources: + requests: + cpu: 500m + memory: 1Gi + limits: + memory: 2Gi + volumeMounts: + - name: scratch + mountPath: /scratch + volumes: + # A run, decrypted: every node's checkpoint. Size it to three + # nodes' stores. + - name: scratch + emptyDir: + sizeLimit: 150Gi diff --git a/docs/README.md b/docs/README.md index 7d04fa4..bf8c035 100644 --- a/docs/README.md +++ b/docs/README.md @@ -8,5 +8,8 @@ changes. - [Message log replication](message-log-replication.md): replicas of stored messages and session logs, hints, and rebuilding a node that lost its disk. +- [Backup and restore](backup-restore.md): checkpoints, the backup run, copies + off the cluster, the security journal, reconciliation after a restore, the + weekly restore test, and the runbooks. - [Rolling deploys](rolling-deploys.md): upgrading a cluster node by node, and the maintenance-window upgrade from v0.3.0. diff --git a/docs/backup-restore.md b/docs/backup-restore.md new file mode 100644 index 0000000..76442ea --- /dev/null +++ b/docs/backup-restore.md @@ -0,0 +1,198 @@ +# Backup and restore + +How a unitdb cluster's data is backed up and restored: checkpoints, the +backup run and its manifest, copies off the cluster, the security journal, +reconciliation after a restore, the weekly restore test, and the runbooks. +Everything here is built; the deploy examples are in `deploy/kubernetes` and +`deploy/aws`. + +## What a backup is + +- **A checkpoint** is a copy of a node's store that opens as the store was + at one moment (`server/internal/db/unitdb/checkpoint.go`). A copy of a + running store's files isn't one: they are copied at different moments, + and the WAL isn't fsynced, so a volume snapshot is like a power cut. The + adapter holds writes back, for up to about 1.5 s, until the DB has written + out the latest; then it syncs, copies the DB's files, and copies memdb's + records (not its files) into a fresh memdb. Reads go on. Last it writes + `checkpoint.json`: the node, the backup run, the time, the ring and + engine versions, the keyring's key ids (never keys), and the copy's + counts. A checkpoint without it is incomplete. +- `POST /_checkpoint` on the monitor port (`monitor_listen`), with + `Authorization: Bearer `, takes one into `CHECKPOINT_DIR`; + off unless both are set. One runs at a time, one per + `CHECKPOINT_MIN_MINUTES` (10); the newest `CHECKPOINT_KEEP` (3) are kept. +- **A backup run** (`server/cmd/backup`, `deploy/kubernetes/backups.yaml`) + checkpoints every node under one run id (`POST /_checkpoint?run=`, + into `ckpt--`), one node after another, each tried three + times, and keeps the run's `manifest.json` (each node's + `checkpoint.json`, or why it failed) in every node's checkpoint + (`PUT /_checkpoint/manifest`). A node that failed fails the run: a + restore needs every node of one run. +- Just before its checkpoint of a run, a node writes **a canary**, the run's + id, into its store as a message and as a memdb record; the restore test + reads it back. + +## Copies off the cluster + +With `-upload`, the backup run has each node upload its own checkpoint (the +checkpoints are on each node's volume): `POST /_checkpoint/upload?run=` +streams it, archived (tar), compressed (zstd) and encrypted with +[age](https://age-encryption.org) to the backup key's public half, to S3, +with the manifest (`server/internal/backup`): + +``` +//manifest.json +//.tar.zst.age +/journal//-