diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index dda06cb9b3..53112eaa61 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -163,6 +163,15 @@ jobs: shellcheck scripts/verify-wedding-backup-denial.sh scripts/verify-wedding-backup-denial-pod.sh scripts/tests/test-verify-wedding-backup-denial.sh bash scripts/tests/test-verify-wedding-backup-denial.sh + # The stale shared copy of the Wedding catalogue is removed on this proof's + # word, and that cannot be undone. These cases keep a missing object, + # different bytes, a live writer or an unreviewed bucket from passing as + # covered, and pin the pod script to commands that only read. + - name: 💒 Validate Wedding shared backup coverage proof + run: | + shellcheck scripts/verify-wedding-shared-backup-coverage.sh scripts/verify-wedding-shared-backup-coverage-pod.sh scripts/tests/test-verify-wedding-shared-backup-coverage.sh + bash scripts/tests/test-verify-wedding-shared-backup-coverage.sh + - name: 💾 Validate two-stage PVC retirement # A merge-group artifact is speculative, but a PVC deletion is not: its # deletionTimestamp survives queue eviction and main cannot undo it. diff --git a/.github/workflows/verify-wedding-shared-backup-coverage.yaml b/.github/workflows/verify-wedding-shared-backup-coverage.yaml new file mode 100644 index 0000000000..ca07736c42 --- /dev/null +++ b/.github/workflows/verify-wedding-shared-backup-coverage.yaml @@ -0,0 +1,94 @@ +name: Verify Wedding Shared Backup Coverage + +# Prove that the dedicated Wedding backup bucket covers everything the stale copy +# of the catalogue in the shared backup bucket still holds (#4481). Removing +# that copy is irreversible, so the decision needs this evidence first. +# +# DORMANT. There is no schedule and no pull_request trigger, and the operator +# must type an exact confirmation phrase, so an accidental dispatch does nothing. +# +# MAIN ONLY. `workflow_dispatch` accepts any branch, and this job holds the +# production kubeconfig. The prod environment already refuses non-main refs; the +# explicit check below fails fast and names the reason. +# +# READ ONLY. The proof lists both buckets and reads the objects it has to hash. +# It never writes to or deletes from either bucket. +on: + workflow_dispatch: + inputs: + confirm: + description: "Type exactly: verify-wedding-shared-backup-coverage" + required: true + type: string + +permissions: {} + +# The proof reads production state and two backup credentials. Serialize with +# every other production mutation so a deploy cannot change the wiring mid-run. +concurrency: + group: prod-deploy + cancel-in-progress: false + queue: max + +jobs: + coverage: + name: 💒 Prove the dedicated bucket covers the shared Wedding catalogue + runs-on: ubuntu-latest + timeout-minutes: 90 + environment: prod + permissions: + contents: read # checkout repository + steps: + # Inputs reach bash through env, never `${{ }}` inside run, so a crafted + # value cannot inject shell. + - name: 🛑 Require main and explicit confirmation + shell: bash + env: + CONFIRM: ${{ inputs.confirm }} + DISPATCH_REF: ${{ github.ref }} + run: | + set -euo pipefail + if [[ "${DISPATCH_REF}" != 'refs/heads/main' ]]; then + printf 'Refusing to run from %s: dispatch this workflow from main.\n' "${DISPATCH_REF}" >&2 + exit 1 + fi + if [[ "${CONFIRM}" != 'verify-wedding-shared-backup-coverage' ]]; then + printf 'Refusing to run: the confirm input must be exactly "verify-wedding-shared-backup-coverage".\n' >&2 + printf 'The proof starts pods beside two production backup credentials. Nothing has been touched.\n' >&2 + exit 1 + fi + printf 'Confirmation accepted.\n' + + - name: 📑 Checkout + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + ref: ${{ github.sha }} + + - name: ⚙️ Setup Go + uses: actions/setup-go@b7ad1dad31e06c5925ef5d2fc7ad053ef454303e # v7.0.0 + with: + go-version-file: go.mod + cache: false + + - name: 🔑 Restore production kubeconfig + shell: bash + env: + KUBE_CONFIG: ${{ secrets.KUBE_CONFIG }} + run: | + set -euo pipefail + test -n "${KUBE_CONFIG}" + umask 077 + mkdir -p ~/.kube + printf '%s' "${KUBE_CONFIG}" >~/.kube/config + chmod 600 ~/.kube/config + + - name: 🎯 Select stable production API endpoint + shell: bash + env: + HCLOUD_TOKEN: ${{ secrets.HCLOUD_TOKEN }} + run: ./scripts/use-prod-stable-api-endpoint.sh + + - name: 💒 Compare the shared catalogue with the dedicated one + shell: bash + run: ./scripts/verify-wedding-shared-backup-coverage.sh --confirm diff --git a/docs/dr/velero-cnpg.md b/docs/dr/velero-cnpg.md index f24f2476b2..ab2491288a 100644 --- a/docs/dr/velero-cnpg.md +++ b/docs/dr/velero-cnpg.md @@ -46,7 +46,16 @@ The `Mirror Wedding Backup Catalogue` and `Verify Wedding Backup Denial` workflows both need the shared credential inside `wedding-app`, so neither can run any more; #4482 re-homes the denial proof and removes the finished cutover tooling, and #4481 removes the predecessor catalogue that still sits in the -shared bucket. `Verify Wedding Backup Cutover` +shared bucket. That copy is no longer written to or pruned, and every holder of +the shared credential can read it. Before it is removed, dispatch `Verify +Wedding Shared Backup Coverage` on `main` with +`confirm=verify-wedding-shared-backup-coverage`. It only reads: a pod beside the +shared credential in `umami` and one beside the dedicated credential in +`wedding-app` each list their side, and the verdict is `COVERED` only when every +object in the shared copy is in the dedicated bucket with matching content, or +was provably removed there by the 30-day retention. Remove the shared copy only +after a `COVERED` run, and run the proof again afterwards: it then reports +`EMPTY`. `Verify Wedding Backup Cutover` is a main-only, protected production dispatch: confirm `verify-wedding-backup-cutover` to request a fresh online primary backup. It refuses a shared archive, unhealthy database, changed database identity, or diff --git a/scripts/mirror-wedding-backup-catalogue/coverage.go b/scripts/mirror-wedding-backup-catalogue/coverage.go new file mode 100644 index 0000000000..c107b49597 --- /dev/null +++ b/scripts/mirror-wedding-backup-catalogue/coverage.go @@ -0,0 +1,303 @@ +package main + +// The coverage evaluation decides whether the dedicated Wedding bucket holds +// everything the stale shared copy of the catalogue holds (#4481). It only +// reads: it is the evidence a person needs before deciding to delete the shared +// copy, and it deletes nothing itself. +// +// Usage: +// +// mirror-wedding-backup-catalogue coverage-multipart-keys +// mirror-wedding-backup-catalogue evaluate-coverage +// +// The two listings come from two pods, because after #3253 no namespace holds +// both credentials. coverage-multipart-keys prints the shared keys whose content +// an ETag cannot prove on at least one side; the caller has each pod hash those +// keys and passes the results as "\t" lines. +// +// The dedicated store applies its retention policy and the shared copy no longer +// does, so the shared copy can hold objects the dedicated store has since +// pruned. Such an object is accepted only when the oldest complete base +// backup the dedicated store keeps in the same server directory predates the +// retention window, and the object was written before that backup started. +// Every other shared object must be present with matching content. + +import ( + "bufio" + "errors" + "fmt" + "io" + "os" + "sort" + "strings" + "time" +) + +var ( + ErrNotCovered = errors.New("shared object is not covered by the dedicated catalogue") + ErrMalformedSums = errors.New("malformed checksum list") + ErrStaleListing = errors.New("dedicated listing does not follow the shared listing") +) + +// CoverageSummary is the non-secret evidence a passing coverage evaluation reports. +type CoverageSummary struct { + SharedObjects int `json:"sharedObjects"` + MatchedObjects int `json:"matchedObjects"` + RetentionPrunedObjects int `json:"retentionPrunedObjects"` + Covered bool `json:"covered"` + SharedNewestBaseBackup string `json:"sharedNewestBaseBackup"` + DedicatedNewestBaseBackup string `json:"dedicatedNewestBaseBackup"` +} + +// MultipartKeys returns, sorted, every shared key whose content cannot be +// compared by ETag because it is multipart on either side. A key the dedicated +// catalogue does not hold is left out: there is nothing to compare it with. +func MultipartKeys(shared, dedicated []Object) ([]string, error) { + dedicatedIndex, err := index(dedicated) + if err != nil { + return nil, err + } + if _, err := index(shared); err != nil { + return nil, err + } + var keys []string + for _, object := range shared { + copied, ok := dedicatedIndex[object.Key] + if ok && (isMultipart(object.ETag) || isMultipart(copied.ETag)) { + keys = append(keys, object.Key) + } + } + sort.Strings(keys) + return keys, nil +} + +// ApplySums sets the SHA256 of every object named in a "\t" list. +// A key the listing does not hold, a repeated key or a malformed digest is +// refused, so a list produced for another listing cannot vouch for this one. +func ApplySums(objects []Object, r io.Reader) error { + positions := make(map[string]int, len(objects)) + for i, object := range objects { + positions[object.Key] = i + } + seen := map[string]bool{} + scanner := bufio.NewScanner(r) + scanner.Buffer(make([]byte, 0, 64*1024), 1024*1024) + for scanner.Scan() { + line := scanner.Text() + if line == "" { + continue + } + key, sum, ok := strings.Cut(line, "\t") + if !ok || !sha256Hex.MatchString(sum) { + return fmt.Errorf("%w: line %q", ErrMalformedSums, line) + } + position, listed := positions[key] + if !listed || seen[key] { + return fmt.Errorf("%w: unexpected key %s", ErrMalformedSums, key) + } + seen[key] = true + objects[position].SHA256 = sum + } + if err := scanner.Err(); err != nil { + return fmt.Errorf("%w: %w", ErrMalformedSums, err) + } + return nil +} + +// retentionWindow is the dedicated ObjectStore's retention policy. The runner +// refuses a store that declares another one. +const retentionWindow = 30 * 24 * time.Hour + +// emptyDigest is the sha256 of no bytes. A non-empty object that hashes to it +// was not read. +const emptyDigest = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + +// oldestCompleteBackups returns, per server directory, the start time of the +// oldest base backup that has both its backup.info and a non-empty data archive. +// A backup ID that is not a real time is refused: skipping it would move the +// floor later and accept more as pruned. +func oldestCompleteBackups(objects map[string]Object) (map[string]time.Time, error) { + hasInfo := map[string]bool{} + hasData := map[string]bool{} + for key, object := range objects { + id := baseBackupIDOf(key) + parts := strings.Split(key, "/") + if id == "" || len(parts) != 4 { + continue + } + backup := parts[0] + "/" + id + switch { + case parts[3] == "backup.info": + hasInfo[backup] = true + case dataArchive.MatchString(parts[3]) && object.Size > 0: + hasData[backup] = true + } + } + oldest := map[string]time.Time{} + for backup := range hasInfo { + if !hasData[backup] { + continue + } + server, id, _ := strings.Cut(backup, "/") + started, err := time.Parse("20060102T150405", id) + if err != nil { + return nil, fmt.Errorf("%w: backup ID %s", ErrMalformedListing, backup) + } + if current, ok := oldest[server]; !ok || started.Before(current) { + oldest[server] = started + } + } + return oldest, nil +} + +// baseBackupIDOf returns the backup ID of a key inside a base backup directory. +func baseBackupIDOf(key string) string { + parts := strings.Split(key, "/") + if len(parts) < 4 || parts[1] != "base" || !backupID.MatchString(parts[2]) { + return "" + } + return parts[2] +} + +// walPositionOf returns the segment a WAL-directory object belongs to: the +// segment itself, or the one a backup label or partial file is named after. A +// timeline history file has no position and is never pruned by retention. +func walPositionOf(key string) string { + if segment := walSegmentOf(key); segment != "" { + return segment + } + parts := strings.Split(key, "/") + if len(parts) != 4 || parts[1] != "wals" || len(parts[3]) < 25 || parts[3][24] != '.' { + return "" + } + segment := parts[3][:24] + if !walSegment.MatchString(segment) || parts[2] != segment[:16] { + return "" + } + return segment +} + +// clockMargin covers the difference between the database clock that stamps a +// backup ID and the object store clock that stamps an object. +const clockMargin = time.Hour + +// prunedByRetention reports whether a shared object the dedicated store lacks +// can only be missing because retention removed it. Retention keeps the newest +// backup older than the window and everything written since that backup +// started. So nothing can have been pruned from a server directory whose oldest +// complete dedicated backup is still inside the window: an older object missing +// there was never copied. Where that backup does predate the window, only an +// object written before it started can be gone. It is judged by when it was +// written, not by its name: segment names do not order across timelines. A +// server directory with no complete dedicated backup has nothing to measure +// against, so nothing in it is accepted. +func prunedByRetention(object Object, oldest map[string]time.Time, listed time.Time) bool { + if baseBackupIDOf(object.Key) == "" && walPositionOf(object.Key) == "" { + return false + } + started, ok := oldest[strings.Split(object.Key, "/")[0]] + return ok && !started.After(listed.Add(-retentionWindow)) && + object.LastModified.Before(started.Add(-clockMargin)) +} + +// EvaluateCoverage proves the dedicated catalogue covers the shared one. Every +// shared object must be in the dedicated catalogue with matching content, or be +// older than what the dedicated store's retention still keeps. The dedicated +// catalogue's newest complete base backup must be at least as new as the shared +// one's, so the shared copy never holds the most recent restore point. +func EvaluateCoverage(shared, dedicated Listing) (CoverageSummary, error) { + if shared.Started.IsZero() || dedicated.Started.IsZero() || + dedicated.Started.Before(shared.Started) { + return CoverageSummary{}, ErrStaleListing + } + sharedIndex, err := index(shared.Objects) + if err != nil { + return CoverageSummary{}, err + } + dedicatedIndex, err := index(dedicated.Objects) + if err != nil { + return CoverageSummary{}, err + } + if len(sharedIndex) == 0 { + return CoverageSummary{}, ErrEmptySource + } + + oldest, err := oldestCompleteBackups(dedicatedIndex) + if err != nil { + return CoverageSummary{}, err + } + matched, pruned := 0, 0 + for key, source := range sharedIndex { + copied, ok := dedicatedIndex[key] + if !ok { + if !prunedByRetention(source, oldest, dedicated.Started) { + return CoverageSummary{}, fmt.Errorf("%w: %s", ErrNotCovered, key) + } + pruned++ + continue + } + if copied.Size != source.Size { + return CoverageSummary{}, fmt.Errorf("%w: %s", ErrPartialCopy, key) + } + if source.Size > 0 && (source.SHA256 == emptyDigest || copied.SHA256 == emptyDigest) { + return CoverageSummary{}, fmt.Errorf("%w: %s was hashed as empty", ErrUnverifiable, key) + } + if err := sameContent(source, copied); err != nil { + return CoverageSummary{}, fmt.Errorf("%w: %s", err, key) + } + matched++ + } + + sharedBackup := newestBaseBackup(sharedIndex) + dedicatedBackup := newestBaseBackup(dedicatedIndex) + if dedicatedBackup == "" { + return CoverageSummary{}, ErrNoBaseBackup + } + _, sharedID, _ := strings.Cut(sharedBackup, "/") + _, dedicatedID, _ := strings.Cut(dedicatedBackup, "/") + if dedicatedID < sharedID { + return CoverageSummary{}, fmt.Errorf("%w: the newest dedicated base backup %s is older than the shared %s", + ErrNotCovered, dedicatedBackup, sharedBackup) + } + + return CoverageSummary{ + SharedObjects: len(sharedIndex), + MatchedObjects: matched, + RetentionPrunedObjects: pruned, + Covered: true, + SharedNewestBaseBackup: sharedBackup, + DedicatedNewestBaseBackup: dedicatedBackup, + }, nil +} + +// readCoverageListings reads the shared and dedicated listings for a coverage evaluation. +func readCoverageListings(sharedBucket, sharedName, dedicatedName string) (Listing, Listing, error) { + if sharedBucket == "" || sharedBucket == destinationBucket { + return Listing{}, Listing{}, fmt.Errorf("%w: shared bucket %q", ErrWrongSource, sharedBucket) + } + shared, err := readListing(sharedName, sharedBucket+"/"+sourcePrefix) + if err != nil { + return Listing{}, Listing{}, fmt.Errorf("%s: %w", sharedName, err) + } + dedicated, err := readListing(dedicatedName, destinationBucket+"/"+destinationPrefix) + if err != nil { + return Listing{}, Listing{}, fmt.Errorf("%s: %w", dedicatedName, err) + } + return shared, dedicated, nil +} + +// applySumsFile applies one checksum list named on the command line. +func applySumsFile(objects []Object, name string) error { + file, err := os.Open(name) + if err != nil { + return err + } + applyErr := ApplySums(objects, file) + if closeErr := file.Close(); closeErr != nil && applyErr == nil { + return closeErr + } + if applyErr != nil { + return fmt.Errorf("%s: %w", name, applyErr) + } + return nil +} diff --git a/scripts/mirror-wedding-backup-catalogue/coverage_test.go b/scripts/mirror-wedding-backup-catalogue/coverage_test.go new file mode 100644 index 0000000000..92b9d078d0 --- /dev/null +++ b/scripts/mirror-wedding-backup-catalogue/coverage_test.go @@ -0,0 +1,252 @@ +package main + +import ( + "bytes" + "encoding/json" + "errors" + "os" + "path/filepath" + "strings" + "testing" + "time" +) + +const ( + oldServerInfo = "wedding-db/base/20260813T030000/backup.info" + oldServerData = "wedding-db/base/20260813T030000/data.tar.gz" + oldServerWAL = "wedding-db/wals/0000000100000000/000000010000000000000007.gz" + + prunedInfo = "wedding-db-20260909/base/20260901T030000/backup.info" + prunedData = "wedding-db-20260909/base/20260901T030000/data.tar.gz" + prunedWAL = "wedding-db-20260909/wals/0000000200000000/0000000200000000000000FE.gz" + prunedLabel = "wedding-db-20260909/wals/0000000200000000/0000000200000000000000FE.00000028.backup" + history = "wedding-db-20260909/wals/00000003.history" + postSwitch = "wedding-db-20260909/wals/0000000300000001/000000030000000100000009.gz" +) + +var ( + sharedListed = time.Date(2026, 10, 9, 9, 0, 0, 0, time.UTC) + dedicatedListed = sharedListed.Add(time.Minute) + // Written before the oldest dedicated base backup started and before the + // retention window began. + longAgo = time.Date(2026, 9, 2, 0, 0, 0, 0, time.UTC) +) + +func sharedCatalogue() Listing { + return Listing{Started: sharedListed, Objects: sourceListing()} +} + +// The dedicated store keeps archiving after the switch, so it holds more than +// the shared copy. +func dedicatedCatalogue() Listing { + objects := append(destinationListing(), + Object{Key: postSwitch, Size: 4300, ETag: "f6", LastModified: sharedListed}) + return Listing{Started: dedicatedListed, Objects: objects} +} + +func TestEvaluateCoverageProvesTheDedicatedCatalogueCoversTheSharedOne(t *testing.T) { + summary, err := EvaluateCoverage(sharedCatalogue(), dedicatedCatalogue()) + if err != nil { + t.Fatalf("EvaluateCoverage() = %v, want nil", err) + } + if summary.SharedObjects != 4 || summary.MatchedObjects != 4 || + summary.RetentionPrunedObjects != 0 || !summary.Covered { + t.Fatalf("summary = %+v, want 4 shared, 4 matched, 0 pruned, covered", summary) + } +} + +// Retention runs on the dedicated store only, so the shared copy outlives +// backups the dedicated store has pruned. Those are older than everything the +// dedicated store keeps and are not a reason to keep the shared copy. +func TestEvaluateCoverageAcceptsObjectsTheDedicatedStorePrunedByRetention(t *testing.T) { + shared := sharedCatalogue() + shared.Objects = append(shared.Objects, + Object{Key: prunedInfo, Size: 1100, ETag: "p1", LastModified: longAgo}, + Object{Key: prunedData, Size: 80_000_000, ETag: "p2-5", LastModified: longAgo}, + Object{Key: prunedWAL, Size: 3900, ETag: "p3", LastModified: longAgo}, + Object{Key: prunedLabel, Size: 300, ETag: "p4", LastModified: longAgo}, + ) + summary, err := EvaluateCoverage(shared, dedicatedCatalogue()) + if err != nil { + t.Fatalf("EvaluateCoverage() = %v, want nil", err) + } + if summary.MatchedObjects != 4 || summary.RetentionPrunedObjects != 4 { + t.Fatalf("summary = %+v, want 4 matched and 4 pruned", summary) + } +} + +func TestEvaluateCoverageRefusals(t *testing.T) { + tests := []struct { + name string + mutate func(shared, dedicated *Listing) + want error + }{ + {"an object the dedicated store never held", func(shared, _ *Listing) { + shared.Objects = append(shared.Objects, Object{Key: lateWAL, Size: 4200, ETag: "e5", LastModified: beforeRun}) + }, ErrNotCovered}, + {"a missing object inside the retained range", func(_, dedicated *Listing) { + dedicated.Objects = dedicated.Objects[:3] + }, ErrNotCovered}, + {"a server directory the dedicated store holds nothing of", func(shared, _ *Listing) { + shared.Objects = append(shared.Objects, + Object{Key: oldServerInfo, Size: 900, ETag: "o1", LastModified: longAgo}, + Object{Key: oldServerData, Size: 900, ETag: "o2", LastModified: longAgo}, + Object{Key: oldServerWAL, Size: 900, ETag: "o3", LastModified: longAgo}) + }, ErrNotCovered}, + {"a timeline history file is never pruned", func(shared, _ *Listing) { + shared.Objects = append(shared.Objects, Object{Key: history, Size: 80, ETag: "h1", LastModified: longAgo}) + }, ErrNotCovered}, + {"WAL archived after the oldest dedicated backup started", func(shared, _ *Listing) { + shared.Objects = append(shared.Objects, Object{Key: prunedWAL, Size: 3900, ETag: "p3", LastModified: beforeRun}) + }, ErrNotCovered}, + {"an old backup missing while the oldest dedicated backup is inside the window", func(shared, dedicated *Listing) { + shared.Started = time.Date(2026, 10, 5, 9, 0, 0, 0, time.UTC) + dedicated.Started = shared.Started.Add(time.Minute) + shared.Objects = append(shared.Objects, + Object{Key: prunedInfo, Size: 1100, ETag: "p1", LastModified: longAgo}, + Object{Key: prunedData, Size: 80_000_000, ETag: "p2-5", LastModified: longAgo}) + }, ErrNotCovered}, + {"an object written within the clock margin of the oldest dedicated backup", func(shared, _ *Listing) { + shared.Objects = append(shared.Objects, + Object{Key: prunedWAL, Size: 3900, ETag: "p3", LastModified: time.Date(2026, 9, 8, 2, 30, 0, 0, time.UTC)}) + }, ErrNotCovered}, + {"a dedicated backup ID that is not a real time", func(_, dedicated *Listing) { + dedicated.Objects = append(dedicated.Objects, + Object{Key: "wedding-db-20260909/base/20260231T030000/backup.info", Size: 1, ETag: "x1", LastModified: beforeRun}, + Object{Key: "wedding-db-20260909/base/20260231T030000/data.tar.gz", Size: 1, ETag: "x2", LastModified: beforeRun}) + }, ErrMalformedListing}, + {"a non-empty object hashed as empty", func(shared, dedicated *Listing) { + shared.Objects[1].SHA256 = emptyDigest + dedicated.Objects[1].SHA256 = emptyDigest + }, ErrUnverifiable}, + {"a different size", func(_, dedicated *Listing) { + dedicated.Objects[0].Size++ + }, ErrPartialCopy}, + {"a different single-part checksum", func(_, dedicated *Listing) { + dedicated.Objects[0].ETag = "zz" + }, ErrChecksumMismatch}, + {"a different content digest", func(_, dedicated *Listing) { + dedicated.Objects[1].SHA256 = otherDigest + }, ErrChecksumMismatch}, + {"a multipart object nobody hashed", func(shared, _ *Listing) { + shared.Objects[1].SHA256 = "" + }, ErrUnverifiable}, + {"an empty shared listing", func(shared, _ *Listing) { + shared.Objects = nil + }, ErrEmptySource}, + {"a dedicated listing taken before the shared one", func(_, dedicated *Listing) { + dedicated.Started = sharedListed.Add(-time.Second) + }, ErrStaleListing}, + {"a dedicated listing with no start time", func(_, dedicated *Listing) { + dedicated.Started = time.Time{} + }, ErrStaleListing}, + {"a shared listing with no start time", func(shared, _ *Listing) { + shared.Started = time.Time{} + }, ErrStaleListing}, + {"a repeated shared key", func(shared, _ *Listing) { + shared.Objects = append(shared.Objects, shared.Objects[0]) + }, ErrMalformedListing}, + {"a dedicated catalogue with no complete base backup", func(shared, dedicated *Listing) { + shared.Objects = shared.Objects[2:] + dedicated.Objects = dedicated.Objects[2:] + }, ErrNoBaseBackup}, + {"a dedicated catalogue whose newest base backup is older", func(shared, _ *Listing) { + shared.Objects = append(shared.Objects, + Object{Key: "wedding-db-20260909/base/20260920T030000/backup.info", Size: 1, ETag: "n1", LastModified: beforeRun}, + Object{Key: "wedding-db-20260909/base/20260920T030000/data.tar.gz", Size: 1, ETag: "n2", LastModified: beforeRun}) + }, ErrNotCovered}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + shared, dedicated := sharedCatalogue(), dedicatedCatalogue() + tt.mutate(&shared, &dedicated) + if _, err := EvaluateCoverage(shared, dedicated); !errors.Is(err, tt.want) { + t.Fatalf("EvaluateCoverage() = %v, want %v", err, tt.want) + } + }) + } +} + +func TestMultipartKeysNamesSharedKeysAnETagCannotProve(t *testing.T) { + shared := []Object{ + {Key: baseInfo, ETag: "a1"}, + {Key: baseData, ETag: "b2-6"}, + {Key: olderWAL, ETag: "c3"}, + {Key: newerWAL, ETag: "d4-2"}, + } + dedicated := []Object{ + {Key: baseInfo, ETag: "a1"}, + {Key: baseData, ETag: "ff"}, + {Key: olderWAL, ETag: "c3-2"}, + {Key: postSwitch, ETag: "g7-2"}, + } + keys, err := MultipartKeys(shared, dedicated) + if err != nil { + t.Fatalf("MultipartKeys() = %v", err) + } + if got, want := strings.Join(keys, ","), baseData+","+olderWAL; got != want { + t.Fatalf("MultipartKeys() = %q, want %q", got, want) + } +} + +func TestApplySums(t *testing.T) { + objects := sourceListing() + objects[1].SHA256 = "" + if err := ApplySums(objects, strings.NewReader(baseData+"\t"+dataDigest+"\n\n")); err != nil { + t.Fatalf("ApplySums() = %v, want nil", err) + } + if objects[1].SHA256 != dataDigest { + t.Fatalf("SHA256 = %q, want %q", objects[1].SHA256, dataDigest) + } + for name, sums := range map[string]string{ + "an unlisted key": lateWAL + "\t" + dataDigest + "\n", + "a repeated key": baseData + "\t" + dataDigest + "\n" + baseData + "\t" + dataDigest + "\n", + "a malformed digest": baseData + "\tnot-a-digest\n", + "no separator": baseData + "\n", + } { + t.Run(name, func(t *testing.T) { + if err := ApplySums(sourceListing(), strings.NewReader(sums)); !errors.Is(err, ErrMalformedSums) { + t.Fatalf("ApplySums() = %v, want %v", err, ErrMalformedSums) + } + }) + } +} + +func TestRunEvaluateCoverageEndToEnd(t *testing.T) { + keys := []string{baseInfo, baseData, olderWAL} + shared := writeListing(t, "shared", sourceLocation, "2026-10-05T09:00:00Z", keys...) + dedicated := writeListing(t, "dedicated", destinationLocation, "2026-10-05T09:01:00Z", keys...) + empty := filepath.Join(t.TempDir(), "empty") + if err := os.WriteFile(empty, nil, 0o600); err != nil { + t.Fatal(err) + } + + var keysOut bytes.Buffer + if err := run([]string{"coverage-multipart-keys", "platform-backups", shared, dedicated}, &keysOut); err != nil { + t.Fatalf("run(coverage-multipart-keys) = %v, want nil", err) + } + + var out bytes.Buffer + if err := run([]string{"evaluate-coverage", "platform-backups", shared, dedicated, empty, empty}, &out); err != nil { + t.Fatalf("run(evaluate-coverage) = %v, want nil", err) + } + var summary CoverageSummary + if err := json.Unmarshal(out.Bytes(), &summary); err != nil { + t.Fatalf("summary is not JSON: %v", err) + } + if !summary.Covered || summary.SharedObjects != len(keys) { + t.Fatalf("summary = %+v", summary) + } + + // The listings are bound to their buckets, so swapping them, or naming the + // dedicated bucket as the shared one, is refused. + if err := run([]string{"evaluate-coverage", "platform-backups", dedicated, shared, empty, empty}, &out); !errors.Is(err, ErrListingLocation) { + t.Fatalf("run(evaluate-coverage, swapped) = %v, want %v", err, ErrListingLocation) + } + if err := run([]string{"evaluate-coverage", destinationBucket, shared, dedicated, empty, empty}, &out); !errors.Is(err, ErrWrongSource) { + t.Fatalf("run(evaluate-coverage, dedicated as shared) = %v, want %v", err, ErrWrongSource) + } + if err := run([]string{"evaluate-coverage", "platform-backups", shared, dedicated, empty, filepath.Join(t.TempDir(), "absent")}, &out); err == nil { + t.Fatal("run(evaluate-coverage, missing sums) = nil, want an error") + } +} diff --git a/scripts/mirror-wedding-backup-catalogue/main.go b/scripts/mirror-wedding-backup-catalogue/main.go index d2260dd7b2..40f1b0e8d6 100644 --- a/scripts/mirror-wedding-backup-catalogue/main.go +++ b/scripts/mirror-wedding-backup-catalogue/main.go @@ -829,7 +829,40 @@ func run(args []string, stdout io.Writer) error { } return json.NewEncoder(stdout).Encode(summary) } - return errors.New("usage: mirror-wedding-backup-catalogue validate-plan | evaluate | evaluate-catch-up | validate-switch-time ") + if len(args) == 4 && args[0] == "coverage-multipart-keys" { + shared, dedicated, err := readCoverageListings(args[1], args[2], args[3]) + if err != nil { + return err + } + keys, err := MultipartKeys(shared.Objects, dedicated.Objects) + if err != nil { + return err + } + for _, key := range keys { + if _, err := fmt.Fprintln(stdout, key); err != nil { + return err + } + } + return nil + } + if len(args) == 6 && args[0] == "evaluate-coverage" { + shared, dedicated, err := readCoverageListings(args[1], args[2], args[3]) + if err != nil { + return err + } + if err := applySumsFile(shared.Objects, args[4]); err != nil { + return err + } + if err := applySumsFile(dedicated.Objects, args[5]); err != nil { + return err + } + summary, err := EvaluateCoverage(shared, dedicated) + if err != nil { + return err + } + return json.NewEncoder(stdout).Encode(summary) + } + return errors.New("usage: mirror-wedding-backup-catalogue coverage-multipart-keys | evaluate-coverage | validate-plan | evaluate | evaluate-catch-up | validate-switch-time ") } // main exits non-zero with the refusal reason when a plan or mirror is not trusted. diff --git a/scripts/tests/test-verify-wedding-shared-backup-coverage.sh b/scripts/tests/test-verify-wedding-shared-backup-coverage.sh new file mode 100755 index 0000000000..8b690ec88e --- /dev/null +++ b/scripts/tests/test-verify-wedding-shared-backup-coverage.sh @@ -0,0 +1,560 @@ +#!/usr/bin/env bash + +# Pin the behaviour of the shared Wedding backup coverage proof: +# scripts/verify-wedding-shared-backup-coverage.sh (runs on the runner) and +# scripts/verify-wedding-shared-backup-coverage-pod.sh (runs in the cluster). +# +# WHY THIS EXISTS. The stale shared copy of the Wedding catalogue is removed on +# this proof's word (#4481), and that removal cannot be undone. Its dangerous +# mistakes are quiet ones: +# +# * reporting COVERED when the dedicated bucket lacks an object, holds different +# bytes, or was listed before the shared one; +# * reporting COVERED while something can still write to the shared copy; +# * handing a credential to an endpoint or bucket nobody reviewed; +# * writing to or deleting from either bucket. +# +# Every case runs the REAL runner, the REAL pod script and the REAL evaluator +# against a fake kubectl and a fake mc, so the three are proven to agree on the +# listing and checksum formats. Needs Go and jq, no cluster and no credentials. + +set -euo pipefail + +root_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +readonly root_dir +readonly runner="${root_dir}/scripts/verify-wedding-shared-backup-coverage.sh" +readonly pod_script="${root_dir}/scripts/verify-wedding-shared-backup-coverage-pod.sh" +readonly mirror="${root_dir}/scripts/mirror-wedding-backup-catalogue.sh" + +work_dir="$(mktemp -d)" +readonly work_dir +# shellcheck disable=SC2317,SC2329 # Invoked indirectly by the EXIT trap. +cleanup() { + rm -rf "${work_dir}" +} +trap cleanup EXIT + +command -v go >/dev/null 2>&1 || { + printf 'FAIL: go is required to build the real evaluator\n' >&2 + exit 1 +} +readonly evaluator="${work_dir}/evaluator" +(cd "${root_dir}" && go build -o "${evaluator}" ./scripts/mirror-wedding-backup-catalogue) + +readonly bin="${work_dir}/bin" +mkdir -p "${bin}" + +cases_run=0 + +fail() { + printf '\nFAIL: %s\n' "$1" >&2 + exit 1 +} + +require_text() { + printf '%s' "$1" | grep -qF -- "$2" || fail "$3: expected '$2'. Got: $1" +} + +refute_text() { + if printf '%s' "$1" | grep -qF -- "$2"; then + fail "$3: did not expect '$2'. Got: $1" + fi +} + +# --------------------------------------------------------------------------- +# Fake mc. Only `alias set`, `ls` and `cat` are served. Anything else is recorded +# as forbidden, and every case asserts nothing forbidden was attempted. Every +# failure writes the endpoint host to stderr, so redaction is exercised. +# --------------------------------------------------------------------------- +cat >"${bin}/mc" <<'FAKE' +#!/bin/sh +set -u +f="${FAKE_MC}" +leak() { + printf 'mc: request to https://abc123.r2.cloudflarestorage.com/x failed\n' >&2 + printf 'mc: dial tcp: lookup abc123.r2.cloudflarestorage.com on 10.96.0.10:53: no such host\n' >&2 +} +case "$1" in + alias) + if [ "$2" != set ]; then printf '%s\n' "$*" >>"${f}/forbidden"; exit 64; fi + printf '%s %s\n' "$3" "$5" >>"${f}/aliases" + exit 0 + ;; + ls) + printf '%s\n' "$4" >>"${f}/listed" + if [ -e "${f}/ls-rc" ]; then leak; exit "$(cat "${f}/ls-rc")"; fi + cat "${f}/ls" + exit 0 + ;; + cat) + printf '%s\n' "$2" >>"${f}/catted" + if [ -e "${f}/cat-rc" ]; then leak; exit "$(cat "${f}/cat-rc")"; fi + if [ -e "${f}/content" ]; then cat "${f}/content"; else printf 'bytes of %s' "${2##*/}"; fi + exit 0 + ;; +esac +printf '%s\n' "$*" >>"${f}/forbidden" +exit 64 +FAKE +chmod +x "${bin}/mc" + +# --------------------------------------------------------------------------- +# Fake kubectl. `exec` runs the real pod script with the environment the runner +# wrote into that namespace's pod manifest, so the manifest wiring is part of +# what each case proves. +# --------------------------------------------------------------------------- +cat >"${bin}/kubectl" <<'FAKE' +#!/usr/bin/env bash +set -u +k="${FAKE_K}" +namespace='' +args=() +while [ "$#" -gt 0 ]; do + case "$1" in + --context) shift 2 ;; + --namespace) namespace="$2"; shift 2 ;; + --request-timeout=*) shift ;; + *) args+=("$1"); shift ;; + esac +done +set -- "${args[@]}" +printf '%s %s\n' "${namespace:-all}" "$*" >>"${k}/calls" + +manifest_env() { + awk -v want="$2" '$0 ~ "- name: " want "$" { getline; gsub(/^ *value: *"?|"?$/, ""); print; exit }' "${k}/pod-$1.yaml" +} + +case "$1" in + get) + case "$2" in + pod) + cat "${k}/phase-${namespace}" 2>/dev/null || printf 'Running' + ;; + clusters.postgresql.cnpg.io | objectstores.barmancloud.cnpg.io | externalsecrets.external-secrets.io | secrets) + if [ "$3" = --all-namespaces ]; then + cat "${k}/all-$2.json" || exit 1 + elif [ "${4:-}" = --ignore-not-found ]; then + if [ -e "${k}/absence-rc" ]; then exit "$(cat "${k}/absence-rc")"; fi + if [ -e "${k}/present-$2-$3" ]; then printf '%s/%s\n' "$2" "$3"; fi + else + cat "${k}/${namespace}-$2-$3.json" 2>/dev/null || exit 1 + fi + ;; + *) exit 64 ;; + esac + ;; + create) + if [ "$2" = configmap ]; then + exit 0 + fi + cat >"${k}/pod-${namespace}.yaml" + ;; + logs) + if [ ! -e "${k}/never-ready-${namespace}" ]; then printf '==== COVERAGE POD READY ====\n'; fi + ;; + delete) + printf '%s %s\n' "${namespace}" "$2" >>"${k}/deleted" + ;; + exec) + while [ "$1" != -- ]; do shift; done + shift + if [ "$1" = cat ]; then + cat "${k}/work-${namespace}/${2#/work/}" + exit $? + fi + [ "$1 $2" = '/tools/sh /coverage/coverage.sh' ] || exit 64 + PATH="${FAKE_BIN}:${PATH}" FAKE_MC="${k}/mc-${namespace}" \ + ENDPOINT="$(manifest_env "${namespace}" ENDPOINT)" ROLE="$(manifest_env "${namespace}" ROLE)" \ + BUCKET="$(manifest_env "${namespace}" BUCKET)" PREFIX="$(manifest_env "${namespace}" PREFIX)" \ + WORK_DIR="${k}/work-${namespace}" CREDENTIALS_DIR="${k}/credentials" \ + sh "${POD_SCRIPT}" "$3" + ;; + *) exit 64 ;; +esac +FAKE +chmod +x "${bin}/kubectl" + +readonly endpoint='https://abc123.r2.cloudflarestorage.com' +readonly old='2026-09-01T00:00:00Z' +# The oldest dedicated backup predates the 30-day retention window, as it does +# once retention has started pruning. +readonly base='wedding-db-20260909/base/20260808T030000' +readonly wal='wedding-db-20260909/wals/0000000200000001/000000020000000100000003.gz' +readonly later_wal='wedding-db-20260909/wals/0000000300000001/000000030000000100000009.gz' + +# record [last-modified] +record() { + printf '{"status":"success","type":"file","lastModified":"%s","size":%s,"key":"%s","etag":"%s","url":"%s","versionOrdinal":1,"storageClass":"STANDARD"}\n' \ + "${4:-${old}}" "$2" "$1" "$3" "${endpoint}" +} + +# store_json [endpoint] [retention] +store_json() { + printf '{"metadata":{},"spec":{"retentionPolicy":"%s","configuration":{"destinationPath":"%s","endpointURL":"%s","s3Credentials":{"accessKeyId":{"name":"%s","key":"ACCESS_KEY_ID"},"secretAccessKey":{"name":"%s","key":"SECRET_ACCESS_KEY"}}}}}\n' \ + "${4:-30d}" "$1" "${3:-${endpoint}}" "$2" "$2" +} + +# cluster_json +cluster_json() { + printf '{"metadata":{"uid":"cluster-uid-1","generation":3},"spec":{"instances":1,"plugins":[{"name":"barman-cloud.cloudnative-pg.io","enabled":true,"isWALArchiver":true,"parameters":{"barmanObjectName":"%s"}}]},"status":{"readyInstances":1,"conditions":[{"type":"Ready","status":"True"},{"type":"ContinuousArchiving","status":"True"}]}}\n' "$1" +} + +# new_case prepares a covered scenario; each case then breaks one thing. +new_case() { + cases_run=$((cases_run + 1)) + k="${work_dir}/case-${cases_run}" + mkdir -p "${k}/mc-umami" "${k}/mc-wedding-app" "${k}/work-umami" "${k}/work-wedding-app" "${k}/credentials" + printf 'data:\n r2_endpoint: %s\n r2_bucket: platform-backups\n' "${endpoint}" >"${k}/config-map.yaml" + printf 'AKIDEXAMPLE' >"${k}/credentials/ACCESS_KEY_ID" + printf 'secret-value' >"${k}/credentials/SECRET_ACCESS_KEY" + cluster_json wedding-db-dedicated >"${k}/wedding-app-clusters.postgresql.cnpg.io-wedding-db.json" + store_json s3://wedding-db-backups/cnpg/wedding-db wedding-db-backup-r2-dedicated \ + >"${k}/wedding-app-objectstores.barmancloud.cnpg.io-wedding-db-dedicated.json" + store_json s3://platform-backups/cnpg/umami-db umami-db-backup-r2 \ + >"${k}/umami-objectstores.barmancloud.cnpg.io-umami-db.json" + printf '{"items":[{"spec":{"configuration":{"destinationPath":"s3://platform-backups/cnpg/umami-db"}}},{"spec":{"configuration":{"destinationPath":"s3://wedding-db-backups/cnpg/wedding-db"}}}]}\n' \ + >"${k}/all-objectstores.barmancloud.cnpg.io.json" + printf '{"items":[{"spec":{"instances":1}}]}\n' >"${k}/all-clusters.postgresql.cnpg.io.json" + { + record "${base}/backup.info" 1200 a1 + record "${base}/data.tar.gz" 90000000 b2-6 + record "${wal}" 4000 c3 + } >"${k}/mc-umami/ls" + # The mirror uploaded the archive in a different number of parts, and the + # dedicated store has archived more since the switch. + { + record "${base}/backup.info" 1200 a1 + record "${base}/data.tar.gz" 90000000 ff-4 + record "${wal}" 4000 c3 + record "${later_wal}" 4300 f6 + } >"${k}/mc-wedding-app/ls" +} + +# run_case [runner arguments...] sets out, err and rc. +run_case() { + rc=0 + FAKE_K="${k}" FAKE_BIN="${bin}" POD_SCRIPT="${pod_script}" KUBECTL="${bin}/kubectl" \ + COVERAGE_EVALUATOR="${evaluator}" COVERAGE_BOOTSTRAP_CONFIG="${k}/config-map.yaml" \ + COVERAGE_POLL_INTERVAL=0 COVERAGE_POLL_LIMIT=2 GITHUB_RUN_ID=77 GITHUB_RUN_ATTEMPT=1 \ + bash "${runner}" "$@" >"${k}/out" 2>"${k}/err" || rc=$? + out="$(cat "${k}/out")" + err="$(cat "${k}/err")" + for side in umami wedding-app; do + [[ ! -e "${k}/mc-${side}/forbidden" ]] || + fail "case ${cases_run}: the pod ran an mc command other than alias set, ls or cat: $(cat "${k}/mc-${side}/forbidden")" + done +} + +# require_untouched asserts a refusal happened before anything was created. +require_untouched() { + [[ "${rc}" -eq 1 ]] || fail "$1: expected exit 1, got ${rc}. stderr: ${err}" + if grep -q ' create ' "${k}/calls" 2>/dev/null; then + fail "$1: something was created before the refusal" + fi + refute_text "${out}" COVERED "$1" +} + +# require_cleaned asserts both pods and both scripts were removed. +require_cleaned() { + for entry in 'umami pod' 'umami configmap' 'wedding-app pod' 'wedding-app configmap'; do + grep -qxF "${entry}" "${k}/deleted" || fail "$1: ${entry} was not removed" + done +} + +# --------------------------------------------------------------------------- +# Covered. +# --------------------------------------------------------------------------- +new_case +run_case --confirm +[[ "${rc}" -eq 0 ]] || fail "covered: expected exit 0, got ${rc}. stderr: ${err}" +require_text "${out}" 'COVERED: of the 3 objects in the shared Wedding catalogue, the dedicated bucket holds 3 with matching content.' 'covered' +require_text "${out}" '"sharedObjects":3,"matchedObjects":3,"retentionPrunedObjects":0,"covered":true' 'covered' +require_cleaned 'covered' +# One credential per pod, each beside its own namespace's Secret. +require_text "$(cat "${k}/pod-umami.yaml")" 'secretName: umami-db-backup-r2' 'shared pod credential' +refute_text "$(cat "${k}/pod-umami.yaml")" 'wedding-db-backup-r2' 'shared pod credential' +require_text "$(cat "${k}/pod-wedding-app.yaml")" 'secretName: wedding-db-backup-r2-dedicated' 'dedicated pod credential' +refute_text "$(cat "${k}/pod-wedding-app.yaml")" 'umami-db-backup-r2' 'dedicated pod credential' +[[ "$(grep -c 'secretName:' "${k}/pod-umami.yaml")" -eq 1 && "$(grep -c 'secretName:' "${k}/pod-wedding-app.yaml")" -eq 1 ]] || + fail 'a pod mounts more than one credential' +# Each pod lists only the Wedding catalogue of its own bucket. +[[ "$(cat "${k}/mc-umami/listed")" == 'store/platform-backups/cnpg/wedding-db/' ]] || fail 'the shared pod listed something else' +[[ "$(cat "${k}/mc-wedding-app/listed")" == 'store/wedding-db-backups/cnpg/wedding-db/' ]] || fail 'the dedicated pod listed something else' +# Only the object an ETag cannot prove is read, on both sides. +[[ "$(cat "${k}/mc-umami/catted")" == "store/platform-backups/cnpg/wedding-db/${base}/data.tar.gz" ]] || fail 'the shared pod hashed the wrong objects' +[[ "$(cat "${k}/mc-wedding-app/catted")" == "store/wedding-db-backups/cnpg/wedding-db/${base}/data.tar.gz" ]] || fail 'the dedicated pod hashed the wrong objects' +refute_text "${out}${err}" 'abc123' 'covered: endpoint host leaked' +printf 'PASS: a covered catalogue is reported as covered and both pods are removed\n' + +# Retention runs on the dedicated store only, so the shared copy outlives backups +# the dedicated store has pruned. +new_case +{ + record 'wedding-db-20260909/base/20260720T030000/backup.info' 1100 p1 2026-07-20T03:10:00Z + record 'wedding-db-20260909/base/20260720T030000/data.tar.gz' 80000000 p2-5 2026-07-20T03:10:00Z + record 'wedding-db-20260909/wals/0000000200000000/0000000200000000000000FE.gz' 3900 p3 2026-07-21T00:00:00Z +} >>"${k}/mc-umami/ls" +run_case --confirm +[[ "${rc}" -eq 0 ]] || fail "retention: expected exit 0, got ${rc}. stderr: ${err}" +require_text "${out}" '"sharedObjects":6,"matchedObjects":3,"retentionPrunedObjects":3,"covered":true' 'retention' +require_text "${out}" 'The other 3 were written before the oldest backup the dedicated store keeps' 'retention' +printf 'PASS: objects the dedicated store pruned by retention do not block the verdict\n' + +# A segment only the shared copy holds, archived after the oldest backup the +# dedicated store keeps, is needed to restore that backup: it was never copied. +new_case +printf '{"status":"success","type":"file","lastModified":"2026-09-20T00:00:00Z","size":3900,"key":"wedding-db-20260909/wals/0000000200000000/0000000200000000000000FE.gz","etag":"p3"}\n' >>"${k}/mc-umami/ls" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "uncopied WAL: expected exit 1, got ${rc}" +require_text "${err}" 'NOT COVERED' 'uncopied WAL' +printf 'PASS: WAL the dedicated store never received is not mistaken for pruned WAL\n' + +# After the shared copy is removed, the same proof reports that. +new_case +: >"${k}/mc-umami/ls" +run_case --confirm +[[ "${rc}" -eq 0 ]] || fail "empty: expected exit 0, got ${rc}. stderr: ${err}" +require_text "${out}" 'EMPTY: the shared bucket holds no Wedding catalogue' 'empty' +refute_text "${out}" 'COVERED' 'empty' +require_cleaned 'empty' +printf 'PASS: an empty shared prefix is reported as empty, not as covered\n' + +# --------------------------------------------------------------------------- +# Not covered. +# --------------------------------------------------------------------------- +new_case +record "${later_wal}" 4300 f6 2026-09-20T00:00:00Z >>"${k}/mc-umami/ls" +{ + record "${base}/backup.info" 1200 a1 + record "${base}/data.tar.gz" 90000000 ff-4 + record "${wal}" 4000 c3 +} >"${k}/mc-wedding-app/ls" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "missing object: expected exit 1, got ${rc}" +require_text "${err}" 'NOT COVERED' 'missing object' +refute_text "${out}" 'COVERED:' 'missing object' +require_cleaned 'missing object' +printf 'PASS: a shared object the dedicated bucket lacks is not covered\n' + +new_case +printf 'different bytes' >"${k}/mc-wedding-app/content" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "different bytes: expected exit 1, got ${rc}" +require_text "${err}" 'checksum mismatch' 'different bytes' +require_text "${err}" 'NOT COVERED' 'different bytes' +printf 'PASS: a multipart object with different bytes is not covered\n' + +new_case +{ + record "${base}/backup.info" 1200 zz + record "${base}/data.tar.gz" 90000000 ff-4 + record "${wal}" 4000 c3 +} >"${k}/mc-wedding-app/ls" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "different etag: expected exit 1, got ${rc}" +require_text "${err}" 'NOT COVERED' 'different etag' +printf 'PASS: a single-part object with a different ETag is not covered\n' + +new_case +printf '1' >"${k}/mc-wedding-app/cat-rc" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "unreadable object: expected exit 1, got ${rc}" +require_text "${err}" 'the dedicated pod could not hash its objects' 'unreadable object' +refute_text "${out}${err}" 'abc123' 'unreadable object: endpoint host leaked' +refute_text "${err}" '10.96.0.10' 'unreadable object: address leaked' +require_text "${err}" '' 'unreadable object' +printf 'PASS: an object that cannot be read is not covered, and the endpoint host is redacted\n' + +new_case +printf '1' >"${k}/mc-umami/ls-rc" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "failed listing: expected exit 1, got ${rc}" +require_text "${err}" 'the shared pod could not list the shared catalogue' 'failed listing' +refute_text "${out}" 'EMPTY' 'failed listing' +require_cleaned 'failed listing' +printf 'PASS: a failed listing is not an empty catalogue\n' + +new_case +printf '{"status":"error","error":{"message":"denied"}}\n' >>"${k}/mc-umami/ls" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "error record: expected exit 1, got ${rc}" +require_text "${err}" 'returned a record that is not a success' 'error record' +printf 'PASS: a listing carrying an error record is refused\n' + +new_case +touch "${k}/never-ready-umami" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "never ready: expected exit 1, got ${rc}" +require_text "${err}" 'the shared pod never became ready' 'never ready' +require_cleaned 'never ready' +printf 'PASS: a pod that never becomes ready fails the proof and is removed\n' + +# --------------------------------------------------------------------------- +# Refusals before anything is created. +# --------------------------------------------------------------------------- +new_case +run_case +require_untouched 'no confirmation' +require_text "${err}" 'refusing to run: use --confirm' 'no confirmation' +[[ ! -e "${k}/calls" ]] || fail 'no confirmation: the cluster was contacted' + +new_case +run_case --confirm --delete +require_untouched 'extra argument' + +new_case +cluster_json wedding-db >"${k}/wedding-app-clusters.postgresql.cnpg.io-wedding-db.json" +run_case --confirm +require_untouched 'still archiving to the shared store' +require_text "${err}" 'does not archive through wedding-db-dedicated' 'still archiving to the shared store' + +for present in objectstores.barmancloud.cnpg.io-wedding-db externalsecrets.external-secrets.io-wedding-db-backup-r2 secrets-wedding-db-backup-r2; do + new_case + touch "${k}/present-${present}" + run_case --confirm + require_untouched "wedding-app still holds ${present}" + require_text "${err}" 'shared backup access has not been retired' "wedding-app still holds ${present}" +done + +# A failed read is not an absence. +new_case +printf '1' >"${k}/absence-rc" +run_case --confirm +require_untouched 'absence check failed' +require_text "${err}" 'could not check whether wedding-app still holds' 'absence check failed' + +new_case +printf '{"items":[{"spec":{"configuration":{"destinationPath":"s3://platform-backups/cnpg/wedding-db"}}}]}\n' \ + >"${k}/all-objectstores.barmancloud.cnpg.io.json" +run_case --confirm +require_untouched 'a store still names the shared catalogue' +require_text "${err}" 'still names the shared Wedding catalogue' 'a store still names the shared catalogue' + +new_case +printf '{"items":[{"spec":{"externalClusters":[{"barmanObjectStore":{"destinationPath":"s3://platform-backups/cnpg/wedding-db/"}}]}}]}\n' \ + >"${k}/all-clusters.postgresql.cnpg.io.json" +run_case --confirm +require_untouched 'a Cluster still names the shared catalogue' + +# A sibling catalogue whose name merely starts the same way is not a reference. +new_case +printf '{"items":[{"spec":{"configuration":{"destinationPath":"s3://platform-backups/cnpg/wedding-db-archive"}}}]}\n' \ + >"${k}/all-objectstores.barmancloud.cnpg.io.json" +run_case --confirm +[[ "${rc}" -eq 0 ]] || fail "sibling catalogue: expected exit 0, got ${rc}. stderr: ${err}" + +new_case +rm "${k}/all-clusters.postgresql.cnpg.io.json" +run_case --confirm +require_untouched 'reference listing failed' +require_text "${err}" 'could not list every clusters.postgresql.cnpg.io' 'reference listing failed' + +new_case +store_json s3://wedding-db-backups/cnpg/other wedding-db-backup-r2-dedicated \ + >"${k}/wedding-app-objectstores.barmancloud.cnpg.io-wedding-db-dedicated.json" +run_case --confirm +require_untouched 'dedicated store drifted' + +new_case +store_json s3://other-bucket/cnpg/umami-db umami-db-backup-r2 >"${k}/umami-objectstores.barmancloud.cnpg.io-umami-db.json" +run_case --confirm +require_untouched 'shared reference store names another bucket' + +new_case +store_json s3://wedding-db-backups/cnpg/wedding-db wedding-db-backup-r2-dedicated "" 7d \ + >"${k}/wedding-app-objectstores.barmancloud.cnpg.io-wedding-db-dedicated.json" +run_case --confirm +require_untouched 'dedicated store keeps another retention' +require_text "${err}" 'does not keep the reviewed 30-day retention' 'dedicated store keeps another retention' + +# A store rooted above the catalogue can write into it without naming it. +new_case +printf '{"items":[{"spec":{"configuration":{"destinationPath":"s3://platform-backups/cnpg/"}}}]}\n' \ + >"${k}/all-objectstores.barmancloud.cnpg.io.json" +run_case --confirm +require_untouched 'a store is rooted above the shared catalogue, with a trailing slash' + +new_case +printf '{"items":[{"spec":{"configuration":{"destinationPath":"s3://platform-backups/cnpg"}}}]}\n' \ + >"${k}/all-objectstores.barmancloud.cnpg.io.json" +run_case --confirm +require_untouched 'a store is rooted above the shared catalogue' + +new_case +store_json s3://platform-backups/cnpg/umami-db other-secret >"${k}/umami-objectstores.barmancloud.cnpg.io-umami-db.json" +run_case --confirm +require_untouched 'shared reference store names another Secret' + +new_case +store_json s3://platform-backups/cnpg/umami-db umami-db-backup-r2 https://evil.example.com \ + >"${k}/umami-objectstores.barmancloud.cnpg.io-umami-db.json" +run_case --confirm +require_untouched 'shared reference store names another endpoint' + +new_case +printf 'data:\n r2_endpoint: %s\n r2_bucket: wedding-db-backups\n' "${endpoint}" >"${k}/config-map.yaml" +run_case --confirm +require_untouched 'committed shared bucket is the dedicated bucket' + +new_case +printf 'data:\n r2_endpoint: http://plain.example.com\n r2_bucket: platform-backups\n' >"${k}/config-map.yaml" +run_case --confirm +require_untouched 'committed endpoint is not https' +printf 'PASS: every refusal happens before a pod or script is created\n' + +# --------------------------------------------------------------------------- +# The pod script on its own. +# --------------------------------------------------------------------------- +# pod sets pod_out and pod_rc. +pod() { + pod_rc=0 + pod_out="$(PATH="${bin}:${PATH}" FAKE_MC="${k}/mc-umami" ENDPOINT="${POD_ENDPOINT:-${endpoint}}" \ + ROLE="$1" BUCKET="$2" PREFIX="$3" WORK_DIR="${k}/work-umami" CREDENTIALS_DIR="${k}/credentials" \ + SERVE_TIMEOUT=0 sh "${pod_script}" "$4" 2>&1 &2 + exit 1 +} + +readonly reviewed_prefix='cnpg/wedding-db' +readonly dedicated_bucket='wedding-db-backups' + +for tool in mc sed grep tail sha256sum cut date cat sleep mkdir rm; do + command -v "${tool}" >/dev/null 2>&1 || fail "required tool missing from the image: ${tool}" +done + +: "${ENDPOINT:?ENDPOINT is required}" +: "${ROLE:?ROLE is required}" +: "${BUCKET:?BUCKET is required}" +: "${PREFIX:?PREFIX is required}" + +case "${ENDPOINT}" in + https://*) ;; + *) fail "the R2 endpoint must be https" ;; +esac +[ "${PREFIX}" = "${reviewed_prefix}" ] || fail "the catalogue prefix is not the reviewed one" +case "${ROLE}" in + shared) + [ "${BUCKET}" != "${dedicated_bucket}" ] || fail "the shared side names the dedicated bucket" + ;; + dedicated) + [ "${BUCKET}" = "${dedicated_bucket}" ] || fail "the dedicated side names another bucket" + ;; + *) fail "the role must be shared or dedicated" ;; +esac + +credentials="${CREDENTIALS_DIR:-/credentials}" +work="${WORK_DIR:-/tmp/coverage}" +mkdir -p "${work}" +export MC_CONFIG_DIR="${MC_CONFIG_DIR:-${work}/.mc}" +mc_err="${work}/mc.err" + +# redact_errors prints the last lines of an mc error log without the endpoint +# host (with or without its scheme), any IPv4 address, or any long hex token such +# as an access key ID or request ID. These lines reach a public workflow log. +host="${ENDPOINT#https://}" +redact_errors() { + sed -e 's#https\{0,1\}://[^/ "`]*##g' -e "s#${host}##g" \ + -e 's/[0-9a-fA-F]\{32,\}//g' \ + -e 's/[0-9]\{1,3\}\(\.[0-9]\{1,3\}\)\{3\}\(:[0-9]\{1,5\}\)\{0,1\}//g' "${mc_err}" | tail -n 5 >&2 +} + +command="${1:-}" +case "${command}" in + serve) + [ "$#" -eq 1 ] || fail "serve takes no arguments" + access_key="$(cat "${credentials}/ACCESS_KEY_ID")" + [ -n "${access_key}" ] || fail "the credential is empty" + if ! mc alias set store "${ENDPOINT}" "${access_key}" \ + "$(cat "${credentials}/SECRET_ACCESS_KEY")" --api S3v4 >/dev/null 2>"${mc_err}"; then + redact_errors + fail "could not configure the ${ROLE} credential" + fi + unset access_key + printf '==== COVERAGE POD READY ====\n' + waited=0 + while [ "${waited}" -lt "${SERVE_TIMEOUT:-3000}" ]; do + sleep 5 + waited=$((waited + 5)) + done + ;; + + list) + [ "$#" -eq 1 ] || fail "list takes no arguments" + output="${work}/listing" + # Truncating to the second can only move the start earlier, which is the + # side the evaluator's ordering check is strict about. + started="$(date -u +%Y-%m-%dT%H:%M:%SZ)" + if ! mc ls --json --recursive "store/${BUCKET}/${PREFIX}/" >"${output}.raw" 2>"${mc_err}" "${output}.partial" + files=0 + while IFS= read -r line; do + [ -n "${line}" ] || continue + case "${line}" in + *'"status":"success"'*) ;; + *) fail "listing the ${ROLE} catalogue returned a record that is not a success" ;; + esac + case "${line}" in + *'"type":"folder"'*) continue ;; + *'"type":"file"'*) ;; + *) fail "listing the ${ROLE} catalogue returned an unexpected record type" ;; + esac + + line="$(printf '%s' "${line}" | sed -e 's/"url":"[^"]*",//' -e 's/,"url":"[^"]*"//')" + key="$(printf '%s' "${line}" | sed -n 's/.*"key":"\([^"\\]*\)".*/\1/p')" + [ -n "${key}" ] || fail "listing the ${ROLE} catalogue returned an object without a plain key" + + printf '%s\n' "${line}" >>"${output}.partial" + files=$((files + 1)) + done <"${output}.raw" + + printf '{"status":"success","type":"listing-complete","location":"%s/%s","started":"%s","files":%d}\n' \ + "${BUCKET}" "${PREFIX}" "${started}" "${files}" >>"${output}.partial" + cat "${output}.partial" >"${output}" + printf 'coverage-pod: listed %s objects on the %s side\n' "${files}" "${ROLE}" + ;; + + hash) + [ "$#" -eq 1 ] || fail "hash takes no arguments" + [ -e "${work}/listing" ] || fail "hash needs a listing" + : >"${work}/sums.partial" + hashed=0 + while IFS= read -r key; do + [ -n "${key}" ] || continue + grep -qF "\"key\":\"${key}\"" "${work}/listing" || fail "asked to hash a key the listing does not hold" + rm -f "${work}/cat-failed" + sum="$({ mc cat "store/${BUCKET}/${PREFIX}/${key}" "${mc_err}" || + : >"${work}/cat-failed"; } | sha256sum | cut -d ' ' -f 1)" + if [ -e "${work}/cat-failed" ]; then + redact_errors + fail "could not read ${key} to checksum it" + fi + printf '%s' "${sum}" | grep -Eq '^[0-9a-f]{64}$' || fail "malformed checksum for ${key}" + printf '%s\t%s\n' "${key}" "${sum}" >>"${work}/sums.partial" + hashed=$((hashed + 1)) + done + cat "${work}/sums.partial" >"${work}/sums" + printf 'coverage-pod: hashed %s objects on the %s side\n' "${hashed}" "${ROLE}" + ;; + + *) fail "unknown command" ;; +esac diff --git a/scripts/verify-wedding-shared-backup-coverage.sh b/scripts/verify-wedding-shared-backup-coverage.sh new file mode 100755 index 0000000000..aa07767030 --- /dev/null +++ b/scripts/verify-wedding-shared-backup-coverage.sh @@ -0,0 +1,447 @@ +#!/usr/bin/env bash + +# Prove that the dedicated Wedding backup bucket covers everything the stale +# copy of the catalogue in the shared backup bucket still holds (#4481). +# +# Until the cutover the Wedding database archived into the shared backup bucket. +# That copy was mirrored into the dedicated bucket, and #3253 removed the store +# that pointed at it, so nothing applies retention to it any more and every +# holder of the shared backup credential can still read it. Removing it is +# irreversible, so the decision needs current evidence that nothing would be +# lost. This is that evidence, and it is all this does. +# +# READ ONLY. It lists both buckets and reads the objects it has to hash. It +# never writes to or deletes from either bucket. +# +# WHAT IT DOES, in order, refusing at the first thing it cannot prove: +# 1. Confirms the wedding-db Cluster archives through wedding-db-dedicated with +# a healthy WAL archiver, that the wedding-app namespace no longer holds the +# shared store or its credential, and that no ObjectStore or Cluster anywhere +# still names the shared catalogue. +# 2. Reads the dedicated ObjectStore and the Umami ObjectStore, whose Secret is +# the shared backup credential, and requires the reviewed destinations at the +# committed endpoint and bucket. +# 3. Starts one pod beside each credential (scripts/verify-wedding-shared-backup- +# coverage-pod.sh). No namespace holds both since #3253. The shared pod lists +# first; the dedicated pod lists afterwards. +# 4. Has both pods hash every object an ETag cannot prove, and runs the reviewed +# evaluator (`evaluate-coverage`): every shared object must be in the +# dedicated catalogue with matching content, or be provably removed there by +# the dedicated store's 30-day retention. +# 5. Repeats step 1. +# +# Exit status: 0 covered; 1 refused, failed or not covered. Needs --confirm, +# because it starts pods beside two production backup credentials. + +set -euo pipefail + +readonly context='admin@prod' +readonly tenant_namespace='wedding-app' +readonly shared_namespace='umami' +readonly cluster='wedding-db' +readonly retired_store='wedding-db' +readonly retired_secret='wedding-db-backup-r2' +readonly dedicated_store='wedding-db-dedicated' +readonly dedicated_secret='wedding-db-backup-r2-dedicated' +readonly dedicated_bucket='wedding-db-backups' +readonly shared_reference_store='umami-db' +readonly shared_reference_prefix='cnpg/umami-db' +readonly shared_secret='umami-db-backup-r2' +readonly catalogue_prefix='cnpg/wedding-db' +readonly plugin='barman-cloud.cloudnative-pg.io' +readonly ready_marker='==== COVERAGE POD READY ====' +# The same digest-pinned images as the catalogue mirror, whose runtime test +# exercises them; the coverage test asserts the two scripts stay identical here. +readonly mc_image='quay.io/minio/aistor/mc:RELEASE.2026-03-12T04-18-55Z@sha256:6c33dc0fbf65c362be95003cd010ed95a41c556500833ea139f86de40c4c4e9f' +readonly tools_image='docker.io/library/busybox:1.38.0-musl@sha256:ea2b9914a16a4ac1981994af97b318f7c7d4db76b580c56177f08bf76f4a0be8' + +root_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +readonly root_dir +readonly pod_script="${root_dir}/scripts/verify-wedding-shared-backup-coverage-pod.sh" + +kubectl_bin="${KUBECTL:-kubectl}" +readonly kubectl_bin +poll_interval="${COVERAGE_POLL_INTERVAL:-10}" +poll_limit="${COVERAGE_POLL_LIMIT:-60}" +readonly poll_interval poll_limit + +work_dir="$(mktemp -d)" +readonly work_dir +# Cleanup removes only what this run created, never a same-named object it found. +created=() +name='' +# Invoked by the EXIT trap below. shellcheck reports this trap-only handler as +# unused (SC2329 on 0.11) or unreachable (SC2317 on the older CI runner); the +# test suite asserts the deletions actually happen. +# shellcheck disable=SC2317,SC2329 +cleanup() { + local entry + for entry in ${created[@]+"${created[@]}"}; do + kube "${entry%%/*}" delete "${entry#*/}" "${name}" --ignore-not-found --wait=false >/dev/null 2>&1 || true + done + rm -rf "${work_dir}" +} +trap cleanup EXIT + +fail() { + printf 'verify-wedding-shared-backup-coverage: %s\n' "$1" >&2 + exit 1 +} + +[[ "$#" -eq 1 && "$1" == '--confirm' ]] || + fail 'refusing to run: use --confirm. The proof starts pods beside two production backup credentials. Nothing has been touched.' + +# A local run is named by its start time, so it never shares a name with an +# earlier run's pod. +run_id="${GITHUB_RUN_ID:-local$(date +%s)}-${GITHUB_RUN_ATTEMPT:-1}" +[[ "${run_id}" =~ ^[a-z0-9-]{1,40}$ ]] || fail 'the run identifier is not a valid name fragment' +name="wedding-backup-coverage-${run_id}" +readonly name + +# kube +kube() { + local namespace="$1" + shift + "${kubectl_bin}" --context "${context}" --namespace "${namespace}" --request-timeout=20s "$@" +} + +evaluator="${COVERAGE_EVALUATOR:-}" +if [[ -z "${evaluator}" ]]; then + evaluator="${work_dir}/evaluator" + (cd "${root_dir}" && go build -o "${evaluator}" ./scripts/mirror-wedding-backup-catalogue) || + fail 'could not build the evaluator' +fi +readonly evaluator + +# The committed bootstrap ConfigMap names the shared bucket and the R2 endpoint. +# A live value is not trusted on its own: a drifted store could name a foreign +# host or bucket, and a credential is about to be handed to that endpoint. +bootstrap_config="${COVERAGE_BOOTSTRAP_CONFIG:-${root_dir}/k8s/bases/bootstrap/config-map.yaml}" +committed() { + local values + values="$(sed -n "s/^ $1: *//p" "${bootstrap_config}")" || fail "could not read the committed $1" + [[ -n "${values}" && "${values}" != *$'\n'* ]] || fail "the bootstrap ConfigMap does not commit exactly one $1" + printf '%s' "${values}" +} +endpoint="$(committed r2_endpoint)" +shared_bucket="$(committed r2_bucket)" +readonly endpoint shared_bucket +[[ "${endpoint}" =~ ^https://[a-z0-9][a-z0-9.-]*$ ]] || + fail 'the committed R2 endpoint is not an https URL with a bare host' +[[ "${shared_bucket}" =~ ^[a-z0-9][a-z0-9.-]*$ ]] || + fail 'the committed shared backup bucket is not a valid bucket name' +[[ "${shared_bucket}" != "${dedicated_bucket}" ]] || + fail 'the committed shared backup bucket is the dedicated bucket' +readonly shared_catalogue="s3://${shared_bucket}/${catalogue_prefix}" + +# require_cluster refuses unless the Cluster archives through the dedicated store +# with a healthy active WAL archiver, and is the same Cluster as on the first call. +cluster_uid='' +require_cluster() { + local uid + kube "${tenant_namespace}" get clusters.postgresql.cnpg.io "${cluster}" -o json >"${work_dir}/cluster.json" 2>/dev/null || + fail "could not read the ${cluster} Cluster" + jq -e --arg plugin "${plugin}" --arg store "${dedicated_store}" ' + .metadata.generation as $generation | + def healthy($kind): + [.status.conditions[]? | select(.type == $kind)] | + length == 1 and .[0].status == "True" and + (.[0].observedGeneration == null or .[0].observedGeneration == $generation); + .metadata.deletionTimestamp == null and + (.metadata.uid | type == "string" and length > 0) and + (.spec.instances | type == "number" and . > 0 and floor == .) and + .status.readyInstances == .spec.instances and + ([.spec.plugins[]? | select(.name == $plugin)] | + length == 1 and .[0].enabled == true and .[0].isWALArchiver == true and + .[0].parameters.barmanObjectName == $store) and + healthy("Ready") and healthy("ContinuousArchiving") + ' "${work_dir}/cluster.json" >/dev/null 2>&1 || + fail "the ${cluster} Cluster does not archive through ${dedicated_store} with a healthy active WAL archiver" + uid="$(jq -r '.metadata.uid' "${work_dir}/cluster.json")" + if [[ -z "${cluster_uid}" ]]; then + cluster_uid="${uid}" + elif [[ "${uid}" != "${cluster_uid}" ]]; then + fail "the ${cluster} Cluster was replaced during the run" + fi +} + +# require_absent refuses unless the wedding-app namespace is +# observed not to hold the object. A failed read is not an absence. +require_absent() { + local found + found="$(kube "${tenant_namespace}" get "$1" "$2" --ignore-not-found --output=name 2>/dev/null)" || + fail "could not check whether ${tenant_namespace} still holds $1/$2" + [[ -z "${found}" ]] || + fail "${tenant_namespace} still holds $1/$2: shared backup access has not been retired (#3253)" +} + +# require_unreferenced refuses when any ObjectStore or Cluster in the cluster +# still names the shared catalogue or a directory above it, in any field: something could then still be +# writing to it, and the coverage shown here would not last. +require_unreferenced() { + local kind + for kind in objectstores.barmancloud.cnpg.io clusters.postgresql.cnpg.io; do + "${kubectl_bin}" --context "${context}" --request-timeout=20s get "${kind}" --all-namespaces -o json \ + >"${work_dir}/references.json" 2>/dev/null || + fail "could not list every ${kind}" + jq -e --arg path "${shared_catalogue}" ' + (.items | type == "array") and + ([.items[].spec | .. | strings | select(. == $path or startswith($path + "/") or (rtrimstr("/") as $root | $path | startswith($root + "/")))] | length == 0) + ' "${work_dir}/references.json" >/dev/null 2>&1 || + fail "a ${kind} still names the shared Wedding catalogue" + done +} + +require_retired() { + require_cluster + require_absent objectstores.barmancloud.cnpg.io "${retired_store}" + require_absent externalsecrets.external-secrets.io "${retired_secret}" + require_absent secrets "${retired_secret}" + require_unreferenced +} + +# require_store refuses unless the +# live ObjectStore writes to the reviewed destination through exactly the +# reviewed Secret keys the pod mounts, at the committed endpoint. +require_store() { + local namespace="$1" store="$2" destination="$3" secret="$4" + kube "${namespace}" get objectstores.barmancloud.cnpg.io "${store}" -o json >"${work_dir}/store.json" 2>/dev/null || + fail "could not read the ${store} ObjectStore" + jq -e --arg path "${destination}" --arg endpoint "${endpoint}" --arg secret "${secret}" ' + .metadata.deletionTimestamp == null and + .spec.configuration.destinationPath == $path and + .spec.configuration.endpointURL == $endpoint and + .spec.configuration.s3Credentials.accessKeyId == {name:$secret,key:"ACCESS_KEY_ID"} and + .spec.configuration.s3Credentials.secretAccessKey == {name:$secret,key:"SECRET_ACCESS_KEY"} + ' "${work_dir}/store.json" >/dev/null 2>&1 || + fail "the ${store} ObjectStore is not wired to the reviewed destination and credential" +} + +require_retired +require_store "${tenant_namespace}" "${dedicated_store}" "s3://${dedicated_bucket}/${catalogue_prefix}" "${dedicated_secret}" +# The evaluator accepts a shared object the dedicated store lacks only when it +# is older than this retention window, so the live store must declare the same one. +jq -e '.spec.retentionPolicy == "30d"' "${work_dir}/store.json" >/dev/null 2>&1 || + fail "the ${dedicated_store} ObjectStore does not keep the reviewed 30-day retention" +# The Umami store proves which Secret holds the credential for the committed +# shared bucket. Its own catalogue is a sibling of the Wedding one and is never +# listed. +require_store "${shared_namespace}" "${shared_reference_store}" "s3://${shared_bucket}/${shared_reference_prefix}" "${shared_secret}" + +# start_pod +start_pod() { + local namespace="$1" role="$2" secret="$3" bucket="$4" manifest + kube "${namespace}" create configmap "${name}" --from-file="coverage.sh=${pod_script}" >/dev/null || + fail "could not stage the coverage script in ${namespace}" + created+=("${namespace}/configmap") + + # The heredoc is quoted, so nothing in it expands, and each placeholder is + # replaced by a value validated above. Every one of those values is restricted + # to characters that cannot end a YAML string or act as a replacement pattern + # (no quote, newline, backslash or `&`), which is what keeps the unquoted + # replacements below portable across bash 3.2 and 5.2. + manifest="$( + cat <<'MANIFEST' +apiVersion: v1 +kind: Pod +metadata: + name: __NAME__ + namespace: __NAMESPACE__ + labels: + app.kubernetes.io/name: wedding-backup-coverage + app.kubernetes.io/managed-by: verify-wedding-shared-backup-coverage +spec: + restartPolicy: Never + automountServiceAccountToken: false + activeDeadlineSeconds: 3600 + securityContext: + runAsNonRoot: true + runAsUser: 65532 + runAsGroup: 65532 + fsGroup: 65532 + seccompProfile: + type: RuntimeDefault + volumes: + - name: tools + emptyDir: + sizeLimit: 8Mi + - name: script + configMap: + name: __NAME__ + - name: credential + secret: + secretName: __SECRET__ + items: + - key: ACCESS_KEY_ID + path: ACCESS_KEY_ID + - key: SECRET_ACCESS_KEY + path: SECRET_ACCESS_KEY + - name: work + emptyDir: + sizeLimit: 1Gi + # `mc alias set` writes the credential into its config directory. Keep that + # in memory so the key never lands on node storage. + - name: mc-config + emptyDir: + medium: Memory + sizeLimit: 1Mi + initContainers: + - name: install-tools + image: __TOOLS_IMAGE__ + command: + - /bin/sh + - -ec + - cp /bin/busybox /tools/busybox; /tools/busybox --install -s /tools + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: ["ALL"] + resources: + requests: + cpu: 10m + memory: 8Mi + limits: + cpu: 100m + memory: 32Mi + volumeMounts: + - name: tools + mountPath: /tools + containers: + - name: coverage + image: __IMAGE__ + command: ["/tools/sh", "/coverage/coverage.sh", "serve"] + securityContext: + runAsNonRoot: true + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: ["ALL"] + env: + - name: PATH + value: /tools:/usr/local/bin:/usr/bin:/bin + - name: ENDPOINT + value: "__ENDPOINT__" + - name: ROLE + value: "__ROLE__" + - name: BUCKET + value: "__BUCKET__" + - name: PREFIX + value: "__PREFIX__" + - name: CREDENTIALS_DIR + value: /credentials + - name: WORK_DIR + value: /work + - name: MC_CONFIG_DIR + value: /mc-config + resources: + requests: + cpu: 50m + memory: 64Mi + limits: + memory: 512Mi + volumeMounts: + - name: tools + mountPath: /tools + readOnly: true + - name: script + mountPath: /coverage + readOnly: true + - name: credential + mountPath: /credentials + readOnly: true + - name: work + mountPath: /work + - name: mc-config + mountPath: /mc-config +MANIFEST + )" + manifest="${manifest//__NAME__/${name}}" + manifest="${manifest//__NAMESPACE__/${namespace}}" + manifest="${manifest//__IMAGE__/${mc_image}}" + manifest="${manifest//__TOOLS_IMAGE__/${tools_image}}" + manifest="${manifest//__SECRET__/${secret}}" + manifest="${manifest//__ENDPOINT__/${endpoint}}" + manifest="${manifest//__ROLE__/${role}}" + manifest="${manifest//__BUCKET__/${bucket}}" + manifest="${manifest//__PREFIX__/${catalogue_prefix}}" + + printf '%s\n' "${manifest}" | kube "${namespace}" create -f - >/dev/null || + fail "could not start the ${role} pod" + created+=("${namespace}/pod") +} + +# wait_ready waits for the pod to store its credential. +wait_ready() { + local namespace="$1" role="$2" phase='' attempt + for ((attempt = 0; attempt < poll_limit; attempt++)); do + phase="$(kube "${namespace}" get pod "${name}" -o 'jsonpath={.status.phase}')" || phase='' + [[ "${phase}" == Succeeded || "${phase}" == Failed ]] && break + if [[ "${phase}" == Running ]] && + kube "${namespace}" logs "pod/${name}" -c coverage 2>/dev/null | grep -qxF "${ready_marker}"; then + return 0 + fi + sleep "${poll_interval}" + done + kube "${namespace}" logs "pod/${name}" -c coverage 2>/dev/null | grep '^coverage-pod: ' >&2 || true + fail "the ${role} pod never became ready (phase '${phase:-unknown}')" +} + +# in_pod runs one command of the pod script. The long +# request timeout covers hashing a base backup. +in_pod() { + "${kubectl_bin}" --context "${context}" --namespace "$1" --request-timeout=45m \ + exec -i "${name}" -c coverage -- /tools/sh /coverage/coverage.sh "$2" +} + +# collect +collect() { + "${kubectl_bin}" --context "${context}" --namespace "$1" --request-timeout=10m \ + exec "${name}" -c coverage -- cat "/work/$2" >"${work_dir}/$3" || + fail "could not collect $2 from the pod in $1" +} + +start_pod "${shared_namespace}" shared "${shared_secret}" "${shared_bucket}" +start_pod "${tenant_namespace}" dedicated "${dedicated_secret}" "${dedicated_bucket}" +wait_ready "${shared_namespace}" shared +wait_ready "${tenant_namespace}" dedicated + +# The dedicated listing is taken after the shared one, because the evaluator +# refuses a dedicated listing that started earlier. +in_pod "${shared_namespace}" list "${work_dir}/multipart-keys" || fail 'the evaluator refused the listings' +in_pod "${shared_namespace}" hash <"${work_dir}/multipart-keys" || fail 'the shared pod could not hash its objects' +in_pod "${tenant_namespace}" hash <"${work_dir}/multipart-keys" || fail 'the dedicated pod could not hash its objects' +collect "${shared_namespace}" sums shared-sums +collect "${tenant_namespace}" sums dedicated-sums + +"${evaluator}" evaluate-coverage "${shared_bucket}" "${work_dir}/shared" "${work_dir}/dedicated" \ + "${work_dir}/shared-sums" "${work_dir}/dedicated-sums" >"${work_dir}/summary.json" || + fail 'NOT COVERED: the dedicated catalogue does not hold everything the shared copy holds. Keep the shared copy.' +jq -e '.covered == true and (.sharedObjects | type == "number" and . > 0)' \ + "${work_dir}/summary.json" >/dev/null 2>&1 || + fail 'the evaluator summary is not a covered verdict' + +require_retired + +cat "${work_dir}/summary.json" +printf 'COVERED: of the %s objects in the shared Wedding catalogue, the dedicated bucket holds %s with matching content.\n' \ + "$(jq -r '.sharedObjects' "${work_dir}/summary.json")" "$(jq -r '.matchedObjects' "${work_dir}/summary.json")" +printf 'The other %s were written before the oldest backup the dedicated store keeps, which itself predates its 30-day retention window, so retention has removed them there.\n' \ + "$(jq -r '.retentionPrunedObjects' "${work_dir}/summary.json")" +printf 'Nothing was changed. This holds as of this run: the shared copy is no longer written to.\n'