From 397ead19d85a5f7dcaa7fdaa136b4b92ca80c1e5 Mon Sep 17 00:00:00 2001 From: Nikolai Emil Damm Date: Mon, 5 Oct 2026 00:24:30 +0200 Subject: [PATCH 1/4] feat(backup): prove the dedicated bucket covers the stale shared Wedding catalogue Removing the shared copy of the Wedding catalogue is irreversible, so the decision needs current evidence that nothing would be lost. Add a read-only, dispatch-only proof: one pod beside each credential lists its side, both hash the objects an ETag cannot prove, and the reviewed evaluator requires every shared object to be in the dedicated catalogue with matching content or to be older than what the dedicated store's retention still keeps. Part of #4481 Co-Authored-By: Claude Opus 5.5 --- .github/workflows/ci.yaml | 9 + ...verify-wedding-shared-backup-coverage.yaml | 94 ++++ .../coverage.go | 270 +++++++++ .../coverage_test.go | 226 ++++++++ .../mirror-wedding-backup-catalogue/main.go | 35 +- ...t-verify-wedding-shared-backup-coverage.sh | 526 ++++++++++++++++++ ...rify-wedding-shared-backup-coverage-pod.sh | 149 +++++ .../verify-wedding-shared-backup-coverage.sh | 441 +++++++++++++++ 8 files changed, 1749 insertions(+), 1 deletion(-) create mode 100644 .github/workflows/verify-wedding-shared-backup-coverage.yaml create mode 100644 scripts/mirror-wedding-backup-catalogue/coverage.go create mode 100644 scripts/mirror-wedding-backup-catalogue/coverage_test.go create mode 100755 scripts/tests/test-verify-wedding-shared-backup-coverage.sh create mode 100755 scripts/verify-wedding-shared-backup-coverage-pod.sh create mode 100755 scripts/verify-wedding-shared-backup-coverage.sh diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index df35c5a8d0..b981bd7f23 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -160,6 +160,15 @@ jobs: shellcheck scripts/verify-wedding-backup-denial.sh scripts/verify-wedding-backup-denial-pod.sh scripts/tests/test-verify-wedding-backup-denial.sh bash scripts/tests/test-verify-wedding-backup-denial.sh + # The stale shared copy of the Wedding catalogue is removed on this proof's + # word, and that cannot be undone. These cases keep a missing object, + # different bytes, a live writer or an unreviewed bucket from passing as + # covered, and pin the pod script to commands that only read. + - name: 💒 Validate Wedding shared backup coverage proof + run: | + shellcheck scripts/verify-wedding-shared-backup-coverage.sh scripts/verify-wedding-shared-backup-coverage-pod.sh scripts/tests/test-verify-wedding-shared-backup-coverage.sh + bash scripts/tests/test-verify-wedding-shared-backup-coverage.sh + - name: 💾 Validate two-stage PVC retirement # A merge-group artifact is speculative, but a PVC deletion is not: its # deletionTimestamp survives queue eviction and main cannot undo it. diff --git a/.github/workflows/verify-wedding-shared-backup-coverage.yaml b/.github/workflows/verify-wedding-shared-backup-coverage.yaml new file mode 100644 index 0000000000..ca07736c42 --- /dev/null +++ b/.github/workflows/verify-wedding-shared-backup-coverage.yaml @@ -0,0 +1,94 @@ +name: Verify Wedding Shared Backup Coverage + +# Prove that the dedicated Wedding backup bucket covers everything the stale copy +# of the catalogue in the shared backup bucket still holds (#4481). Removing +# that copy is irreversible, so the decision needs this evidence first. +# +# DORMANT. There is no schedule and no pull_request trigger, and the operator +# must type an exact confirmation phrase, so an accidental dispatch does nothing. +# +# MAIN ONLY. `workflow_dispatch` accepts any branch, and this job holds the +# production kubeconfig. The prod environment already refuses non-main refs; the +# explicit check below fails fast and names the reason. +# +# READ ONLY. The proof lists both buckets and reads the objects it has to hash. +# It never writes to or deletes from either bucket. +on: + workflow_dispatch: + inputs: + confirm: + description: "Type exactly: verify-wedding-shared-backup-coverage" + required: true + type: string + +permissions: {} + +# The proof reads production state and two backup credentials. Serialize with +# every other production mutation so a deploy cannot change the wiring mid-run. +concurrency: + group: prod-deploy + cancel-in-progress: false + queue: max + +jobs: + coverage: + name: 💒 Prove the dedicated bucket covers the shared Wedding catalogue + runs-on: ubuntu-latest + timeout-minutes: 90 + environment: prod + permissions: + contents: read # checkout repository + steps: + # Inputs reach bash through env, never `${{ }}` inside run, so a crafted + # value cannot inject shell. + - name: 🛑 Require main and explicit confirmation + shell: bash + env: + CONFIRM: ${{ inputs.confirm }} + DISPATCH_REF: ${{ github.ref }} + run: | + set -euo pipefail + if [[ "${DISPATCH_REF}" != 'refs/heads/main' ]]; then + printf 'Refusing to run from %s: dispatch this workflow from main.\n' "${DISPATCH_REF}" >&2 + exit 1 + fi + if [[ "${CONFIRM}" != 'verify-wedding-shared-backup-coverage' ]]; then + printf 'Refusing to run: the confirm input must be exactly "verify-wedding-shared-backup-coverage".\n' >&2 + printf 'The proof starts pods beside two production backup credentials. Nothing has been touched.\n' >&2 + exit 1 + fi + printf 'Confirmation accepted.\n' + + - name: 📑 Checkout + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + ref: ${{ github.sha }} + + - name: ⚙️ Setup Go + uses: actions/setup-go@b7ad1dad31e06c5925ef5d2fc7ad053ef454303e # v7.0.0 + with: + go-version-file: go.mod + cache: false + + - name: 🔑 Restore production kubeconfig + shell: bash + env: + KUBE_CONFIG: ${{ secrets.KUBE_CONFIG }} + run: | + set -euo pipefail + test -n "${KUBE_CONFIG}" + umask 077 + mkdir -p ~/.kube + printf '%s' "${KUBE_CONFIG}" >~/.kube/config + chmod 600 ~/.kube/config + + - name: 🎯 Select stable production API endpoint + shell: bash + env: + HCLOUD_TOKEN: ${{ secrets.HCLOUD_TOKEN }} + run: ./scripts/use-prod-stable-api-endpoint.sh + + - name: 💒 Compare the shared catalogue with the dedicated one + shell: bash + run: ./scripts/verify-wedding-shared-backup-coverage.sh --confirm diff --git a/scripts/mirror-wedding-backup-catalogue/coverage.go b/scripts/mirror-wedding-backup-catalogue/coverage.go new file mode 100644 index 0000000000..28ddf8be13 --- /dev/null +++ b/scripts/mirror-wedding-backup-catalogue/coverage.go @@ -0,0 +1,270 @@ +package main + +// The coverage evaluation decides whether the dedicated Wedding bucket holds +// everything the stale shared copy of the catalogue holds (#4481). It only +// reads: it is the evidence a person needs before deciding to delete the shared +// copy, and it deletes nothing itself. +// +// Usage: +// +// mirror-wedding-backup-catalogue coverage-multipart-keys +// mirror-wedding-backup-catalogue evaluate-coverage +// +// The two listings come from two pods, because after #3253 no namespace holds +// both credentials. coverage-multipart-keys prints the shared keys whose content +// an ETag cannot prove on at least one side; the caller has each pod hash those +// keys and passes the results as "\t" lines. +// +// The dedicated store applies its retention policy and the shared copy no longer +// does, so the shared copy can hold objects the dedicated store has since +// pruned. Such an object is accepted only when it is older than everything the +// dedicated store still keeps of the same kind in the same server directory. +// Every other shared object must be present with matching content. + +import ( + "bufio" + "errors" + "fmt" + "io" + "os" + "sort" + "strings" +) + +var ( + ErrNotCovered = errors.New("shared object is not covered by the dedicated catalogue") + ErrMalformedSums = errors.New("malformed checksum list") + ErrStaleListing = errors.New("dedicated listing does not follow the shared listing") +) + +// CoverageSummary is the non-secret evidence a passing coverage evaluation reports. +type CoverageSummary struct { + SharedObjects int `json:"sharedObjects"` + MatchedObjects int `json:"matchedObjects"` + RetentionPrunedObjects int `json:"retentionPrunedObjects"` + Covered bool `json:"covered"` + SharedNewestBaseBackup string `json:"sharedNewestBaseBackup"` + DedicatedNewestBaseBackup string `json:"dedicatedNewestBaseBackup"` +} + +// MultipartKeys returns, sorted, every shared key whose content cannot be +// compared by ETag because it is multipart on either side. A key the dedicated +// catalogue does not hold is left out: there is nothing to compare it with. +func MultipartKeys(shared, dedicated []Object) ([]string, error) { + dedicatedIndex, err := index(dedicated) + if err != nil { + return nil, err + } + if _, err := index(shared); err != nil { + return nil, err + } + var keys []string + for _, object := range shared { + copied, ok := dedicatedIndex[object.Key] + if ok && (isMultipart(object.ETag) || isMultipart(copied.ETag)) { + keys = append(keys, object.Key) + } + } + sort.Strings(keys) + return keys, nil +} + +// ApplySums sets the SHA256 of every object named in a "\t" list. +// A key the listing does not hold, a repeated key or a malformed digest is +// refused, so a list produced for another listing cannot vouch for this one. +func ApplySums(objects []Object, r io.Reader) error { + positions := make(map[string]int, len(objects)) + for i, object := range objects { + positions[object.Key] = i + } + seen := map[string]bool{} + scanner := bufio.NewScanner(r) + scanner.Buffer(make([]byte, 0, 64*1024), 1024*1024) + for scanner.Scan() { + line := scanner.Text() + if line == "" { + continue + } + key, sum, ok := strings.Cut(line, "\t") + if !ok || !sha256Hex.MatchString(sum) { + return fmt.Errorf("%w: line %q", ErrMalformedSums, line) + } + position, listed := positions[key] + if !listed || seen[key] { + return fmt.Errorf("%w: unexpected key %s", ErrMalformedSums, key) + } + seen[key] = true + objects[position].SHA256 = sum + } + if err := scanner.Err(); err != nil { + return fmt.Errorf("%w: %w", ErrMalformedSums, err) + } + return nil +} + +// retentionFloor records, per server directory, the oldest base backup and the +// oldest WAL segment the dedicated store still keeps. Retention only ever +// removes from the old end, so anything older than these is gone by design. +type retentionFloor struct { + backup map[string]string + wal map[string]string +} + +func retentionFloorOf(objects map[string]Object) retentionFloor { + floor := retentionFloor{backup: map[string]string{}, wal: map[string]string{}} + for key := range objects { + parts := strings.Split(key, "/") + server := parts[0] + if id := baseBackupIDOf(key); id != "" { + if current, ok := floor.backup[server]; !ok || id < current { + floor.backup[server] = id + } + } + if segment := walSegmentOf(key); segment != "" { + if current, ok := floor.wal[server]; !ok || segment < current { + floor.wal[server] = segment + } + } + } + return floor +} + +// baseBackupIDOf returns the backup ID of a key inside a base backup directory. +func baseBackupIDOf(key string) string { + parts := strings.Split(key, "/") + if len(parts) < 4 || parts[1] != "base" || !backupID.MatchString(parts[2]) { + return "" + } + return parts[2] +} + +// walPositionOf returns the segment a WAL-directory object belongs to: the +// segment itself, or the one a backup label or partial file is named after. A +// timeline history file has no position and is never pruned by retention. +func walPositionOf(key string) string { + if segment := walSegmentOf(key); segment != "" { + return segment + } + parts := strings.Split(key, "/") + if len(parts) != 4 || parts[1] != "wals" || len(parts[3]) < 25 || parts[3][24] != '.' { + return "" + } + segment := parts[3][:24] + if !walSegment.MatchString(segment) || parts[2] != segment[:16] { + return "" + } + return segment +} + +// prunedByRetention reports whether a shared object the dedicated store lacks is +// older than everything that store still keeps of its kind in its server +// directory. A server directory the dedicated store holds nothing of has no +// floor, so nothing in it is accepted as pruned. +func (floor retentionFloor) prunedByRetention(key string) bool { + server := strings.Split(key, "/")[0] + if id := baseBackupIDOf(key); id != "" { + oldest, ok := floor.backup[server] + return ok && id < oldest + } + if position := walPositionOf(key); position != "" { + oldest, ok := floor.wal[server] + return ok && position < oldest + } + return false +} + +// EvaluateCoverage proves the dedicated catalogue covers the shared one. Every +// shared object must be in the dedicated catalogue with matching content, or be +// older than what the dedicated store's retention still keeps. The dedicated +// catalogue's newest complete base backup must be at least as new as the shared +// one's, so the shared copy never holds the most recent restore point. +func EvaluateCoverage(shared, dedicated Listing) (CoverageSummary, error) { + if shared.Started.IsZero() || dedicated.Started.IsZero() || + dedicated.Started.Before(shared.Started) { + return CoverageSummary{}, ErrStaleListing + } + sharedIndex, err := index(shared.Objects) + if err != nil { + return CoverageSummary{}, err + } + dedicatedIndex, err := index(dedicated.Objects) + if err != nil { + return CoverageSummary{}, err + } + if len(sharedIndex) == 0 { + return CoverageSummary{}, ErrEmptySource + } + + floor := retentionFloorOf(dedicatedIndex) + matched, pruned := 0, 0 + for key, source := range sharedIndex { + copied, ok := dedicatedIndex[key] + if !ok { + if !floor.prunedByRetention(key) { + return CoverageSummary{}, fmt.Errorf("%w: %s", ErrNotCovered, key) + } + pruned++ + continue + } + if copied.Size != source.Size { + return CoverageSummary{}, fmt.Errorf("%w: %s", ErrPartialCopy, key) + } + if err := sameContent(source, copied); err != nil { + return CoverageSummary{}, fmt.Errorf("%w: %s", err, key) + } + matched++ + } + + sharedBackup := newestBaseBackup(sharedIndex) + dedicatedBackup := newestBaseBackup(dedicatedIndex) + if dedicatedBackup == "" { + return CoverageSummary{}, ErrNoBaseBackup + } + _, sharedID, _ := strings.Cut(sharedBackup, "/") + _, dedicatedID, _ := strings.Cut(dedicatedBackup, "/") + if dedicatedID < sharedID { + return CoverageSummary{}, fmt.Errorf("%w: the newest dedicated base backup %s is older than the shared %s", + ErrNotCovered, dedicatedBackup, sharedBackup) + } + + return CoverageSummary{ + SharedObjects: len(sharedIndex), + MatchedObjects: matched, + RetentionPrunedObjects: pruned, + Covered: true, + SharedNewestBaseBackup: sharedBackup, + DedicatedNewestBaseBackup: dedicatedBackup, + }, nil +} + +// readCoverageListings reads the shared and dedicated listings for a coverage evaluation. +func readCoverageListings(sharedBucket, sharedName, dedicatedName string) (Listing, Listing, error) { + if sharedBucket == "" || sharedBucket == destinationBucket { + return Listing{}, Listing{}, fmt.Errorf("%w: shared bucket %q", ErrWrongSource, sharedBucket) + } + shared, err := readListing(sharedName, sharedBucket+"/"+sourcePrefix) + if err != nil { + return Listing{}, Listing{}, fmt.Errorf("%s: %w", sharedName, err) + } + dedicated, err := readListing(dedicatedName, destinationBucket+"/"+destinationPrefix) + if err != nil { + return Listing{}, Listing{}, fmt.Errorf("%s: %w", dedicatedName, err) + } + return shared, dedicated, nil +} + +// applySumsFile applies one checksum list named on the command line. +func applySumsFile(objects []Object, name string) error { + file, err := os.Open(name) + if err != nil { + return err + } + applyErr := ApplySums(objects, file) + if closeErr := file.Close(); closeErr != nil && applyErr == nil { + return closeErr + } + if applyErr != nil { + return fmt.Errorf("%s: %w", name, applyErr) + } + return nil +} diff --git a/scripts/mirror-wedding-backup-catalogue/coverage_test.go b/scripts/mirror-wedding-backup-catalogue/coverage_test.go new file mode 100644 index 0000000000..af4b64f6b4 --- /dev/null +++ b/scripts/mirror-wedding-backup-catalogue/coverage_test.go @@ -0,0 +1,226 @@ +package main + +import ( + "bytes" + "encoding/json" + "errors" + "os" + "path/filepath" + "strings" + "testing" + "time" +) + +const ( + oldServerInfo = "wedding-db/base/20260813T030000/backup.info" + oldServerData = "wedding-db/base/20260813T030000/data.tar.gz" + oldServerWAL = "wedding-db/wals/0000000100000000/000000010000000000000007.gz" + + prunedInfo = "wedding-db-20260909/base/20260901T030000/backup.info" + prunedData = "wedding-db-20260909/base/20260901T030000/data.tar.gz" + prunedWAL = "wedding-db-20260909/wals/0000000200000000/0000000200000000000000FE.gz" + prunedLabel = "wedding-db-20260909/wals/0000000200000000/0000000200000000000000FE.00000028.backup" + history = "wedding-db-20260909/wals/00000003.history" + postSwitch = "wedding-db-20260909/wals/0000000300000001/000000030000000100000009.gz" +) + +var ( + sharedListed = time.Date(2026, 10, 5, 9, 0, 0, 0, time.UTC) + dedicatedListed = sharedListed.Add(time.Minute) +) + +func sharedCatalogue() Listing { + return Listing{Started: sharedListed, Objects: sourceListing()} +} + +// The dedicated store keeps archiving after the switch, so it holds more than +// the shared copy. +func dedicatedCatalogue() Listing { + objects := append(destinationListing(), + Object{Key: postSwitch, Size: 4300, ETag: "f6", LastModified: sharedListed}) + return Listing{Started: dedicatedListed, Objects: objects} +} + +func TestEvaluateCoverageProvesTheDedicatedCatalogueCoversTheSharedOne(t *testing.T) { + summary, err := EvaluateCoverage(sharedCatalogue(), dedicatedCatalogue()) + if err != nil { + t.Fatalf("EvaluateCoverage() = %v, want nil", err) + } + if summary.SharedObjects != 4 || summary.MatchedObjects != 4 || + summary.RetentionPrunedObjects != 0 || !summary.Covered { + t.Fatalf("summary = %+v, want 4 shared, 4 matched, 0 pruned, covered", summary) + } +} + +// Retention runs on the dedicated store only, so the shared copy outlives +// backups the dedicated store has pruned. Those are older than everything the +// dedicated store keeps and are not a reason to keep the shared copy. +func TestEvaluateCoverageAcceptsObjectsTheDedicatedStorePrunedByRetention(t *testing.T) { + shared := sharedCatalogue() + shared.Objects = append(shared.Objects, + Object{Key: prunedInfo, Size: 1100, ETag: "p1", LastModified: beforeRun}, + Object{Key: prunedData, Size: 80_000_000, ETag: "p2-5", LastModified: beforeRun}, + Object{Key: prunedWAL, Size: 3900, ETag: "p3", LastModified: beforeRun}, + Object{Key: prunedLabel, Size: 300, ETag: "p4", LastModified: beforeRun}, + ) + summary, err := EvaluateCoverage(shared, dedicatedCatalogue()) + if err != nil { + t.Fatalf("EvaluateCoverage() = %v, want nil", err) + } + if summary.MatchedObjects != 4 || summary.RetentionPrunedObjects != 4 { + t.Fatalf("summary = %+v, want 4 matched and 4 pruned", summary) + } +} + +func TestEvaluateCoverageRefusals(t *testing.T) { + tests := []struct { + name string + mutate func(shared, dedicated *Listing) + want error + }{ + {"an object the dedicated store never held", func(shared, _ *Listing) { + shared.Objects = append(shared.Objects, Object{Key: lateWAL, Size: 4200, ETag: "e5", LastModified: beforeRun}) + }, ErrNotCovered}, + {"a missing object inside the retained range", func(_, dedicated *Listing) { + dedicated.Objects = dedicated.Objects[:3] + }, ErrNotCovered}, + {"a server directory the dedicated store holds nothing of", func(shared, _ *Listing) { + shared.Objects = append(shared.Objects, + Object{Key: oldServerInfo, Size: 900, ETag: "o1", LastModified: beforeRun}, + Object{Key: oldServerData, Size: 900, ETag: "o2", LastModified: beforeRun}, + Object{Key: oldServerWAL, Size: 900, ETag: "o3", LastModified: beforeRun}) + }, ErrNotCovered}, + {"a timeline history file is never pruned", func(shared, _ *Listing) { + shared.Objects = append(shared.Objects, Object{Key: history, Size: 80, ETag: "h1", LastModified: beforeRun}) + }, ErrNotCovered}, + {"a different size", func(_, dedicated *Listing) { + dedicated.Objects[0].Size++ + }, ErrPartialCopy}, + {"a different single-part checksum", func(_, dedicated *Listing) { + dedicated.Objects[0].ETag = "zz" + }, ErrChecksumMismatch}, + {"a different content digest", func(_, dedicated *Listing) { + dedicated.Objects[1].SHA256 = otherDigest + }, ErrChecksumMismatch}, + {"a multipart object nobody hashed", func(shared, _ *Listing) { + shared.Objects[1].SHA256 = "" + }, ErrUnverifiable}, + {"an empty shared listing", func(shared, _ *Listing) { + shared.Objects = nil + }, ErrEmptySource}, + {"a dedicated listing taken before the shared one", func(_, dedicated *Listing) { + dedicated.Started = sharedListed.Add(-time.Second) + }, ErrStaleListing}, + {"a dedicated listing with no start time", func(_, dedicated *Listing) { + dedicated.Started = time.Time{} + }, ErrStaleListing}, + {"a shared listing with no start time", func(shared, _ *Listing) { + shared.Started = time.Time{} + }, ErrStaleListing}, + {"a repeated shared key", func(shared, _ *Listing) { + shared.Objects = append(shared.Objects, shared.Objects[0]) + }, ErrMalformedListing}, + {"a dedicated catalogue with no complete base backup", func(shared, dedicated *Listing) { + shared.Objects = shared.Objects[2:] + dedicated.Objects = dedicated.Objects[2:] + }, ErrNoBaseBackup}, + {"a dedicated catalogue whose newest base backup is older", func(shared, _ *Listing) { + shared.Objects = append(shared.Objects, + Object{Key: "wedding-db-20260909/base/20260920T030000/backup.info", Size: 1, ETag: "n1", LastModified: beforeRun}, + Object{Key: "wedding-db-20260909/base/20260920T030000/data.tar.gz", Size: 1, ETag: "n2", LastModified: beforeRun}) + }, ErrNotCovered}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + shared, dedicated := sharedCatalogue(), dedicatedCatalogue() + tt.mutate(&shared, &dedicated) + if _, err := EvaluateCoverage(shared, dedicated); !errors.Is(err, tt.want) { + t.Fatalf("EvaluateCoverage() = %v, want %v", err, tt.want) + } + }) + } +} + +func TestMultipartKeysNamesSharedKeysAnETagCannotProve(t *testing.T) { + shared := []Object{ + {Key: baseInfo, ETag: "a1"}, + {Key: baseData, ETag: "b2-6"}, + {Key: olderWAL, ETag: "c3"}, + {Key: newerWAL, ETag: "d4-2"}, + } + dedicated := []Object{ + {Key: baseInfo, ETag: "a1"}, + {Key: baseData, ETag: "ff"}, + {Key: olderWAL, ETag: "c3-2"}, + {Key: postSwitch, ETag: "g7-2"}, + } + keys, err := MultipartKeys(shared, dedicated) + if err != nil { + t.Fatalf("MultipartKeys() = %v", err) + } + if got, want := strings.Join(keys, ","), baseData+","+olderWAL; got != want { + t.Fatalf("MultipartKeys() = %q, want %q", got, want) + } +} + +func TestApplySums(t *testing.T) { + objects := sourceListing() + objects[1].SHA256 = "" + if err := ApplySums(objects, strings.NewReader(baseData+"\t"+dataDigest+"\n\n")); err != nil { + t.Fatalf("ApplySums() = %v, want nil", err) + } + if objects[1].SHA256 != dataDigest { + t.Fatalf("SHA256 = %q, want %q", objects[1].SHA256, dataDigest) + } + for name, sums := range map[string]string{ + "an unlisted key": lateWAL + "\t" + dataDigest + "\n", + "a repeated key": baseData + "\t" + dataDigest + "\n" + baseData + "\t" + dataDigest + "\n", + "a malformed digest": baseData + "\tnot-a-digest\n", + "no separator": baseData + "\n", + } { + t.Run(name, func(t *testing.T) { + if err := ApplySums(sourceListing(), strings.NewReader(sums)); !errors.Is(err, ErrMalformedSums) { + t.Fatalf("ApplySums() = %v, want %v", err, ErrMalformedSums) + } + }) + } +} + +func TestRunEvaluateCoverageEndToEnd(t *testing.T) { + keys := []string{baseInfo, baseData, olderWAL} + shared := writeListing(t, "shared", sourceLocation, "2026-10-05T09:00:00Z", keys...) + dedicated := writeListing(t, "dedicated", destinationLocation, "2026-10-05T09:01:00Z", keys...) + empty := filepath.Join(t.TempDir(), "empty") + if err := os.WriteFile(empty, nil, 0o600); err != nil { + t.Fatal(err) + } + + var keysOut bytes.Buffer + if err := run([]string{"coverage-multipart-keys", "platform-backups", shared, dedicated}, &keysOut); err != nil { + t.Fatalf("run(coverage-multipart-keys) = %v, want nil", err) + } + + var out bytes.Buffer + if err := run([]string{"evaluate-coverage", "platform-backups", shared, dedicated, empty, empty}, &out); err != nil { + t.Fatalf("run(evaluate-coverage) = %v, want nil", err) + } + var summary CoverageSummary + if err := json.Unmarshal(out.Bytes(), &summary); err != nil { + t.Fatalf("summary is not JSON: %v", err) + } + if !summary.Covered || summary.SharedObjects != len(keys) { + t.Fatalf("summary = %+v", summary) + } + + // The listings are bound to their buckets, so swapping them, or naming the + // dedicated bucket as the shared one, is refused. + if err := run([]string{"evaluate-coverage", "platform-backups", dedicated, shared, empty, empty}, &out); !errors.Is(err, ErrListingLocation) { + t.Fatalf("run(evaluate-coverage, swapped) = %v, want %v", err, ErrListingLocation) + } + if err := run([]string{"evaluate-coverage", destinationBucket, shared, dedicated, empty, empty}, &out); !errors.Is(err, ErrWrongSource) { + t.Fatalf("run(evaluate-coverage, dedicated as shared) = %v, want %v", err, ErrWrongSource) + } + if err := run([]string{"evaluate-coverage", "platform-backups", shared, dedicated, empty, filepath.Join(t.TempDir(), "absent")}, &out); err == nil { + t.Fatal("run(evaluate-coverage, missing sums) = nil, want an error") + } +} diff --git a/scripts/mirror-wedding-backup-catalogue/main.go b/scripts/mirror-wedding-backup-catalogue/main.go index d2260dd7b2..40f1b0e8d6 100644 --- a/scripts/mirror-wedding-backup-catalogue/main.go +++ b/scripts/mirror-wedding-backup-catalogue/main.go @@ -829,7 +829,40 @@ func run(args []string, stdout io.Writer) error { } return json.NewEncoder(stdout).Encode(summary) } - return errors.New("usage: mirror-wedding-backup-catalogue validate-plan | evaluate | evaluate-catch-up | validate-switch-time ") + if len(args) == 4 && args[0] == "coverage-multipart-keys" { + shared, dedicated, err := readCoverageListings(args[1], args[2], args[3]) + if err != nil { + return err + } + keys, err := MultipartKeys(shared.Objects, dedicated.Objects) + if err != nil { + return err + } + for _, key := range keys { + if _, err := fmt.Fprintln(stdout, key); err != nil { + return err + } + } + return nil + } + if len(args) == 6 && args[0] == "evaluate-coverage" { + shared, dedicated, err := readCoverageListings(args[1], args[2], args[3]) + if err != nil { + return err + } + if err := applySumsFile(shared.Objects, args[4]); err != nil { + return err + } + if err := applySumsFile(dedicated.Objects, args[5]); err != nil { + return err + } + summary, err := EvaluateCoverage(shared, dedicated) + if err != nil { + return err + } + return json.NewEncoder(stdout).Encode(summary) + } + return errors.New("usage: mirror-wedding-backup-catalogue coverage-multipart-keys | evaluate-coverage | validate-plan | evaluate | evaluate-catch-up | validate-switch-time ") } // main exits non-zero with the refusal reason when a plan or mirror is not trusted. diff --git a/scripts/tests/test-verify-wedding-shared-backup-coverage.sh b/scripts/tests/test-verify-wedding-shared-backup-coverage.sh new file mode 100755 index 0000000000..b9552fb083 --- /dev/null +++ b/scripts/tests/test-verify-wedding-shared-backup-coverage.sh @@ -0,0 +1,526 @@ +#!/usr/bin/env bash + +# Pin the behaviour of the shared Wedding backup coverage proof: +# scripts/verify-wedding-shared-backup-coverage.sh (runs on the runner) and +# scripts/verify-wedding-shared-backup-coverage-pod.sh (runs in the cluster). +# +# WHY THIS EXISTS. The stale shared copy of the Wedding catalogue is removed on +# this proof's word (#4481), and that removal cannot be undone. Its dangerous +# mistakes are quiet ones: +# +# * reporting COVERED when the dedicated bucket lacks an object, holds different +# bytes, or was listed before the shared one; +# * reporting COVERED while something can still write to the shared copy; +# * handing a credential to an endpoint or bucket nobody reviewed; +# * writing to or deleting from either bucket. +# +# Every case runs the REAL runner, the REAL pod script and the REAL evaluator +# against a fake kubectl and a fake mc, so the three are proven to agree on the +# listing and checksum formats. Needs Go and jq, no cluster and no credentials. + +set -euo pipefail + +root_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +readonly root_dir +readonly runner="${root_dir}/scripts/verify-wedding-shared-backup-coverage.sh" +readonly pod_script="${root_dir}/scripts/verify-wedding-shared-backup-coverage-pod.sh" +readonly mirror="${root_dir}/scripts/mirror-wedding-backup-catalogue.sh" + +work_dir="$(mktemp -d)" +readonly work_dir +# shellcheck disable=SC2317,SC2329 # Invoked indirectly by the EXIT trap. +cleanup() { + rm -rf "${work_dir}" +} +trap cleanup EXIT + +command -v go >/dev/null 2>&1 || { + printf 'FAIL: go is required to build the real evaluator\n' >&2 + exit 1 +} +readonly evaluator="${work_dir}/evaluator" +(cd "${root_dir}" && go build -o "${evaluator}" ./scripts/mirror-wedding-backup-catalogue) + +readonly bin="${work_dir}/bin" +mkdir -p "${bin}" + +cases_run=0 + +fail() { + printf '\nFAIL: %s\n' "$1" >&2 + exit 1 +} + +require_text() { + printf '%s' "$1" | grep -qF -- "$2" || fail "$3: expected '$2'. Got: $1" +} + +refute_text() { + if printf '%s' "$1" | grep -qF -- "$2"; then + fail "$3: did not expect '$2'. Got: $1" + fi +} + +# --------------------------------------------------------------------------- +# Fake mc. Only `alias set`, `ls` and `cat` are served. Anything else is recorded +# as forbidden, and every case asserts nothing forbidden was attempted. Every +# failure writes the endpoint host to stderr, so redaction is exercised. +# --------------------------------------------------------------------------- +cat >"${bin}/mc" <<'FAKE' +#!/bin/sh +set -u +f="${FAKE_MC}" +leak() { + printf 'mc: request to https://abc123.r2.cloudflarestorage.com/x failed\n' >&2 +} +case "$1" in + alias) + if [ "$2" != set ]; then printf '%s\n' "$*" >>"${f}/forbidden"; exit 64; fi + printf '%s %s\n' "$3" "$5" >>"${f}/aliases" + exit 0 + ;; + ls) + printf '%s\n' "$4" >>"${f}/listed" + if [ -e "${f}/ls-rc" ]; then leak; exit "$(cat "${f}/ls-rc")"; fi + cat "${f}/ls" + exit 0 + ;; + cat) + printf '%s\n' "$2" >>"${f}/catted" + if [ -e "${f}/cat-rc" ]; then leak; exit "$(cat "${f}/cat-rc")"; fi + if [ -e "${f}/content" ]; then cat "${f}/content"; else printf 'bytes of %s' "${2##*/}"; fi + exit 0 + ;; +esac +printf '%s\n' "$*" >>"${f}/forbidden" +exit 64 +FAKE +chmod +x "${bin}/mc" + +# --------------------------------------------------------------------------- +# Fake kubectl. `exec` runs the real pod script with the environment the runner +# wrote into that namespace's pod manifest, so the manifest wiring is part of +# what each case proves. +# --------------------------------------------------------------------------- +cat >"${bin}/kubectl" <<'FAKE' +#!/usr/bin/env bash +set -u +k="${FAKE_K}" +namespace='' +args=() +while [ "$#" -gt 0 ]; do + case "$1" in + --context) shift 2 ;; + --namespace) namespace="$2"; shift 2 ;; + --request-timeout=*) shift ;; + *) args+=("$1"); shift ;; + esac +done +set -- "${args[@]}" +printf '%s %s\n' "${namespace:-all}" "$*" >>"${k}/calls" + +manifest_env() { + awk -v want="$2" '$0 ~ "- name: " want "$" { getline; gsub(/^ *value: *"?|"?$/, ""); print; exit }' "${k}/pod-$1.yaml" +} + +case "$1" in + get) + case "$2" in + pod) + cat "${k}/phase-${namespace}" 2>/dev/null || printf 'Running' + ;; + clusters.postgresql.cnpg.io | objectstores.barmancloud.cnpg.io | externalsecrets.external-secrets.io | secrets) + if [ "$3" = --all-namespaces ]; then + cat "${k}/all-$2.json" || exit 1 + elif [ "${4:-}" = --ignore-not-found ]; then + if [ -e "${k}/absence-rc" ]; then exit "$(cat "${k}/absence-rc")"; fi + if [ -e "${k}/present-$2-$3" ]; then printf '%s/%s\n' "$2" "$3"; fi + else + cat "${k}/${namespace}-$2-$3.json" 2>/dev/null || exit 1 + fi + ;; + *) exit 64 ;; + esac + ;; + create) + if [ "$2" = configmap ]; then + exit 0 + fi + cat >"${k}/pod-${namespace}.yaml" + ;; + logs) + if [ ! -e "${k}/never-ready-${namespace}" ]; then printf '==== COVERAGE POD READY ====\n'; fi + ;; + delete) + printf '%s %s\n' "${namespace}" "$2" >>"${k}/deleted" + ;; + exec) + while [ "$1" != -- ]; do shift; done + shift + if [ "$1" = cat ]; then + cat "${k}/work-${namespace}/${2#/work/}" + exit $? + fi + [ "$1 $2" = '/tools/sh /coverage/coverage.sh' ] || exit 64 + PATH="${FAKE_BIN}:${PATH}" FAKE_MC="${k}/mc-${namespace}" \ + ENDPOINT="$(manifest_env "${namespace}" ENDPOINT)" ROLE="$(manifest_env "${namespace}" ROLE)" \ + BUCKET="$(manifest_env "${namespace}" BUCKET)" PREFIX="$(manifest_env "${namespace}" PREFIX)" \ + WORK_DIR="${k}/work-${namespace}" CREDENTIALS_DIR="${k}/credentials" \ + sh "${POD_SCRIPT}" "$3" + ;; + *) exit 64 ;; +esac +FAKE +chmod +x "${bin}/kubectl" + +readonly endpoint='https://abc123.r2.cloudflarestorage.com' +readonly old='2026-09-01T00:00:00Z' +readonly base='wedding-db-20260909/base/20260908T030000' +readonly wal='wedding-db-20260909/wals/0000000200000001/000000020000000100000003.gz' +readonly later_wal='wedding-db-20260909/wals/0000000300000001/000000030000000100000009.gz' + +# record +record() { + printf '{"status":"success","type":"file","lastModified":"%s","size":%s,"key":"%s","etag":"%s","url":"%s","versionOrdinal":1,"storageClass":"STANDARD"}\n' \ + "${old}" "$2" "$1" "$3" "${endpoint}" +} + +# store_json [endpoint] +store_json() { + printf '{"metadata":{},"spec":{"configuration":{"destinationPath":"%s","endpointURL":"%s","s3Credentials":{"accessKeyId":{"name":"%s","key":"ACCESS_KEY_ID"},"secretAccessKey":{"name":"%s","key":"SECRET_ACCESS_KEY"}}}}}\n' \ + "$1" "${3:-${endpoint}}" "$2" "$2" +} + +# cluster_json +cluster_json() { + printf '{"metadata":{"uid":"cluster-uid-1","generation":3},"spec":{"instances":1,"plugins":[{"name":"barman-cloud.cloudnative-pg.io","enabled":true,"isWALArchiver":true,"parameters":{"barmanObjectName":"%s"}}]},"status":{"readyInstances":1,"conditions":[{"type":"Ready","status":"True"},{"type":"ContinuousArchiving","status":"True"}]}}\n' "$1" +} + +# new_case prepares a covered scenario; each case then breaks one thing. +new_case() { + cases_run=$((cases_run + 1)) + k="${work_dir}/case-${cases_run}" + mkdir -p "${k}/mc-umami" "${k}/mc-wedding-app" "${k}/work-umami" "${k}/work-wedding-app" "${k}/credentials" + printf 'data:\n r2_endpoint: %s\n r2_bucket: platform-backups\n' "${endpoint}" >"${k}/config-map.yaml" + printf 'AKIDEXAMPLE' >"${k}/credentials/ACCESS_KEY_ID" + printf 'secret-value' >"${k}/credentials/SECRET_ACCESS_KEY" + cluster_json wedding-db-dedicated >"${k}/wedding-app-clusters.postgresql.cnpg.io-wedding-db.json" + store_json s3://wedding-db-backups/cnpg/wedding-db wedding-db-backup-r2-dedicated \ + >"${k}/wedding-app-objectstores.barmancloud.cnpg.io-wedding-db-dedicated.json" + store_json s3://platform-backups/cnpg/umami-db umami-db-backup-r2 \ + >"${k}/umami-objectstores.barmancloud.cnpg.io-umami-db.json" + printf '{"items":[{"spec":{"configuration":{"destinationPath":"s3://platform-backups/cnpg/umami-db"}}},{"spec":{"configuration":{"destinationPath":"s3://wedding-db-backups/cnpg/wedding-db"}}}]}\n' \ + >"${k}/all-objectstores.barmancloud.cnpg.io.json" + printf '{"items":[{"spec":{"instances":1}}]}\n' >"${k}/all-clusters.postgresql.cnpg.io.json" + { + record "${base}/backup.info" 1200 a1 + record "${base}/data.tar.gz" 90000000 b2-6 + record "${wal}" 4000 c3 + } >"${k}/mc-umami/ls" + # The mirror uploaded the archive in a different number of parts, and the + # dedicated store has archived more since the switch. + { + record "${base}/backup.info" 1200 a1 + record "${base}/data.tar.gz" 90000000 ff-4 + record "${wal}" 4000 c3 + record "${later_wal}" 4300 f6 + } >"${k}/mc-wedding-app/ls" +} + +# run_case [runner arguments...] sets out, err and rc. +run_case() { + rc=0 + FAKE_K="${k}" FAKE_BIN="${bin}" POD_SCRIPT="${pod_script}" KUBECTL="${bin}/kubectl" \ + COVERAGE_EVALUATOR="${evaluator}" COVERAGE_BOOTSTRAP_CONFIG="${k}/config-map.yaml" \ + COVERAGE_POLL_INTERVAL=0 COVERAGE_POLL_LIMIT=2 GITHUB_RUN_ID=77 GITHUB_RUN_ATTEMPT=1 \ + bash "${runner}" "$@" >"${k}/out" 2>"${k}/err" || rc=$? + out="$(cat "${k}/out")" + err="$(cat "${k}/err")" + for side in umami wedding-app; do + [[ ! -e "${k}/mc-${side}/forbidden" ]] || + fail "case ${cases_run}: the pod ran an mc command other than alias set, ls or cat: $(cat "${k}/mc-${side}/forbidden")" + done +} + +# require_untouched asserts a refusal happened before anything was created. +require_untouched() { + [[ "${rc}" -eq 1 ]] || fail "$1: expected exit 1, got ${rc}. stderr: ${err}" + if grep -q ' create ' "${k}/calls" 2>/dev/null; then + fail "$1: something was created before the refusal" + fi + refute_text "${out}" COVERED "$1" +} + +# require_cleaned asserts both pods and both scripts were removed. +require_cleaned() { + for entry in 'umami pod' 'umami configmap' 'wedding-app pod' 'wedding-app configmap'; do + grep -qxF "${entry}" "${k}/deleted" || fail "$1: ${entry} was not removed" + done +} + +# --------------------------------------------------------------------------- +# Covered. +# --------------------------------------------------------------------------- +new_case +run_case --confirm +[[ "${rc}" -eq 0 ]] || fail "covered: expected exit 0, got ${rc}. stderr: ${err}" +require_text "${out}" 'COVERED: the dedicated bucket holds every object of the shared Wedding catalogue (3 objects)' 'covered' +require_text "${out}" '"sharedObjects":3,"matchedObjects":3,"retentionPrunedObjects":0,"covered":true' 'covered' +require_cleaned 'covered' +# One credential per pod, each beside its own namespace's Secret. +require_text "$(cat "${k}/pod-umami.yaml")" 'secretName: umami-db-backup-r2' 'shared pod credential' +refute_text "$(cat "${k}/pod-umami.yaml")" 'wedding-db-backup-r2' 'shared pod credential' +require_text "$(cat "${k}/pod-wedding-app.yaml")" 'secretName: wedding-db-backup-r2-dedicated' 'dedicated pod credential' +refute_text "$(cat "${k}/pod-wedding-app.yaml")" 'umami-db-backup-r2' 'dedicated pod credential' +[[ "$(grep -c 'secretName:' "${k}/pod-umami.yaml")" -eq 1 && "$(grep -c 'secretName:' "${k}/pod-wedding-app.yaml")" -eq 1 ]] || + fail 'a pod mounts more than one credential' +# Each pod lists only the Wedding catalogue of its own bucket. +[[ "$(cat "${k}/mc-umami/listed")" == 'store/platform-backups/cnpg/wedding-db/' ]] || fail 'the shared pod listed something else' +[[ "$(cat "${k}/mc-wedding-app/listed")" == 'store/wedding-db-backups/cnpg/wedding-db/' ]] || fail 'the dedicated pod listed something else' +# Only the object an ETag cannot prove is read, on both sides. +[[ "$(cat "${k}/mc-umami/catted")" == "store/platform-backups/cnpg/wedding-db/${base}/data.tar.gz" ]] || fail 'the shared pod hashed the wrong objects' +[[ "$(cat "${k}/mc-wedding-app/catted")" == "store/wedding-db-backups/cnpg/wedding-db/${base}/data.tar.gz" ]] || fail 'the dedicated pod hashed the wrong objects' +refute_text "${out}${err}" 'abc123' 'covered: endpoint host leaked' +printf 'PASS: a covered catalogue is reported as covered and both pods are removed\n' + +# Retention runs on the dedicated store only, so the shared copy outlives backups +# the dedicated store has pruned. +new_case +{ + record 'wedding-db-20260909/base/20260801T030000/backup.info' 1100 p1 + record 'wedding-db-20260909/base/20260801T030000/data.tar.gz' 80000000 p2-5 + record 'wedding-db-20260909/wals/0000000200000000/0000000200000000000000FE.gz' 3900 p3 +} >>"${k}/mc-umami/ls" +run_case --confirm +[[ "${rc}" -eq 0 ]] || fail "retention: expected exit 0, got ${rc}. stderr: ${err}" +require_text "${out}" '"sharedObjects":6,"matchedObjects":3,"retentionPrunedObjects":3,"covered":true' 'retention' +printf 'PASS: objects the dedicated store pruned by retention do not block the verdict\n' + +# After the shared copy is removed, the same proof reports that. +new_case +: >"${k}/mc-umami/ls" +run_case --confirm +[[ "${rc}" -eq 0 ]] || fail "empty: expected exit 0, got ${rc}. stderr: ${err}" +require_text "${out}" 'EMPTY: the shared bucket holds no Wedding catalogue' 'empty' +refute_text "${out}" 'COVERED' 'empty' +require_cleaned 'empty' +printf 'PASS: an empty shared prefix is reported as empty, not as covered\n' + +# --------------------------------------------------------------------------- +# Not covered. +# --------------------------------------------------------------------------- +new_case +record "${later_wal}" 4300 f6 >>"${k}/mc-umami/ls" +{ + record "${base}/backup.info" 1200 a1 + record "${base}/data.tar.gz" 90000000 ff-4 + record "${wal}" 4000 c3 +} >"${k}/mc-wedding-app/ls" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "missing object: expected exit 1, got ${rc}" +require_text "${err}" 'NOT COVERED' 'missing object' +refute_text "${out}" 'COVERED:' 'missing object' +require_cleaned 'missing object' +printf 'PASS: a shared object the dedicated bucket lacks is not covered\n' + +new_case +printf 'different bytes' >"${k}/mc-wedding-app/content" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "different bytes: expected exit 1, got ${rc}" +require_text "${err}" 'checksum mismatch' 'different bytes' +require_text "${err}" 'NOT COVERED' 'different bytes' +printf 'PASS: a multipart object with different bytes is not covered\n' + +new_case +{ + record "${base}/backup.info" 1200 zz + record "${base}/data.tar.gz" 90000000 ff-4 + record "${wal}" 4000 c3 +} >"${k}/mc-wedding-app/ls" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "different etag: expected exit 1, got ${rc}" +require_text "${err}" 'NOT COVERED' 'different etag' +printf 'PASS: a single-part object with a different ETag is not covered\n' + +new_case +printf '1' >"${k}/mc-wedding-app/cat-rc" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "unreadable object: expected exit 1, got ${rc}" +require_text "${err}" 'the dedicated pod could not hash its objects' 'unreadable object' +refute_text "${out}${err}" 'abc123' 'unreadable object: endpoint host leaked' +require_text "${err}" '' 'unreadable object' +printf 'PASS: an object that cannot be read is not covered, and the endpoint host is redacted\n' + +new_case +printf '1' >"${k}/mc-umami/ls-rc" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "failed listing: expected exit 1, got ${rc}" +require_text "${err}" 'the shared pod could not list the shared catalogue' 'failed listing' +refute_text "${out}" 'EMPTY' 'failed listing' +require_cleaned 'failed listing' +printf 'PASS: a failed listing is not an empty catalogue\n' + +new_case +printf '{"status":"error","error":{"message":"denied"}}\n' >>"${k}/mc-umami/ls" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "error record: expected exit 1, got ${rc}" +require_text "${err}" 'returned a record that is not a success' 'error record' +printf 'PASS: a listing carrying an error record is refused\n' + +new_case +touch "${k}/never-ready-umami" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "never ready: expected exit 1, got ${rc}" +require_text "${err}" 'the shared pod never became ready' 'never ready' +require_cleaned 'never ready' +printf 'PASS: a pod that never becomes ready fails the proof and is removed\n' + +# --------------------------------------------------------------------------- +# Refusals before anything is created. +# --------------------------------------------------------------------------- +new_case +run_case +require_untouched 'no confirmation' +require_text "${err}" 'refusing to run: use --confirm' 'no confirmation' +[[ ! -e "${k}/calls" ]] || fail 'no confirmation: the cluster was contacted' + +new_case +run_case --confirm --delete +require_untouched 'extra argument' + +new_case +cluster_json wedding-db >"${k}/wedding-app-clusters.postgresql.cnpg.io-wedding-db.json" +run_case --confirm +require_untouched 'still archiving to the shared store' +require_text "${err}" 'does not archive through wedding-db-dedicated' 'still archiving to the shared store' + +for present in objectstores.barmancloud.cnpg.io-wedding-db externalsecrets.external-secrets.io-wedding-db-backup-r2 secrets-wedding-db-backup-r2; do + new_case + touch "${k}/present-${present}" + run_case --confirm + require_untouched "wedding-app still holds ${present}" + require_text "${err}" 'shared backup access has not been retired' "wedding-app still holds ${present}" +done + +# A failed read is not an absence. +new_case +printf '1' >"${k}/absence-rc" +run_case --confirm +require_untouched 'absence check failed' +require_text "${err}" 'could not check whether wedding-app still holds' 'absence check failed' + +new_case +printf '{"items":[{"spec":{"configuration":{"destinationPath":"s3://platform-backups/cnpg/wedding-db"}}}]}\n' \ + >"${k}/all-objectstores.barmancloud.cnpg.io.json" +run_case --confirm +require_untouched 'a store still names the shared catalogue' +require_text "${err}" 'still names the shared Wedding catalogue' 'a store still names the shared catalogue' + +new_case +printf '{"items":[{"spec":{"externalClusters":[{"barmanObjectStore":{"destinationPath":"s3://platform-backups/cnpg/wedding-db/"}}]}}]}\n' \ + >"${k}/all-clusters.postgresql.cnpg.io.json" +run_case --confirm +require_untouched 'a Cluster still names the shared catalogue' + +# A sibling catalogue whose name merely starts the same way is not a reference. +new_case +printf '{"items":[{"spec":{"configuration":{"destinationPath":"s3://platform-backups/cnpg/wedding-db-archive"}}}]}\n' \ + >"${k}/all-objectstores.barmancloud.cnpg.io.json" +run_case --confirm +[[ "${rc}" -eq 0 ]] || fail "sibling catalogue: expected exit 0, got ${rc}. stderr: ${err}" + +new_case +rm "${k}/all-clusters.postgresql.cnpg.io.json" +run_case --confirm +require_untouched 'reference listing failed' +require_text "${err}" 'could not list every clusters.postgresql.cnpg.io' 'reference listing failed' + +new_case +store_json s3://wedding-db-backups/cnpg/other wedding-db-backup-r2-dedicated \ + >"${k}/wedding-app-objectstores.barmancloud.cnpg.io-wedding-db-dedicated.json" +run_case --confirm +require_untouched 'dedicated store drifted' + +new_case +store_json s3://other-bucket/cnpg/umami-db umami-db-backup-r2 >"${k}/umami-objectstores.barmancloud.cnpg.io-umami-db.json" +run_case --confirm +require_untouched 'shared reference store names another bucket' + +new_case +store_json s3://platform-backups/cnpg/umami-db other-secret >"${k}/umami-objectstores.barmancloud.cnpg.io-umami-db.json" +run_case --confirm +require_untouched 'shared reference store names another Secret' + +new_case +store_json s3://platform-backups/cnpg/umami-db umami-db-backup-r2 https://evil.example.com \ + >"${k}/umami-objectstores.barmancloud.cnpg.io-umami-db.json" +run_case --confirm +require_untouched 'shared reference store names another endpoint' + +new_case +printf 'data:\n r2_endpoint: %s\n r2_bucket: wedding-db-backups\n' "${endpoint}" >"${k}/config-map.yaml" +run_case --confirm +require_untouched 'committed shared bucket is the dedicated bucket' + +new_case +printf 'data:\n r2_endpoint: http://plain.example.com\n r2_bucket: platform-backups\n' >"${k}/config-map.yaml" +run_case --confirm +require_untouched 'committed endpoint is not https' +printf 'PASS: every refusal happens before a pod or script is created\n' + +# --------------------------------------------------------------------------- +# The pod script on its own. +# --------------------------------------------------------------------------- +# pod sets pod_out and pod_rc. +pod() { + pod_rc=0 + pod_out="$(PATH="${bin}:${PATH}" FAKE_MC="${k}/mc-umami" ENDPOINT="${POD_ENDPOINT:-${endpoint}}" \ + ROLE="$1" BUCKET="$2" PREFIX="$3" WORK_DIR="${k}/work-umami" CREDENTIALS_DIR="${k}/credentials" \ + SERVE_TIMEOUT=0 sh "${pod_script}" "$4" 2>&1 &2 + exit 1 +} + +readonly reviewed_prefix='cnpg/wedding-db' +readonly dedicated_bucket='wedding-db-backups' + +for tool in mc sed grep tail sha256sum cut date cat sleep mkdir rm; do + command -v "${tool}" >/dev/null 2>&1 || fail "required tool missing from the image: ${tool}" +done + +: "${ENDPOINT:?ENDPOINT is required}" +: "${ROLE:?ROLE is required}" +: "${BUCKET:?BUCKET is required}" +: "${PREFIX:?PREFIX is required}" + +case "${ENDPOINT}" in + https://*) ;; + *) fail "the R2 endpoint must be https" ;; +esac +[ "${PREFIX}" = "${reviewed_prefix}" ] || fail "the catalogue prefix is not the reviewed one" +case "${ROLE}" in + shared) + [ "${BUCKET}" != "${dedicated_bucket}" ] || fail "the shared side names the dedicated bucket" + ;; + dedicated) + [ "${BUCKET}" = "${dedicated_bucket}" ] || fail "the dedicated side names another bucket" + ;; + *) fail "the role must be shared or dedicated" ;; +esac + +credentials="${CREDENTIALS_DIR:-/credentials}" +work="${WORK_DIR:-/tmp/coverage}" +mkdir -p "${work}" +export MC_CONFIG_DIR="${MC_CONFIG_DIR:-${work}/.mc}" +mc_err="${work}/mc.err" + +# redact_errors prints the last lines of an mc error log without the endpoint host. +redact_errors() { + sed -e 's#https://[^/ ]*##g' "${mc_err}" | tail -n 5 >&2 +} + +command="${1:-}" +case "${command}" in + serve) + [ "$#" -eq 1 ] || fail "serve takes no arguments" + access_key="$(cat "${credentials}/ACCESS_KEY_ID")" + [ -n "${access_key}" ] || fail "the credential is empty" + if ! mc alias set store "${ENDPOINT}" "${access_key}" \ + "$(cat "${credentials}/SECRET_ACCESS_KEY")" --api S3v4 >/dev/null 2>"${mc_err}"; then + redact_errors + fail "could not configure the ${ROLE} credential" + fi + unset access_key + printf '==== COVERAGE POD READY ====\n' + waited=0 + while [ "${waited}" -lt "${SERVE_TIMEOUT:-3000}" ]; do + sleep 5 + waited=$((waited + 5)) + done + ;; + + list) + [ "$#" -eq 1 ] || fail "list takes no arguments" + output="${work}/listing" + # Truncating to the second can only move the start earlier, which is the + # side the evaluator's ordering check is strict about. + started="$(date -u +%Y-%m-%dT%H:%M:%SZ)" + if ! mc ls --json --recursive "store/${BUCKET}/${PREFIX}/" >"${output}.raw" 2>"${mc_err}" "${output}.partial" + files=0 + while IFS= read -r line; do + [ -n "${line}" ] || continue + case "${line}" in + *'"status":"success"'*) ;; + *) fail "listing the ${ROLE} catalogue returned a record that is not a success" ;; + esac + case "${line}" in + *'"type":"folder"'*) continue ;; + *'"type":"file"'*) ;; + *) fail "listing the ${ROLE} catalogue returned an unexpected record type" ;; + esac + + line="$(printf '%s' "${line}" | sed -e 's/"url":"[^"]*",//' -e 's/,"url":"[^"]*"//')" + key="$(printf '%s' "${line}" | sed -n 's/.*"key":"\([^"\\]*\)".*/\1/p')" + [ -n "${key}" ] || fail "listing the ${ROLE} catalogue returned an object without a plain key" + + printf '%s\n' "${line}" >>"${output}.partial" + files=$((files + 1)) + done <"${output}.raw" + + printf '{"status":"success","type":"listing-complete","location":"%s/%s","started":"%s","files":%d}\n' \ + "${BUCKET}" "${PREFIX}" "${started}" "${files}" >>"${output}.partial" + cat "${output}.partial" >"${output}" + printf 'coverage-pod: listed %s objects on the %s side\n' "${files}" "${ROLE}" + ;; + + hash) + [ "$#" -eq 1 ] || fail "hash takes no arguments" + [ -e "${work}/listing" ] || fail "hash needs a listing" + : >"${work}/sums.partial" + hashed=0 + while IFS= read -r key; do + [ -n "${key}" ] || continue + grep -qF "\"key\":\"${key}\"" "${work}/listing" || fail "asked to hash a key the listing does not hold" + rm -f "${work}/cat-failed" + sum="$({ mc cat "store/${BUCKET}/${PREFIX}/${key}" "${mc_err}" || + : >"${work}/cat-failed"; } | sha256sum | cut -d ' ' -f 1)" + if [ -e "${work}/cat-failed" ]; then + redact_errors + fail "could not read ${key} to checksum it" + fi + printf '%s' "${sum}" | grep -Eq '^[0-9a-f]{64}$' || fail "malformed checksum for ${key}" + printf '%s\t%s\n' "${key}" "${sum}" >>"${work}/sums.partial" + hashed=$((hashed + 1)) + done + cat "${work}/sums.partial" >"${work}/sums" + printf 'coverage-pod: hashed %s objects on the %s side\n' "${hashed}" "${ROLE}" + ;; + + *) fail "unknown command" ;; +esac diff --git a/scripts/verify-wedding-shared-backup-coverage.sh b/scripts/verify-wedding-shared-backup-coverage.sh new file mode 100755 index 0000000000..c608d431cd --- /dev/null +++ b/scripts/verify-wedding-shared-backup-coverage.sh @@ -0,0 +1,441 @@ +#!/usr/bin/env bash + +# Prove that the dedicated Wedding backup bucket covers everything the stale +# copy of the catalogue in the shared backup bucket still holds (#4481). +# +# Until the cutover the Wedding database archived into the shared backup bucket. +# That copy was mirrored into the dedicated bucket, and #3253 removed the store +# that pointed at it, so nothing applies retention to it any more and every +# holder of the shared backup credential can still read it. Removing it is +# irreversible, so the decision needs current evidence that nothing would be +# lost. This is that evidence, and it is all this does. +# +# READ ONLY. It lists both buckets and reads the objects it has to hash. It +# never writes to or deletes from either bucket. +# +# WHAT IT DOES, in order, refusing at the first thing it cannot prove: +# 1. Confirms the wedding-db Cluster archives through wedding-db-dedicated with +# a healthy WAL archiver, that the wedding-app namespace no longer holds the +# shared store or its credential, and that no ObjectStore or Cluster anywhere +# still names the shared catalogue. +# 2. Reads the dedicated ObjectStore and the Umami ObjectStore, whose Secret is +# the shared backup credential, and requires the reviewed destinations at the +# committed endpoint and bucket. +# 3. Starts one pod beside each credential (scripts/verify-wedding-shared-backup- +# coverage-pod.sh). No namespace holds both since #3253. The shared pod lists +# first; the dedicated pod lists afterwards. +# 4. Has both pods hash every object an ETag cannot prove, and runs the reviewed +# evaluator (`evaluate-coverage`): every shared object must be in the +# dedicated catalogue with matching content, or be older than what the +# dedicated store's retention still keeps. +# 5. Repeats step 1. +# +# Exit status: 0 covered; 1 refused, failed or not covered. Needs --confirm, +# because it starts pods beside two production backup credentials. + +set -euo pipefail + +readonly context='admin@prod' +readonly tenant_namespace='wedding-app' +readonly shared_namespace='umami' +readonly cluster='wedding-db' +readonly retired_store='wedding-db' +readonly retired_secret='wedding-db-backup-r2' +readonly dedicated_store='wedding-db-dedicated' +readonly dedicated_secret='wedding-db-backup-r2-dedicated' +readonly dedicated_bucket='wedding-db-backups' +readonly shared_reference_store='umami-db' +readonly shared_reference_prefix='cnpg/umami-db' +readonly shared_secret='umami-db-backup-r2' +readonly catalogue_prefix='cnpg/wedding-db' +readonly plugin='barman-cloud.cloudnative-pg.io' +readonly ready_marker='==== COVERAGE POD READY ====' +# The same digest-pinned images as the catalogue mirror, whose runtime test +# exercises them; the coverage test asserts the two scripts stay identical here. +readonly mc_image='quay.io/minio/aistor/mc:RELEASE.2026-03-12T04-18-55Z@sha256:6c33dc0fbf65c362be95003cd010ed95a41c556500833ea139f86de40c4c4e9f' +readonly tools_image='docker.io/library/busybox:1.38.0-musl@sha256:ea2b9914a16a4ac1981994af97b318f7c7d4db76b580c56177f08bf76f4a0be8' + +root_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +readonly root_dir +readonly pod_script="${root_dir}/scripts/verify-wedding-shared-backup-coverage-pod.sh" + +kubectl_bin="${KUBECTL:-kubectl}" +readonly kubectl_bin +poll_interval="${COVERAGE_POLL_INTERVAL:-10}" +poll_limit="${COVERAGE_POLL_LIMIT:-60}" +readonly poll_interval poll_limit + +work_dir="$(mktemp -d)" +readonly work_dir +# Cleanup removes only what this run created, never a same-named object it found. +created=() +name='' +# Invoked by the EXIT trap below. shellcheck reports this trap-only handler as +# unused (SC2329 on 0.11) or unreachable (SC2317 on the older CI runner); the +# test suite asserts the deletions actually happen. +# shellcheck disable=SC2317,SC2329 +cleanup() { + local entry + for entry in ${created[@]+"${created[@]}"}; do + kube "${entry%%/*}" delete "${entry#*/}" "${name}" --ignore-not-found --wait=false >/dev/null 2>&1 || true + done + rm -rf "${work_dir}" +} +trap cleanup EXIT + +fail() { + printf 'verify-wedding-shared-backup-coverage: %s\n' "$1" >&2 + exit 1 +} + +[[ "$#" -eq 1 && "$1" == '--confirm' ]] || + fail 'refusing to run: use --confirm. The proof starts pods beside two production backup credentials. Nothing has been touched.' + +# A local run is named by its start time, so it never shares a name with an +# earlier run's pod. +run_id="${GITHUB_RUN_ID:-local$(date +%s)}-${GITHUB_RUN_ATTEMPT:-1}" +[[ "${run_id}" =~ ^[a-z0-9-]{1,40}$ ]] || fail 'the run identifier is not a valid name fragment' +name="wedding-backup-coverage-${run_id}" +readonly name + +# kube +kube() { + local namespace="$1" + shift + "${kubectl_bin}" --context "${context}" --namespace "${namespace}" --request-timeout=20s "$@" +} + +evaluator="${COVERAGE_EVALUATOR:-}" +if [[ -z "${evaluator}" ]]; then + evaluator="${work_dir}/evaluator" + (cd "${root_dir}" && go build -o "${evaluator}" ./scripts/mirror-wedding-backup-catalogue) || + fail 'could not build the evaluator' +fi +readonly evaluator + +# The committed bootstrap ConfigMap names the shared bucket and the R2 endpoint. +# A live value is not trusted on its own: a drifted store could name a foreign +# host or bucket, and a credential is about to be handed to that endpoint. +bootstrap_config="${COVERAGE_BOOTSTRAP_CONFIG:-${root_dir}/k8s/bases/bootstrap/config-map.yaml}" +committed() { + local values + values="$(sed -n "s/^ $1: *//p" "${bootstrap_config}")" || fail "could not read the committed $1" + [[ -n "${values}" && "${values}" != *$'\n'* ]] || fail "the bootstrap ConfigMap does not commit exactly one $1" + printf '%s' "${values}" +} +endpoint="$(committed r2_endpoint)" +shared_bucket="$(committed r2_bucket)" +readonly endpoint shared_bucket +[[ "${endpoint}" =~ ^https://[a-z0-9][a-z0-9.-]*$ ]] || + fail 'the committed R2 endpoint is not an https URL with a bare host' +[[ "${shared_bucket}" =~ ^[a-z0-9][a-z0-9.-]*$ ]] || + fail 'the committed shared backup bucket is not a valid bucket name' +[[ "${shared_bucket}" != "${dedicated_bucket}" ]] || + fail 'the committed shared backup bucket is the dedicated bucket' +readonly shared_catalogue="s3://${shared_bucket}/${catalogue_prefix}" + +# require_cluster refuses unless the Cluster archives through the dedicated store +# with a healthy active WAL archiver, and is the same Cluster as on the first call. +cluster_uid='' +require_cluster() { + local uid + kube "${tenant_namespace}" get clusters.postgresql.cnpg.io "${cluster}" -o json >"${work_dir}/cluster.json" 2>/dev/null || + fail "could not read the ${cluster} Cluster" + jq -e --arg plugin "${plugin}" --arg store "${dedicated_store}" ' + .metadata.generation as $generation | + def healthy($kind): + [.status.conditions[]? | select(.type == $kind)] | + length == 1 and .[0].status == "True" and + (.[0].observedGeneration == null or .[0].observedGeneration == $generation); + .metadata.deletionTimestamp == null and + (.metadata.uid | type == "string" and length > 0) and + (.spec.instances | type == "number" and . > 0 and floor == .) and + .status.readyInstances == .spec.instances and + ([.spec.plugins[]? | select(.name == $plugin)] | + length == 1 and .[0].enabled == true and .[0].isWALArchiver == true and + .[0].parameters.barmanObjectName == $store) and + healthy("Ready") and healthy("ContinuousArchiving") + ' "${work_dir}/cluster.json" >/dev/null 2>&1 || + fail "the ${cluster} Cluster does not archive through ${dedicated_store} with a healthy active WAL archiver" + uid="$(jq -r '.metadata.uid' "${work_dir}/cluster.json")" + if [[ -z "${cluster_uid}" ]]; then + cluster_uid="${uid}" + elif [[ "${uid}" != "${cluster_uid}" ]]; then + fail "the ${cluster} Cluster was replaced during the run" + fi +} + +# require_absent refuses unless the wedding-app namespace is +# observed not to hold the object. A failed read is not an absence. +require_absent() { + local found + found="$(kube "${tenant_namespace}" get "$1" "$2" --ignore-not-found --output=name 2>/dev/null)" || + fail "could not check whether ${tenant_namespace} still holds $1/$2" + [[ -z "${found}" ]] || + fail "${tenant_namespace} still holds $1/$2: shared backup access has not been retired (#3253)" +} + +# require_unreferenced refuses when any ObjectStore or Cluster in the cluster +# still names the shared catalogue, in any field: something could then still be +# writing to it, and the coverage shown here would not last. +require_unreferenced() { + local kind + for kind in objectstores.barmancloud.cnpg.io clusters.postgresql.cnpg.io; do + "${kubectl_bin}" --context "${context}" --request-timeout=20s get "${kind}" --all-namespaces -o json \ + >"${work_dir}/references.json" 2>/dev/null || + fail "could not list every ${kind}" + jq -e --arg path "${shared_catalogue}" ' + (.items | type == "array") and + ([.items[].spec | .. | strings | select(. == $path or startswith($path + "/"))] | length == 0) + ' "${work_dir}/references.json" >/dev/null 2>&1 || + fail "a ${kind} still names the shared Wedding catalogue" + done +} + +require_retired() { + require_cluster + require_absent objectstores.barmancloud.cnpg.io "${retired_store}" + require_absent externalsecrets.external-secrets.io "${retired_secret}" + require_absent secrets "${retired_secret}" + require_unreferenced +} + +# require_store refuses unless the +# live ObjectStore writes to the reviewed destination through exactly the +# reviewed Secret keys the pod mounts, at the committed endpoint. +require_store() { + local namespace="$1" store="$2" destination="$3" secret="$4" + kube "${namespace}" get objectstores.barmancloud.cnpg.io "${store}" -o json >"${work_dir}/store.json" 2>/dev/null || + fail "could not read the ${store} ObjectStore" + jq -e --arg path "${destination}" --arg endpoint "${endpoint}" --arg secret "${secret}" ' + .metadata.deletionTimestamp == null and + .spec.configuration.destinationPath == $path and + .spec.configuration.endpointURL == $endpoint and + .spec.configuration.s3Credentials.accessKeyId == {name:$secret,key:"ACCESS_KEY_ID"} and + .spec.configuration.s3Credentials.secretAccessKey == {name:$secret,key:"SECRET_ACCESS_KEY"} + ' "${work_dir}/store.json" >/dev/null 2>&1 || + fail "the ${store} ObjectStore is not wired to the reviewed destination and credential" +} + +require_retired +require_store "${tenant_namespace}" "${dedicated_store}" "s3://${dedicated_bucket}/${catalogue_prefix}" "${dedicated_secret}" +# The Umami store proves which Secret holds the credential for the committed +# shared bucket. Its own catalogue is a sibling of the Wedding one and is never +# listed. +require_store "${shared_namespace}" "${shared_reference_store}" "s3://${shared_bucket}/${shared_reference_prefix}" "${shared_secret}" + +# start_pod +start_pod() { + local namespace="$1" role="$2" secret="$3" bucket="$4" manifest + kube "${namespace}" create configmap "${name}" --from-file="coverage.sh=${pod_script}" >/dev/null || + fail "could not stage the coverage script in ${namespace}" + created+=("${namespace}/configmap") + + # The heredoc is quoted, so nothing in it expands, and each placeholder is + # replaced by a value validated above. Every one of those values is restricted + # to characters that cannot end a YAML string or act as a replacement pattern + # (no quote, newline, backslash or `&`), which is what keeps the unquoted + # replacements below portable across bash 3.2 and 5.2. + manifest="$( + cat <<'MANIFEST' +apiVersion: v1 +kind: Pod +metadata: + name: __NAME__ + namespace: __NAMESPACE__ + labels: + app.kubernetes.io/name: wedding-backup-coverage + app.kubernetes.io/managed-by: verify-wedding-shared-backup-coverage +spec: + restartPolicy: Never + automountServiceAccountToken: false + activeDeadlineSeconds: 3600 + securityContext: + runAsNonRoot: true + runAsUser: 65532 + runAsGroup: 65532 + fsGroup: 65532 + seccompProfile: + type: RuntimeDefault + volumes: + - name: tools + emptyDir: + sizeLimit: 8Mi + - name: script + configMap: + name: __NAME__ + - name: credential + secret: + secretName: __SECRET__ + items: + - key: ACCESS_KEY_ID + path: ACCESS_KEY_ID + - key: SECRET_ACCESS_KEY + path: SECRET_ACCESS_KEY + - name: work + emptyDir: + sizeLimit: 1Gi + # `mc alias set` writes the credential into its config directory. Keep that + # in memory so the key never lands on node storage. + - name: mc-config + emptyDir: + medium: Memory + sizeLimit: 1Mi + initContainers: + - name: install-tools + image: __TOOLS_IMAGE__ + command: + - /bin/sh + - -ec + - cp /bin/busybox /tools/busybox; /tools/busybox --install -s /tools + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: ["ALL"] + resources: + requests: + cpu: 10m + memory: 8Mi + limits: + cpu: 100m + memory: 32Mi + volumeMounts: + - name: tools + mountPath: /tools + containers: + - name: coverage + image: __IMAGE__ + command: ["/tools/sh", "/coverage/coverage.sh", "serve"] + securityContext: + runAsNonRoot: true + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: ["ALL"] + env: + - name: PATH + value: /tools:/usr/local/bin:/usr/bin:/bin + - name: ENDPOINT + value: "__ENDPOINT__" + - name: ROLE + value: "__ROLE__" + - name: BUCKET + value: "__BUCKET__" + - name: PREFIX + value: "__PREFIX__" + - name: CREDENTIALS_DIR + value: /credentials + - name: WORK_DIR + value: /work + - name: MC_CONFIG_DIR + value: /mc-config + resources: + requests: + cpu: 50m + memory: 64Mi + limits: + memory: 512Mi + volumeMounts: + - name: tools + mountPath: /tools + readOnly: true + - name: script + mountPath: /coverage + readOnly: true + - name: credential + mountPath: /credentials + readOnly: true + - name: work + mountPath: /work + - name: mc-config + mountPath: /mc-config +MANIFEST + )" + manifest="${manifest//__NAME__/${name}}" + manifest="${manifest//__NAMESPACE__/${namespace}}" + manifest="${manifest//__IMAGE__/${mc_image}}" + manifest="${manifest//__TOOLS_IMAGE__/${tools_image}}" + manifest="${manifest//__SECRET__/${secret}}" + manifest="${manifest//__ENDPOINT__/${endpoint}}" + manifest="${manifest//__ROLE__/${role}}" + manifest="${manifest//__BUCKET__/${bucket}}" + manifest="${manifest//__PREFIX__/${catalogue_prefix}}" + + created+=("${namespace}/pod") + printf '%s\n' "${manifest}" | kube "${namespace}" create -f - >/dev/null || + fail "could not start the ${role} pod" +} + +# wait_ready waits for the pod to store its credential. +wait_ready() { + local namespace="$1" role="$2" phase='' attempt + for ((attempt = 0; attempt < poll_limit; attempt++)); do + phase="$(kube "${namespace}" get pod "${name}" -o 'jsonpath={.status.phase}')" || phase='' + [[ "${phase}" == Succeeded || "${phase}" == Failed ]] && break + if [[ "${phase}" == Running ]] && + kube "${namespace}" logs "pod/${name}" -c coverage 2>/dev/null | grep -qxF "${ready_marker}"; then + return 0 + fi + sleep "${poll_interval}" + done + kube "${namespace}" logs "pod/${name}" -c coverage 2>/dev/null | grep '^coverage-pod: ' >&2 || true + fail "the ${role} pod never became ready (phase '${phase:-unknown}')" +} + +# in_pod runs one command of the pod script. The long +# request timeout covers hashing a base backup. +in_pod() { + "${kubectl_bin}" --context "${context}" --namespace "$1" --request-timeout=45m \ + exec -i "${name}" -c coverage -- /tools/sh /coverage/coverage.sh "$2" +} + +# collect +collect() { + "${kubectl_bin}" --context "${context}" --namespace "$1" --request-timeout=10m \ + exec "${name}" -c coverage -- cat "/work/$2" >"${work_dir}/$3" || + fail "could not collect $2 from the pod in $1" +} + +start_pod "${shared_namespace}" shared "${shared_secret}" "${shared_bucket}" +start_pod "${tenant_namespace}" dedicated "${dedicated_secret}" "${dedicated_bucket}" +wait_ready "${shared_namespace}" shared +wait_ready "${tenant_namespace}" dedicated + +# The dedicated listing is taken after the shared one, because the evaluator +# refuses a dedicated listing that started earlier. +in_pod "${shared_namespace}" list "${work_dir}/multipart-keys" || fail 'the evaluator refused the listings' +in_pod "${shared_namespace}" hash <"${work_dir}/multipart-keys" || fail 'the shared pod could not hash its objects' +in_pod "${tenant_namespace}" hash <"${work_dir}/multipart-keys" || fail 'the dedicated pod could not hash its objects' +collect "${shared_namespace}" sums shared-sums +collect "${tenant_namespace}" sums dedicated-sums + +"${evaluator}" evaluate-coverage "${shared_bucket}" "${work_dir}/shared" "${work_dir}/dedicated" \ + "${work_dir}/shared-sums" "${work_dir}/dedicated-sums" >"${work_dir}/summary.json" || + fail 'NOT COVERED: the dedicated catalogue does not hold everything the shared copy holds. Keep the shared copy.' +jq -e '.covered == true and (.sharedObjects | type == "number" and . > 0)' \ + "${work_dir}/summary.json" >/dev/null 2>&1 || + fail 'the evaluator summary is not a covered verdict' + +require_retired + +cat "${work_dir}/summary.json" +printf 'COVERED: the dedicated bucket holds every object of the shared Wedding catalogue (%s objects), or has pruned it by retention.\n' \ + "$(jq -r '.sharedObjects' "${work_dir}/summary.json")" +printf 'Nothing was changed. This holds as of this run: the shared copy is no longer written to.\n' From 4f87679094887302fdb181b10fa47bc5e11d8de4 Mon Sep 17 00:00:00 2001 From: Nikolai Emil Damm Date: Mon, 5 Oct 2026 00:32:09 +0200 Subject: [PATCH 2/4] fix(backup): accept a missing shared object only when retention must have removed it Review found that the tolerance for retention-pruned objects compared names against whatever the dedicated bucket happened to hold, so WAL or a base backup that was never copied passed as pruned. Judge by when the object was written instead: before the oldest complete backup the dedicated store keeps and before its 30-day retention window, and require the live store to declare that window. Also refuse a non-empty object hashed as empty, refuse a store rooted above the catalogue, redact bare hosts and addresses from pod errors, and report how many objects were matched and how many were accepted as pruned. Part of #4481 Co-Authored-By: Claude Opus 5.5 --- .../coverage.go | 99 ++++++++++++------- .../coverage_test.go | 30 ++++-- ...t-verify-wedding-shared-backup-coverage.sh | 40 ++++++-- ...rify-wedding-shared-backup-coverage-pod.sh | 9 +- .../verify-wedding-shared-backup-coverage.sh | 16 ++- 5 files changed, 134 insertions(+), 60 deletions(-) diff --git a/scripts/mirror-wedding-backup-catalogue/coverage.go b/scripts/mirror-wedding-backup-catalogue/coverage.go index 28ddf8be13..e06bb71cdc 100644 --- a/scripts/mirror-wedding-backup-catalogue/coverage.go +++ b/scripts/mirror-wedding-backup-catalogue/coverage.go @@ -17,8 +17,9 @@ package main // // The dedicated store applies its retention policy and the shared copy no longer // does, so the shared copy can hold objects the dedicated store has since -// pruned. Such an object is accepted only when it is older than everything the -// dedicated store still keeps of the same kind in the same server directory. +// pruned. Such an object is accepted only when it was written before the oldest +// complete base backup the dedicated store keeps in the same server directory +// and before the retention window began. // Every other shared object must be present with matching content. import ( @@ -29,6 +30,7 @@ import ( "os" "sort" "strings" + "time" ) var ( @@ -102,31 +104,48 @@ func ApplySums(objects []Object, r io.Reader) error { return nil } -// retentionFloor records, per server directory, the oldest base backup and the -// oldest WAL segment the dedicated store still keeps. Retention only ever -// removes from the old end, so anything older than these is gone by design. -type retentionFloor struct { - backup map[string]string - wal map[string]string -} +// retentionWindow is the dedicated ObjectStore's retention policy. The runner +// refuses a store that declares another one. +const retentionWindow = 30 * 24 * time.Hour + +// emptyDigest is the sha256 of no bytes. A non-empty object that hashes to it +// was not read. +const emptyDigest = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" -func retentionFloorOf(objects map[string]Object) retentionFloor { - floor := retentionFloor{backup: map[string]string{}, wal: map[string]string{}} - for key := range objects { +// oldestCompleteBackups returns, per server directory, the start time of the +// oldest base backup that has both its backup.info and a non-empty data archive. +func oldestCompleteBackups(objects map[string]Object) map[string]time.Time { + hasInfo := map[string]bool{} + hasData := map[string]bool{} + for key, object := range objects { + id := baseBackupIDOf(key) parts := strings.Split(key, "/") - server := parts[0] - if id := baseBackupIDOf(key); id != "" { - if current, ok := floor.backup[server]; !ok || id < current { - floor.backup[server] = id - } + if id == "" || len(parts) != 4 { + continue } - if segment := walSegmentOf(key); segment != "" { - if current, ok := floor.wal[server]; !ok || segment < current { - floor.wal[server] = segment - } + backup := parts[0] + "/" + id + switch { + case parts[3] == "backup.info": + hasInfo[backup] = true + case dataArchive.MatchString(parts[3]) && object.Size > 0: + hasData[backup] = true } } - return floor + oldest := map[string]time.Time{} + for backup := range hasInfo { + if !hasData[backup] { + continue + } + server, id, _ := strings.Cut(backup, "/") + started, err := time.Parse("20060102T150405", id) + if err != nil { + continue + } + if current, ok := oldest[server]; !ok || started.Before(current) { + oldest[server] = started + } + } + return oldest } // baseBackupIDOf returns the backup ID of a key inside a base backup directory. @@ -156,21 +175,22 @@ func walPositionOf(key string) string { return segment } -// prunedByRetention reports whether a shared object the dedicated store lacks is -// older than everything that store still keeps of its kind in its server -// directory. A server directory the dedicated store holds nothing of has no -// floor, so nothing in it is accepted as pruned. -func (floor retentionFloor) prunedByRetention(key string) bool { - server := strings.Split(key, "/")[0] - if id := baseBackupIDOf(key); id != "" { - oldest, ok := floor.backup[server] - return ok && id < oldest - } - if position := walPositionOf(key); position != "" { - oldest, ok := floor.wal[server] - return ok && position < oldest +// prunedByRetention reports whether a shared object the dedicated store lacks +// can only be missing because retention removed it. Retention never removes +// anything written at or after the start of the oldest backup it keeps, and +// never removes anything still inside the retention window, so the object must +// have been written before both. It is judged by when it was written, not by its +// name: segment names do not order across timelines, and an object that was +// never copied has a name just like a pruned one. A server directory with no +// complete dedicated backup has nothing to measure against, so nothing in it is +// accepted. +func prunedByRetention(object Object, oldest map[string]time.Time, listed time.Time) bool { + if baseBackupIDOf(object.Key) == "" && walPositionOf(object.Key) == "" { + return false } - return false + started, ok := oldest[strings.Split(object.Key, "/")[0]] + return ok && object.LastModified.Before(started) && + object.LastModified.Before(listed.Add(-retentionWindow)) } // EvaluateCoverage proves the dedicated catalogue covers the shared one. Every @@ -195,12 +215,12 @@ func EvaluateCoverage(shared, dedicated Listing) (CoverageSummary, error) { return CoverageSummary{}, ErrEmptySource } - floor := retentionFloorOf(dedicatedIndex) + oldest := oldestCompleteBackups(dedicatedIndex) matched, pruned := 0, 0 for key, source := range sharedIndex { copied, ok := dedicatedIndex[key] if !ok { - if !floor.prunedByRetention(key) { + if !prunedByRetention(source, oldest, dedicated.Started) { return CoverageSummary{}, fmt.Errorf("%w: %s", ErrNotCovered, key) } pruned++ @@ -209,6 +229,9 @@ func EvaluateCoverage(shared, dedicated Listing) (CoverageSummary, error) { if copied.Size != source.Size { return CoverageSummary{}, fmt.Errorf("%w: %s", ErrPartialCopy, key) } + if source.Size > 0 && (source.SHA256 == emptyDigest || copied.SHA256 == emptyDigest) { + return CoverageSummary{}, fmt.Errorf("%w: %s was hashed as empty", ErrUnverifiable, key) + } if err := sameContent(source, copied); err != nil { return CoverageSummary{}, fmt.Errorf("%w: %s", err, key) } diff --git a/scripts/mirror-wedding-backup-catalogue/coverage_test.go b/scripts/mirror-wedding-backup-catalogue/coverage_test.go index af4b64f6b4..f39a91cc57 100644 --- a/scripts/mirror-wedding-backup-catalogue/coverage_test.go +++ b/scripts/mirror-wedding-backup-catalogue/coverage_test.go @@ -27,6 +27,9 @@ const ( var ( sharedListed = time.Date(2026, 10, 5, 9, 0, 0, 0, time.UTC) dedicatedListed = sharedListed.Add(time.Minute) + // Written before the oldest dedicated base backup started and before the + // retention window began. + longAgo = time.Date(2026, 9, 2, 0, 0, 0, 0, time.UTC) ) func sharedCatalogue() Listing { @@ -58,10 +61,10 @@ func TestEvaluateCoverageProvesTheDedicatedCatalogueCoversTheSharedOne(t *testin func TestEvaluateCoverageAcceptsObjectsTheDedicatedStorePrunedByRetention(t *testing.T) { shared := sharedCatalogue() shared.Objects = append(shared.Objects, - Object{Key: prunedInfo, Size: 1100, ETag: "p1", LastModified: beforeRun}, - Object{Key: prunedData, Size: 80_000_000, ETag: "p2-5", LastModified: beforeRun}, - Object{Key: prunedWAL, Size: 3900, ETag: "p3", LastModified: beforeRun}, - Object{Key: prunedLabel, Size: 300, ETag: "p4", LastModified: beforeRun}, + Object{Key: prunedInfo, Size: 1100, ETag: "p1", LastModified: longAgo}, + Object{Key: prunedData, Size: 80_000_000, ETag: "p2-5", LastModified: longAgo}, + Object{Key: prunedWAL, Size: 3900, ETag: "p3", LastModified: longAgo}, + Object{Key: prunedLabel, Size: 300, ETag: "p4", LastModified: longAgo}, ) summary, err := EvaluateCoverage(shared, dedicatedCatalogue()) if err != nil { @@ -86,13 +89,24 @@ func TestEvaluateCoverageRefusals(t *testing.T) { }, ErrNotCovered}, {"a server directory the dedicated store holds nothing of", func(shared, _ *Listing) { shared.Objects = append(shared.Objects, - Object{Key: oldServerInfo, Size: 900, ETag: "o1", LastModified: beforeRun}, - Object{Key: oldServerData, Size: 900, ETag: "o2", LastModified: beforeRun}, - Object{Key: oldServerWAL, Size: 900, ETag: "o3", LastModified: beforeRun}) + Object{Key: oldServerInfo, Size: 900, ETag: "o1", LastModified: longAgo}, + Object{Key: oldServerData, Size: 900, ETag: "o2", LastModified: longAgo}, + Object{Key: oldServerWAL, Size: 900, ETag: "o3", LastModified: longAgo}) }, ErrNotCovered}, {"a timeline history file is never pruned", func(shared, _ *Listing) { - shared.Objects = append(shared.Objects, Object{Key: history, Size: 80, ETag: "h1", LastModified: beforeRun}) + shared.Objects = append(shared.Objects, Object{Key: history, Size: 80, ETag: "h1", LastModified: longAgo}) }, ErrNotCovered}, + {"WAL archived after the oldest dedicated backup started", func(shared, _ *Listing) { + shared.Objects = append(shared.Objects, Object{Key: prunedWAL, Size: 3900, ETag: "p3", LastModified: beforeRun}) + }, ErrNotCovered}, + {"an older backup still inside the retention window", func(shared, _ *Listing) { + shared.Objects = append(shared.Objects, + Object{Key: prunedInfo, Size: 1100, ETag: "p1", LastModified: time.Date(2026, 9, 7, 0, 0, 0, 0, time.UTC)}) + }, ErrNotCovered}, + {"a non-empty object hashed as empty", func(shared, dedicated *Listing) { + shared.Objects[1].SHA256 = emptyDigest + dedicated.Objects[1].SHA256 = emptyDigest + }, ErrUnverifiable}, {"a different size", func(_, dedicated *Listing) { dedicated.Objects[0].Size++ }, ErrPartialCopy}, diff --git a/scripts/tests/test-verify-wedding-shared-backup-coverage.sh b/scripts/tests/test-verify-wedding-shared-backup-coverage.sh index b9552fb083..0143fb5d09 100755 --- a/scripts/tests/test-verify-wedding-shared-backup-coverage.sh +++ b/scripts/tests/test-verify-wedding-shared-backup-coverage.sh @@ -72,6 +72,7 @@ set -u f="${FAKE_MC}" leak() { printf 'mc: request to https://abc123.r2.cloudflarestorage.com/x failed\n' >&2 + printf 'mc: dial tcp: lookup abc123.r2.cloudflarestorage.com on 10.96.0.10:53: no such host\n' >&2 } case "$1" in alias) @@ -179,16 +180,16 @@ readonly base='wedding-db-20260909/base/20260908T030000' readonly wal='wedding-db-20260909/wals/0000000200000001/000000020000000100000003.gz' readonly later_wal='wedding-db-20260909/wals/0000000300000001/000000030000000100000009.gz' -# record +# record [last-modified] record() { printf '{"status":"success","type":"file","lastModified":"%s","size":%s,"key":"%s","etag":"%s","url":"%s","versionOrdinal":1,"storageClass":"STANDARD"}\n' \ - "${old}" "$2" "$1" "$3" "${endpoint}" + "${4:-${old}}" "$2" "$1" "$3" "${endpoint}" } -# store_json [endpoint] +# store_json [endpoint] [retention] store_json() { - printf '{"metadata":{},"spec":{"configuration":{"destinationPath":"%s","endpointURL":"%s","s3Credentials":{"accessKeyId":{"name":"%s","key":"ACCESS_KEY_ID"},"secretAccessKey":{"name":"%s","key":"SECRET_ACCESS_KEY"}}}}}\n' \ - "$1" "${3:-${endpoint}}" "$2" "$2" + printf '{"metadata":{},"spec":{"retentionPolicy":"%s","configuration":{"destinationPath":"%s","endpointURL":"%s","s3Credentials":{"accessKeyId":{"name":"%s","key":"ACCESS_KEY_ID"},"secretAccessKey":{"name":"%s","key":"SECRET_ACCESS_KEY"}}}}}\n' \ + "${4:-30d}" "$1" "${3:-${endpoint}}" "$2" "$2" } # cluster_json @@ -264,7 +265,7 @@ require_cleaned() { new_case run_case --confirm [[ "${rc}" -eq 0 ]] || fail "covered: expected exit 0, got ${rc}. stderr: ${err}" -require_text "${out}" 'COVERED: the dedicated bucket holds every object of the shared Wedding catalogue (3 objects)' 'covered' +require_text "${out}" 'COVERED: of the 3 objects in the shared Wedding catalogue, the dedicated bucket holds 3 with matching content.' 'covered' require_text "${out}" '"sharedObjects":3,"matchedObjects":3,"retentionPrunedObjects":0,"covered":true' 'covered' require_cleaned 'covered' # One credential per pod, each beside its own namespace's Secret. @@ -294,8 +295,18 @@ new_case run_case --confirm [[ "${rc}" -eq 0 ]] || fail "retention: expected exit 0, got ${rc}. stderr: ${err}" require_text "${out}" '"sharedObjects":6,"matchedObjects":3,"retentionPrunedObjects":3,"covered":true' 'retention' +require_text "${out}" 'The other 3 predate both the oldest backup the dedicated store keeps' 'retention' printf 'PASS: objects the dedicated store pruned by retention do not block the verdict\n' +# A segment only the shared copy holds, archived after the oldest backup the +# dedicated store keeps, is needed to restore that backup: it was never copied. +new_case +printf '{"status":"success","type":"file","lastModified":"2026-09-20T00:00:00Z","size":3900,"key":"wedding-db-20260909/wals/0000000200000000/0000000200000000000000FE.gz","etag":"p3"}\n' >>"${k}/mc-umami/ls" +run_case --confirm +[[ "${rc}" -eq 1 ]] || fail "uncopied WAL: expected exit 1, got ${rc}" +require_text "${err}" 'NOT COVERED' 'uncopied WAL' +printf 'PASS: WAL the dedicated store never received is not mistaken for pruned WAL\n' + # After the shared copy is removed, the same proof reports that. new_case : >"${k}/mc-umami/ls" @@ -310,7 +321,7 @@ printf 'PASS: an empty shared prefix is reported as empty, not as covered\n' # Not covered. # --------------------------------------------------------------------------- new_case -record "${later_wal}" 4300 f6 >>"${k}/mc-umami/ls" +record "${later_wal}" 4300 f6 2026-09-20T00:00:00Z >>"${k}/mc-umami/ls" { record "${base}/backup.info" 1200 a1 record "${base}/data.tar.gz" 90000000 ff-4 @@ -348,6 +359,7 @@ run_case --confirm [[ "${rc}" -eq 1 ]] || fail "unreadable object: expected exit 1, got ${rc}" require_text "${err}" 'the dedicated pod could not hash its objects' 'unreadable object' refute_text "${out}${err}" 'abc123' 'unreadable object: endpoint host leaked' +refute_text "${err}" '10.96.0.10' 'unreadable object: address leaked' require_text "${err}" '' 'unreadable object' printf 'PASS: an object that cannot be read is not covered, and the endpoint host is redacted\n' @@ -446,6 +458,20 @@ store_json s3://other-bucket/cnpg/umami-db umami-db-backup-r2 >"${k}/umami-objec run_case --confirm require_untouched 'shared reference store names another bucket' +new_case +store_json s3://wedding-db-backups/cnpg/wedding-db wedding-db-backup-r2-dedicated "" 7d \ + >"${k}/wedding-app-objectstores.barmancloud.cnpg.io-wedding-db-dedicated.json" +run_case --confirm +require_untouched 'dedicated store keeps another retention' +require_text "${err}" 'does not keep the reviewed 30-day retention' 'dedicated store keeps another retention' + +# A store rooted above the catalogue can write into it without naming it. +new_case +printf '{"items":[{"spec":{"configuration":{"destinationPath":"s3://platform-backups/cnpg"}}}]}\n' \ + >"${k}/all-objectstores.barmancloud.cnpg.io.json" +run_case --confirm +require_untouched 'a store is rooted above the shared catalogue' + new_case store_json s3://platform-backups/cnpg/umami-db other-secret >"${k}/umami-objectstores.barmancloud.cnpg.io-umami-db.json" run_case --confirm diff --git a/scripts/verify-wedding-shared-backup-coverage-pod.sh b/scripts/verify-wedding-shared-backup-coverage-pod.sh index 401e91d2cc..c79f9e83d1 100755 --- a/scripts/verify-wedding-shared-backup-coverage-pod.sh +++ b/scripts/verify-wedding-shared-backup-coverage-pod.sh @@ -58,9 +58,14 @@ mkdir -p "${work}" export MC_CONFIG_DIR="${MC_CONFIG_DIR:-${work}/.mc}" mc_err="${work}/mc.err" -# redact_errors prints the last lines of an mc error log without the endpoint host. +# redact_errors prints the last lines of an mc error log without the endpoint +# host (with or without its scheme), any IPv4 address, or any long hex token such +# as an access key ID or request ID. These lines reach a public workflow log. +host="${ENDPOINT#https://}" redact_errors() { - sed -e 's#https://[^/ ]*##g' "${mc_err}" | tail -n 5 >&2 + sed -e 's#https\{0,1\}://[^/ "`]*##g' -e "s#${host}##g" \ + -e 's/[0-9a-fA-F]\{32,\}//g' \ + -e 's/[0-9]\{1,3\}\(\.[0-9]\{1,3\}\)\{3\}\(:[0-9]\{1,5\}\)\{0,1\}//g' "${mc_err}" | tail -n 5 >&2 } command="${1:-}" diff --git a/scripts/verify-wedding-shared-backup-coverage.sh b/scripts/verify-wedding-shared-backup-coverage.sh index c608d431cd..aab3587168 100755 --- a/scripts/verify-wedding-shared-backup-coverage.sh +++ b/scripts/verify-wedding-shared-backup-coverage.sh @@ -176,7 +176,7 @@ require_absent() { } # require_unreferenced refuses when any ObjectStore or Cluster in the cluster -# still names the shared catalogue, in any field: something could then still be +# still names the shared catalogue or a directory above it, in any field: something could then still be # writing to it, and the coverage shown here would not last. require_unreferenced() { local kind @@ -186,7 +186,7 @@ require_unreferenced() { fail "could not list every ${kind}" jq -e --arg path "${shared_catalogue}" ' (.items | type == "array") and - ([.items[].spec | .. | strings | select(. == $path or startswith($path + "/"))] | length == 0) + ([.items[].spec | .. | strings | select(. == $path or startswith($path + "/") or (. as $root | $path | startswith($root + "/")))] | length == 0) ' "${work_dir}/references.json" >/dev/null 2>&1 || fail "a ${kind} still names the shared Wedding catalogue" done @@ -219,6 +219,10 @@ require_store() { require_retired require_store "${tenant_namespace}" "${dedicated_store}" "s3://${dedicated_bucket}/${catalogue_prefix}" "${dedicated_secret}" +# The evaluator accepts a shared object the dedicated store lacks only when it +# is older than this retention window, so the live store must declare the same one. +jq -e '.spec.retentionPolicy == "30d"' "${work_dir}/store.json" >/dev/null 2>&1 || + fail "the ${dedicated_store} ObjectStore does not keep the reviewed 30-day retention" # The Umami store proves which Secret holds the credential for the committed # shared bucket. Its own catalogue is a sibling of the Wedding one and is never # listed. @@ -362,9 +366,9 @@ MANIFEST manifest="${manifest//__BUCKET__/${bucket}}" manifest="${manifest//__PREFIX__/${catalogue_prefix}}" - created+=("${namespace}/pod") printf '%s\n' "${manifest}" | kube "${namespace}" create -f - >/dev/null || fail "could not start the ${role} pod" + created+=("${namespace}/pod") } # wait_ready waits for the pod to store its credential. @@ -436,6 +440,8 @@ jq -e '.covered == true and (.sharedObjects | type == "number" and . > 0)' \ require_retired cat "${work_dir}/summary.json" -printf 'COVERED: the dedicated bucket holds every object of the shared Wedding catalogue (%s objects), or has pruned it by retention.\n' \ - "$(jq -r '.sharedObjects' "${work_dir}/summary.json")" +printf 'COVERED: of the %s objects in the shared Wedding catalogue, the dedicated bucket holds %s with matching content.\n' \ + "$(jq -r '.sharedObjects' "${work_dir}/summary.json")" "$(jq -r '.matchedObjects' "${work_dir}/summary.json")" +printf 'The other %s predate both the oldest backup the dedicated store keeps and its 30-day retention window, so retention has removed them there.\n' \ + "$(jq -r '.retentionPrunedObjects' "${work_dir}/summary.json")" printf 'Nothing was changed. This holds as of this run: the shared copy is no longer written to.\n' From 50b0702cc62279877dcaf63fef9386abbd85f5b5 Mon Sep 17 00:00:00 2001 From: Nikolai Emil Damm Date: Mon, 5 Oct 2026 00:35:16 +0200 Subject: [PATCH 3/4] fix(backup): accept pruning only where the oldest dedicated backup predates the window Retention keeps the newest backup older than its window and everything written since. While the oldest complete dedicated backup is still inside the window, nothing has been pruned, so an older shared object missing there was never copied. Require that backup to predate the window before accepting any missing object, keep a clock margin against its start, refuse a backup ID that is not a real time, and match a store rooted above the catalogue with a trailing slash. Part of #4481 Co-Authored-By: Claude Opus 5.5 --- .../coverage.go | 44 ++++++++++++------- .../coverage_test.go | 18 ++++++-- ...t-verify-wedding-shared-backup-coverage.sh | 18 +++++--- .../verify-wedding-shared-backup-coverage.sh | 8 ++-- 4 files changed, 59 insertions(+), 29 deletions(-) diff --git a/scripts/mirror-wedding-backup-catalogue/coverage.go b/scripts/mirror-wedding-backup-catalogue/coverage.go index e06bb71cdc..c107b49597 100644 --- a/scripts/mirror-wedding-backup-catalogue/coverage.go +++ b/scripts/mirror-wedding-backup-catalogue/coverage.go @@ -17,9 +17,9 @@ package main // // The dedicated store applies its retention policy and the shared copy no longer // does, so the shared copy can hold objects the dedicated store has since -// pruned. Such an object is accepted only when it was written before the oldest -// complete base backup the dedicated store keeps in the same server directory -// and before the retention window began. +// pruned. Such an object is accepted only when the oldest complete base +// backup the dedicated store keeps in the same server directory predates the +// retention window, and the object was written before that backup started. // Every other shared object must be present with matching content. import ( @@ -114,7 +114,9 @@ const emptyDigest = "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b785 // oldestCompleteBackups returns, per server directory, the start time of the // oldest base backup that has both its backup.info and a non-empty data archive. -func oldestCompleteBackups(objects map[string]Object) map[string]time.Time { +// A backup ID that is not a real time is refused: skipping it would move the +// floor later and accept more as pruned. +func oldestCompleteBackups(objects map[string]Object) (map[string]time.Time, error) { hasInfo := map[string]bool{} hasData := map[string]bool{} for key, object := range objects { @@ -139,13 +141,13 @@ func oldestCompleteBackups(objects map[string]Object) map[string]time.Time { server, id, _ := strings.Cut(backup, "/") started, err := time.Parse("20060102T150405", id) if err != nil { - continue + return nil, fmt.Errorf("%w: backup ID %s", ErrMalformedListing, backup) } if current, ok := oldest[server]; !ok || started.Before(current) { oldest[server] = started } } - return oldest + return oldest, nil } // baseBackupIDOf returns the backup ID of a key inside a base backup directory. @@ -175,22 +177,27 @@ func walPositionOf(key string) string { return segment } +// clockMargin covers the difference between the database clock that stamps a +// backup ID and the object store clock that stamps an object. +const clockMargin = time.Hour + // prunedByRetention reports whether a shared object the dedicated store lacks -// can only be missing because retention removed it. Retention never removes -// anything written at or after the start of the oldest backup it keeps, and -// never removes anything still inside the retention window, so the object must -// have been written before both. It is judged by when it was written, not by its -// name: segment names do not order across timelines, and an object that was -// never copied has a name just like a pruned one. A server directory with no -// complete dedicated backup has nothing to measure against, so nothing in it is -// accepted. +// can only be missing because retention removed it. Retention keeps the newest +// backup older than the window and everything written since that backup +// started. So nothing can have been pruned from a server directory whose oldest +// complete dedicated backup is still inside the window: an older object missing +// there was never copied. Where that backup does predate the window, only an +// object written before it started can be gone. It is judged by when it was +// written, not by its name: segment names do not order across timelines. A +// server directory with no complete dedicated backup has nothing to measure +// against, so nothing in it is accepted. func prunedByRetention(object Object, oldest map[string]time.Time, listed time.Time) bool { if baseBackupIDOf(object.Key) == "" && walPositionOf(object.Key) == "" { return false } started, ok := oldest[strings.Split(object.Key, "/")[0]] - return ok && object.LastModified.Before(started) && - object.LastModified.Before(listed.Add(-retentionWindow)) + return ok && !started.After(listed.Add(-retentionWindow)) && + object.LastModified.Before(started.Add(-clockMargin)) } // EvaluateCoverage proves the dedicated catalogue covers the shared one. Every @@ -215,7 +222,10 @@ func EvaluateCoverage(shared, dedicated Listing) (CoverageSummary, error) { return CoverageSummary{}, ErrEmptySource } - oldest := oldestCompleteBackups(dedicatedIndex) + oldest, err := oldestCompleteBackups(dedicatedIndex) + if err != nil { + return CoverageSummary{}, err + } matched, pruned := 0, 0 for key, source := range sharedIndex { copied, ok := dedicatedIndex[key] diff --git a/scripts/mirror-wedding-backup-catalogue/coverage_test.go b/scripts/mirror-wedding-backup-catalogue/coverage_test.go index f39a91cc57..92b9d078d0 100644 --- a/scripts/mirror-wedding-backup-catalogue/coverage_test.go +++ b/scripts/mirror-wedding-backup-catalogue/coverage_test.go @@ -25,7 +25,7 @@ const ( ) var ( - sharedListed = time.Date(2026, 10, 5, 9, 0, 0, 0, time.UTC) + sharedListed = time.Date(2026, 10, 9, 9, 0, 0, 0, time.UTC) dedicatedListed = sharedListed.Add(time.Minute) // Written before the oldest dedicated base backup started and before the // retention window began. @@ -99,10 +99,22 @@ func TestEvaluateCoverageRefusals(t *testing.T) { {"WAL archived after the oldest dedicated backup started", func(shared, _ *Listing) { shared.Objects = append(shared.Objects, Object{Key: prunedWAL, Size: 3900, ETag: "p3", LastModified: beforeRun}) }, ErrNotCovered}, - {"an older backup still inside the retention window", func(shared, _ *Listing) { + {"an old backup missing while the oldest dedicated backup is inside the window", func(shared, dedicated *Listing) { + shared.Started = time.Date(2026, 10, 5, 9, 0, 0, 0, time.UTC) + dedicated.Started = shared.Started.Add(time.Minute) shared.Objects = append(shared.Objects, - Object{Key: prunedInfo, Size: 1100, ETag: "p1", LastModified: time.Date(2026, 9, 7, 0, 0, 0, 0, time.UTC)}) + Object{Key: prunedInfo, Size: 1100, ETag: "p1", LastModified: longAgo}, + Object{Key: prunedData, Size: 80_000_000, ETag: "p2-5", LastModified: longAgo}) }, ErrNotCovered}, + {"an object written within the clock margin of the oldest dedicated backup", func(shared, _ *Listing) { + shared.Objects = append(shared.Objects, + Object{Key: prunedWAL, Size: 3900, ETag: "p3", LastModified: time.Date(2026, 9, 8, 2, 30, 0, 0, time.UTC)}) + }, ErrNotCovered}, + {"a dedicated backup ID that is not a real time", func(_, dedicated *Listing) { + dedicated.Objects = append(dedicated.Objects, + Object{Key: "wedding-db-20260909/base/20260231T030000/backup.info", Size: 1, ETag: "x1", LastModified: beforeRun}, + Object{Key: "wedding-db-20260909/base/20260231T030000/data.tar.gz", Size: 1, ETag: "x2", LastModified: beforeRun}) + }, ErrMalformedListing}, {"a non-empty object hashed as empty", func(shared, dedicated *Listing) { shared.Objects[1].SHA256 = emptyDigest dedicated.Objects[1].SHA256 = emptyDigest diff --git a/scripts/tests/test-verify-wedding-shared-backup-coverage.sh b/scripts/tests/test-verify-wedding-shared-backup-coverage.sh index 0143fb5d09..8b690ec88e 100755 --- a/scripts/tests/test-verify-wedding-shared-backup-coverage.sh +++ b/scripts/tests/test-verify-wedding-shared-backup-coverage.sh @@ -176,7 +176,9 @@ chmod +x "${bin}/kubectl" readonly endpoint='https://abc123.r2.cloudflarestorage.com' readonly old='2026-09-01T00:00:00Z' -readonly base='wedding-db-20260909/base/20260908T030000' +# The oldest dedicated backup predates the 30-day retention window, as it does +# once retention has started pruning. +readonly base='wedding-db-20260909/base/20260808T030000' readonly wal='wedding-db-20260909/wals/0000000200000001/000000020000000100000003.gz' readonly later_wal='wedding-db-20260909/wals/0000000300000001/000000030000000100000009.gz' @@ -288,14 +290,14 @@ printf 'PASS: a covered catalogue is reported as covered and both pods are remov # the dedicated store has pruned. new_case { - record 'wedding-db-20260909/base/20260801T030000/backup.info' 1100 p1 - record 'wedding-db-20260909/base/20260801T030000/data.tar.gz' 80000000 p2-5 - record 'wedding-db-20260909/wals/0000000200000000/0000000200000000000000FE.gz' 3900 p3 + record 'wedding-db-20260909/base/20260720T030000/backup.info' 1100 p1 2026-07-20T03:10:00Z + record 'wedding-db-20260909/base/20260720T030000/data.tar.gz' 80000000 p2-5 2026-07-20T03:10:00Z + record 'wedding-db-20260909/wals/0000000200000000/0000000200000000000000FE.gz' 3900 p3 2026-07-21T00:00:00Z } >>"${k}/mc-umami/ls" run_case --confirm [[ "${rc}" -eq 0 ]] || fail "retention: expected exit 0, got ${rc}. stderr: ${err}" require_text "${out}" '"sharedObjects":6,"matchedObjects":3,"retentionPrunedObjects":3,"covered":true' 'retention' -require_text "${out}" 'The other 3 predate both the oldest backup the dedicated store keeps' 'retention' +require_text "${out}" 'The other 3 were written before the oldest backup the dedicated store keeps' 'retention' printf 'PASS: objects the dedicated store pruned by retention do not block the verdict\n' # A segment only the shared copy holds, archived after the oldest backup the @@ -466,6 +468,12 @@ require_untouched 'dedicated store keeps another retention' require_text "${err}" 'does not keep the reviewed 30-day retention' 'dedicated store keeps another retention' # A store rooted above the catalogue can write into it without naming it. +new_case +printf '{"items":[{"spec":{"configuration":{"destinationPath":"s3://platform-backups/cnpg/"}}}]}\n' \ + >"${k}/all-objectstores.barmancloud.cnpg.io.json" +run_case --confirm +require_untouched 'a store is rooted above the shared catalogue, with a trailing slash' + new_case printf '{"items":[{"spec":{"configuration":{"destinationPath":"s3://platform-backups/cnpg"}}}]}\n' \ >"${k}/all-objectstores.barmancloud.cnpg.io.json" diff --git a/scripts/verify-wedding-shared-backup-coverage.sh b/scripts/verify-wedding-shared-backup-coverage.sh index aab3587168..aa07767030 100755 --- a/scripts/verify-wedding-shared-backup-coverage.sh +++ b/scripts/verify-wedding-shared-backup-coverage.sh @@ -26,8 +26,8 @@ # first; the dedicated pod lists afterwards. # 4. Has both pods hash every object an ETag cannot prove, and runs the reviewed # evaluator (`evaluate-coverage`): every shared object must be in the -# dedicated catalogue with matching content, or be older than what the -# dedicated store's retention still keeps. +# dedicated catalogue with matching content, or be provably removed there by +# the dedicated store's 30-day retention. # 5. Repeats step 1. # # Exit status: 0 covered; 1 refused, failed or not covered. Needs --confirm, @@ -186,7 +186,7 @@ require_unreferenced() { fail "could not list every ${kind}" jq -e --arg path "${shared_catalogue}" ' (.items | type == "array") and - ([.items[].spec | .. | strings | select(. == $path or startswith($path + "/") or (. as $root | $path | startswith($root + "/")))] | length == 0) + ([.items[].spec | .. | strings | select(. == $path or startswith($path + "/") or (rtrimstr("/") as $root | $path | startswith($root + "/")))] | length == 0) ' "${work_dir}/references.json" >/dev/null 2>&1 || fail "a ${kind} still names the shared Wedding catalogue" done @@ -442,6 +442,6 @@ require_retired cat "${work_dir}/summary.json" printf 'COVERED: of the %s objects in the shared Wedding catalogue, the dedicated bucket holds %s with matching content.\n' \ "$(jq -r '.sharedObjects' "${work_dir}/summary.json")" "$(jq -r '.matchedObjects' "${work_dir}/summary.json")" -printf 'The other %s predate both the oldest backup the dedicated store keeps and its 30-day retention window, so retention has removed them there.\n' \ +printf 'The other %s were written before the oldest backup the dedicated store keeps, which itself predates its 30-day retention window, so retention has removed them there.\n' \ "$(jq -r '.retentionPrunedObjects' "${work_dir}/summary.json")" printf 'Nothing was changed. This holds as of this run: the shared copy is no longer written to.\n' From 948fdc63331e62ee9b211bea94ae97088413ecd6 Mon Sep 17 00:00:00 2001 From: Nikolai Emil Damm Date: Mon, 5 Oct 2026 00:39:54 +0200 Subject: [PATCH 4/4] docs(backup): describe the shared Wedding catalogue coverage proof Part of #4481 Co-Authored-By: Claude Opus 5.5 --- docs/dr/velero-cnpg.md | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/docs/dr/velero-cnpg.md b/docs/dr/velero-cnpg.md index f24f2476b2..ab2491288a 100644 --- a/docs/dr/velero-cnpg.md +++ b/docs/dr/velero-cnpg.md @@ -46,7 +46,16 @@ The `Mirror Wedding Backup Catalogue` and `Verify Wedding Backup Denial` workflows both need the shared credential inside `wedding-app`, so neither can run any more; #4482 re-homes the denial proof and removes the finished cutover tooling, and #4481 removes the predecessor catalogue that still sits in the -shared bucket. `Verify Wedding Backup Cutover` +shared bucket. That copy is no longer written to or pruned, and every holder of +the shared credential can read it. Before it is removed, dispatch `Verify +Wedding Shared Backup Coverage` on `main` with +`confirm=verify-wedding-shared-backup-coverage`. It only reads: a pod beside the +shared credential in `umami` and one beside the dedicated credential in +`wedding-app` each list their side, and the verdict is `COVERED` only when every +object in the shared copy is in the dedicated bucket with matching content, or +was provably removed there by the 30-day retention. Remove the shared copy only +after a `COVERED` run, and run the proof again afterwards: it then reports +`EMPTY`. `Verify Wedding Backup Cutover` is a main-only, protected production dispatch: confirm `verify-wedding-backup-cutover` to request a fresh online primary backup. It refuses a shared archive, unhealthy database, changed database identity, or