diff --git a/cmd/opensave-app/frontend/src/components/ConflictModal.svelte b/cmd/opensave-app/frontend/src/components/ConflictModal.svelte index 4a9311ff..f24a7baa 100644 --- a/cmd/opensave-app/frontend/src/components/ConflictModal.svelte +++ b/cmd/opensave-app/frontend/src/components/ConflictModal.svelte @@ -1,6 +1,6 @@
@@ -99,12 +108,22 @@ {peer.status} + {#if peer.needsUpdate} + + needs update + + {/if}
{peer.address === 'relay' ? 'internet relay' : `${peer.address}:${peer.port}`} · last synced {fmtTime(peer.lastSynced)} {#if peer.appVersion}· OpenSave {peer.appVersion}{/if}
+ {#if peer.link} +
+ {linkLabel(peer.link)} +
+ {/if}
diff --git a/cmd/opensave-app/frontend/src/views/game/GameHeader.svelte b/cmd/opensave-app/frontend/src/views/game/GameHeader.svelte index 19750020..ceafac2a 100644 --- a/cmd/opensave-app/frontend/src/views/game/GameHeader.svelte +++ b/cmd/opensave-app/frontend/src/views/game/GameHeader.svelte @@ -13,10 +13,12 @@ import ArrowLeft from 'lucide-svelte/icons/arrow-left'; import Play from 'lucide-svelte/icons/play'; import RefreshCw from 'lucide-svelte/icons/refresh-cw'; + import Upload from 'lucide-svelte/icons/upload'; import FolderOpen from 'lucide-svelte/icons/folder-open'; import Pencil from 'lucide-svelte/icons/pencil'; import Star from 'lucide-svelte/icons/star'; import TriangleAlert from 'lucide-svelte/icons/triangle-alert'; + import Info from 'lucide-svelte/icons/info'; import { collections, isFavourite, toggleFavourite } from '../../lib/collections.js'; export let game; @@ -42,6 +44,20 @@ const syncNow = () => run('Sync triggered', () => api.post(`/api/games/${game.id}/sync`)); const launchGame = () => run('Launching…', () => api.post(`/api/games/${game.id}/launch`)); + // For when the devices hold states nobody can rank — a copy that arrived + // only half-way, a mixture of two saves — and the person knows which device + // has the right one. Every other device takes this save whole, keeping a + // snapshot of its own first; one that changed the save on its own since is + // still asked. + async function useEverywhere() { + const ok = await askConfirm( + `Make this device's save of ${game.name} the one every other device uses? They each replace theirs with it — every file, including removing files this device does not have — and keep a snapshot of their own first. A device where the save was changed after this update is asked instead.`, + { title: 'Use this save everywhere?', confirmText: 'Use this save everywhere', danger: true } + ); + if (!ok) return; + await run(`${game.name}: this save goes to your other devices`, () => api.post(`/api/games/${game.id}/use-everywhere`)); + } + let editPath = false; let pathDraft = ''; @@ -137,6 +153,7 @@ {#if canLaunch(game)} {/if} +
@@ -188,7 +205,18 @@ {/if} -{#if game.savePathMissing && !editPath} +{#if game.savePathMissing && game.saveDriveMissing && !editPath} + +
+ +
+ This game isn't on this device. + Its saves are on {game.savePath.slice(0, 2)}, which this PC doesn't have, so it's skipped here and keeps syncing + between your other devices. If it is installed here, point the game at its save folder with Edit. +
+
+{:else if game.savePathMissing && !editPath} @@ -288,6 +316,13 @@ .missing strong { color: var(--text); } + .missing.elsewhere { + border-color: var(--border); + background: transparent; + } + .missing.elsewhere :global(svg) { + color: var(--text-faint); + } .emptied-actions { display: flex; flex-wrap: wrap; diff --git a/cmd/opensave-app/main.go b/cmd/opensave-app/main.go index 09e46f07..4f8ca678 100644 --- a/cmd/opensave-app/main.go +++ b/cmd/opensave-app/main.go @@ -7,6 +7,7 @@ import ( "fmt" "os" "runtime" + "runtime/debug" "github.com/wailsapp/wails/v2" "github.com/wailsapp/wails/v2/pkg/options" @@ -45,6 +46,8 @@ func main() { // removes its leftover binary) before claiming the single-instance lock. cleanupReplacedBinary() + tuneGC() + app := NewApp() err := wails.Run(&options.App{ @@ -97,3 +100,17 @@ func main() { os.Exit(1) } } + +// tuneGC trades a little CPU for a smaller footprint. OpenSave is a +// background app, and Go's default (let the heap double before collecting) +// meant a large save's manifests showed up as twice their size in Task +// Manager. The soft limit makes the runtime collect and hand memory back to +// the OS harder as it approaches 1GiB. Explicit GOGC / GOMEMLIMIT win. +func tuneGC() { + if os.Getenv("GOGC") == "" { + debug.SetGCPercent(50) + } + if os.Getenv("GOMEMLIMIT") == "" { + debug.SetMemoryLimit(1 << 30) + } +} diff --git a/e2e/batch_bench_test.go b/e2e/batch_bench_test.go new file mode 100644 index 00000000..0299c4e8 --- /dev/null +++ b/e2e/batch_bench_test.go @@ -0,0 +1,59 @@ +package e2e + +import ( + "fmt" + "net/http" + "os" + "path/filepath" + "strconv" + "strings" + "testing" + "time" + + "github.com/opensave/opensave/internal/p2p" + "github.com/opensave/opensave/internal/p2p/syncengine" + "github.com/opensave/opensave/testutil" +) + +// TestBench_ManySmallFiles times pulling a save of many small files between +// two real daemons, batched and file by file. Opt-in, because it is a +// measurement rather than a check: OPENSAVE_BENCH_FILES=5000 go test ./e2e -run TestBench -v +func TestBench_ManySmallFiles(t *testing.T) { + n, _ := strconv.Atoi(os.Getenv("OPENSAVE_BENCH_FILES")) + if n <= 0 { + t.Skip("set OPENSAVE_BENCH_FILES to run") + } + for _, mode := range []struct { + name string + proto int + }{{"batched", syncengine.ProtoBatchFiles}, {"file-by-file", syncengine.ProtoMultiRoot}} { + t.Run(mode.name, func(t *testing.T) { + restore := p2p.SetServedProto(mode.proto) + defer p2p.SetServedProto(restore) + + a := testutil.NewTestDaemon(t, "Bench-A") + b := testutil.NewTestDaemon(t, "Bench-B") + a.PairWith(b) + chunk := strings.Repeat("x", 3500) // a Project Zomboid map chunk is ~3.5KB + for i := 0; i < n; i++ { + a.WriteSave(fmt.Sprintf("map/map_%d_%d.bin", i/100, i%100), chunk+strconv.Itoa(i)) + } + restoreSyncOnTrack := suppressSyncOnTrack(a, b) + gameID := a.TrackGame("Bench") + b.API(http.MethodPost, "/api/games", map[string]string{"name": "Bench", "savePath": b.SaveDir}, nil) + restoreSyncOnTrack() + + start := time.Now() + b.API(http.MethodPost, "/api/games/"+gameID+"/sync", nil, nil) + count := func() int { + m, _ := filepath.Glob(filepath.Join(b.SaveDir, "map", "*.bin")) + return len(m) + } + if !testutil.WaitFor(60*time.Minute, func() bool { return count() >= n }) { + t.Fatalf("only %d of %d files arrived", count(), n) + } + el := time.Since(start) + t.Logf("%s: %d files in %s = %.2f ms/file", mode.name, n, el.Round(time.Millisecond), float64(el.Microseconds())/1000/float64(n)) + }) + } +} diff --git a/e2e/cli_features_test.go b/e2e/cli_features_test.go index 6c656fa1..15877e00 100644 --- a/e2e/cli_features_test.go +++ b/e2e/cli_features_test.go @@ -176,10 +176,10 @@ func TestCLI_SnapshotAndRollbackRestoresEverything(t *testing.T) { c.mustRun("add", "History Game", dir) c.mustRun("snapshot", "history-game", "-m", "known good") - // Two: the one tracking took automatically, and the one just asked for. - // The manual one is newest, and is the one to roll back to. + // One: the one asked for holds the same files as the one tracking took, + // and takes its place (snapshot/content.go). It is the one to roll back to. ids := c.snapshotIDs("history-game") - if len(ids) != 2 { + if len(ids) != 1 { t.Fatalf("expected the initial snapshot plus the manual one, got %v", ids) } snap := ids[0] @@ -410,6 +410,9 @@ func TestCLI_SnapshotAll(t *testing.T) { c.mustRun("add", "All One", one) c.mustRun("add", "All Two", two) before := len(c.snapshotIDs("all-one")) + len(c.snapshotIDs("all-two")) + // Played on: a snapshot of the same files would be the one already there. + c.saveDir("all-one", map[string]string{"a.sav": "1, played on"}) + c.saveDir("all-two", map[string]string{"b.sav": "2, played on"}) out := c.mustRun("snapshot", "--all", "before", "the", "reinstall") if !strings.Contains(out, "2 game") { @@ -630,7 +633,9 @@ func TestCLI_SnapshotDeleteAndPrune(t *testing.T) { dir := c.saveDir("pruning", map[string]string{"slot1.sav": "v1"}) c.mustRun("add", "Pruning", dir) - for _, v := range []string{"v1", "v2", "v3"} { + // Each a change: a snapshot of the same files as one already there would + // be that one (snapshot/content.go). + for _, v := range []string{"v2", "v3", "v4"} { c.saveDir("pruning", map[string]string{"slot1.sav": v}) c.mustRun("snapshot", "pruning", "-m", v) } @@ -671,6 +676,7 @@ func TestCLI_UntrackAllRequiresConfirmation(t *testing.T) { dir := c.saveDir("guard", map[string]string{"slot1.sav": "keep me"}) c.mustRun("add", "Guarded", dir) + c.saveDir("guard", map[string]string{"slot1.sav": "keep me, played on"}) c.mustRun("snapshot", "guarded") c.mustFail("untrack-all") @@ -694,7 +700,7 @@ func TestCLI_UntrackAllRequiresConfirmation(t *testing.T) { if out := c.mustRun("status"); strings.Contains(out, "Guarded") { t.Errorf("untrack-all --yes did not clear the library:\n%s", out) } - if got := c.readSave(dir, "slot1.sav"); got != "keep me" { + if got := c.readSave(dir, "slot1.sav"); got != "keep me, played on" { t.Errorf("untrack-all touched the save on disk: %q", got) } } diff --git a/e2e/cloud_and_peers_test.go b/e2e/cloud_and_peers_test.go index eaf6ee1f..5fdb30a5 100644 --- a/e2e/cloud_and_peers_test.go +++ b/e2e/cloud_and_peers_test.go @@ -197,6 +197,9 @@ func TestCloud_SyncLocalRepairsATruncatedRemoteCopy(t *testing.T) { a.WriteSave("slot1.sav", strings.Repeat("save data ", 500)) gameID := a.TrackGame("Repair Game") + // A change first: a snapshot of the same files as the one tracking took + // would be that one (snapshot/content.go). + a.WriteSave("slot1.sav", strings.Repeat("more save data ", 500)) a.API(http.MethodPost, "/api/games/"+gameID+"/snapshot", map[string]any{"comment": "full"}, nil) name := waitForUpload(t, cloudDir)[0] diff --git a/e2e/history_test.go b/e2e/history_test.go index b2b7bcff..e5e98b70 100644 --- a/e2e/history_test.go +++ b/e2e/history_test.go @@ -1,11 +1,15 @@ package e2e import ( + "archive/zip" "fmt" + "io" "net/http" "testing" "time" + "github.com/opensave/opensave/internal/snapshot" + "github.com/opensave/opensave/testutil" ) @@ -14,6 +18,7 @@ type snapshotWire struct { Comment string `json:"comment"` Branch string `json:"branch"` IsSystemAuto bool `json:"isSystemAuto"` + ZipPath string `json:"zipPath"` } type gameWire struct { @@ -82,11 +87,49 @@ func TestHistory_SnapshotAndRollback(t *testing.T) { } // Rolling back is itself non-destructive: the pre-rollback state must - // still be reachable, or "undo" would be a one-way door. - snaps := snapshotsOn(a, gameID, "main") - if len(snaps) < 3 { - t.Errorf("expected a safety snapshot of the pre-rollback state, have %d snapshots", len(snaps)) + // still be reachable, or "undo" would be a one-way door. Asked of the + // snapshots themselves, not counted: a state already snapshotted is not + // archived a second time. + if !snapshotHolds(t, a, gameID, "slot1.sav", "level-9") || !snapshotHolds(t, a, gameID, "inventory.dat", "cursed sword") { + t.Error("the pre-rollback state is in no snapshot") + } +} + +// snapshotHolds reports whether any snapshot on a game's main branch holds the +// archive entry name (a path in the save; ".opensave-locations//" +// in another location) with exactly content. +func snapshotHolds(t *testing.T, td *testutil.TestDaemon, gameID, name, content string) bool { + t.Helper() + for _, s := range snapshotsOn(td, gameID, "main") { + path, done, err := snapshot.OpenArchive(s.ZipPath) + if err != nil { + continue + } + zr, err := zip.OpenReader(path) + if err != nil { + done() + continue + } + for _, f := range zr.File { + if f.Name != name { + continue + } + rc, err := f.Open() + if err != nil { + continue + } + b, _ := io.ReadAll(rc) + rc.Close() + if string(b) == content { + zr.Close() + done() + return true + } + } + zr.Close() + done() } + return false } // Branches let two playthroughs coexist. Switching between them has to carry diff --git a/e2e/keep_both_test.go b/e2e/keep_both_test.go index 897bedd8..c3b5f35e 100644 --- a/e2e/keep_both_test.go +++ b/e2e/keep_both_test.go @@ -54,9 +54,12 @@ func branchHolds(t *testing.T, st branchState, branch, rel string) string { if len(snaps) == 0 { return "" } - path, done, err := snapshot.OpenArchive(snaps[0].ZipPath) + // The API lists a branch's snapshots oldest first (gamePayload), so the + // newest — what a switch puts back — is the last. + newest := snaps[len(snaps)-1] + path, done, err := snapshot.OpenArchive(newest.ZipPath) if err != nil { - t.Fatalf("open %s: %v", snaps[0].ZipPath, err) + t.Fatalf("open %s: %v", newest.ZipPath, err) } defer done() zr, err := zip.OpenReader(path) diff --git a/e2e/multiroot_scenarios_test.go b/e2e/multiroot_scenarios_test.go index e6659650..3ab77668 100644 --- a/e2e/multiroot_scenarios_test.go +++ b/e2e/multiroot_scenarios_test.go @@ -10,6 +10,7 @@ import ( "github.com/opensave/opensave/internal/delta" "github.com/opensave/opensave/internal/p2p/syncengine" + "github.com/opensave/opensave/internal/snapshot" "github.com/opensave/opensave/testutil" ) @@ -426,9 +427,11 @@ func TestMultiRootScenario_ResolveALocationConflictKeepingTheirs(t *testing.T) { if len(a.Daemon.P2P.Sync.ActiveRootConflicts()) != 0 { t.Error("the conflict is still open after being resolved") } - // Undoable: a snapshot was taken before anything was overwritten. - if after := len(snapshotsOn(a, gameID, "main")); after <= before { - t.Errorf("snapshots %d -> %d; no safety snapshot was taken before overwriting", before, after) + // Undoable: what was overwritten is in a snapshot — taken before the + // overwrite, or already there from when it was written. + _ = before + if !snapshotHolds(t, a, gameID, snapshot.RootPrefix+"config/settings.ini", "A-version") { + t.Error("the overwritten version is in no snapshot") } if got := a.ReadSave("save.sav"); got != "primary v1" { t.Errorf("resolving a location changed the save folder: %q", got) diff --git a/e2e/presence_timing_test.go b/e2e/presence_timing_test.go new file mode 100644 index 00000000..4d9402d6 --- /dev/null +++ b/e2e/presence_timing_test.go @@ -0,0 +1,53 @@ +package e2e + +import ( + "os" + "testing" + "time" + + "github.com/opensave/opensave/testutil" +) + +// TestPresence_GoneDeviceShowsOffline measures how long a device that went +// away keeps being shown online. Opt-in measurement: +// OPENSAVE_PRESENCE_TIMING=1 go test ./e2e -run TestPresence_ -v +func TestPresence_GoneDeviceShowsOffline(t *testing.T) { + if os.Getenv("OPENSAVE_PRESENCE_TIMING") == "" { + t.Skip("set OPENSAVE_PRESENCE_TIMING to run") + } + a := testutil.NewTestDaemon(t, "Presence-A") + b := testutil.NewTestDaemon(t, "Presence-B") + a.PairWith(b) + bID := b.NodeID() + + status := func() string { + p, err := a.Daemon.Store.GetPeer(bID) + if err != nil { + return "?" + } + return p.Status + } + if !testutil.WaitFor(60*time.Second, func() bool { return status() == "online" }) { + t.Fatalf("B never showed online on A: %q", status()) + } + + gone := time.Now() + b.Server.Stop() + b.Daemon.Stop() + + last := status() + for time.Since(gone) < 10*time.Minute { + if s := status(); s != last { + t.Logf("%6.1fs after B stopped: %s -> %s", time.Since(gone).Seconds(), last, s) + last = s + } + if last == "offline" { + break + } + time.Sleep(500 * time.Millisecond) + } + t.Logf("B shown online for %.1fs after it stopped", time.Since(gone).Seconds()) + if last != "offline" { + t.Fatalf("B still %q 10 minutes after it stopped", last) + } +} diff --git a/internal/api/routes.go b/internal/api/routes.go index 65a28d0c..c13e5b1c 100644 --- a/internal/api/routes.go +++ b/internal/api/routes.go @@ -1,6 +1,7 @@ package api import ( + "context" "encoding/json" "errors" "fmt" @@ -49,6 +50,7 @@ func (s *Server) routes(r chi.Router) { r.Post("/api/games/{gameId}/snapshot", s.handleCreateSnapshot) r.Post("/api/games/{gameId}/rollback", s.handleRollback) + r.Post("/api/games/{gameId}/use-everywhere", s.handleUseSaveEverywhere) r.Get("/api/games/{gameId}/save-files", s.handleGameSaveFiles) r.Get("/api/games/{gameId}/snapshot/{snapshotId}/files", s.handleSnapshotFiles) r.Post("/api/games/{gameId}/snapshot/{snapshotId}/restore-file", s.handleRestoreFile) @@ -56,6 +58,7 @@ func (s *Server) routes(r chi.Router) { r.Patch("/api/games/{gameId}/snapshot/{snapshotId}", s.handleEditSnapshot) r.Get("/api/games/{gameId}/sessions", s.handleGameSessions) r.Get("/api/snapshots/check", s.handleSnapshotChecks) + r.Post("/api/snapshots/merge-duplicates", s.handleMergeDuplicates) r.Get("/api/games/{gameId}/snapshot/{snapshotId}/compare/{otherId}", s.handleCompareSnapshots) r.Post("/api/snapshots/check", s.handleVerifySnapshots) r.Post("/api/games/{gameId}/session", s.handleMarkSession) @@ -637,6 +640,50 @@ func (s *Server) handleRollback(w http.ResponseWriter, r *http.Request) { writeJSON(w, http.StatusOK, snap) } +// snapshotID is the request's snapshot id, followed when it is kept as an +// alias of the snapshot holding the same files (snapshot/content.go). For +// reading a snapshot only: deleting or editing takes the id as given, so an +// old id can never reach the snapshot that took its place. +func (s *Server) snapshotID(r *http.Request) string { + return s.resolveID(chi.URLParam(r, "snapshotId")) +} + +func (s *Server) resolveID(id string) string { + if resolved, ok := s.Daemon.Store.ResolveSnapshotID(id); ok { + return resolved + } + return id +} + +// handleMergeDuplicates keeps every game's identical snapshots once, now +// rather than at the next start (daemon.MergeSnapshotDuplicates). +func (s *Server) handleMergeDuplicates(w http.ResponseWriter, r *http.Request) { + merged, freed := s.Daemon.MergeSnapshotDuplicates(true) + s.BroadcastGamesUpdate() + writeJSON(w, http.StatusOK, map[string]any{"merged": merged, "freedBytes": freed}) +} + +// handleUseSaveEverywhere makes this device's save the one every other device +// takes (syncengine.UseSaveEverywhere), and starts a sync to hand it out. The +// sync runs in the background: a large save takes a while, and the answer the +// person needs is that it was accepted. A paused device hands it out when it +// resumes. +func (s *Server) handleUseSaveEverywhere(w http.ResponseWriter, r *http.Request) { + gameID := chi.URLParam(r, "gameId") + vec, err := s.Daemon.P2P.Sync.UseSaveEverywhere(gameID) + if err != nil { + writeError(w, http.StatusConflict, err.Error()) + return + } + s.Daemon.P2P.GoSync(func(ctx context.Context) { + ctx, cancel := context.WithTimeout(ctx, 30*time.Minute) + defer cancel() + _, _ = s.Daemon.P2P.SyncGame(ctx, gameID) + }) + s.BroadcastGamesUpdate() + writeJSON(w, http.StatusOK, map[string]any{"version": vec.String()}) +} + func (s *Server) handleCreateBranch(w http.ResponseWriter, r *http.Request) { gameID := chi.URLParam(r, "gameId") // copyCurrentSave defaults to true when the field is absent: a caller that @@ -973,7 +1020,7 @@ func (s *Server) handleSyncResume(w http.ResponseWriter, r *http.Request) { // handlePreviewRestore says what restoring a snapshot would change, file by // file, without changing anything. func (s *Server) handlePreviewRestore(w http.ResponseWriter, r *http.Request) { - preview, err := s.Daemon.Snapshots.PreviewRestore(chi.URLParam(r, "gameId"), chi.URLParam(r, "snapshotId")) + preview, err := s.Daemon.Snapshots.PreviewRestore(chi.URLParam(r, "gameId"), s.snapshotID(r)) if err != nil { writeError(w, notFoundToStatus(err), err.Error()) return diff --git a/internal/api/routes_backup.go b/internal/api/routes_backup.go index 2be3bcaa..3d4125d0 100644 --- a/internal/api/routes_backup.go +++ b/internal/api/routes_backup.go @@ -25,7 +25,7 @@ import ( // granular-restore browser in the UI). func (s *Server) handleSnapshotFiles(w http.ResponseWriter, r *http.Request) { gameID := chi.URLParam(r, "gameId") - snapshotID := chi.URLParam(r, "snapshotId") + snapshotID := s.snapshotID(r) snap, err := s.Daemon.Store.GetSnapshot(snapshotID) if err != nil || snap.GameID != gameID { @@ -59,7 +59,7 @@ func (s *Server) handleSnapshotFiles(w http.ResponseWriter, r *http.Request) { // save location, taking a safety snapshot first. func (s *Server) handleRestoreFile(w http.ResponseWriter, r *http.Request) { gameID := chi.URLParam(r, "gameId") - snapshotID := chi.URLParam(r, "snapshotId") + snapshotID := s.snapshotID(r) var body struct { RelPath string `json:"relPath"` diff --git a/internal/api/routes_verify.go b/internal/api/routes_verify.go index af9641ea..e3732368 100644 --- a/internal/api/routes_verify.go +++ b/internal/api/routes_verify.go @@ -31,7 +31,7 @@ func (s *Server) handleVerifySnapshots(w http.ResponseWriter, r *http.Request) { // handleCompareSnapshots says what changed from one of a game's snapshots to // another (snapshot.Manager.Compare). func (s *Server) handleCompareSnapshots(w http.ResponseWriter, r *http.Request) { - c, err := s.Daemon.Snapshots.Compare(chi.URLParam(r, "gameId"), chi.URLParam(r, "snapshotId"), chi.URLParam(r, "otherId")) + c, err := s.Daemon.Snapshots.Compare(chi.URLParam(r, "gameId"), s.snapshotID(r), s.resolveID(chi.URLParam(r, "otherId"))) if err != nil { writeError(w, http.StatusBadRequest, err.Error()) return diff --git a/internal/api/server.go b/internal/api/server.go index 66a32562..c9dc694a 100644 --- a/internal/api/server.go +++ b/internal/api/server.go @@ -98,12 +98,20 @@ func (s *Server) peersPayload() map[string]any { AppVersion string `json:"appVersion,omitempty"` BuildTimeMs int64 `json:"buildTimeMs,omitempty"` HasNewerBuild bool `json:"hasNewerBuild,omitempty"` + // NeedsUpdate: its OpenSave keeps no save versions, so this + // device takes nothing from it until it is updated. + NeedsUpdate bool `json:"needsUpdate,omitempty"` + // Link is how fast this device has found its connection to it, + // and what kind of address it is reached on: what decides the + // order devices are synced in. + Link syncengine.LinkStat `json:"link"` // What actually protects traffic with this device. Sent per peer // rather than described once in the interface, because the answer // differs per pairing and the reader cannot work out which case // they are in from a general statement. p2p.PeerProtection - }{Peer: p, PeerProtection: s.Daemon.P2P.PeerProtection(p)} + }{Peer: p, PeerProtection: s.Daemon.P2P.PeerProtection(p), NeedsUpdate: s.Daemon.P2P.Sync.PeerNeedsUpdate(p.ID), + Link: s.Daemon.P2P.Sync.Link(syncengine.Peer{ID: p.ID, Name: p.Name, Address: p.Address, Port: p.Port, IsWan: p.Address == "relay"})} if b, ok := builds[p.ID]; ok { entry.AppVersion = b.AppVersion entry.BuildTimeMs = b.BuildTimeMs @@ -567,10 +575,14 @@ func (s *Server) gamePayload(g store.Game) map[string]any { // The save folder is not there — gone, moved, or on a drive not // plugged in. Nothing is watched or synced for it until it is back. "savePathMissing": daemon.SaveFolderMissing(g.SavePath), - "lastPlayedAt": lastPlayed, - "playingSince": playingSince, - "playtimeMs": play.PlaytimeMs, - "playSessions": play.Sessions, + // …and of those, the ones whose whole drive is absent here: the game + // lives on another device's D:\ or E:\, which is not a problem to + // warn about (daemon.SaveDriveMissing). + "saveDriveMissing": daemon.SaveFolderMissing(g.SavePath) && daemon.SaveDriveMissing(g.SavePath), + "lastPlayedAt": lastPlayed, + "playingSince": playingSince, + "playtimeMs": play.PlaytimeMs, + "playSessions": play.Sessions, // Whether the game is installed on this device: "found", "not-found", // or "" when there is no telling (daemon.InstallState). A game that is // not here is one nobody plays here, which is why it shows no play. diff --git a/internal/api/server_test.go b/internal/api/server_test.go index f7a9ebd5..dfefa6a0 100644 --- a/internal/api/server_test.go +++ b/internal/api/server_test.go @@ -578,7 +578,11 @@ func TestDeleteSnapshotAndBranch(t *testing.T) { ts.do(t, http.MethodPost, "/api/games", map[string]string{"name": "Del Game", "savePath": ts.saveDir}) waitInitialSnapshot(t, ts, "del-game") - // A manual second snapshot so we have one to delete. + // A manual second snapshot so we have one to delete — of a changed save: + // one of the same files would be the first snapshot, kept once. + if err := os.WriteFile(filepath.Join(ts.saveDir, "s.sav"), []byte("y"), 0o666); err != nil { + t.Fatal(err) + } ts.do(t, http.MethodPost, "/api/games/del-game/snapshot", map[string]string{"comment": "manual"}) snaps, _ := ts.daemon.Store.ListSnapshots("del-game", "main") if len(snaps) < 2 { diff --git a/internal/cliapp/syncsummary.go b/internal/cliapp/syncsummary.go index b5518341..ff13d852 100644 --- a/internal/cliapp/syncsummary.go +++ b/internal/cliapp/syncsummary.go @@ -101,6 +101,12 @@ func outcomeFromPeers(gameID string, peers map[string]peerSyncResult) syncOutcom waiting = who + " is waiting for a folder to be chosen for it" case "peer_holding": waiting = who + " is holding it back: its save was emptied there" + case "peer_outdated_app": + waiting = who + " runs an older OpenSave without save versions — update it there" + case "waiting": + if waiting == "" { + waiting = "a newer version is on a device that is not online" + } case "peer_missing": if waiting == "" { waiting = who + " does not track it" diff --git a/internal/daemon/daemon.go b/internal/daemon/daemon.go index 306ce6ff..86ffa570 100644 --- a/internal/daemon/daemon.go +++ b/internal/daemon/daemon.go @@ -62,6 +62,9 @@ type Daemon struct { // Games whose save folder is not there; see missing.go. missingMu sync.Mutex missing map[string]bool + // relocateChecked is when each missing save folder was last looked for on + // another drive (missing.go). + relocateChecked map[string]time.Time // Play sessions; see sessions.go. sessions sessionState @@ -80,6 +83,9 @@ type Daemon struct { newGames newGameState // scanMu runs one save scan at a time. See ScanForSaves. scanMu sync.Mutex + // mergeMu runs one merge of identical snapshots at a time + // (MergeSnapshotDuplicates). + mergeMu sync.Mutex // uploads counts cloud mirrors still running, so Stop can wait for them // rather than letting process exit truncate one. @@ -145,6 +151,9 @@ func New(opts Options) (*Daemon, error) { snaps := snapshot.New(s) snaps.Log = log.Log + // Each new snapshot takes the place of older automatic ones it holds + // whole (snapshot/content.go). + snaps.PruneOnCreate = true d := &Daemon{ Paths: paths, @@ -229,7 +238,16 @@ func New(opts Options) (*Daemon, error) { _, err := snaps.Create(gameID, "", true) return err }, + SyncWriting: func(gameID string) bool { + return d.P2P != nil && d.P2P.Sync != nil && d.P2P.Sync.BeingWritten(gameID) + }, + SyncWritingNow: func(gameID string) bool { + return d.P2P != nil && d.P2P.Sync != nil && d.P2P.Sync.WritingNow(gameID) + }, OnChanged: func(gameID string) { + // A change made here, by the game or the user: a new version of + // the save, recorded before anyone is offered it. + d.P2P.Sync.NoteLocalChange(gameID) // An emptied save is noticed as it happens, and said on screen, // even with no other device online to hold it back from. _, _ = d.P2P.Sync.CheckHold(gameID, false) @@ -253,6 +271,41 @@ func New(opts Options) (*Daemon, error) { return d, nil } +// adoptNeverSyncedView re-takes, once per change of delta.NeverSyncedList, +// the hashes recorded for every game under the previous list — each only when +// the save is the one it was recorded for, so a real change still counts. +// +// Without it the first build with a new list sees every save holding one of +// those files as changed: an auto-snapshot of a game nobody played, and a new +// version of its own on every device at once, which the devices then disagree +// over. +func (d *Daemon) adoptNeverSyncedView(games []store.Game) { + const mark = "never_synced" + if d.Store.Mark(mark) == delta.NeverSyncedList { + return + } + for _, game := range games { + extra, err := d.Store.GameRootPaths(game.ID) + if err != nil { + extra = nil + } + m, failures, err := delta.BuildMultiManifest(game.SavePath, extra) + if err != nil || len(failures) > 0 { + continue // unreadable now; compared as it is when it is back + } + if game.LastManifestHash != "" && + game.LastManifestHash == watcher.ContentHashBeforeNeverSynced(m, game.SyncIgnore) { + _ = d.Store.SetLastManifestHash(game.ID, watcher.ContentHash(m, game.SyncIgnore)) + } + if d.P2P != nil && d.P2P.Sync != nil { + primary := m + primary.Extra = nil + d.P2P.Sync.AdoptNeverSyncedView(game.ID, primary) + } + } + _ = d.Store.SetMark(mark, delta.NeverSyncedList) +} + // Start begins watching every tracked game with auto-sync enabled. func (d *Daemon) Start() error { // Deletion records expire. Swept once per launch rather than on a timer: @@ -266,6 +319,10 @@ func (d *Daemon) Start() error { if err != nil { return err } + // Snapshots that hold the same files are kept once (snapshot/content.go). + // In the background, once things have settled: naming a snapshot taken + // before contents were recorded reads its whole archive. + time.AfterFunc(2*time.Minute, func() { d.MergeSnapshotDuplicates(false) }) // Switch games tracked under a made-up name get their real one, when an // emulator here knows it by now. d.nameSwitchGames() @@ -276,6 +333,9 @@ func (d *Daemon) Start() error { // then evicts entries it is about to want and starts re-reading saves. delta.SetHashCacheBudgetForGames(len(games)) + // Before anything is watched or compared. + d.adoptNeverSyncedView(games) + for _, game := range games { // Backfill cover art for games tracked before covers existed (or // migrated from the JS app without one). @@ -985,6 +1045,40 @@ func (d *Daemon) checkSavePathShape(abs string) error { return nil } +// MergeSnapshotDuplicates keeps every game's identical snapshots once +// (snapshot.Manager.MergeDuplicates), one game at a time. Returns how many +// were merged and the bytes freed. all looks at every game; otherwise only at +// games with snapshots whose content is not named yet. +func (d *Daemon) MergeSnapshotDuplicates(all bool) (merged int, freed int64) { + d.mergeMu.Lock() + defer d.mergeMu.Unlock() + games, err := d.Store.ListGames() + if err != nil { + return 0, 0 + } + for _, g := range games { + // Once per game: snapshots taken since are named as they are taken, + // and checked against the older ones then (PruneOnCreate). A game + // whose snapshots all have their content named has had its pass. + if unnamed, err := d.Store.SnapshotsWithoutContentHash(g.ID); !all && (err != nil || len(unnamed) == 0) { + continue + } + if n, f, err := d.Snapshots.MergeDuplicates(g.ID); err == nil { + merged += n + freed += f + } + // And steps on the way to a newer snapshot that holds them whole. + if n, f, err := d.Snapshots.PruneContained(g.ID); err == nil { + merged += n + freed += f + } + } + if merged > 0 { + d.Log.Log("info", fmt.Sprintf("snapshots cleaned up: %d identical or held whole by a newer one, %d MB freed", merged, freed>>20)) + } + return merged, freed +} + // EnsureImportedSnapshot registers a snapshot restored from an .sscb // backup: creates the branch if missing and inserts the metadata row // unless that snapshot id is already known (idempotent re-imports). diff --git a/internal/daemon/missing.go b/internal/daemon/missing.go index aa083447..b55c7b1f 100644 --- a/internal/daemon/missing.go +++ b/internal/daemon/missing.go @@ -6,6 +6,8 @@ import ( "io/fs" "os" "path/filepath" + "strings" + "time" "github.com/opensave/opensave/internal/delta" "github.com/opensave/opensave/internal/logging" @@ -34,10 +36,126 @@ func SaveFolderMissing(savePath string) bool { return errors.Is(err, fs.ErrNotExist) } +// SaveDriveMissing reports whether a save path lives on a drive this device +// does not have at all — a D:\ or E:\ Steam library that exists on another +// PC. That is where a game auto-tracked from a peer lands when the peer's +// library is on a drive letter this machine lacks, and it is not something +// to warn about: nothing was lost here, the game just isn't on this device. +// Only a volume that is plainly absent counts; an unreachable network share +// reports some other error and stays a warning. +func SaveDriveMissing(savePath string) bool { + vol := filepath.VolumeName(savePath) + if vol == "" { + return false + } + _, err := os.Stat(vol + string(filepath.Separator)) + return errors.Is(err, fs.ErrNotExist) +} + +// The same save folder under another drive letter. +// +// A save inside a game's install folder — a Steam library, most often — sits +// on whichever drive that library is on, and that differs between devices: D: +// on one, E: on another. A game tracked on one device and auto-tracked on +// another arrives with the first device's drive letter, and on the second the +// folder is simply "missing" though it is right there on another drive. +// +// So a missing folder is looked for at the identical path on every other drive +// letter. Exactly one match is adopted for this device, and said once. None, or +// more than one (which would be a guess), is said once too, and the search is +// not repeated more often than relocateRecheck: it touches every drive, and the +// minute-long watch reconcile asks about missing folders constantly. +const relocateRecheck = time.Hour + +// relocateCandidates lists the folders at savePath's exact path under the other +// drive letters that exist here. Windows drive letters only. +func relocateCandidates(savePath string) []string { + vol := filepath.VolumeName(savePath) + if len(vol) != 2 || vol[1] != ':' { + return nil + } + rest := savePath[2:] + own := strings.ToUpper(vol[:1]) + var found []string + for c := 'C'; c <= 'Z'; c++ { + letter := string(c) + if letter == own { + continue + } + candidate := letter + ":" + rest + if !SaveFolderMissing(candidate) { + found = append(found, candidate) + } + } + return found +} + +// relocateToOtherDrive switches a game whose save folder is missing to the same +// path on another drive, when there is exactly one. Reports whether it did. +func (d *Daemon) relocateToOtherDrive(gameID, name, savePath string) bool { + d.missingMu.Lock() + last, checked := d.relocateChecked[gameID] + if checked && time.Since(last) < relocateRecheck { + d.missingMu.Unlock() + return false + } + if d.relocateChecked == nil { + d.relocateChecked = map[string]time.Time{} + } + d.relocateChecked[gameID] = time.Now() + d.missingMu.Unlock() + + if name == "" { + name = gameID + } + found := relocateCandidates(savePath) + switch { + case len(found) == 0: + if !checked { + d.Log.Log("info", fmt.Sprintf("the save folder of %q is not on another drive here either (looked for %s on every drive letter)", + name, logging.Quote(savePath[min(2, len(savePath)):]))) + } + return false + case len(found) > 1: + if !checked { + d.Log.Log("warn", fmt.Sprintf("the save folder of %q is missing at %s, and the same path exists on %d other drives (%s) — not guessing which; set it with Edit", + name, logging.Quote(savePath), len(found), strings.Join(found, ", "))) + } + return false + } + + newPath, err := d.ValidateSavePath(found[0]) + if err != nil { + d.Log.Log("warn", fmt.Sprintf("the save folder of %q exists at %s, but it cannot be used: %v", name, logging.Quote(found[0]), err)) + return false + } + game, err := d.Store.GetGame(gameID) + if err != nil || game.SavePath != savePath { + return false // changed meanwhile; the next reconcile sees the new path + } + game.SavePath = newPath + if err := d.Store.UpdateGame(game); err != nil { + d.Log.Log("warn", fmt.Sprintf("could not switch %q to %s: %v", name, logging.Quote(newPath), err)) + return false + } + d.Log.Log("success", fmt.Sprintf("the save folder of %q was not at %s but is at %s on this device — using that", + name, logging.Quote(savePath), logging.Quote(newPath))) + d.missingMu.Lock() + delete(d.missing, gameID) + d.missingMu.Unlock() + if d.OnGameChanged != nil { + d.OnGameChanged(gameID) + } + return true +} + // noteMissing records whether a game's save folder is missing, and says so // once when that changes — not on every minute's check — so the log and the // screen follow the folder going and coming back. func (d *Daemon) noteMissing(gameID, name, savePath string, missing bool) { + if missing && d.relocateToOtherDrive(gameID, name, savePath) { + return // found under another drive letter and switched to it + } d.missingMu.Lock() was := d.missing[gameID] if missing { @@ -55,7 +173,10 @@ func (d *Daemon) noteMissing(gameID, name, savePath string, missing bool) { if name == "" { name = gameID } - if missing { + if missing && SaveDriveMissing(savePath) { + d.Log.Log("info", fmt.Sprintf("the save folder of %q is on %s, which this device doesn't have (%s) — the game is skipped here "+ + "and keeps syncing between the devices that have it", name, filepath.VolumeName(savePath), logging.Quote(savePath))) + } else if missing { d.Log.Log("warn", fmt.Sprintf("the save folder of %q is not there (%s) — it is not watched or synced until it is back; "+ "OpenSave will not create it, since an empty folder in its place would read as every file deleted", name, logging.Quote(savePath))) } else { diff --git a/internal/daemon/missing_test.go b/internal/daemon/missing_test.go new file mode 100644 index 00000000..0b5fdf6c --- /dev/null +++ b/internal/daemon/missing_test.go @@ -0,0 +1,40 @@ +package daemon + +import ( + "os" + "path/filepath" + "runtime" + "testing" +) + +// A save path on a drive letter this machine does not have is "not on this +// device", not "folder gone": the UI informs rather than warns about it. +func TestSaveDriveMissing(t *testing.T) { + existing := t.TempDir() + if SaveDriveMissing(filepath.Join(existing, "saves")) { + t.Error("a folder on an existing drive was reported as on a missing drive") + } + if SaveDriveMissing("relative/path") { + t.Error("a path without a volume was reported as on a missing drive") + } + + if runtime.GOOS != "windows" { + return + } + // Find a drive letter that is not mounted. + for c := 'Z'; c >= 'D'; c-- { + root := string(c) + `:\` + if _, err := os.Stat(root); err == nil { + continue + } + p := string(c) + `:\SteamLibrary\steamapps\common\Game\save` + if !SaveDriveMissing(p) { + t.Errorf("%s: drive %c: is absent but was not reported missing", p, c) + } + if !SaveFolderMissing(p) { + t.Errorf("%s: SaveFolderMissing should still report the folder missing", p) + } + return + } + t.Skip("every drive letter D-Z is mounted") +} diff --git a/internal/daemon/relocate_test.go b/internal/daemon/relocate_test.go new file mode 100644 index 00000000..152755fc --- /dev/null +++ b/internal/daemon/relocate_test.go @@ -0,0 +1,82 @@ +package daemon + +import ( + "os" + "path/filepath" + "runtime" + "strings" + "testing" + + "github.com/opensave/opensave/internal/store" +) + +// freeDriveLetter returns a drive letter not mounted here, or "". +func freeDriveLetter() string { + for c := 'Z'; c >= 'D'; c-- { + if _, err := os.Stat(string(c) + `:\`); err != nil { + return string(c) + } + } + return "" +} + +// A game tracked elsewhere with its save on a drive this device lacks, while +// the identical path exists on another drive here, is switched to it — once. +func TestRelocateToOtherDrive(t *testing.T) { + if runtime.GOOS != "windows" { + t.Skip("drive letters are a Windows thing") + } + free := freeDriveLetter() + if free == "" { + t.Skip("no free drive letter") + } + d := newTestDaemon(t) + real := filepath.Join(t.TempDir(), "SteamLibrary", "steamapps", "common", "Game", "save") + if err := os.MkdirAll(real, 0o777); err != nil { + t.Fatal(err) + } + elsewhere := free + ":" + real[2:] // same path, absent drive + if err := d.Store.CreateGame(store.Game{ID: "g", Name: "Game", SavePath: elsewhere, ActiveBranch: "main", AutoSync: true}); err != nil { + t.Fatal(err) + } + + d.noteMissing("g", "Game", elsewhere, true) + + g, err := d.Store.GetGame("g") + if err != nil { + t.Fatal(err) + } + if !strings.EqualFold(g.SavePath, real) { + t.Fatalf("save path = %s, want it switched to %s", g.SavePath, real) + } + d.missingMu.Lock() + stillMissing := d.missing["g"] + d.missingMu.Unlock() + if stillMissing { + t.Error("the game is still marked missing after being found") + } +} + +// Not found: said once, and not searched again on every reconcile. +func TestRelocateNotFound_IsNotRepeated(t *testing.T) { + if runtime.GOOS != "windows" { + t.Skip("drive letters are a Windows thing") + } + free := freeDriveLetter() + if free == "" { + t.Skip("no free drive letter") + } + d := newTestDaemon(t) + path := free + `:\NoSuchLibrary\steamapps\common\Nothing\save` + if err := d.Store.CreateGame(store.Game{ID: "n", Name: "Nothing", SavePath: path, ActiveBranch: "main", AutoSync: true}); err != nil { + t.Fatal(err) + } + if d.relocateToOtherDrive("n", "Nothing", path) { + t.Fatal("relocated to a folder that does not exist") + } + first := d.relocateChecked["n"] + d.relocateToOtherDrive("n", "Nothing", path) + if !d.relocateChecked["n"].Equal(first) { + t.Error("searched again straight away; it should wait relocateRecheck") + } +} diff --git a/internal/daemon/verify.go b/internal/daemon/verify.go index ebf57cbd..47552baf 100644 --- a/internal/daemon/verify.go +++ b/internal/daemon/verify.go @@ -64,6 +64,11 @@ func (d *Daemon) VerifySnapshots(ctx context.Context, gameID string, pace time.D } was := snap.Problem if err := d.Snapshots.Verify(snap); err != nil { + if _, gerr := d.Store.GetSnapshot(snap.ID); gerr != nil { + // Removed while this ran (a newer snapshot holding + // the same files took its place): not damaged, gone. + continue + } report.Damaged = append(report.Damaged, DamagedSnapshot{ GameID: g.ID, GameName: g.Name, SnapshotID: snap.ID, Timestamp: snap.Timestamp, Problem: err.Error(), }) diff --git a/internal/delta/hashcache.go b/internal/delta/hashcache.go index 5ee3b611..61adcda8 100644 --- a/internal/delta/hashcache.go +++ b/internal/delta/hashcache.go @@ -4,6 +4,7 @@ import ( "os" "path/filepath" "runtime" + "sort" "strings" "sync" "time" @@ -64,11 +65,19 @@ type cacheEntry struct { // the old behaviour. const ( // cacheMinBytes is the floor, and the value a small library gets. - cacheMinBytes = 64 << 20 + // + // A budget, not an allocation: the cache only ever holds entries for + // files that exist, so a small library never comes near it. It has to + // cover the rare library with one enormous save, though — a game that + // writes every map chunk as its own file (Project Zomboid: ~240k files, + // ~53MB by entryCost) filled a 64MB budget on its own, and the cache then + // evicted and re-read that save on every pass, which is the churn it + // exists to prevent. + cacheMinBytes = 256 << 20 // cacheMaxCeiling is the most this will ever hold, however many games are // tracked. A cache is meant to replace repeated reading, not to become // the memory problem it was added to fix. - cacheMaxCeiling = 384 << 20 + cacheMaxCeiling = 512 << 20 // cachePerGameBytes is how much the budget grows per tracked game. // // From measurement rather than taste: a game's manifest costs roughly @@ -77,7 +86,10 @@ const ( // library and is not enough for someone tracking 350 games — the cache // sits at its cap and evicts entries it is about to want again, which // quietly reinstates the repeated reading it exists to remove. - cachePerGameBytes = 512 << 10 + // + // Doubled alongside the floor so a large library still gets more than a + // small one (350 games -> 350MB). + cachePerGameBytes = 1 << 20 ) // cacheMaxBytes is the current budget. Not a constant: it scales with how many @@ -193,6 +205,12 @@ func FileEntryFor(path string) (FileEntry, error) { return cachedHashFile(path, info) } +// FileEntryForInfo is FileEntryFor with the FileInfo a directory walk already +// has, so a cache hit costs no syscall at all. +func FileEntryForInfo(path string, info os.FileInfo) (FileEntry, error) { + return cachedHashFile(path, info) +} + func cachedHashFile(path string, info os.FileInfo) (FileEntry, error) { key := cacheKeyFor(path) stamp := cacheStamp{size: info.Size(), mtimeNs: info.ModTime().UnixNano()} @@ -302,13 +320,10 @@ func evictLocked() { for k, v := range hashCache.entries { all = append(all, aged{k, v.usedAt}) } - // Partial selection would be faster, but this runs only on overflow and - // clarity is worth more here than the microseconds. - for i := 1; i < len(all); i++ { - for j := i; j > 0 && all[j].at.Before(all[j-1].at); j-- { - all[j], all[j-1] = all[j-1], all[j] - } - } + // sort.Slice, not an insertion sort: map iteration order is random, so + // insertion sort was quadratic in the entry count — ~10^10 comparisons for + // a quarter-million-file save, on every overflow. + sort.Slice(all, func(i, j int) bool { return all[i].at.Before(all[j].at) }) target := cacheMaxBytes * 3 / 4 for _, a := range all { if hashCache.bytes <= target { diff --git a/internal/delta/manifest.go b/internal/delta/manifest.go index f70016e5..c4f195bc 100644 --- a/internal/delta/manifest.go +++ b/internal/delta/manifest.go @@ -10,7 +10,9 @@ import ( "encoding/json" "fmt" "io" + "io/fs" "os" + "path" "path/filepath" "sort" "strconv" @@ -28,6 +30,32 @@ import ( // peer as a real file (and cascade into name.opensave.tmp.opensave.tmp). const TmpSuffix = ".opensave.tmp" +// NeverSynced reports whether a file in a save folder is never part of the +// save, on any device. +// +// remotecache.vdf is Steam's record of what Steam Cloud has uploaded for a +// game, kept beside the game's files in Steam's userdata folder — which is +// where a game whose saves live in Steam Cloud is tracked. It describes this +// device's Steam, not the save: syncing it hands Steam another machine's +// bookkeeping. And that folder is under Program Files, where writing needs +// rights OpenSave does not have, so the copy failed with "access denied" on +// every pass, from every device. +// +// Left out where decisions are made — what to sync, what a version is, what +// a snapshot holds — and NOT from the manifest a device builds and serves: a +// device on an older build would read the gap as the file deleted, and +// delete its own copy (the same reasoning as exclusion rules, +// syncengine/ignore.go). +func NeverSynced(rel string) bool { + return strings.EqualFold(path.Base(rel), "remotecache.vdf") +} + +// NeverSyncedList names what NeverSynced leaves out. It goes into the tag of +// a version's hash, so changing the list re-takes the hash quietly on every +// device, like changing a game's exclusion rules — instead of every device +// calling the same save a new version of its own at once. +const NeverSyncedList = "remotecache.vdf" + // staleTmpAge is how old a leftover TmpSuffix file must be before the // manifest walk garbage-collects it. Generous enough that a patch actively // writing its temp file is never deleted out from under it. @@ -217,9 +245,17 @@ func HashFile(path string) (FileEntry, error) { } } + wholeHash := hex.EncodeToString(whole.Sum(nil)) + // A single-block file's block hash is the whole-file hash; share the one + // string. Most save files are single-block, and on a save of a quarter + // million of them the duplicate hex strings cost ~20MB per manifest. + if len(blocks) == 1 { + blocks[0].Hash = wholeHash + } + return FileEntry{ Size: info.Size(), - Hash: hex.EncodeToString(whole.Sum(nil)), + Hash: wholeHash, Blocks: blocks, BlockSize: blockSize, MtimeMs: Milli(info.ModTime().UnixMilli()), @@ -263,7 +299,20 @@ func buildManifest(root string) (Manifest, error) { } var dirs []string - err = filepath.Walk(root, func(path string, walkInfo os.FileInfo, walkErr error) error { + // WalkDir, not Walk: Walk issues an extra Lstat per entry, which on + // Windows opens every file and dominated the cost of a warm build of a + // large save (13-19s vs ~1.2s for 240k files). WalkDir takes size and + // mtime from the directory listing itself. + err = filepath.WalkDir(root, func(path string, d fs.DirEntry, walkErr error) error { + var walkInfo os.FileInfo + if walkErr == nil && path != root { + // Reparse points are rejected below from the listing's type bits, + // before Info is asked for anything. + if d.Type()&(os.ModeSymlink|os.ModeIrregular) != 0 { + return nil + } + walkInfo, walkErr = d.Info() + } if walkErr != nil { // The save folder itself could not be read — its listing refused, // or the folder gone since it was looked at. That is not an empty @@ -282,7 +331,7 @@ func buildManifest(root string) (Manifest, error) { // skipped instead of failing the whole manifest — one // unreadable directory must not abort every sync of the game. if os.IsPermission(walkErr) { - if walkInfo != nil && walkInfo.IsDir() { + if d != nil && d.IsDir() { return filepath.SkipDir } return nil @@ -341,7 +390,6 @@ func buildManifest(root string) (Manifest, error) { } return nil } - entry, err := cachedHashFile(path, walkInfo) if err != nil { // Deleted after the walk listed it and before it could be read. diff --git a/internal/delta/neversynced_test.go b/internal/delta/neversynced_test.go new file mode 100644 index 00000000..1bcf8c21 --- /dev/null +++ b/internal/delta/neversynced_test.go @@ -0,0 +1,17 @@ +package delta + +import "testing" + +func TestNeverSynced(t *testing.T) { + for p, want := range map[string]bool{ + "remotecache.vdf": true, + "remote/RemoteCache.VDF": true, + "remote/save.sav": false, + "remotecache.vdf.bak": false, + "myremotecache.vdf": false, + } { + if got := NeverSynced(p); got != want { + t.Errorf("NeverSynced(%q) = %v, want %v", p, got, want) + } + } +} diff --git a/internal/delta/patch.go b/internal/delta/patch.go index 7976fc45..98e0104b 100644 --- a/internal/delta/patch.go +++ b/internal/delta/patch.go @@ -9,6 +9,7 @@ import ( "time" "github.com/opensave/opensave/internal/fsx" + "github.com/opensave/opensave/internal/owntouch" ) // BlockSource supplies the bytes for one block index, either from the @@ -77,7 +78,11 @@ func replaceWithRetry(tmpPath, filePath string) error { delay *= 2 } clearReadOnlyIfSet(filePath) + // Marked before the rename, so the watcher's event finds it already + // known as a change OpenSave made (owntouch). + owntouch.Mark(filePath) if err = renameFile(tmpPath, filePath); err == nil { + owntouch.Settled(filePath) return nil } } diff --git a/internal/owntouch/owntouch.go b/internal/owntouch/owntouch.go new file mode 100644 index 00000000..e63987a5 --- /dev/null +++ b/internal/owntouch/owntouch.go @@ -0,0 +1,137 @@ +// Package owntouch remembers which paths in save folders OpenSave itself has +// just written or removed, so the watcher can tell those changes apart from +// ones a game or a person made. +// +// Without it every change OpenSave applied — a file pulled from a peer, a +// deletion a peer asked for, the mtimes "keep mine" touches — looked to the +// watcher exactly like the game saving. A peer deleting files one request at +// a time over several minutes produced a full auto-snapshot every couple of +// seconds, each of a save that was neither the old one nor the new one, and +// those filled the retention budget and pushed out the snapshots that mattered. +package owntouch + +import ( + "os" + "path/filepath" + "runtime" + "strings" + "sync" + "time" +) + +// window is how long a mark counts. Long enough to cover the watcher's +// debounce and a slow burst of events behind it; short enough that the same +// path changed again by a game a little later is not mistaken for ours. +const window = 2 * time.Minute + +var ( + mu sync.Mutex + marked = map[string]mark{} +) + +// mark is when OpenSave touched a path, whether it removed it, and - once +// OpenSave is done with a file (Settled) - the time and size it left it with. +type mark struct { + at time.Time + removed bool + settled bool + mod time.Time + size int64 +} + +func key(path string) string { + p := filepath.Clean(path) + if runtime.GOOS == "windows" { + p = strings.ToLower(p) + } + return p +} + +// Mark records that OpenSave itself just created, wrote, removed or re-dated +// path. Its parent folder is marked too: creating or removing an entry is an +// event on the folder as well. +func Mark(path string) { record(path, false) } + +// MarkRemoved records that OpenSave itself is about to remove path. Only a +// path marked this way counts as OpenSave's when it is found gone: a file +// OpenSave has just written and a game or a person then deleted is their +// change, and has to sync and be snapshotted as one. +func MarkRemoved(path string) { record(path, true) } + +// Settled records the time and size OpenSave has left a file it marked with, +// once it is done with it (written, renamed into place, re-dated). From then on +// the file is OpenSave's only while it still has exactly those: a game saving +// over it a moment later - inside any slack a time comparison would need - is +// the game's change. +func Settled(path string) { + fi, err := os.Lstat(path) + if err != nil || fi.IsDir() { + return + } + k := key(path) + mu.Lock() + defer mu.Unlock() + if m, ok := marked[k]; ok { + m.settled, m.mod, m.size = true, fi.ModTime(), fi.Size() + marked[k] = m + } +} + +func record(path string, removed bool) { + if path == "" { + return + } + now := time.Now() + mu.Lock() + defer mu.Unlock() + marked[key(path)] = mark{at: now, removed: removed} + marked[key(filepath.Dir(path))] = mark{at: now} + if len(marked) > 50000 { + for k, m := range marked { + if now.Sub(m.at) > window { + delete(marked, k) + } + } + } +} + +// slack is how much later than its mark a file OpenSave wrote may be dated: +// re-dating to now happens just after marking, and some filesystems keep +// times to two seconds. +const slack = 2 * time.Second + +// Recent reports whether the change just seen at path is one OpenSave made: +// the path was marked within the window, and nothing has written it since. +// +// The second half matters as much as the first. A game often saves straight +// after a sync — the same file OpenSave has just put there — and a mark alone +// would take that save for the sync's for two minutes: no snapshot of it, and +// no new version to hand to anyone. Whatever OpenSave leaves behind is dated +// no later than its mark (a pulled file keeps the time it was written to its +// temporary name, or is given the peer's; re-dating sets now), so a file +// dated later was written by someone else. Once OpenSave is done with a file +// (Settled) the test is exact: the time and size it left, or not ours - the +// slack would take a game's save within two seconds of a pull for the pull's. +// A path that is gone is OpenSave's +// only if OpenSave removed it (MarkRemoved): one it wrote and someone then +// deleted is theirs. A folder carries no save of its own, so its mark is +// enough. +func Recent(path string) bool { + mu.Lock() + m, ok := marked[key(path)] + mu.Unlock() + if !ok || time.Since(m.at) > window { + return false + } + fi, err := os.Lstat(path) + if err != nil { + return m.removed + } + if fi.IsDir() { + return true + } + if m.settled { + return fi.ModTime().Equal(m.mod) && fi.Size() == m.size + } + return !fi.ModTime().After(m.at.Add(slack)) +} diff --git a/internal/owntouch/owntouch_test.go b/internal/owntouch/owntouch_test.go new file mode 100644 index 00000000..824f04a4 --- /dev/null +++ b/internal/owntouch/owntouch_test.go @@ -0,0 +1,132 @@ +package owntouch + +import ( + "os" + "path/filepath" + "testing" + "time" +) + +func TestMarkAndRecent(t *testing.T) { + dir := t.TempDir() + file := filepath.Join(dir, "sub", "slot1.sav") + if Recent(file) { + t.Fatal("unmarked path reported as ours") + } + Mark(file) + if err := os.MkdirAll(filepath.Dir(file), 0o777); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(file, []byte("pulled"), 0o666); err != nil { + t.Fatal(err) + } + if !Recent(file) { + t.Error("marked path not reported as ours") + } + if !Recent(filepath.Dir(file)) { + t.Error("the parent folder of a marked path not reported as ours") + } + if Recent(filepath.Join(dir, "sub", "slot2.sav")) { + t.Error("a sibling of a marked path reported as ours") + } +} + +// A game saving over a file OpenSave has just written is the game's change, +// however soon after. +func TestRecent_AWriteAfterTheMarkIsNotOurs(t *testing.T) { + dir := t.TempDir() + file := filepath.Join(dir, "slot1.sav") + Mark(file) + if err := os.WriteFile(file, []byte("pulled"), 0o666); err != nil { + t.Fatal(err) + } + if !Recent(file) { + t.Fatal("the file OpenSave wrote is not reported as ours") + } + // The game saves a little later. + later := time.Now().Add(5 * time.Second) + if err := os.WriteFile(file, []byte("played"), 0o666); err != nil { + t.Fatal(err) + } + if err := os.Chtimes(file, later, later); err != nil { + t.Fatal(err) + } + if Recent(file) { + t.Error("a save written after OpenSave's was taken for OpenSave's") + } +} + +// A file OpenSave has just written, then deleted by a game or a person, is +// their deletion: it has to sync and be snapshotted. Only a path OpenSave +// removed itself is its own when found gone. +func TestRecent_AFileDeletedAfterOurWriteIsNotOurs(t *testing.T) { + dir := t.TempDir() + pulled := filepath.Join(dir, "pulled.sav") + Mark(pulled) + if err := os.WriteFile(pulled, []byte("from the peer"), 0o666); err != nil { + t.Fatal(err) + } + if err := os.Remove(pulled); err != nil { + t.Fatal(err) + } + if Recent(pulled) { + t.Error("a file deleted by someone else after OpenSave wrote it was taken for OpenSave's") + } + + removed := filepath.Join(dir, "removed.sav") + if err := os.WriteFile(removed, []byte("old"), 0o666); err != nil { + t.Fatal(err) + } + MarkRemoved(removed) + if err := os.Remove(removed); err != nil { + t.Fatal(err) + } + if !Recent(removed) { + t.Error("a file OpenSave removed was not reported as its own") + } +} + +// A game saving over a file OpenSave has just pulled is the game's change +// however soon it comes - inside the slack a time comparison allows too. +func TestRecent_AWriteRightAfterASettledPullIsNotOurs(t *testing.T) { + file := filepath.Join(t.TempDir(), "progress.sav") + Mark(file) + if err := os.WriteFile(file, []byte("from A"), 0o666); err != nil { + t.Fatal(err) + } + Settled(file) + if !Recent(file) { + t.Fatal("the file OpenSave pulled is not reported as ours") + } + // Same size, a moment later: only the time tells them apart. + time.Sleep(20 * time.Millisecond) + if err := os.WriteFile(file, []byte("from B"), 0o666); err != nil { + t.Fatal(err) + } + if Recent(file) { + t.Error("a save written straight after the pull was taken for OpenSave's") + } +} + +// A pull renames a file into place, then re-dates it to the peer's time. In +// between, and before it is Settled again, the file is still OpenSave's. +func TestRecent_ARedatedPullBeforeItSettlesAgainIsOurs(t *testing.T) { + file := filepath.Join(t.TempDir(), "auto.sav") + Mark(file) + if err := os.WriteFile(file, []byte("pulled"), 0o666); err != nil { + t.Fatal(err) + } + Settled(file) + Mark(file) + peers := time.Now().Add(-time.Hour) + if err := os.Chtimes(file, peers, peers); err != nil { + t.Fatal(err) + } + if !Recent(file) { + t.Error("a pulled file re-dated to the peer's time was taken for someone else's change") + } + Settled(file) + if !Recent(file) { + t.Error("a pulled file settled at the peer's time was taken for someone else's change") + } +} diff --git a/internal/p2p/deletion_race_test.go b/internal/p2p/deletion_race_test.go index 09f46f32..952fcb19 100644 --- a/internal/p2p/deletion_race_test.go +++ b/internal/p2p/deletion_race_test.go @@ -58,7 +58,7 @@ func (a *deviceA) FetchManifest(ctx context.Context, peer syncengine.Peer, gameI if err != nil { return syncengine.ManifestResponse{}, err } - return syncengine.ManifestResponse{Manifest: m, ActiveBranch: "main", Proto: syncengine.ProtoMultiRoot}, nil + return syncengine.ManifestResponse{Manifest: m, ActiveBranch: "main", Proto: syncengine.ProtoVersions}, nil } func (a *deviceA) FetchBlocks(ctx context.Context, peer syncengine.Peer, ref syncengine.FileRef, blockIndices []int, blockSize int) ([]syncengine.BlockData, error) { @@ -159,7 +159,7 @@ func writeFile(t *testing.T, path, body string) { // deleteRequest is one of A's deletion requests arriving at B's LAN route. func (f *raceFixture) deleteRequest(t *testing.T, rel, root string) { t.Helper() - body, _ := json.Marshal(map[string]string{"relPath": rel, "root": root}) + body, _ := json.Marshal(map[string]any{"relPath": rel, "root": root, "versioned": true}) r := httptest.NewRequest(http.MethodPost, "/api/p2p/delete-file/"+f.gameID, bytes.NewReader(body)) r.RemoteAddr = "198.51.100.7:40000" // not a paired device's address: no lineage refresh races the test rc := chi.NewRouteContext() @@ -172,6 +172,26 @@ func (f *raceFixture) deleteRequest(t *testing.T, rel, root string) { } } +// A deletion asked for by an OpenSave that keeps no save versions is refused: +// its old file comparison reads whatever its copy lacks as deleted. +func TestADeletionFromAnOldBuildIsRefused(t *testing.T) { + f := newRaceFixture(t, false) + body, _ := json.Marshal(map[string]any{"relPath": "slot1.sav"}) + r := httptest.NewRequest(http.MethodPost, "/api/p2p/delete-file/"+f.gameID, bytes.NewReader(body)) + r.RemoteAddr = "198.51.100.7:40000" + rc := chi.NewRouteContext() + rc.URLParams.Add("gameId", f.gameID) + r = r.WithContext(context.WithValue(r.Context(), chi.RouteCtxKey, rc)) + w := httptest.NewRecorder() + f.e.handleDeleteFile(w, r) + if w.Code != http.StatusConflict { + t.Errorf("a deletion from an old build: %d %s, want %d", w.Code, w.Body.String(), http.StatusConflict) + } + if !f.has(f.bDir, "slot1.sav") { + t.Error("the file was deleted on an old build's word") + } +} + func (f *raceFixture) has(dir, rel string) bool { _, err := os.Stat(filepath.Join(dir, filepath.FromSlash(rel))) return err == nil @@ -212,7 +232,7 @@ func TestASyncBetweenTwoRelayedDeletionsIsNotAConflict(t *testing.T) { os.Remove(filepath.Join(f.a.dir, "slot1.sav")) os.Remove(filepath.Join(f.a.dir, "slot2.sav")) - body, _ := json.Marshal(map[string]string{"relPath": "slot1.sav"}) + body, _ := json.Marshal(map[string]any{"relPath": "slot1.sav", "versioned": true}) if code, out := w.serveDeleteFile("/delete-file/"+f.gameID, body, "node_unknown"); code != 200 { t.Fatalf("relayed delete: %d %v", code, out) } diff --git a/internal/p2p/engine.go b/internal/p2p/engine.go index b1777e2e..83e68ea7 100644 --- a/internal/p2p/engine.go +++ b/internal/p2p/engine.go @@ -9,11 +9,15 @@ import ( "encoding/json" "errors" "fmt" + "io/fs" "net/http" + "os" + "path/filepath" "strings" "sync" "time" + "github.com/opensave/opensave/internal/delta" "github.com/opensave/opensave/internal/e2ee" "github.com/opensave/opensave/internal/p2p/discovery" "github.com/opensave/opensave/internal/p2p/pairing" @@ -119,6 +123,20 @@ type Engine struct { // that a device has gone away — see offlineStrikes. pingMu sync.Mutex pingMisses map[string]int + // probeMu serializes PingPairedPeers rounds. + probeMu sync.Mutex + // probingLinks: peers whose connection is being speed-tested now + // (probeLinkSoon). + probeLinksMu sync.Mutex + probingLinks map[string]bool + // When each game last got a safety snapshot before a peer's deletions + // (snapshotBeforePeerDeletions). + peerDeleteSnapMu sync.Mutex + peerDeleteSnap map[string]time.Time + // answered holds the peers that answered a probe during this run, and + // startedMs is when the run began (heardThisRun). + answered map[string]bool + startedMs int64 // Lineage refreshes already running after a peer-applied deletion, keyed // by game+peer. Deletions arrive one file at a time, and each one leaves @@ -283,6 +301,7 @@ func New(s *store.Store, snaps *snapshot.Manager, logf func(level, msg string)) Log: logf, ctx: ctx, cancel: cancel, + startedMs: time.Now().UnixMilli(), } e.Wan = newWanClient(e) e.RelayHost = NewRelayHost(logf) @@ -379,9 +398,34 @@ func (e *Engine) clearPingMisses(peerID string) { delete(e.pingMisses, peerID) } +// noteAnswered records that a peer answered a probe during this run. +func (e *Engine) noteAnswered(peerID string) { + e.pingMu.Lock() + defer e.pingMu.Unlock() + if e.answered == nil { + e.answered = map[string]bool{} + } + e.answered[peerID] = true +} + +// heardThisRun reports whether there is any sign of a peer since this engine +// started: a probe it answered, or a sighting (discovery, a request from it, +// the relay) recent enough to postdate the start. +func (e *Engine) heardThisRun(p store.Peer) bool { + e.pingMu.Lock() + answered := e.answered[p.ID] + e.pingMu.Unlock() + return answered || p.LastSeenMs >= e.startedMs +} + // PingPairedPeers probes every paired LAN peer and updates their // online/offline status. func (e *Engine) PingPairedPeers(ctx context.Context) { + // One probe round at a time: the presence loop and a sync starting can + // both ask, and two overlapping rounds would count one outage as two + // misses toward offlineStrikes. + e.probeMu.Lock() + defer e.probeMu.Unlock() peers, err := e.Store.ListPeers() if err != nil { return @@ -414,11 +458,19 @@ func (e *Engine) PingPairedPeers(ctx context.Context) { // failed cannot lose anything: a device that really has gone is // declared offline a few probes later, and the only cost is a sync // attempt that fails the way it would have anyway. + // + // That patience is for a device seen during this run. A status + // carried over from the previous run is not evidence of anything: a + // PC switched off overnight came back "online" from the database on + // every start, and stayed so until three probes had failed. A device + // not heard from since this start goes offline on its first miss. newStatus := p.Status if ok { e.clearPingMisses(p.ID) + e.noteAnswered(p.ID) newStatus = "online" - } else if e.notePingMiss(p.ID) >= offlineStrikes { + e.probeLinkSoon(p) + } else if misses := e.notePingMiss(p.ID); misses >= offlineStrikes || !e.heardThisRun(p) { newStatus = "offline" } @@ -434,6 +486,35 @@ func (e *Engine) PingPairedPeers(ctx context.Context) { } } +// probeLinkSoon tests the connection to a peer that just answered, in the +// background, when nothing has measured it for a while +// (syncengine/linkspeed.go). One test per peer at a time. +func (e *Engine) probeLinkSoon(p store.Peer) { + if e.Sync == nil || p.Address == "relay" { + return + } + e.probeLinksMu.Lock() + if e.probingLinks == nil { + e.probingLinks = map[string]bool{} + } + if e.probingLinks[p.ID] { + e.probeLinksMu.Unlock() + return + } + e.probingLinks[p.ID] = true + e.probeLinksMu.Unlock() + go func() { + defer func() { + e.probeLinksMu.Lock() + delete(e.probingLinks, p.ID) + e.probeLinksMu.Unlock() + }() + if e.Sync.ProbeLinkIfDue(context.Background(), syncengine.Peer{ID: p.ID, Name: p.Name, Address: p.Address, Port: p.Port}) { + e.notifyPeerUpdate() + } + }() +} + // ErrNoPeersOnline is a sync with no other device to sync with: none of the // paired devices answered. Not a failure of anything — the save goes over when // one is back — which is why it is told apart from errors that are. @@ -542,6 +623,8 @@ func (e *Engine) StartResyncLoop() { stop := e.stopRetry e.pendingMu.Unlock() + e.startPresenceLoop(stop) + // Tracked on the engine's lifecycle so a shutdown lands between ticks // rather than half-way through a transfer. e.GoSync(func(ctx context.Context) { @@ -571,6 +654,43 @@ func (e *Engine) StartResyncLoop() { }) } +// presenceInterval is how often paired LAN peers are probed just to know +// whether they are there. +// +// Probing used to happen only as a side effect: before the 60-second +// reconcile, before a retry, before a manual sync. The reconcile runs on the +// same goroutine as the sync-everything pass it starts, so while a big +// library synced the next probe waited minutes, and with offlineStrikes +// that was minutes more before a device that had gone showed as gone. Peers +// seen by UDP broadcast hid this; one reached over a VPN such as Tailscale, +// which does not carry broadcasts, depends on probes alone. +// +// 10s makes "back" show within ~10s and "gone" within ~30s. A probe is one +// small GET per peer. A var only so a test can shorten it. +var presenceInterval = 10 * time.Second + +// startPresenceLoop probes paired peers on its own ticker, independent of +// whatever sync work is running. It stops with the resync loop. +func (e *Engine) startPresenceLoop(stop chan struct{}) { + e.GoSync(func(ctx context.Context) { + // At once, then on the ticker: until the first round, every peer + // shows whatever the previous run left in the database. + e.PingPairedPeers(ctx) + ticker := time.NewTicker(presenceInterval) + defer ticker.Stop() + for { + select { + case <-stop: + return + case <-ctx.Done(): + return + case <-ticker.C: + e.PingPairedPeers(ctx) + } + } + }) +} + // reconcileAllGames refreshes peer status and re-syncs every game — the // periodic backstop against missed sync triggers. func (e *Engine) reconcileAllGames(ctx context.Context) { @@ -602,6 +722,13 @@ func (e *Engine) retryPendingResyncs(ctx context.Context) { if ctx.Err() != nil { return // shutting down; the failsafe picks these up next start } + // A save folder that is not here cannot be synced, and retrying it + // every tick only logs the same failure. Dropped from the queue; the + // periodic reconcile picks it up once the folder is back. + if g, err := e.Store.GetGame(id); err == nil && saveFolderAbsent(g.SavePath) { + e.ClearPendingResync(id) + continue + } e.Log("info", fmt.Sprintf("retrying interrupted sync for %s", id)) results, err := e.Sync.SyncGame(ctx, id, online) if err == nil { @@ -616,6 +743,22 @@ func (e *Engine) retryPendingResyncs(ctx context.Context) { } } +// saveFolderAbsent reports whether a game's save folder is not on this device +// — the same test as daemon.SaveFolderMissing, which this package cannot +// import. For a single-file save it is the folder the file goes in that +// counts: the file itself may not have been written yet. +func saveFolderAbsent(savePath string) bool { + if savePath == "" { + return false + } + folder := savePath + if isFile, err := delta.ResolveLocalSaveFilePath(savePath); err == nil && isFile { + folder = filepath.Dir(savePath) + } + _, err := os.Stat(folder) + return errors.Is(err, fs.ErrNotExist) +} + // SyncAllGames syncs every tracked game (used when a peer comes online). func (e *Engine) SyncAllGames(ctx context.Context) { if e.Pause.Paused() { @@ -633,6 +776,12 @@ func (e *Engine) SyncAllGames(ctx context.Context) { if !g.AutoSync { continue } + // Not on this device (a drive it doesn't have, a folder that is + // gone): nothing to sync, and it is shown as such on screen. Trying + // anyway logged an error for it against every peer, every pass. + if saveFolderAbsent(g.SavePath) { + continue + } results, err := e.Sync.SyncGame(ctx, g.ID, online) if errors.Is(err, syncengine.ErrHeld) { continue // said once, when it was held; asked about on screen diff --git a/internal/p2p/manifest_gzip_test.go b/internal/p2p/manifest_gzip_test.go new file mode 100644 index 00000000..cca409cf --- /dev/null +++ b/internal/p2p/manifest_gzip_test.go @@ -0,0 +1,63 @@ +package p2p + +import ( + "crypto/sha256" + "encoding/hex" + "fmt" + "net/http" + "net/http/httptest" + "sync/atomic" + "testing" + + "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/p2p/syncengine" +) + +// A large manifest goes over the wire gzipped, and an ordinary Go client — +// which is what every OpenSave version uses — reads it without knowing. +func TestManifestIsCompressedAndReadable(t *testing.T) { + m := delta.Manifest{Files: map[string]delta.FileEntry{}} + for i := 0; i < 5000; i++ { + sum := sha256.Sum256([]byte(fmt.Sprint(i))) // random-looking, like real hashes + h := hex.EncodeToString(sum[:]) + m.Files[fmt.Sprintf("Saves/world/map_%d_%d.bin", i/100, i%100)] = delta.FileEntry{ + Size: 3500, Hash: h, BlockSize: 65536, Blocks: []delta.Block{{Index: 0, Hash: h, Length: 3500}}, + } + } + resp := syncengine.ManifestResponse{Manifest: m, ActiveBranch: "main"} + + var wire atomic.Int64 + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + cw := &countingResponse{ResponseWriter: w} + jsonOKCompressed(cw, r, resp) + wire.Store(cw.n) + })) + defer srv.Close() + + req, _ := http.NewRequest(http.MethodGet, srv.URL, nil) + var got syncengine.ManifestResponse + if err := doJSONWith(lanBulkClient, req, &got); err != nil { + t.Fatal(err) + } + if len(got.Manifest.Files) != len(m.Files) || got.Manifest.ManifestHash() != m.ManifestHash() { + t.Fatalf("manifest came back with %d files and a different hash", len(got.Manifest.Files)) + } + + plain := httptest.NewRecorder() + jsonOK(plain, resp) + if wire.Load()*2 > int64(plain.Body.Len()) { + t.Errorf("compressed %d bytes vs %d plain — expected at least 2x smaller", wire.Load(), plain.Body.Len()) + } + t.Logf("manifest of %d files: %d bytes plain, %d on the wire", len(m.Files), plain.Body.Len(), wire.Load()) +} + +type countingResponse struct { + http.ResponseWriter + n int64 +} + +func (c *countingResponse) Write(b []byte) (int, error) { + n, err := c.ResponseWriter.Write(b) + c.n += int64(n) + return n, err +} diff --git a/internal/p2p/missing_skip_test.go b/internal/p2p/missing_skip_test.go new file mode 100644 index 00000000..b5de9828 --- /dev/null +++ b/internal/p2p/missing_skip_test.go @@ -0,0 +1,28 @@ +package p2p + +import ( + "os" + "path/filepath" + "testing" +) + +func TestSaveFolderAbsent(t *testing.T) { + dir := t.TempDir() + if saveFolderAbsent(dir) { + t.Error("an existing folder was reported absent") + } + if !saveFolderAbsent(filepath.Join(dir, "gone")) { + t.Error("a missing folder was not reported absent") + } + // A single-file save not written yet: its folder is what counts. + file := filepath.Join(dir, "save.sav") + if err := os.WriteFile(file, []byte("x"), 0o666); err != nil { + t.Fatal(err) + } + if saveFolderAbsent(file) { + t.Error("an existing single-file save was reported absent") + } + if saveFolderAbsent("") { + t.Error("an empty path was reported absent") + } +} diff --git a/internal/p2p/presence_test.go b/internal/p2p/presence_test.go new file mode 100644 index 00000000..aca90a06 --- /dev/null +++ b/internal/p2p/presence_test.go @@ -0,0 +1,125 @@ +package p2p + +import ( + "net" + "net/http" + "net/http/httptest" + "path/filepath" + "strconv" + "sync/atomic" + "testing" + "time" + + "github.com/opensave/opensave/internal/snapshot" + "github.com/opensave/opensave/internal/store" +) + +func waitUntil(d time.Duration, cond func() bool) bool { + deadline := time.Now().Add(d) + for time.Now().Before(deadline) { + if cond() { + return true + } + time.Sleep(20 * time.Millisecond) + } + return cond() +} + +// A PC switched off since the last run is in the database as "online". It must +// not stay so for the three probes a device seen this run is allowed: the +// first unanswered probe after start says it is not there. +func TestPresence_StaleOnlineFromLastRunClearsAtOnce(t *testing.T) { + old := presenceInterval + presenceInterval = time.Hour // only the probe at start + t.Cleanup(func() { presenceInterval = old }) + + s, err := store.Open(filepath.Join(t.TempDir(), "opensave.db")) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { s.Close() }) + if err := s.EnsureDefaultSettings(t.TempDir(), t.TempDir()); err != nil { + t.Fatal(err) + } + // Nothing listens here: the PC is off. + ln, err := net.Listen("tcp", "127.0.0.1:0") + if err != nil { + t.Fatal(err) + } + port := ln.Addr().(*net.TCPAddr).Port + ln.Close() + lastRun := time.Now().Add(-10 * time.Hour).UnixMilli() + if err := s.UpsertPeer(store.Peer{ID: "node_off", Name: "Switched off", Address: "127.0.0.1", Port: port, Status: "online", LastSeenMs: lastRun}); err != nil { + t.Fatal(err) + } + + e := New(s, snapshot.New(s), func(string, string) {}) + stop := make(chan struct{}) + start := time.Now() + e.startPresenceLoop(stop) + t.Cleanup(func() { close(stop); e.cancel(); e.bgSyncWG.Wait() }) + + if !waitUntil(5*time.Second, func() bool { + p, err := s.GetPeer("node_off") + return err == nil && p.Status == "offline" + }) { + t.Fatal("a peer online only in the last run's record was still online after the first probe") + } + t.Logf("shown offline %.1fs after start", time.Since(start).Seconds()) +} + +// A peer's online state follows it on its own ticker, whatever else the +// engine is doing — not only when a sync or the minute-long reconcile happens +// to probe. A peer reached over a VPN gets no UDP broadcasts, so without this +// its indicator lagged by minutes. +func TestPresenceLoop_FollowsPeer(t *testing.T) { + old := presenceInterval + presenceInterval = 50 * time.Millisecond + t.Cleanup(func() { presenceInterval = old }) + + var up atomic.Bool + up.Store(true) + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if !up.Load() { + w.WriteHeader(http.StatusServiceUnavailable) + return + } + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{"status":"ok","appVersion":"2.4.0"}`)) + })) + t.Cleanup(srv.Close) + host, portStr, _ := net.SplitHostPort(srv.Listener.Addr().String()) + port, _ := strconv.Atoi(portStr) + + s, err := store.Open(filepath.Join(t.TempDir(), "opensave.db")) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { s.Close() }) + if err := s.EnsureDefaultSettings(t.TempDir(), t.TempDir()); err != nil { + t.Fatal(err) + } + if err := s.UpsertPeer(store.Peer{ID: "node_vpn", Name: "Over VPN", Address: host, Port: port, Status: "offline"}); err != nil { + t.Fatal(err) + } + + e := New(s, snapshot.New(s), func(string, string) {}) + stop := make(chan struct{}) + e.startPresenceLoop(stop) + t.Cleanup(func() { close(stop); e.cancel(); e.bgSyncWG.Wait() }) + + status := func() string { + p, err := s.GetPeer("node_vpn") + if err != nil { + return "" + } + return p.Status + } + if !waitUntil(3*time.Second, func() bool { return status() == "online" }) { + t.Fatalf("peer answering pings still %q", status()) + } + up.Store(false) + if !waitUntil(3*time.Second, func() bool { return status() == "offline" }) { + t.Fatalf("peer no longer answering still %q", status()) + } +} diff --git a/internal/p2p/routes.go b/internal/p2p/routes.go index 67d11a5c..a66ec9d7 100644 --- a/internal/p2p/routes.go +++ b/internal/p2p/routes.go @@ -1,7 +1,9 @@ package p2p import ( + "compress/gzip" "context" + "crypto/rand" "encoding/json" "errors" "fmt" @@ -11,12 +13,14 @@ import ( "os" "path/filepath" "strconv" + "strings" "sync/atomic" "time" "github.com/go-chi/chi/v5" "github.com/opensave/opensave/internal/delta" "github.com/opensave/opensave/internal/logging" + "github.com/opensave/opensave/internal/owntouch" "github.com/opensave/opensave/internal/p2p/pairing" "github.com/opensave/opensave/internal/p2p/syncengine" "github.com/opensave/opensave/internal/store" @@ -40,18 +44,48 @@ func (e *Engine) RegisterRoutes(r chi.Router) { r.Get("/api/p2p/games", e.handlePeerGameList) r.Post("/api/p2p/sync-event/{gameId}", e.handleSyncEvent) r.Get("/api/p2p/app-binary", e.handleAppBinary) + r.Get("/api/p2p/speedtest", e.handleSpeedTest) // What moves save data, or starts a sync: refused while paused. r.Group(func(r chi.Router) { r.Use(e.refuseWhilePaused) r.Get("/api/p2p/manifest/{gameId}", e.handleManifest) r.Post("/api/p2p/blocks/{gameId}", e.handleBlocks) + r.Post("/api/p2p/files/{gameId}", e.handleFileBatch) r.Post("/api/p2p/delete-file/{gameId}", e.handleDeleteFile) r.Get("/api/sync/trigger/{gameId}", e.handleSyncTrigger) }) }) } +// speedTestMax bounds one speed test, and speedTestData is what it sends: +// random, so nothing along the way can compress it into a better figure. +const speedTestMax = 8 << 20 + +var speedTestData = func() []byte { + b := make([]byte, 1<<20) + _, _ = rand.Read(b) + return b +}() + +// handleSpeedTest sends n bytes (at most speedTestMax) for a peer to time: +// how it learns how close this device is (syncengine/linkspeed.go). +func (e *Engine) handleSpeedTest(w http.ResponseWriter, r *http.Request) { + n, _ := strconv.Atoi(r.URL.Query().Get("n")) + if n <= 0 || n > speedTestMax { + n = speedTestMax + } + w.Header().Set("Content-Type", "application/octet-stream") + w.Header().Set("Content-Length", strconv.Itoa(n)) + for n > 0 { + chunk := min(n, len(speedTestData)) + if _, err := w.Write(speedTestData[:chunk]); err != nil { + return + } + n -= chunk + } +} + func clientIP(r *http.Request) string { host, _, err := net.SplitHostPort(r.RemoteAddr) if err != nil { @@ -720,11 +754,25 @@ func (e *Engine) handleManifest(w http.ResponseWriter, r *http.Request) { ActiveBranch: game.ActiveBranch, Proto: ServedProto(), DeletionConfirmed: e.Sync.DeletionConfirmed(game.ID), + Version: e.Sync.LocalVersion(game, manifest), + Versions: true, } if latest, err := e.Snapshots.LatestSnapshot(gameID, ""); err == nil { - resp.LatestSnapshot = &syncengine.SnapshotInfo{ID: latest.ID, Timestamp: latest.Timestamp, Comment: latest.Comment} - } - jsonOK(w, resp) + resp.LatestSnapshot = &syncengine.SnapshotInfo{ID: latest.ID, Timestamp: latest.Timestamp, Comment: latest.Comment, ContentHash: latest.ContentHash} + } + // The asker already holds the state both devices last agreed on and says + // so; if this device still holds it too, "unchanged" is the whole answer. + // For a save of a quarter-million files the full manifest is tens of MB + // of JSON, and the minute-long reconcile used to fetch it from every peer + // every minute to learn that nothing had changed. Single-location games + // only: extra locations carry their own bases (see RootHash). + if ifHash := r.URL.Query().Get("ifHash"); ifHash != "" && len(manifest.Extra) == 0 && + manifest.ManifestHash() == ifHash { + resp.Manifest = delta.Manifest{} + resp.Unchanged = true + resp.ManifestHash = ifHash + } + jsonOKCompressed(w, r, resp) } // holdForServing holds the game's save still for a manifest to be served from @@ -787,11 +835,141 @@ func (e *Engine) handleBlocks(w http.ResponseWriter, r *http.Request) { jsonOK(w, map[string]any{"blocks": encodeBlocks(blocks, wantsGzip(body.Encodings))}) } +// Bounds on one batch request, so a peer cannot make this device read and +// hold an unbounded amount in a single response. The requester stays well +// under them (syncengine.batchMaxFiles / batchMaxBytes). +const ( + fileBatchMaxFiles = 1024 + fileBatchMaxBytes = 16 << 20 +) + +// handleFileBatch serves the blocks of many files of one save location in a +// single response: handleBlocks for a list. One round trip per file was what +// made a save of many small files take hours; see syncengine.ProtoBatchFiles. +// +// A file that cannot be read is reported in its own entry rather than failing +// the batch, so the requester can say which one it was. +func (e *Engine) handleFileBatch(w http.ResponseWriter, r *http.Request) { + gameID := chi.URLParam(r, "gameId") + var body struct { + Root string `json:"root"` + Files []syncengine.FileBlocksRequest `json:"files"` + Encodings []string `json:"encodings"` + } + if err := json.NewDecoder(r.Body).Decode(&body); err != nil || len(body.Files) == 0 { + jsonError(w, http.StatusBadRequest, "files are required") + return + } + if len(body.Files) > fileBatchMaxFiles { + jsonError(w, http.StatusBadRequest, fmt.Sprintf("at most %d files per batch", fileBatchMaxFiles)) + return + } + game, err := e.trackedGameForPeer(gameID) + if err != nil { + jsonError(w, http.StatusNotFound, "Game not found.") + return + } + base, ok := e.resolveServeRoot(gameID, game, body.Root) + if !ok { + jsonError(w, http.StatusNotFound, "This device has no save location named "+strconv.Quote(body.Root)+" for that game.") + return + } + singleFile, _ := delta.ResolveLocalSaveFilePath(base) + + // What the answer will hold: each file's requested blocks at its block + // size, but never more than the file itself. Counting every block at full + // size took 256 files of 4 KB for 16 MB — the whole limit — and refused + // any batch of small files with a few larger ones among them, which failed + // whole pulls of saves made of many small files. + var requested int64 + for _, f := range body.Files { + bs := f.BlockSize + if bs <= 0 { + bs = 64 * 1024 + } + n := int64(len(f.BlockIndices)) * int64(bs) + if f.RelPath != "" && delta.IsSafePath(base, f.RelPath) { + p := delta.LocalNameFor(base, f.RelPath) + if singleFile { + p = base + } + if info, err := os.Stat(p); err == nil && info.Size() < n { + n = info.Size() + } + } + requested += n + } + if requested > fileBatchMaxBytes { + jsonError(w, http.StatusBadRequest, "batch too large") + return + } + gz := wantsGzip(body.Encodings) + + out := make([]syncengine.FileBlocks, 0, len(body.Files)) + for _, f := range body.Files { + entry := syncengine.FileBlocks{RelPath: f.RelPath} + if f.RelPath == "" || !delta.IsSafePath(base, f.RelPath) { + entry.Error = "invalid path" + out = append(out, entry) + continue + } + // Resolved, not joined — see handleBlocks. + fullPath := delta.LocalNameFor(base, f.RelPath) + if singleFile { + fullPath = base + } + blocks, err := delta.ReadBlocks(fullPath, f.BlockIndices, f.BlockSize) + if err != nil { + entry.Error = "read blocks failed: " + err.Error() + } else { + entry.Blocks = encodeBlocks(blocks, gz) + } + out = append(out, entry) + } + jsonOK(w, map[string]any{"files": out}) +} + +// peerDeleteSnapshotEvery is how far apart two safety snapshots before a +// peer's deletions may be: one for a burst of deletions, however many files +// it removes one request at a time. +const peerDeleteSnapshotEvery = 10 * time.Minute + +// snapshotBeforePeerDeletions takes a safety snapshot of a game before the +// first of a burst of deletions a peer asks for. +func (e *Engine) snapshotBeforePeerDeletions(gameID string, r *http.Request) { + if e.Snapshots == nil { + return + } + e.peerDeleteSnapMu.Lock() + if e.peerDeleteSnap == nil { + e.peerDeleteSnap = map[string]time.Time{} + } + if last, ok := e.peerDeleteSnap[gameID]; ok && time.Since(last) < peerDeleteSnapshotEvery { + e.peerDeleteSnapMu.Unlock() + return + } + e.peerDeleteSnap[gameID] = time.Now() + e.peerDeleteSnapMu.Unlock() + + asker := "another device" + if peer, ok := e.peerByAddress(clientIP(r)); ok { + asker = peer.Name + } + if _, err := e.Snapshots.CreateBeforeReplacing(gameID, fmt.Sprintf("Before %s deleted files here", asker)); err != nil { + e.Log("warn", fmt.Sprintf("safety snapshot before %s's deletions failed: %v", asker, err)) + } +} + func (e *Engine) handleDeleteFile(w http.ResponseWriter, r *http.Request) { gameID := chi.URLParam(r, "gameId") var body struct { RelPath string `json:"relPath"` Root string `json:"root"` + // Versioned: the asking device keeps save versions. One that does + // not decides deletions by the old file comparison, which reads + // whatever its copy lacks as deleted — the deletions that emptied + // whole saves when a copy had arrived only half-way. + Versioned bool `json:"versioned"` } if err := json.NewDecoder(r.Body).Decode(&body); err != nil || body.RelPath == "" { jsonError(w, http.StatusBadRequest, "relPath is required.") @@ -803,6 +981,15 @@ func (e *Engine) handleDeleteFile(w http.ResponseWriter, r *http.Request) { jsonError(w, http.StatusNotFound, "Game not found.") return } + if game.AutoSync && !body.Versioned { + from := clientIP(r) + if p, ok := e.peerByAddress(from); ok { + from = p.Name + } + e.Sync.RefusedOldBuildDelete(game.Name, from) + jsonError(w, http.StatusConflict, syncengine.OldBuildDeleteMessage) + return + } base, ok := e.resolveServeRoot(gameID, game, body.Root) if !ok { jsonError(w, http.StatusNotFound, "This device has no save location named "+strconv.Quote(body.Root)+" for that game.") @@ -821,6 +1008,11 @@ func (e *Engine) handleDeleteFile(w http.ResponseWriter, r *http.Request) { _ = os.Chmod(full, 0o666) deleting := time.Now() if info, statErr := os.Stat(full); statErr == nil { + // One copy of the save before a peer starts deleting from it, not + // one per file: the watcher no longer snapshots changes OpenSave + // applies (owntouch), and it was those, one every few seconds of a + // half-deleted save, that had been the only copy before. + e.snapshotBeforePeerDeletions(game.ID, r) // What is removed is remembered, so a sync of this device's own that // lands before the rest of the batch does not take it for a change // made here (syncengine/peerdeleted.go). @@ -829,6 +1021,8 @@ func (e *Engine) handleDeleteFile(w http.ResponseWriter, r *http.Request) { entry, _ = delta.FileEntryFor(full) } // Empty dirs only, for a folder, like rmdirSync. + // A change made at a peer's request, not by the game (owntouch). + owntouch.MarkRemoved(full) if os.Remove(full) == nil && (info.IsDir() || entry.Hash != "") { e.Sync.NotePeerDeletion(game.ID, body.Root, body.RelPath, entry, info.IsDir()) } @@ -939,6 +1133,10 @@ func (e *Engine) handleSyncEvent(w http.ResponseWriter, r *http.Request) { if e.Sync.Progress.OnSyncError != nil { e.Sync.Progress.OnSyncError(gameID, ev) } + // A sync here waiting for this peer to finish need not wait on. + if peer, ok := e.peerByAddress(clientIP(r)); ok { + e.Sync.NotePeerPulled(gameID, peer.ID) + } } jsonOK(w, map[string]any{"success": true}) } @@ -979,6 +1177,32 @@ func progressEventFromMap(data map[string]any) syncengine.ProgressEvent { return ev } +// jsonOKCompressed is jsonOK, gzipped when the asker accepts it. +// +// For manifests. A save of a few hundred thousand files makes a manifest of +// tens of MB of JSON, which over a VPN did not arrive inside a peer's request +// timeout — so the peer never learned what this device holds, every sync of +// that game failed, and the two devices never converged. Manifest JSON (paths +// and hex hashes) compresses several times over. Go's HTTP client asks for +// gzip and unpacks it on its own unless told not to, so every version of +// OpenSave reads these, including ones that predate this. +func jsonOKCompressed(w http.ResponseWriter, r *http.Request, v any) { + if !strings.Contains(r.Header.Get("Accept-Encoding"), "gzip") { + jsonOK(w, v) + return + } + w.Header().Set("Content-Type", "application/json") + w.Header().Set("Content-Encoding", "gzip") + w.Header().Add("Vary", "Accept-Encoding") + w.WriteHeader(http.StatusOK) + zw, err := gzip.NewWriterLevel(w, gzip.BestSpeed) + if err != nil { + return + } + _ = json.NewEncoder(zw).Encode(v) + _ = zw.Close() +} + func jsonOK(w http.ResponseWriter, v any) { w.Header().Set("Content-Type", "application/json") w.WriteHeader(http.StatusOK) @@ -1019,7 +1243,7 @@ func (e *Engine) resolveServeRoot(gameID string, game store.Game, root string) ( // argument — the read still happens concurrently. var servedProto atomic.Int64 -func init() { servedProto.Store(syncengine.ProtoMultiRoot) } +func init() { servedProto.Store(syncengine.ProtoVersions) } // ServedProto reports the protocol revision advertised to peers. func ServedProto() int { return int(servedProto.Load()) } @@ -1040,6 +1264,8 @@ func SetServedProto(v int) int { // freshly-pushed files deliberately stay out of it (see persistLineage), so // deleting one here would pull it back instead of propagating the delete. func (e *Engine) peerFinishedPulling(gameID string, peer syncengine.Peer, data map[string]any) { + // A sync here waiting for this peer before slower ones goes on. + defer e.Sync.NotePeerPulled(gameID, peer.ID) // Recorded first, and synchronously: the peer said exactly which files it // wrote, so the lineage can be updated now rather than after a manifest // round trip. That round trip is what left a window in which deleting a diff --git a/internal/p2p/syncengine/batch.go b/internal/p2p/syncengine/batch.go new file mode 100644 index 00000000..c1b1ff23 --- /dev/null +++ b/internal/p2p/syncengine/batch.go @@ -0,0 +1,301 @@ +package syncengine + +import ( + "context" + "errors" + "fmt" + "os" + "strings" + "sync" + "time" + + "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/owntouch" +) + +// Pulling many small files, many to a request. +// +// A pull used to cost one HTTP round trip per file, one after the other. +// Measured on a Windows LAN, that came to ~200ms a file, of which writing the +// file (temp, fsync, rename) was ~2ms — so a save made of a quarter-million +// small files took half a day to move 50MB. Batching removes the round trips; +// the writing is the same as before, file by file, and just as safe: each +// file still goes through a PatchWriter (temp file, content verified against +// the manifest's hash, fsync, rename) and gets the peer's mtime. + +const ( + // batchFileMaxBytes is the most a file may need fetched to go into a + // batch. Anything bigger keeps the per-file path, which already fetches + // its blocks in parallel and has nothing to gain here. + batchFileMaxBytes = 256 << 10 + // batchMaxFiles and batchMaxBytes bound one request, well inside what the + // responder accepts (p2p.fileBatchMaxFiles / fileBatchMaxBytes). + batchMaxFiles = 256 + batchMaxBytes = 4 << 20 + // batchMaxClaimed bounds a request as the responder measures it: every + // block requested at the file's full block size, whatever the file's real + // size. 256 files of 4 KB claim 16 MB that way, which is the responder's + // whole limit (p2p.fileBatchMaxBytes), so any file of two blocks in such + // a batch had it refused as "batch too large" — and the whole pull with + // it, over and over, on saves of many small files. Kept under the limit + // with room to spare. + batchMaxClaimed = 12 << 20 +) + +// claimedBytes is what a file costs a batch as the responder counts it. +func claimedBytes(f delta.FileEntry, indices []int) int64 { + bs := f.BlockSize + if bs <= 0 { + bs = 64 * 1024 + } + return int64(len(indices)) * int64(bs) +} + +// batchWorkers is how many batch requests are in flight at once: enough to +// keep the link and the responder's disk busy while this side writes. A var +// only so a test can make the order deterministic. +var batchWorkers = 4 + +type batchJob struct { + relPath string + localPath string + remote delta.FileEntry + indices []int +} + +// changedBytes is how much of a file a pull has to fetch. +func changedBytes(f delta.FileEntry, indices []int) int64 { + var n int64 + for _, idx := range indices { + if idx >= 0 && idx < len(f.Blocks) { + n += int64(f.Blocks[idx].Length) + } + } + return n +} + +// groupBatches splits jobs into requests of at most batchMaxFiles files, +// batchMaxBytes of blocks, and batchMaxClaimed as the responder counts them. +func groupBatches(jobs []batchJob) [][]batchJob { + var out [][]batchJob + var cur []batchJob + var curBytes, curClaimed int64 + for _, j := range jobs { + b := changedBytes(j.remote, j.indices) + c := claimedBytes(j.remote, j.indices) + if len(cur) > 0 && (len(cur) >= batchMaxFiles || curBytes+b > batchMaxBytes || curClaimed+c > batchMaxClaimed) { + out = append(out, cur) + cur, curBytes, curClaimed = nil, 0, 0 + } + cur = append(cur, j) + curBytes += b + curClaimed += c + } + if len(cur) > 0 { + out = append(out, cur) + } + return out +} + +// pullBatched fetches and writes jobs, many files per request. It returns the +// paths it wrote, including on error, since those files are on disk. +func (e *Engine) pullBatched(ctx context.Context, batcher BatchFetcher, peer Peer, gameID, root string, + jobs []batchJob, throttle *throttler, tracker *progressTracker, onProgress func(force bool)) ([]string, error) { + + ctx, cancel := context.WithCancel(ctx) + defer cancel() + + var ( + mu sync.Mutex + written []string + firstErr error + ) + fail := func(err error) { + mu.Lock() + if firstErr == nil { + firstErr = err + cancel() + } + mu.Unlock() + } + + batches := groupBatches(jobs) + workers := batchWorkers + if workers > len(batches) { + workers = len(batches) + } + work := make(chan []batchJob) + var wg sync.WaitGroup + for i := 0; i < workers; i++ { + wg.Add(1) + go func() { + defer wg.Done() + for batch := range work { + done, err := e.pullOneBatch(ctx, batcher, peer, gameID, root, batch, throttle, tracker, onProgress) + mu.Lock() + written = append(written, done...) + mu.Unlock() + if err != nil { + fail(err) + return + } + // One line per batch instead of one per file: a quarter-million + // "file updated" lines rotated every other message out of the log. + if len(done) > 0 { + e.Log("info", fmt.Sprintf("%d files updated (%s … %s)", len(done), done[0], done[len(done)-1])) + } + onProgress(true) + } + }() + } + for _, b := range batches { + select { + case work <- b: + case <-ctx.Done(): + } + } + close(work) + wg.Wait() + + mu.Lock() + defer mu.Unlock() + if firstErr != nil { + return written, firstErr + } + return written, ctx.Err() +} + +// pullOneBatch fetches one batch and writes its files. +func (e *Engine) pullOneBatch(ctx context.Context, batcher BatchFetcher, peer Peer, gameID, root string, + batch []batchJob, throttle *throttler, tracker *progressTracker, onProgress func(force bool)) ([]string, error) { + + req := make([]FileBlocksRequest, len(batch)) + for i, j := range batch { + req[i] = FileBlocksRequest{RelPath: j.relPath, BlockIndices: j.indices, BlockSize: j.remote.BlockSize} + } + + var resp []FileBlocks + var err error + const maxAttempts = 3 + for attempt := 1; attempt <= maxAttempts; attempt++ { + resp, err = batcher.FetchFileBatch(ctx, peer, gameID, root, req) + if err == nil || errors.Is(err, ErrBatchUnsupported) { + break + } + // Refused as too large — a responder that counts or limits batches + // differently: the same request will be refused again, so it goes + // as two halves instead. Asking again unchanged failed the whole pull. + if strings.Contains(err.Error(), "batch too large") && len(batch) > 1 { + half := len(batch) / 2 + first, err := e.pullOneBatch(ctx, batcher, peer, gameID, root, batch[:half], throttle, tracker, onProgress) + if err != nil { + return first, err + } + second, err := e.pullOneBatch(ctx, batcher, peer, gameID, root, batch[half:], throttle, tracker, onProgress) + return append(first, second...), err + } + e.Log("warn", fmt.Sprintf("batch fetch attempt %d/%d of %d files failed: %v", attempt, maxAttempts, len(batch), err)) + if attempt < maxAttempts { + select { + case <-ctx.Done(): + return nil, ctx.Err() + case <-time.After(time.Duration(attempt) * time.Second): + } + } + } + if errors.Is(err, ErrBatchUnsupported) { + return e.pullBatchOneByOne(ctx, peer, gameID, root, batch, throttle, tracker, onProgress) + } + if err != nil { + return nil, fmt.Errorf("fetch batch of %d files: %w", len(batch), err) + } + + byPath := make(map[string]FileBlocks, len(resp)) + for _, f := range resp { + byPath[f.RelPath] = f + } + + var written []string + for _, j := range batch { + if ctx.Err() != nil { + return written, ctx.Err() + } + f, ok := byPath[j.relPath] + if !ok { + return written, fmt.Errorf("fetch blocks for %s: %s did not send it", j.relPath, peer.Name) + } + if f.Error != "" { + return written, fmt.Errorf("fetch blocks for %s: %s", j.relPath, f.Error) + } + if err := writeBatchedFile(j, f.Blocks); err != nil { + return written, err + } + written = append(written, j.relPath) + n := changedBytes(j.remote, j.indices) + tracker.add(n) + onProgress(false) + throttle.wait(ctx, n) + } + return written, nil +} + +// writeBatchedFile writes one file from its fetched blocks, exactly as +// pullFile does: unchanged blocks from the copy on disk, the rest from the +// peer, verified and renamed into place by the PatchWriter. +func writeBatchedFile(j batchJob, blocks []BlockData) error { + writer, err := delta.NewPatchWriter(j.localPath, j.remote) + if err != nil { + return fmt.Errorf("patch %s: %w", j.relPath, err) + } + committed := false + defer func() { + if !committed { + writer.Abort() + } + }() + incoming := make(map[int]bool, len(j.indices)) + for _, idx := range j.indices { + incoming[idx] = true + } + if err := writer.SeedUnchanged(j.localPath, incoming); err != nil { + return fmt.Errorf("patch %s: %w", j.relPath, err) + } + for _, b := range blocks { + if err := writer.WriteBlock(b.Index, b.Data); err != nil { + return fmt.Errorf("patch %s: %w", j.relPath, err) + } + } + if err := writer.Commit(); err != nil { + return fmt.Errorf("patch %s: %w", j.relPath, err) + } + committed = true + if j.remote.MtimeMs > 0 { + mtime := time.UnixMilli(int64(j.remote.MtimeMs)) + owntouch.Mark(j.localPath) // see pullFiles: re-dated, then Settled + _ = os.Chtimes(j.localPath, mtime, mtime) + owntouch.Settled(j.localPath) + } + return nil +} + +// pullBatchOneByOne is the fallback when the transport cannot batch for this +// peer after all: the same files, through the per-file path. +func (e *Engine) pullBatchOneByOne(ctx context.Context, peer Peer, gameID, root string, + batch []batchJob, throttle *throttler, tracker *progressTracker, onProgress func(force bool)) ([]string, error) { + + var written []string + for _, j := range batch { + if err := e.pullFile(ctx, peer, FileRef{GameID: gameID, Root: root, RelPath: j.relPath}, j.localPath, + j.remote, j.indices, throttle, tracker, onProgress); err != nil { + return written, err + } + if j.remote.MtimeMs > 0 { + mtime := time.UnixMilli(int64(j.remote.MtimeMs)) + owntouch.Mark(j.localPath) + _ = os.Chtimes(j.localPath, mtime, mtime) + owntouch.Settled(j.localPath) + } + written = append(written, j.relPath) + } + return written, nil +} diff --git a/internal/p2p/syncengine/batch_test.go b/internal/p2p/syncengine/batch_test.go new file mode 100644 index 00000000..c468cf6c --- /dev/null +++ b/internal/p2p/syncengine/batch_test.go @@ -0,0 +1,237 @@ +package syncengine + +import ( + "bytes" + "context" + "fmt" + "os" + "path/filepath" + "strings" + "sync" + "testing" + + "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/snapshot" +) + +// batchTransport is fakeTransport plus batched fetches, answering manifests +// with a chosen protocol revision. +type batchTransport struct { + *fakeTransport + proto int + + mu sync.Mutex + batchCalls int + blockCalls int + failPath string // answered with a per-file error + // maxFiles, when set, refuses bigger batches the way a responder that + // counts differently does. + maxFiles int +} + +func (b *batchTransport) FetchManifest(ctx context.Context, peer Peer, gameID string, q ManifestQuery) (ManifestResponse, error) { + resp, err := b.fakeTransport.FetchManifest(ctx, peer, gameID, q) + resp.Proto = b.proto + return resp, err +} + +func (b *batchTransport) FetchBlocks(ctx context.Context, peer Peer, ref FileRef, idx []int, bs int) ([]BlockData, error) { + b.mu.Lock() + b.blockCalls++ + b.mu.Unlock() + return b.fakeTransport.FetchBlocks(ctx, peer, ref, idx, bs) +} + +func (b *batchTransport) FetchFileBatch(ctx context.Context, peer Peer, gameID, root string, files []FileBlocksRequest) ([]FileBlocks, error) { + b.mu.Lock() + b.batchCalls++ + b.mu.Unlock() + if len(files) > batchMaxFiles { + return nil, fmt.Errorf("batch of %d files is over the limit", len(files)) + } + if b.maxFiles > 0 && len(files) > b.maxFiles { + return nil, fmt.Errorf(`peer returned 400: {"error":"batch too large"}`) + } + out := make([]FileBlocks, 0, len(files)) + for _, f := range files { + if f.RelPath == b.failPath { + out = append(out, FileBlocks{RelPath: f.RelPath, Error: "read blocks failed: locked"}) + continue + } + blocks, err := b.fakeTransport.FetchBlocks(ctx, peer, FileRef{GameID: gameID, Root: root, RelPath: f.RelPath}, f.BlockIndices, f.BlockSize) + if err != nil { + out = append(out, FileBlocks{RelPath: f.RelPath, Error: err.Error()}) + continue + } + out = append(out, FileBlocks{RelPath: f.RelPath, Blocks: blocks}) + } + return out, nil +} + +func setupBatchEngine(t *testing.T, proto int) (*engineEnv, *batchTransport) { + t.Helper() + env := setupEngine(t) + bt := &batchTransport{fakeTransport: env.transport, proto: proto} + env.engine = New(env.store, snapshot.New(env.store), bt) + return env, bt +} + +// A save of many small files, plus one big one, in two folders. +func writeManyFiles(t *testing.T, dir string, n int) { + t.Helper() + for i := 0; i < n; i++ { + write(t, dir, fmt.Sprintf("map/chunk_%04d.bin", i), strings.Repeat(fmt.Sprintf("chunk %d;", i), 20)) + } + big := bytes.Repeat([]byte("large save data "), (512<<10)/16) // 512KB: per-file path + if err := os.WriteFile(filepath.Join(dir, "big.sav"), big, 0o666); err != nil { + t.Fatal(err) + } +} + +func assertSameTree(t *testing.T, want, got string) { + t.Helper() + wm, err := delta.BuildManifest(want) + if err != nil { + t.Fatal(err) + } + gm, err := delta.BuildManifest(got) + if err != nil { + t.Fatal(err) + } + if wm.ManifestHash() != gm.ManifestHash() { + t.Fatalf("trees differ: %d files remote, %d local", len(wm.Files), len(gm.Files)) + } + for p, f := range wm.Files { + if gm.Files[p].MtimeMs != f.MtimeMs { + t.Errorf("%s: mtime %d, want the peer's %d", p, gm.Files[p].MtimeMs, f.MtimeMs) + break + } + } +} + +func TestPull_BatchesSmallFiles(t *testing.T) { + env, bt := setupBatchEngine(t, ProtoBatchFiles) + writeManyFiles(t, env.remoteDir, 600) + + res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err != nil { + t.Fatalf("SyncWithPeer: %v", err) + } + if res.Status != "updated" || res.Direction != "pull" { + t.Fatalf("result = %+v, want updated/pull", res) + } + assertSameTree(t, env.remoteDir, env.localDir) + + // 600 one-block files: each claims a 64 KB block as the responder counts, + // so 192 fit a request (batchMaxClaimed) -> 4 requests; the big file alone + // goes block by block. + if bt.batchCalls != 4 { + t.Errorf("batch requests = %d, want 4", bt.batchCalls) + } + if bt.blockCalls != 1 { + t.Errorf("per-file block requests = %d, want 1 (the big file)", bt.blockCalls) + } + if len(env.transport.syncEvents) == 0 || env.transport.syncEvents[len(env.transport.syncEvents)-1] != "sync-complete" { + t.Errorf("sync did not complete: %v", env.transport.syncEvents) + } +} + +// A peer that predates batching is asked file by file, as before. +func TestPull_OldPeerIsNotBatched(t *testing.T) { + env, bt := setupBatchEngine(t, ProtoMultiRoot) + writeManyFiles(t, env.remoteDir, 50) + + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatalf("SyncWithPeer: %v", err) + } + assertSameTree(t, env.remoteDir, env.localDir) + if bt.batchCalls != 0 { + t.Errorf("batch requests to an old peer = %d, want 0", bt.batchCalls) + } + if bt.blockCalls != 51 { + t.Errorf("per-file requests = %d, want 51", bt.blockCalls) + } +} + +// One file the peer cannot read fails the pull and names the file, like the +// per-file path does; nothing half-written is left in its place. +func TestPull_BatchFileErrorNamesTheFile(t *testing.T) { + env, bt := setupBatchEngine(t, ProtoBatchFiles) + writeManyFiles(t, env.remoteDir, 10) + bt.failPath = "map/chunk_0003.bin" + + _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err == nil || !strings.Contains(err.Error(), "map/chunk_0003.bin") { + t.Fatalf("err = %v, want one naming map/chunk_0003.bin", err) + } + if _, statErr := os.Stat(filepath.Join(env.localDir, "map", "chunk_0003.bin")); statErr == nil { + t.Error("the file the peer could not serve was written anyway") + } + leftovers, _ := filepath.Glob(filepath.Join(env.localDir, "map", "*"+delta.TmpSuffix)) + if len(leftovers) > 0 { + t.Errorf("temp files left behind: %v", leftovers) + } +} + +func TestGroupBatches_RespectsLimits(t *testing.T) { + var jobs []batchJob + for i := 0; i < 600; i++ { + jobs = append(jobs, batchJob{relPath: fmt.Sprint(i), + remote: delta.FileEntry{Blocks: []delta.Block{{Index: 0, Length: 32 << 10}}}, + indices: []int{0}}) + } + for _, b := range groupBatches(jobs) { + var n int64 + for _, j := range b { + n += changedBytes(j.remote, j.indices) + } + if len(b) > batchMaxFiles || n > batchMaxBytes { + t.Errorf("batch of %d files / %d bytes is over the limit", len(b), n) + } + } +} + +// Many small files, some of several blocks: what the responder counts — +// every block at full block size — must stay under its limit too. 256 files +// of one 64 KB block already claim 16 MB, and each file of more blocks tipped +// a batch over it: "batch too large", and the whole pull failed. +func TestGroupBatches_StaysUnderWhatTheResponderCounts(t *testing.T) { + var jobs []batchJob + for i := 0; i < 2000; i++ { + blocks := []delta.Block{{Index: 0, Length: 4 << 10}} + indices := []int{0} + if i%10 == 0 { + blocks = []delta.Block{{Index: 0, Length: 64 << 10}, {Index: 1, Length: 64 << 10}, {Index: 2, Length: 10 << 10}} + indices = []int{0, 1, 2} + } + jobs = append(jobs, batchJob{relPath: fmt.Sprint(i), + remote: delta.FileEntry{BlockSize: 64 << 10, Blocks: blocks}, indices: indices}) + } + for _, b := range groupBatches(jobs) { + var claimed int64 + for _, j := range b { + claimed += claimedBytes(j.remote, j.indices) + } + if claimed > 16<<20 { + t.Errorf("a batch of %d files claims %d bytes, over the responder's limit", len(b), claimed) + } + } +} + +// A responder that refuses a batch as too large gets it again in halves, +// rather than the pull failing. +func TestPull_ABatchRefusedAsTooLargeIsSplit(t *testing.T) { + env, bt := setupBatchEngine(t, ProtoBatchFiles) + bt.maxFiles = 10 + for i := 0; i < 40; i++ { + write(t, env.remoteDir, fmt.Sprintf("chunks/c%02d.bin", i), fmt.Sprint("chunk ", i)) + } + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatalf("sync failed though the batches could be split: %v", err) + } + for i := 0; i < 40; i++ { + if _, err := os.Stat(filepath.Join(env.localDir, "chunks", fmt.Sprintf("c%02d.bin", i))); err != nil { + t.Fatalf("c%02d.bin did not arrive", i) + } + } +} diff --git a/internal/p2p/syncengine/changedsince_test.go b/internal/p2p/syncengine/changedsince_test.go new file mode 100644 index 00000000..59c57c10 --- /dev/null +++ b/internal/p2p/syncengine/changedsince_test.go @@ -0,0 +1,93 @@ +package syncengine + +import ( + "os" + "path/filepath" + "testing" + "time" + + "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/owntouch" +) + +// A pull that finished and whose file the game then saved over is not a +// pull that stopped part-way: resuming it would replace the new save with +// the peer's. +func TestChangedSincePulled(t *testing.T) { + dir := t.TempDir() + pulled := filepath.Join(dir, "progress.sav") + pull := func(content string) { + owntouch.Mark(pulled) + if err := os.WriteFile(pulled, []byte(content), 0o666); err != nil { + t.Fatal(err) + } + owntouch.Settled(pulled) + } + build := func() delta.Manifest { + m, err := delta.BuildManifest(dir) + if err != nil { + t.Fatal(err) + } + return m + } + d := Decision{FilesToPull: []string{"progress.sav"}} + + pull("from A") + remote := build() + + // The game saves straight after the pull. + time.Sleep(20 * time.Millisecond) + if err := os.WriteFile(pulled, []byte("from B"), 0o666); err != nil { + t.Fatal(err) + } + delta.InvalidateRoot(dir) + if !changedSincePulled(dir, build(), remote, d) { + t.Error("a save written over a finished pull was taken for the pull stopping part-way") + } + + // The pull wrote an old copy and stopped: still as OpenSave left it. + pull("old") + delta.InvalidateRoot(dir) + if changedSincePulled(dir, build(), remote, d) { + t.Error("a pull that stopped part-way was taken for a change made here") + } + + // A file the pull did not touch was the same on both sides when it was + // planned; differing now, it was changed here meanwhile. + if !changedSincePulled(dir, build(), remote, Decision{}) { + t.Error("a file the pull did not touch, changed since, was taken for the pull stopping part-way") + } +} + +// A file the pull did not touch, deleted here the moment the pull is done +// (TestFullSyncFlow on CI): a deletion made here, not a part of the pull to +// fetch again. +func TestChangedSincePulled_AnUntouchedFileDeleted(t *testing.T) { + dir := t.TempDir() + for name, content := range map[string]string{"slot1.sav": "from B", "config/video.ini": "fullscreen=1"} { + full := filepath.Join(dir, filepath.FromSlash(name)) + if err := os.MkdirAll(filepath.Dir(full), 0o777); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(full, []byte(content), 0o666); err != nil { + t.Fatal(err) + } + } + owntouch.Mark(filepath.Join(dir, "slot1.sav")) + owntouch.Settled(filepath.Join(dir, "slot1.sav")) + remote, err := delta.BuildManifest(dir) + if err != nil { + t.Fatal(err) + } + if err := os.Remove(filepath.Join(dir, "config", "video.ini")); err != nil { + t.Fatal(err) + } + delta.InvalidateRoot(dir) + fresh, err := delta.BuildManifest(dir) + if err != nil { + t.Fatal(err) + } + if !changedSincePulled(dir, fresh, remote, Decision{FilesToPull: []string{"slot1.sav"}}) { + t.Error("a file deleted here right after the pull was taken for the pull stopping part-way, and would be fetched back") + } +} diff --git a/internal/p2p/syncengine/conflict.go b/internal/p2p/syncengine/conflict.go index 66293b5a..742c35f8 100644 --- a/internal/p2p/syncengine/conflict.go +++ b/internal/p2p/syncengine/conflict.go @@ -8,6 +8,7 @@ import ( "time" "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/owntouch" "github.com/opensave/opensave/internal/snapshot" ) @@ -41,9 +42,12 @@ func (e *Engine) ResolveConflict(ctx context.Context, gameID, peerID, resolution // see both sides still differing from the stale base and re-conflict // immediately — the "keep looping" bug. e.Log("info", fmt.Sprintf("conflict on %q resolved: keep LOCAL — our version becomes the shared state", gameID)) - if err := e.markResolvedLocal(gameID, peer); err != nil { + if err := e.markResolvedLocal(ctx, gameID, peer); err != nil { return "", err } + // This device's version becomes newer than both, so the peer — and any + // device holding either side's — takes it whole (version.go). + e.versionAfterResolution(gameID, conflict.remoteVersion, true) e.clearConflict(gameID) e.Transport.TriggerPeerPull(peer, gameID) return "", nil @@ -63,8 +67,18 @@ func (e *Engine) ResolveConflict(ctx context.Context, gameID, peerID, resolution e.Log("warn", fmt.Sprintf("safety snapshot before keep-remote failed: %v", err)) } if err := e.overwriteLocalWithRemote(ctx, gameID, peer, remoteData, "Resolved conflict: Overwrite with remote"); err != nil { + // This device's own version is given up either way; what is still + // missing comes from the peer like any outdated device's would. + e.discardLocalVersion(gameID, remoteData.Version) return "", err } + if game, err := e.Store.GetGame(gameID); err == nil { + if now, err := delta.BuildManifest(game.SavePath); err == nil && sameFiles(now, remoteData.Manifest) { + e.versionAfterResolution(gameID, remoteData.Version, false) + } else { + e.discardLocalVersion(gameID, remoteData.Version) + } + } e.markResolvedConverged(gameID, peer) e.clearConflict(gameID) return "", nil @@ -103,9 +117,10 @@ func (e *Engine) ResolveConflict(ctx context.Context, gameID, peerID, resolution // From here it is "keep mine": this device's version becomes the // shared one, and the peer is asked to take it — which, if it has // changes of its own, asks the person there in turn. - if err := e.markResolvedLocal(gameID, peer); err != nil { + if err := e.markResolvedLocal(ctx, gameID, peer); err != nil { return "", err } + e.versionAfterResolution(gameID, conflict.remoteVersion, true) e.clearConflict(gameID) e.Transport.TriggerPeerPull(peer, gameID) return created, nil @@ -160,7 +175,20 @@ func (e *Engine) markResolvedConverged(gameID string, peer Peer) { // unambiguously the newest, so the sync direction is a push. The peer // then receives our version (its own engine snapshots first, or re-raises // its own conflict if it has newer unsynced work of its own). -func (e *Engine) markResolvedLocal(gameID string, peer Peer) error { +// +// The lineage, though, must say only what both sides verifiably hold. It used +// to record every local file as shared (persistLineage(local, local)) before +// the peer had pulled any of it. When that pull did not finish — a large save, +// a request that timed out, a device that went away — the next sync found +// those files "shared before and missing on the peer now", read that as the +// peer having deleted them, and deleted them HERE: on the device whose version +// the person had just chosen to keep. A save of a quarter-million files lost +// tens of thousands that way, and the folders differing again re-raised the +// conflict, inviting the same answer. Now the record is the intersection of +// the two saves as they stand; a file only this device has is new to the +// peer, not deleted by it, and goes over on the next sync. If the peer cannot +// be asked, nothing is recorded as shared, so nothing can be read as deleted. +func (e *Engine) markResolvedLocal(ctx context.Context, gameID string, peer Peer) error { game, err := e.Store.GetGame(gameID) if err != nil { return err @@ -170,7 +198,18 @@ func (e *Engine) markResolvedLocal(gameID string, peer Peer) error { if err != nil { return err } - e.persistLineage(gameID, peer.ID, local, local) + var files, dirs []string + if remote, err := e.peerStateForResolution(ctx, gameID, peer); err == nil { + files, dirs = IntersectLineage(local, remote.Manifest) + } else { + e.Log("info", fmt.Sprintf("could not read %s's save while resolving %q (%v) — nothing is recorded as shared, so nothing it lacks is taken as deleted", peer.Name, gameID, err)) + } + rules := e.rulesFor(gameID) + files = filterPathList(files, rules) + dirs = filterPathList(dirs, rules) + if err := e.Store.SetSyncState(gameID, peer.ID, files, dirs); err != nil { + e.Log("warn", fmt.Sprintf("persist sync lineage failed: %v", err)) + } _ = e.Store.SetAgreedHash(gameID, peer.ID, local.ManifestHash()) _ = e.Store.SetLastManifestHash(gameID, e.contentHashOf(gameID, local, game.SavePath)) e.recordSynced(gameID, peer.ID) @@ -197,14 +236,18 @@ func (e *Engine) touchSaveMtimes(gameID, root string) { return } if !info.IsDir() { + owntouch.Mark(root) _ = os.Chtimes(root, now, now) + owntouch.Settled(root) return } _ = filepath.Walk(root, func(path string, fi os.FileInfo, walkErr error) error { if walkErr != nil || fi.IsDir() { return nil } + owntouch.Mark(path) _ = os.Chtimes(path, now, now) + owntouch.Settled(path) return nil }) } @@ -337,6 +380,7 @@ func (e *Engine) overwriteLocalWithRemote(ctx context.Context, gameID string, pe } full := filepath.Join(game.SavePath, filepath.FromSlash(relPath)) _ = os.Chmod(full, 0o666) + owntouch.MarkRemoved(full) _ = os.Remove(full) } diff --git a/internal/p2p/syncengine/engine.go b/internal/p2p/syncengine/engine.go index e3b02b00..62d5e84e 100644 --- a/internal/p2p/syncengine/engine.go +++ b/internal/p2p/syncengine/engine.go @@ -12,6 +12,7 @@ import ( "time" "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/owntouch" "github.com/opensave/opensave/internal/snapshot" "github.com/opensave/opensave/internal/store" "github.com/opensave/opensave/internal/syncpause" @@ -27,6 +28,17 @@ type Conflict struct { RemoteStats SideStats `json:"remoteStats"` DiffFiles []DiffFile `json:"diffFiles"` // capped; DiffTotal is the real count DiffTotal int `json:"diffTotal"` + // The uncapped counts by kind. DiffFiles stops at 100, and counting from + // it told someone choosing "keep theirs" on a save of a quarter-million + // files that it would remove a few dozen, when it removed tens of + // thousands: every OnlyLocal file goes. + OnlyLocalTotal int `json:"onlyLocalTotal"` + OnlyRemoteTotal int `json:"onlyRemoteTotal"` + ChangedTotal int `json:"changedTotal"` + + // remoteVersion is the peer's version when the conflict was found, which + // the answer is recorded against (version.go). + remoteVersion *VersionInfo } // SideStats summarises one side's save state for the conflict UI. @@ -73,7 +85,7 @@ type Result struct { // back until told whether that was meant (hold.go). peer_busy: the peer was // still writing the game for a sync of its own; this one runs again when it // has finished (settle.go). - Status string `json:"status"` // in_sync | updated | updated_bidirectional | deletions_synced | triggered_peer_pull | conflict | peer_missing | peer_awaiting_folder | peer_holding | peer_busy + Status string `json:"status"` // in_sync | updated | updated_bidirectional | deletions_synced | triggered_peer_pull | conflict | peer_missing | peer_awaiting_folder | peer_holding | peer_busy | unwritable Direction string `json:"direction"` PeerID string `json:"peerId,omitempty"` PeerName string `json:"peerName,omitempty"` @@ -119,6 +131,8 @@ type Engine struct { // on this device right now (settle.go). settleMu sync.Mutex gates map[string]*saveGate + // writtenAt is when a sync last stopped writing each game (BeingWritten). + writtenAt map[string]time.Time // servedMu guards served: the save states this device recently handed // each peer, per game (served.go). servedMu sync.Mutex @@ -136,6 +150,25 @@ type Engine struct { // rootConflicts holds divergences in a game's EXTRA save locations, keyed // by game and location so several can wait on a decision at once. rootConflicts map[string]*RootConflict + + // linkMu guards links, the measured connection to each peer, and + // pullWaits, syncs waiting for a close peer to finish taking a save + // (linkspeed.go). + linkMu sync.Mutex + links map[string]store.PeerLink + pullWaits map[string]chan struct{} + // versionMu serialises reading and writing the save versions + // (version.go); pullingNow is the games this run is pulling a newer + // version of, and loggedOnce the last "waiting" line said per game/peer. + versionMu sync.Mutex + pullingNow map[string]bool + loggedOnce map[string]string + // peerVersions: per peer, whether its OpenSave keeps save versions + // (PeerNeedsUpdate). + peerVersions map[string]bool + // unwritable: games whose save folder could not be written, and until + // when pulls into it wait (version.go). + unwritable map[string]time.Time } // New creates an Engine. @@ -303,13 +336,32 @@ func (e *Engine) SyncGame(ctx context.Context, gameID string, onlinePeers []Peer results := map[string]Result{} busy := false - for _, peer := range onlinePeers { + // Fastest connections first (linkspeed.go): a device that is behind takes + // the newer save from the closest device holding it, and a newer save + // here reaches the close devices before the slow ones. + onlinePeers = e.OrderBySpeed(onlinePeers) + for i, peer := range onlinePeers { // Hard per-peer cap: a wedged transport must never hold // activeSyncs forever (which would silently block every future // sync of this game until an app restart). peerCtx, cancel := context.WithTimeout(ctx, perPeerSyncTimeout) res, err := e.SyncWithPeer(peerCtx, gameID, peer) cancel() + // A close device asked to take this save is given the time to, before + // a much slower one is asked: it gets it at full speed, and the slow + // one can then take it from whichever device is nearest to it. + if err == nil && res.Status == "triggered_peer_pull" && e.waitsForClose(onlinePeers, i) { + var size int64 + if game, gerr := e.Store.GetGame(gameID); gerr == nil { + if m, merr := e.ReadManifest(ctx, gameID, game.SavePath); merr == nil { + for _, f := range m.Files { + size += f.Size + } + } + } + e.Log("info", fmt.Sprintf("waiting for %s (close by) to take %s before the slower devices", peer.Name, gameID)) + e.waitForPull(ctx, gameID, peer, size) + } if err == nil && res.Status == "peer_busy" { // Nothing was compared, so nothing is stamped, and it is not a // failure: the peer was writing this game for a sync of its own. @@ -334,7 +386,7 @@ func (e *Engine) SyncGame(ctx context.Context, gameID string, onlinePeers []Peer // divergence from the NEXT sync, causing the peer to silently overwrite // its own changes instead of detecting the conflict and asking. switch res.Status { - case "conflict", "error": + case "conflict", "error", versionStatusWaiting, StatusPeerOutdatedApp, "unwritable": case "peer_missing", "peer_awaiting_folder", "peer_holding": // The two devices talked and finished, which is what the // per-device stamp has always recorded. But nothing of THIS game @@ -397,15 +449,34 @@ func (e *Engine) SyncWithPeer(ctx context.Context, gameID string, peer Peer) (Re if err != nil { return Result{}, err } - e.Log("info", fmt.Sprintf("syncing %q with %q (%s)", game.Name, peer.Name, - map[bool]string{true: "WAN relay", false: "direct LAN"}[peer.Wan()])) - - // 1. Fetch remote manifest + branch info. isFile, _ := delta.ResolveLocalSaveFilePath(game.SavePath) - remoteData, err := e.Transport.FetchManifest(ctx, peer, gameID, ManifestQuery{ + query := ManifestQuery{ Name: game.Name, SavePath: game.SavePath, IsFile: isFile, AppID: game.AppID, CoverURL: game.CoverURL, - }) + } + + // 0. Nothing changed on either side since they last agreed: say so and + // stop, without moving either manifest. Quietly, too — this is the + // answer for nearly every game on every periodic pass. + res, done, prefetched := e.quickInSync(ctx, gameID, game, peer, query) + if done { + return res, nil + } + + // Said once there is something to do (after the in-sync check below), + // not here: a line per game per peer per periodic pass, most of them + // followed by "already in sync", rotated everything else out of the log + // within hours. + syncingLine := fmt.Sprintf("syncing %q with %q (%s)", game.Name, peer.Name, + map[bool]string{true: "WAN relay", false: "direct LAN"}[peer.Wan()]) + + // 1. Fetch remote manifest + branch info. + var remoteData ManifestResponse + if prefetched != nil { + remoteData = *prefetched + } else { + remoteData, err = e.Transport.FetchManifest(ctx, peer, gameID, query) + } if err != nil { // The peer simply isn't tracking this game (they untracked it, or // never had it). That's a stable state, not a transient network @@ -436,6 +507,19 @@ func (e *Engine) SyncWithPeer(ctx context.Context, gameID string, peer Peer) (Re return Result{}, fmt.Errorf("fetch remote manifest: %w", err) } + // 1b. A peer whose OpenSave keeps no save versions decides by the old file + // comparison: whatever its copy lacks, it reads as deleted, and a copy that + // arrived half-way spreads its gaps that way. Nothing is taken from it + // until it is updated — it can still take from here, and its deletions are + // refused where they arrive (handleDeleteFile). + e.notePeerApp(peer.ID, remoteData.KeepsVersions()) + if game.AutoSync && !remoteData.KeepsVersions() { + e.logOnce(gameID+"|"+peer.ID, fmt.Sprintf( + "%q: %s runs an OpenSave without save versions — nothing is taken from it until it is updated", + game.Name, peer.Name)) + return Result{Status: StatusPeerOutdatedApp, PeerID: peer.ID, PeerName: peer.Name}, nil + } + // 2. Branch alignment: local follows the remote's active branch. // // From the switch until the peer's save has been pulled into the branch, @@ -495,18 +579,45 @@ func (e *Engine) SyncWithPeer(ctx context.Context, gameID string, peer Peer) (Re // would propagate a deletion of the very file being protected. ignoreRules := e.rulesFor(gameID) unfilteredLocal, unfilteredRemote := localManifest, remoteData.Manifest - if !ignoreRules.Empty() { - localManifest = filterManifest(localManifest, ignoreRules) - remoteData.Manifest = filterManifest(remoteData.Manifest, ignoreRules) - } + localManifest = filterManifest(localManifest, ignoreRules) + remoteData.Manifest = filterManifest(remoteData.Manifest, ignoreRules) - // 3. Existing unresolved conflict blocks further syncing. + // 3. Existing unresolved conflict blocks further syncing — unless a + // version newer than both sides of it has turned up since (version.go): + // it was settled elsewhere, by an answer or "Use this save everywhere", + // and the question here no longer stands. e.mu.Lock() - if existing := e.activeConflicts[gameID]; existing != nil { - e.mu.Unlock() - return Result{Status: "conflict", PeerID: peer.ID, PeerName: peer.Name}, nil - } + existing := e.activeConflicts[gameID] e.mu.Unlock() + if existing != nil { + if remoteData.Version == nil || !game.AutoSync || !e.supersedes(gameID, *remoteData.Version) { + return Result{Status: "conflict", PeerID: peer.ID, PeerName: peer.Name}, nil + } + e.Log("info", fmt.Sprintf("%q: %s holds a version newer than both sides of the open conflict — taking it instead of asking", + game.Name, peer.Name)) + e.clearConflict(gameID) + } + + // 3b. The versions decide, when both devices keep them (version.go): who + // is behind takes the newer save whole, from a device that holds it, and + // nothing is ever taken from — or deleted on the word of — a device that + // is behind. Left to the file comparison below only when the two hold + // the same save, or both saves predate versions. + // + // Not while this device is fetching back a save someone emptied here and + // asked to have put back (hold.go): the emptying was recorded as a version + // of its own, and by versions this device would be the newer one, with + // nothing to fetch. The hold's own rules bring the files back. + fetchingBack := false + if h, ok, err := e.Store.GetDeletionHold(gameID); err == nil && ok && h.State == store.HoldFetching { + fetchingBack = true + } + if remoteData.Version != nil && game.AutoSync && !fetchingBack { + if res, handled, err := e.syncByVersion(ctx, game, peer, localManifest, unfilteredLocal, remoteData); handled { + releaseAligned() + return res, err + } + } // 4. Conflict detection (lineage + skew-tolerant mtimes). lastSyncMs := e.lastSyncTimeMs(peer.ID) @@ -582,7 +693,7 @@ func (e *Engine) SyncWithPeer(ctx context.Context, gameID string, peer Peer) (Re // side's UNFILTERED state still hashes to the base then that side has not // changed since, whichever view the base was written in, and its filtered // hash is the same fact expressed in today's terms. - if !ignoreRules.Empty() && agreedHash != "" { + if agreedHash != "" { switch agreedHash { case unfilteredLocal.ManifestHash(): agreedHash = localManifest.ManifestHash() @@ -607,10 +718,9 @@ func (e *Engine) SyncWithPeer(ctx context.Context, gameID string, peer Peer) (Re // path is still recorded as shared. Leave it there and the decision reads // "we both had this and now I do not" — and propagates a deletion of the // file the rule exists to protect. - if rules := e.rulesFor(gameID); !rules.Empty() { - lineageFiles = filterLineage(lineageFiles, rules) - lineageDirs = filterLineage(lineageDirs, rules) - } + rules := ignoreRules + lineageFiles = filterLineage(lineageFiles, rules) + lineageDirs = filterLineage(lineageDirs, rules) // A side that is merely behind the other has not diverged from it, whatever // the clocks and the base say (OnlyBehind). @@ -634,17 +744,14 @@ func (e *Engine) SyncWithPeer(ctx context.Context, gameID string, peer Peer) (Re // Excluded paths are filtered out for the same reason the lineage is: a // rule change must never be able to reach across and remove a peer's file. deleted := e.recordedDeletions(gameID, delta.PrimaryRoot) - if rules := e.rulesFor(gameID); !rules.Empty() { - for path := range deleted { - if rules.Match(path) { - delete(deleted, path) - } + for path := range deleted { + if excluded(rules, path) { + delete(deleted, path) } } decision := ComputeWithDeletions(localManifest, remoteData.Manifest, lineageFiles, lineageDirs, agreedHash, deleted) if !decision.HasChanges() { - e.Log("success", fmt.Sprintf("%q already in sync with %q", game.Name, peer.Name)) e.persistLineage(gameID, peer.ID, localManifest, remoteData.Manifest) // Both sides verifiably identical: this is a convergence point. _ = e.Store.SetAgreedHash(gameID, peer.ID, localManifest.ManifestHash()) @@ -663,8 +770,23 @@ func (e *Engine) SyncWithPeer(ctx context.Context, gameID string, peer Peer) (Re // are read through the gate — so the branch hold goes first. releaseAligned() e.syncExtraRoots(ctx, gameID, game, peer, remoteData) + // Share the peer's latest history entry here too. A pull does this; + // a sync that found both sides already identical used to skip it, + // so a game whose saves were identical everywhere from the start + // (Steam Cloud, a copied folder) showed a snapshot on the device + // that tracked it and none on any other. The local files ARE the + // peer's at this point, so the mirror describes the same content. + // Once per peer snapshot: recordMirrorSnapshot skips an id it has. + // + // Last, after the other locations have synced: it zips every location, + // and taking that time before them delayed their sync for nothing. + if remoteData.LatestSnapshot != nil && len(localManifest.Files) > 0 && !e.hasSnapshot(gameID, game) { + e.recordMirrorSnapshot(gameID, game, peer, *remoteData.LatestSnapshot, + fmt.Sprintf("Synced from peer: %s (%s)", peer.Name, remoteData.LatestSnapshot.Comment)) + } return Result{Status: "in_sync", Direction: "none"}, nil } + e.Log("info", syncingLine) if emptiedUnconfirmed(remoteData.Manifest.Files, decision, remoteData.DeletionConfirmed) { e.Log("info", fmt.Sprintf("%q holds none of %q's save files now, and has not confirmed deleting them — keeping this device's copies", @@ -897,10 +1019,9 @@ func (e *Engine) persistLineage(gameID, peerID string, local, remote delta.Manif // RefreshLineage rebuilds this from unfiltered manifests, so without a // filter here an excluded file would be written back in — and the next // sync would read it as shared, then as deleted, and propagate that. - if rules := e.rulesFor(gameID); !rules.Empty() { - files = filterPathList(files, rules) - dirs = filterPathList(dirs, rules) - } + rules := e.rulesFor(gameID) + files = filterPathList(files, rules) + dirs = filterPathList(dirs, rules) if err := e.Store.SetSyncState(gameID, peerID, files, dirs); err != nil { e.Log("warn", fmt.Sprintf("persist sync lineage failed: %v", err)) } @@ -1032,9 +1153,7 @@ func (e *Engine) AddConfirmedLineageForRoot(gameID, peerID, root string, files [ // Excluded paths must not enter the record of what both sides hold, for // the same reason persistLineage filters them: the next sync would read an // excluded file as shared, then as deleted, and propagate that. - if rules := e.rulesFor(gameID); !rules.Empty() { - merged = filterPathList(merged, rules) - } + merged = filterPathList(merged, e.rulesFor(gameID)) sort.Strings(merged) if err := e.Store.SetSyncStateForRoot(gameID, peerID, root, merged, existingDirs); err != nil { e.Log("warn", fmt.Sprintf("recording confirmed lineage failed: %v", err)) @@ -1260,6 +1379,10 @@ func (e *Engine) registerConflict(gameID string, peer Peer, localManifest delta. // Capture comparison data while we hold both manifests, so the UI can // show which side is further along and exactly what differs. diffs := diffManifests(localManifest, remoteData.Manifest) + counts := map[string]int{} + for _, d := range diffs { + counts[d.Status]++ + } const maxDiffFiles = 100 total := len(diffs) if len(diffs) > maxDiffFiles { @@ -1269,10 +1392,12 @@ func (e *Engine) registerConflict(gameID string, peer Peer, localManifest delta. e.mu.Lock() e.activeConflicts[gameID] = &Conflict{ Peer: peer, LocalSnap: localSnap, RemoteSnap: remoteSnap, - LocalStats: manifestStats(localManifest), - RemoteStats: manifestStats(remoteData.Manifest), - DiffFiles: diffs, - DiffTotal: total, + LocalStats: manifestStats(localManifest), + RemoteStats: manifestStats(remoteData.Manifest), + DiffFiles: diffs, + DiffTotal: total, + OnlyLocalTotal: counts["only-local"], OnlyRemoteTotal: counts["only-remote"], ChangedTotal: counts["changed"], + remoteVersion: remoteData.Version, } e.mu.Unlock() @@ -1371,6 +1496,7 @@ func (e *Engine) applyLocalDeletions(gameID string, root syncRoot, d Decision) { // spelling this disk actually holds. full := delta.LocalNameFor(root.Path, relPath) _ = os.Chmod(full, 0o666) + owntouch.MarkRemoved(full) if err := os.Remove(full); err == nil { e.Log("info", "deleted locally (peer deleted): "+relPath) } @@ -1385,6 +1511,7 @@ func (e *Engine) applyLocalDeletions(gameID string, root syncRoot, d Decision) { } full := delta.LocalNameFor(root.Path, relDir) if info, err := os.Stat(full); err == nil && info.IsDir() { + owntouch.MarkRemoved(full) if err := os.Remove(full); err == nil { // only removes empty dirs, matching rmdirSync e.Log("info", "deleted directory locally (peer deleted): "+relDir) } @@ -1427,7 +1554,9 @@ func (e *Engine) createPulledDirsIn(gameID string, root syncRoot, dirsToPull []s if !delta.IsSafePath(root.Path, relDir) { continue } - _ = os.MkdirAll(filepath.Join(root.Path, filepath.FromSlash(relDir)), 0o777) + full := filepath.Join(root.Path, filepath.FromSlash(relDir)) + owntouch.Mark(full) + _ = os.MkdirAll(full, 0o777) } } @@ -1475,7 +1604,11 @@ func (e *Engine) pullFiles(ctx context.Context, peer Peer, gameID string, game s // Make sure every remote directory exists before patching into it. for _, dir := range remoteData.Manifest.Dirs { if delta.IsSafePath(root.Path, dir) { - _ = os.MkdirAll(filepath.Join(root.Path, filepath.FromSlash(dir)), 0o777) + full := filepath.Join(root.Path, filepath.FromSlash(dir)) + if _, err := os.Stat(full); err != nil { + owntouch.Mark(full) + _ = os.MkdirAll(full, 0o777) + } } } @@ -1523,6 +1656,24 @@ func (e *Engine) pullFiles(ctx context.Context, peer Peer, gameID string, game s tracker := newProgressTracker(totalBytes) throttle := e.throttleFor(peer.Wan()) + // What this transfer says about the connection (linkspeed.go): every + // measureWindow while it runs — a pull of a large save takes an hour over a + // slow link, and one measurement at its end left the link unmeasured all + // that time — and what is left when it ends, however it ends: a pull cut + // off half-way still moved what it moved. + const measureWindow = 30 * time.Second + var measureMu sync.Mutex + measuredAt, measuredBytes := time.Now(), int64(0) + measure := func(final bool) { + moved, _, _ := tracker.stats() + measureMu.Lock() + defer measureMu.Unlock() + if d := time.Since(measuredAt); final || d >= measureWindow { + e.noteLinkRate(peer, moved-measuredBytes, d) + measuredAt, measuredBytes = time.Now(), moved + } + } + defer measure(true) // Progress reporter shared by the per-file loop and the block-group // loop inside each file. Without in-file reporting, a single large @@ -1539,6 +1690,7 @@ func (e *Engine) pullFiles(ctx context.Context, peer Peer, gameID string, game s } lastReport = time.Now() reportMu.Unlock() + measure(false) bytesPulled, speed, pct := tracker.stats() ev := ProgressEvent{PeerName: peer.Name, BytesTransferred: bytesPulled, TotalBytes: totalBytes, SpeedBytesPerSec: speed, Percentage: pct} @@ -1561,6 +1713,11 @@ func (e *Engine) pullFiles(ctx context.Context, peer Peer, gameID string, game s // can record the lineage from confirmed fact rather than rediscovering it // with a manifest round trip. See the sync-complete event below. var pulled []string + // Small files are collected here and fetched many to a request once the + // per-file checks below have passed; see pullBatched. + batcher, canBatch := e.Transport.(BatchFetcher) + canBatch = canBatch && !peer.Wan() && remoteData.Proto >= ProtoBatchFiles + var batched []batchJob for _, relPath := range filesToPull { if !delta.IsSafePath(root.Path, relPath) { return fmt.Errorf("path traversal attempt on pulled file %s", relPath) @@ -1608,13 +1765,21 @@ func (e *Engine) pullFiles(ctx context.Context, peer Peer, gameID string, game s if isFile, _ := delta.ResolveLocalSaveFilePath(root.Path); isFile { localFilePath = root.Path // single-file save mode } + if canBatch && len(indices) > 0 && changedBytes(remoteFile, indices) <= batchFileMaxBytes { + batched = append(batched, batchJob{relPath: relPath, localPath: localFilePath, remote: remoteFile, indices: indices}) + continue + } if err := e.pullFile(ctx, peer, FileRef{GameID: gameID, Root: root.Name, RelPath: relPath}, localFilePath, remoteFile, indices, throttle, tracker, reportProgress); err != nil { return err } if remoteFile.MtimeMs > 0 { mtime := time.UnixMilli(int64(remoteFile.MtimeMs)) + // Marked again first: until Settled below, the time the rename left + // is not the one to compare (owntouch). + owntouch.Mark(localFilePath) _ = os.Chtimes(localFilePath, mtime, mtime) + owntouch.Settled(localFilePath) } pulled = append(pulled, relPath) e.Log("info", "file updated: "+relPath) @@ -1622,6 +1787,13 @@ func (e *Engine) pullFiles(ctx context.Context, peer Peer, gameID string, game s // File-boundary progress reporting (always fires). reportProgress(true) } + if len(batched) > 0 { + done, err := e.pullBatched(ctx, batcher, peer, gameID, root.Name, batched, throttle, tracker, reportProgress) + pulled = append(pulled, done...) + if err != nil { + return err + } + } // Said once, plainly, at the end. A sync that quietly leaves files behind // is worse than one that fails: the save looks synced and is not, and the @@ -1822,11 +1994,37 @@ func fetchWithRetry(ctx context.Context, t Transport, peer Peer, ref FileRef, return nil, lastErr } +// hasSnapshot reports whether the game has any snapshot on its active branch +// here. +// +// A sync that finds both sides already identical records the peer's latest +// snapshot only for a game with no history at all — saves identical from the +// start (Steam Cloud, a copied folder) would otherwise never get one. It used +// to record it for every peer it compared with: a full copy of the same save +// per device, again with each new snapshot over there, which for a save of a +// quarter-million files was gigabytes of identical archives. +func (e *Engine) hasSnapshot(gameID string, game store.Game) bool { + if e.Snapshots == nil { + return false + } + _, err := e.Snapshots.LatestSnapshot(gameID, game.ActiveBranch) + return err == nil +} + // recordMirrorSnapshot zips the (just-updated) local save under the peer's // snapshot id so both devices show the same history entry. func (e *Engine) recordMirrorSnapshot(gameID string, game store.Game, peer Peer, remoteSnap SnapshotInfo, comment string) { - if _, err := e.Store.GetSnapshot(remoteSnap.ID); err == nil { - return // already mirrored + if _, known := e.Store.ResolveSnapshotID(remoteSnap.ID); known { + return // already mirrored, or known as the same files as one here + } + // The peer says what its snapshot holds: if a snapshot here holds exactly + // that, it is the same snapshot, and the save is not archived again. + if remoteSnap.ContentHash != "" { + if existing, ok, _ := e.Store.SnapshotByContent(gameID, game.ActiveBranch, remoteSnap.ContentHash); ok && + snapshot.ArchiveExists(existing.ZipPath) { + _ = e.Store.AddSnapshotAlias(remoteSnap.ID, existing.ID) + return + } } settings, err := e.Store.GetSettings() @@ -1851,16 +2049,40 @@ func (e *Engine) recordMirrorSnapshot(gameID string, game store.Game, peer Peer, if rootsErr != nil { mirrorRoots = nil } - if _, err := snapshot.ZipRoots(game.SavePath, mirrorRoots, zipPath); err != nil { + var skipped []string + var captured []store.CapturedFile + if e.Snapshots != nil { + // Unchanged files copied from the newest snapshot, not compressed + // again: seconds rather than minutes for a large save. + skipped, captured, err = e.Snapshots.ArchiveSaveTo(gameID, game.ActiveBranch, zipPath) + } else { + skipped, captured, err = snapshot.ZipRootsCapturing(game.SavePath, mirrorRoots, zipPath) + } + if err != nil { + os.Remove(zipPath) e.Log("warn", fmt.Sprintf("mirror snapshot zip failed: %v", err)) return } + // Incomplete is not kept (snapshot.IncompleteError); the next sync with + // the peer records it again. + if len(skipped) > 0 { + os.Remove(zipPath) + e.Log("info", fmt.Sprintf("snapshot of %q from %s discarded: %d file(s) could not be read; taken again at the next sync", + game.Name, peer.Name, len(skipped))) + return + } + key := snapshot.ContentKey(captured) + if existing, ok, _ := e.Store.SnapshotByContent(gameID, game.ActiveBranch, key); ok && snapshot.ArchiveExists(existing.ZipPath) { + os.Remove(zipPath) + _ = e.Store.AddSnapshotAlias(remoteSnap.ID, existing.ID) + return + } info, err := os.Stat(zipPath) if err != nil { return } - _ = e.Store.CreateSnapshot(store.Snapshot{ + if err := e.Store.CreateSnapshot(store.Snapshot{ ID: remoteSnap.ID, GameID: gameID, BranchName: game.ActiveBranch, @@ -1869,7 +2091,15 @@ func (e *Engine) recordMirrorSnapshot(gameID string, game store.Game, peer Peer, IsSystemAuto: true, ZipPath: zipPath, SizeBytes: info.Size(), - }) + ContentHash: key, + }); err == nil { + _ = e.Store.RecordSnapshotFiles(remoteSnap.ID, captured) + if e.Snapshots != nil { + if snap, gerr := e.Store.GetSnapshot(remoteSnap.ID); gerr == nil { + e.Snapshots.PruneContainedBy(snap, captured) + } + } + } } // humanBytes formats a byte count as a short human-readable string. diff --git a/internal/p2p/syncengine/engine_test.go b/internal/p2p/syncengine/engine_test.go index 1ed654ff..22a2be33 100644 --- a/internal/p2p/syncengine/engine_test.go +++ b/internal/p2p/syncengine/engine_test.go @@ -224,6 +224,59 @@ func TestSync_PullNewRemoteFile(t *testing.T) { } } +// Saves identical on both sides from the start (Steam Cloud, a copied folder) +// never pull anything — and used to never share the peer's history entry +// either, so the game showed a snapshot on one device and none on the other. +func TestSync_InSyncMirrorsPeerSnapshot(t *testing.T) { + env := setupEngine(t) + write(t, env.localDir, "save.dat", "same progress") + write(t, env.remoteDir, "save.dat", "same progress") + env.transport.latestSnap = &SnapshotInfo{ID: "snap_888", Timestamp: "2026-06-01T00:00:00.000Z", Comment: "Initial snapshot"} + + res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err != nil { + t.Fatalf("SyncWithPeer error = %v", err) + } + if res.Status != "in_sync" { + t.Fatalf("result = %+v, want in_sync", res) + } + snap, err := env.store.GetSnapshot("snap_888") + if err != nil { + t.Fatalf("peer snapshot not mirrored on an in-sync pass: %v", err) + } + if !strings.Contains(snap.Comment, "Synced from peer: Remote Device") { + t.Errorf("mirror comment = %q", snap.Comment) + } + + // A second in-sync pass must not take it again. + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatal(err) + } + game, err := env.store.GetGame("game1") + if err != nil { + t.Fatal(err) + } + snaps, err := env.store.ListSnapshots("game1", game.ActiveBranch) + if err != nil { + t.Fatal(err) + } + if len(snaps) != 1 { + t.Errorf("snapshots after two in-sync passes = %d, want 1", len(snaps)) + } +} + +// An empty save folder in sync with an empty peer has nothing to mirror. +func TestSync_InSyncEmptyFolderTakesNoMirror(t *testing.T) { + env := setupEngine(t) + env.transport.latestSnap = &SnapshotInfo{ID: "snap_999", Timestamp: "2026-06-01T00:00:00.000Z", Comment: "Initial snapshot"} + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatal(err) + } + if _, err := env.store.GetSnapshot("snap_999"); err == nil { + t.Error("an empty folder was mirrored") + } +} + func TestSync_InsufficientDiskSpace(t *testing.T) { env := setupEngine(t) write(t, env.remoteDir, "save.dat", "remote progress that needs space") diff --git a/internal/p2p/syncengine/hold.go b/internal/p2p/syncengine/hold.go index 7060c241..b444156d 100644 --- a/internal/p2p/syncengine/hold.go +++ b/internal/p2p/syncengine/hold.go @@ -85,6 +85,11 @@ type emptiness struct { func (x emptiness) back(held map[string][]string) bool { for name, files := range held { for _, f := range files { + if delta.NeverSynced(f) { + // Recorded by a build that still counted it; never part of + // the save, so not something that has to come back. + continue + } if _, ok := x.present[name][f]; !ok { return false } diff --git a/internal/p2p/syncengine/ignore.go b/internal/p2p/syncengine/ignore.go index 96901a0e..ee3cc8d6 100644 --- a/internal/p2p/syncengine/ignore.go +++ b/internal/p2p/syncengine/ignore.go @@ -42,10 +42,53 @@ func (e *Engine) rulesFor(gameID string) ignore.Rules { return ignore.Parse(game.SyncIgnore) } +// excluded says whether a path takes no part in syncing: a file no save is +// made of on any device (delta.NeverSynced), or one the game's rules exclude. +// The first is handled exactly like the second — the manifest served still +// lists it, the decision never sees it — so a device on an older build, which +// still counts it, is never told it was deleted. +func excluded(rules ignore.Rules, p string) bool { + return delta.NeverSynced(p) || (!rules.Empty() && rules.Match(p)) +} + +func holdsNeverSynced(m delta.Manifest) bool { + for p := range m.Files { + if delta.NeverSynced(p) { + return true + } + } + for _, root := range m.Extra { + for p := range root.Files { + if delta.NeverSynced(p) { + return true + } + } + } + return false +} + +func anyNeverSynced(paths []string) bool { + for _, p := range paths { + if delta.NeverSynced(p) { + return true + } + } + return false +} + +func anyNeverSyncedIn(paths map[string]struct{}) bool { + for p := range paths { + if delta.NeverSynced(p) { + return true + } + } + return false +} + // filterManifest returns a copy with every excluded path removed. The // original is untouched: it is still what gets served to peers. func filterManifest(m delta.Manifest, rules ignore.Rules) delta.Manifest { - if rules.Empty() { + if rules.Empty() && !holdsNeverSynced(m) { return m } out := delta.Manifest{ @@ -61,7 +104,7 @@ func filterManifest(m delta.Manifest, rules ignore.Rules) delta.Manifest { // would be told their saves diverged, over a file neither of them is // syncing. for p, entry := range m.Files { - if rules.Match(p) { + if excluded(rules, p) { continue } out.Files[p] = entry @@ -70,7 +113,7 @@ func filterManifest(m delta.Manifest, rules ignore.Rules) delta.Manifest { } } for _, d := range m.Dirs { - if rules.Match(d) { + if excluded(rules, d) { continue } out.Dirs = append(out.Dirs, d) @@ -83,13 +126,13 @@ func filterManifest(m delta.Manifest, rules ignore.Rules) delta.Manifest { for name, root := range m.Extra { sub := delta.RootManifest{Files: make(map[string]delta.FileEntry, len(root.Files))} for p, entry := range root.Files { - if rules.Match(p) { + if excluded(rules, p) { continue } sub.Files[p] = entry } for _, d := range root.Dirs { - if rules.Match(d) { + if excluded(rules, d) { continue } sub.Dirs = append(sub.Dirs, d) @@ -107,12 +150,12 @@ func filterManifest(m delta.Manifest, rules ignore.Rules) delta.Manifest { // decision reads "we both had this and now I don't" and propagates a deletion // to the peer. func filterLineage(paths map[string]struct{}, rules ignore.Rules) map[string]struct{} { - if rules.Empty() { + if rules.Empty() && !anyNeverSyncedIn(paths) { return paths } out := make(map[string]struct{}, len(paths)) for p := range paths { - if rules.Match(p) { + if excluded(rules, p) { continue } out[p] = struct{}{} @@ -123,12 +166,12 @@ func filterLineage(paths map[string]struct{}, rules ignore.Rules) map[string]str // filterPathList drops excluded paths from a plain list, for the lineage that // is persisted back after a sync. func filterPathList(paths []string, rules ignore.Rules) []string { - if rules.Empty() { + if rules.Empty() && !anyNeverSynced(paths) { return paths } out := make([]string, 0, len(paths)) for _, p := range paths { - if rules.Match(p) { + if excluded(rules, p) { continue } out = append(out, p) diff --git a/internal/p2p/syncengine/keeplocal_safety_test.go b/internal/p2p/syncengine/keeplocal_safety_test.go new file mode 100644 index 00000000..064298f5 --- /dev/null +++ b/internal/p2p/syncengine/keeplocal_safety_test.go @@ -0,0 +1,34 @@ +package syncengine + +import ( + "context" + "fmt" + "testing" +) + +// "Keep mine" against a peer holding part of the save, where the peer never +// gets to pull (a pull that times out, a device that goes away). The next +// sync must not read the peer's missing files as deletions and remove them +// from the device whose version the person chose to keep. +func TestKeepLocal_PeerThatNeverPulled_DeletesNothingHere(t *testing.T) { + env := setupEngine(t) + for i := 0; i < 300; i++ { + write(t, env.localDir, fmt.Sprintf("world/map_%04d.bin", i), fmt.Sprintf("chunk %d", i)) + } + for i := 0; i < 100; i++ { + write(t, env.remoteDir, fmt.Sprintf("world/map_%04d.bin", i), fmt.Sprintf("chunk %d", i)) + } + + // The person answers the conflict with "Keep mine". + if err := env.engine.markResolvedLocal(context.Background(), "game1", env.peer); err != nil { + t.Fatal(err) + } + // The peer's pull never happens; its folder stays as it was. + + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatalf("SyncWithPeer: %v", err) + } + if n := countFiles(t, env.localDir); n != 300 { + t.Fatalf("after \"keep mine\" this device went from 300 to %d files — the peer's partial copy was read as deletions", n) + } +} diff --git a/internal/p2p/syncengine/linkspeed.go b/internal/p2p/syncengine/linkspeed.go new file mode 100644 index 00000000..21af61f7 --- /dev/null +++ b/internal/p2p/syncengine/linkspeed.go @@ -0,0 +1,297 @@ +package syncengine + +import ( + "context" + "fmt" + "net" + "sort" + "time" + + "github.com/opensave/opensave/internal/store" +) + +// Network-aware syncing. +// +// Devices are not equally close. Two on the same home network move a save at +// tens of MB a second; one reached over a VPN relay or the internet relay may +// manage a hundred KB. A save that goes to every device at once spends hours +// crawling over the slow links while the device next door, which could have +// had it in seconds and then handed it on, waits in the same queue. +// +// So each device keeps how fast it has found its connection to every other — +// measured from the transfers it makes, and from a short speed test when it +// has nothing recent — and works fastest first: a sync goes to the close +// devices first, and a device that is behind takes the newer save from the +// fastest device that holds it. Before anything is measured, the kind of +// address stands in: a home-network address is taken for fast, a VPN address +// (Tailscale's range) for slower, the internet relay for slowest. + +// LinkStat is what this device knows about its connection to a peer. +type LinkStat struct { + // BytesPerSec is the measured rate, 0 when nothing is measured yet. + BytesPerSec float64 `json:"bytesPerSec"` + MeasuredMs int64 `json:"measuredMs"` + // Kind is what the address says: "lan", "vpn", "relay" or "internet". + Kind string `json:"kind"` +} + +const ( + // minMeasureBytes and minMeasureTime: smaller transfers say more about + // round trips than about the link, and are not counted. + minMeasureBytes = 512 << 10 + minMeasureTime = 500 * time.Millisecond + // probeEvery is how often a link is tested again when no transfer has + // measured it since; probeBytes how much a test moves. + probeEvery = 6 * time.Hour + probeBytes = 2 << 20 + // probeRetry is how soon a speed test that failed is tried again. + probeRetry = 10 * time.Minute +) + +// prior rates, by address kind, used until a link is measured. +var priorRate = map[string]float64{ + "lan": 40 << 20, + "vpn": 2 << 20, + "internet": 1 << 20, + "relay": 256 << 10, +} + +var ( + _, tailscaleNet, _ = net.ParseCIDR("100.64.0.0/10") + privateNets = func() []*net.IPNet { + var out []*net.IPNet + for _, c := range []string{"10.0.0.0/8", "172.16.0.0/12", "192.168.0.0/16", "169.254.0.0/16", "fc00::/7", "fe80::/10"} { + _, n, _ := net.ParseCIDR(c) + out = append(out, n) + } + return out + }() +) + +// linkKind says what a peer's address is. +func linkKind(peer Peer) string { + if peer.Wan() { + return "relay" + } + ip := net.ParseIP(peer.Address) + if ip == nil { + return "internet" + } + if ip.IsLoopback() { + return "lan" + } + if tailscaleNet.Contains(ip) { + return "vpn" + } + for _, n := range privateNets { + if n.Contains(ip) { + return "lan" + } + } + return "internet" +} + +func (e *Engine) loadLinksLocked() { + if e.links != nil { + return + } + e.links = map[string]store.PeerLink{} + if e.Store == nil { + return + } + if links, err := e.Store.PeerLinks(); err == nil { + e.links = links + } +} + +// Link is what this device knows about its connection to a peer. +func (e *Engine) Link(peer Peer) LinkStat { + e.linkMu.Lock() + defer e.linkMu.Unlock() + e.loadLinksLocked() + l := e.links[peer.ID] + return LinkStat{BytesPerSec: l.BytesPerSec, MeasuredMs: l.MeasuredMs, Kind: linkKind(peer)} +} + +// expectedRate is how fast a transfer with a peer is expected to go: what was +// measured, or the address kind's guess. +func (e *Engine) expectedRate(peer Peer) float64 { + l := e.Link(peer) + if l.BytesPerSec > 0 { + return l.BytesPerSec + } + return priorRate[l.Kind] +} + +// noteLinkRate records a transfer with a peer: bytes in d. +func (e *Engine) noteLinkRate(peer Peer, bytes int64, d time.Duration) { + e.recordLinkRate(peer, bytes, d, minMeasureTime) +} + +// recordLinkRate records bytes in d, when d is at least minTime. +func (e *Engine) recordLinkRate(peer Peer, bytes int64, d, minTime time.Duration) { + if bytes < minMeasureBytes || d < minTime || d <= 0 { + return + } + rate := float64(bytes) / d.Seconds() + e.linkMu.Lock() + defer e.linkMu.Unlock() + e.loadLinksLocked() + l := e.links[peer.ID] + l.PeerID = peer.ID + if l.BytesPerSec > 0 { + // A running average, weighted to the latest: a link changes (a VPN + // that found a direct path, a laptop that moved networks), and the + // order should follow it within a transfer or two. + l.BytesPerSec = 0.4*l.BytesPerSec + 0.6*rate + } else { + l.BytesPerSec = rate + } + l.MeasuredMs = time.Now().UnixMilli() + e.links[peer.ID] = l + if e.Store != nil { + _ = e.Store.SavePeerLink(l) + } +} + +// OrderBySpeed sorts peers fastest first, keeping the given order among +// equals. +func (e *Engine) OrderBySpeed(peers []Peer) []Peer { + out := append([]Peer(nil), peers...) + rates := make(map[string]float64, len(out)) + for _, p := range out { + rates[p.ID] = e.expectedRate(p) + } + sort.SliceStable(out, func(i, j int) bool { return rates[out[i].ID] > rates[out[j].ID] }) + return out +} + +// SpeedProber is implemented by transports that can test a link: fetch n +// bytes of nothing in particular from the peer and say how long it took. +type SpeedProber interface { + ProbeSpeed(ctx context.Context, peer Peer, n int) (int64, time.Duration, error) +} + +// ProbeLinkIfDue tests the link to a peer when nothing has measured it for a +// while. Not over the internet relay, which is the slowest link whatever a +// test says and would only be loaded further. Reports whether it ran. +func (e *Engine) ProbeLinkIfDue(ctx context.Context, peer Peer) bool { + prober, ok := e.Transport.(SpeedProber) + if !ok || peer.Wan() { + return false + } + e.linkMu.Lock() + e.loadLinksLocked() + l := e.links[peer.ID] + now := time.Now() + // A link never measured is tried again soon after a test that failed; + // one measured before only when that has gone stale. + retryAfter := probeEvery + if l.BytesPerSec == 0 { + retryAfter = probeRetry + } + due := now.Sub(time.UnixMilli(l.MeasuredMs)) > probeEvery && now.Sub(time.UnixMilli(l.ProbedMs)) > retryAfter + if due { + l.PeerID = peer.ID + l.ProbedMs = now.UnixMilli() + e.links[peer.ID] = l + if e.Store != nil { + _ = e.Store.SavePeerLink(l) + } + } + e.linkMu.Unlock() + if !due { + return false + } + ctx, cancel := context.WithTimeout(ctx, 60*time.Second) + defer cancel() + n, d, err := prober.ProbeSpeed(ctx, peer, probeBytes) + if err == nil && d < minMeasureTime { + // Over a home network the test is over before it measures much: + // once more, larger, for a figure worth having. + n, d, err = prober.ProbeSpeed(ctx, peer, 4*probeBytes) + } + if err != nil { + // A link never measured is tried again after probeRetry (see due + // above): the peer may simply not have been ready — or not yet on a + // build that answers a speed test, which once left every link + // unmeasured for hours. + e.Log("info", fmt.Sprintf("speed test with %s did not finish: %v", peer.Name, err)) + return true + } + // A test is a measurement however short: it moved nothing but the test, + // so no round trips of a sync's own are in it. + e.recordLinkRate(peer, n, d, time.Millisecond) + e.Log("info", fmt.Sprintf("connection to %s: %s/s (%s)", peer.Name, humanBytes(int64(e.Link(peer).BytesPerSec)), linkKind(peer))) + return true +} + +// Waiting for close devices before slow ones. + +// fastLink is the rate from which a device counts as close, and slowerBy how +// much slower the next device must be for this one to be waited for first. +const ( + fastLink = 4 << 20 + slowerBy = 4.0 + maxWaitFor = 20 * time.Minute +) + +// NotePeerPulled is told that a peer finished (or gave up) pulling a game from +// this device, and lets a sync waiting for it go on. +func (e *Engine) NotePeerPulled(gameID, peerID string) { + e.linkMu.Lock() + ch := e.pullWaits[gameID+"|"+peerID] + delete(e.pullWaits, gameID+"|"+peerID) + e.linkMu.Unlock() + if ch != nil { + close(ch) + } +} + +// waitForPull waits, after asking a close peer to take this device's save, +// until it has — before the slow devices are asked, so the save reaches the +// close one at full speed and the slow ones can then take it from whichever +// device is closest to them. Bounded by how long the transfer should take, +// and given up when the context ends. +func (e *Engine) waitForPull(ctx context.Context, gameID string, peer Peer, bytes int64) { + ch := make(chan struct{}) + e.linkMu.Lock() + if e.pullWaits == nil { + e.pullWaits = map[string]chan struct{}{} + } + e.pullWaits[gameID+"|"+peer.ID] = ch + e.linkMu.Unlock() + defer func() { + e.linkMu.Lock() + if e.pullWaits[gameID+"|"+peer.ID] == ch { + delete(e.pullWaits, gameID+"|"+peer.ID) + } + e.linkMu.Unlock() + }() + wait := 30*time.Second + time.Duration(3*float64(bytes)/e.expectedRate(peer)*float64(time.Second)) + if wait > maxWaitFor { + wait = maxWaitFor + } + select { + case <-ch: + case <-time.After(wait): + e.Log("info", fmt.Sprintf("%s has not finished taking %s yet; going on with the other devices", peer.Name, gameID)) + case <-ctx.Done(): + } +} + +// waitsForClose reports whether, having asked peers[i] to take this device's +// save, the sync should wait for it before going on: it is a close device, +// and a much slower one is still to come. +func (e *Engine) waitsForClose(peers []Peer, i int) bool { + r := e.expectedRate(peers[i]) + if r < fastLink { + return false + } + for _, p := range peers[i+1:] { + if e.expectedRate(p)*slowerBy <= r { + return true + } + } + return false +} diff --git a/internal/p2p/syncengine/linkspeed_test.go b/internal/p2p/syncengine/linkspeed_test.go new file mode 100644 index 00000000..968dcc87 --- /dev/null +++ b/internal/p2p/syncengine/linkspeed_test.go @@ -0,0 +1,175 @@ +package syncengine + +import ( + "context" + "errors" + "testing" + "time" +) + +func TestLinkKind(t *testing.T) { + cases := map[string]string{ + "192.168.178.20": "lan", + "10.0.0.5": "lan", + "127.0.0.1": "lan", + "100.99.121.82": "vpn", + "203.0.113.9": "internet", + } + for addr, want := range cases { + if got := linkKind(Peer{Address: addr}); got != want { + t.Errorf("%s: %s, want %s", addr, got, want) + } + } + if got := linkKind(Peer{Address: "relay"}); got != "relay" { + t.Errorf("relay: %s", got) + } +} + +// Before anything is measured the kind of address orders the devices; once a +// link is measured, the measurement does. +func TestOrderBySpeed(t *testing.T) { + env := setupEngine(t) + e := env.engine + home := Peer{ID: "home", Address: "192.168.178.20"} + vpn := Peer{ID: "vpn", Address: "100.99.121.82"} + relay := Peer{ID: "relay", Address: "relay"} + + got := e.OrderBySpeed([]Peer{relay, vpn, home}) + if got[0].ID != "home" || got[1].ID != "vpn" || got[2].ID != "relay" { + t.Errorf("by address: %v", ids(got)) + } + + // A VPN link that turns out fast, and a home one that turns out slow. + e.noteLinkRate(vpn, 200<<20, 2*time.Second) + e.noteLinkRate(home, 1<<20, 10*time.Second) + got = e.OrderBySpeed([]Peer{home, vpn, relay}) + if got[0].ID != "vpn" { + t.Errorf("by measurement: %v", ids(got)) + } + // Remembered across restarts. + e2 := New(env.store, nil, env.transport) + if e2.Link(vpn).BytesPerSec == 0 { + t.Error("the measured link was not kept") + } +} + +func TestNoteLinkRate_IgnoresTinyTransfers(t *testing.T) { + env := setupEngine(t) + p := Peer{ID: "p", Address: "192.168.1.2"} + env.engine.noteLinkRate(p, 10<<10, 2*time.Second) + if env.engine.Link(p).BytesPerSec != 0 { + t.Error("a transfer of a few KB was taken as the link's speed") + } +} + +// A close device is waited for before a much slower one, and the wait ends as +// soon as it reports that it has finished. +func TestWaitForClose(t *testing.T) { + env := setupEngine(t) + e := env.engine + home := Peer{ID: "home", Address: "192.168.178.20"} + vpn := Peer{ID: "vpn", Address: "100.99.121.82"} + peers := e.OrderBySpeed([]Peer{vpn, home}) + if !e.waitsForClose(peers, 0) { + t.Fatal("a close device is not waited for before a slow one") + } + if e.waitsForClose(peers, 1) { + t.Error("the last device is waited for, with nothing after it") + } + if e.waitsForClose(e.OrderBySpeed([]Peer{home}), 0) { + t.Error("waited with no slower device to follow") + } + + done := make(chan struct{}) + start := time.Now() + go func() { + e.waitForPull(context.Background(), "game1", home, 1<<30) + close(done) + }() + time.Sleep(100 * time.Millisecond) + e.NotePeerPulled("game1", home.ID) + select { + case <-done: + if time.Since(start) > 5*time.Second { + t.Error("the wait did not end when the device reported") + } + case <-time.After(10 * time.Second): + t.Fatal("the wait did not end when the device reported") + } +} + +type probeTransport struct { + *fakeTransport + fail bool + calls int + // perMB is how long each MB takes; a second when unset. + perMB time.Duration +} + +func (p *probeTransport) ProbeSpeed(ctx context.Context, peer Peer, n int) (int64, time.Duration, error) { + p.calls++ + if p.fail { + return 0, 0, errors.New("peer returned 404") + } + if p.perMB > 0 { + return int64(n), time.Duration(n>>20) * p.perMB, nil + } + return int64(n), time.Second, nil +} + +// A speed test that fails — a peer not yet updated to answer one — is tried +// again after minutes, not hours; a measured link is not tested again soon. +func TestProbeLink_RetriesAFailureSoon(t *testing.T) { + env := setupEngine(t) + pt := &probeTransport{fakeTransport: env.transport, fail: true} + env.engine.Transport = pt + p := Peer{ID: "p", Name: "P", Address: "192.168.1.9"} + + if !env.engine.ProbeLinkIfDue(context.Background(), p) { + t.Fatal("an unmeasured link was not tested") + } + if env.engine.ProbeLinkIfDue(context.Background(), p) { + t.Fatal("tested again straight after a failure") + } + // Eleven minutes later. + env.engine.linkMu.Lock() + l := env.engine.links[p.ID] + l.ProbedMs = time.Now().Add(-11 * time.Minute).UnixMilli() + env.engine.links[p.ID] = l + env.engine.linkMu.Unlock() + pt.fail = false + if !env.engine.ProbeLinkIfDue(context.Background(), p) { + t.Fatal("a failed test was not tried again after ten minutes") + } + if env.engine.Link(p).BytesPerSec == 0 { + t.Fatal("the successful test was not recorded") + } + if env.engine.ProbeLinkIfDue(context.Background(), p) { + t.Error("a freshly measured link was tested again") + } +} + +// A home network finishes the test before it has measured much: it is run +// again larger, and the result is kept — not discarded as too short, which +// left the fastest link unmeasured and sorted behind the slow ones. +func TestProbeLink_AFastLinkIsMeasured(t *testing.T) { + env := setupEngine(t) + pt := &probeTransport{fakeTransport: env.transport, perMB: 20 * time.Millisecond} + env.engine.Transport = pt + p := Peer{ID: "lan", Name: "LAN", Address: "192.168.178.20"} + env.engine.ProbeLinkIfDue(context.Background(), p) + if got := env.engine.Link(p).BytesPerSec; got < 20<<20 { + t.Errorf("a 50 MB/s link measured as %.0f B/s", got) + } + if pt.calls != 2 { + t.Errorf("%d test(s), want the short one and a larger one", pt.calls) + } +} + +func ids(ps []Peer) []string { + out := make([]string, len(ps)) + for i, p := range ps { + out[i] = p.ID + } + return out +} diff --git a/internal/p2p/syncengine/multiroot.go b/internal/p2p/syncengine/multiroot.go index c763a459..544e7e53 100644 --- a/internal/p2p/syncengine/multiroot.go +++ b/internal/p2p/syncengine/multiroot.go @@ -114,24 +114,20 @@ func (e *Engine) syncOneRoot(ctx context.Context, gameID string, game store.Game rules := e.rulesFor(gameID) remote := sr.remote unfilteredLocal, unfilteredRemote := local, remote - if !rules.Empty() { - local = filterManifest(local, rules) - remote = filterManifest(remote, rules) - } + local = filterManifest(local, rules) + remote = filterManifest(remote, rules) fileList, dirList, err := e.Store.GetSyncStateForRoot(gameID, peer.ID, sr.root.Name) if err != nil { return err } lineageFiles, lineageDirs := toSet(fileList), toSet(dirList) - if !rules.Empty() { - // Without this an exclusion added to a location that had already - // synced reads as "we both had this file and now I do not", and the - // deletion travels to the peer — destroying the copy the rule was - // written to protect. - lineageFiles = filterLineage(lineageFiles, rules) - lineageDirs = filterLineage(lineageDirs, rules) - } + // Without this an exclusion added to a location that had already + // synced reads as "we both had this file and now I do not", and the + // deletion travels to the peer — destroying the copy the rule was + // written to protect. + lineageFiles = filterLineage(lineageFiles, rules) + lineageDirs = filterLineage(lineageDirs, rules) decision := Compute(local, remote, lineageFiles, lineageDirs) if !decision.HasChanges() { @@ -172,7 +168,7 @@ func (e *Engine) syncOneRoot(ctx context.Context, gameID string, game store.Game // adds a rule. If a side's UNFILTERED state still hashes to the base then // that side has not changed, and its filtered hash says the same thing in // today's terms. Same translation the main folder does. - if !rules.Empty() && base != "" { + if base != "" { switch base { case unfilteredLocal.RootHash(delta.PrimaryRoot): base = local.RootHash(delta.PrimaryRoot) diff --git a/internal/p2p/syncengine/multiroot_conflict.go b/internal/p2p/syncengine/multiroot_conflict.go index 1fea1ba6..401d69de 100644 --- a/internal/p2p/syncengine/multiroot_conflict.go +++ b/internal/p2p/syncengine/multiroot_conflict.go @@ -140,7 +140,17 @@ func (e *Engine) ResolveRootConflict(ctx context.Context, gameID, peerID, root, if err := e.Store.SetAgreedHashForRoot(gameID, peerID, root, local.RootHash(delta.PrimaryRoot)); err != nil { return err } - files, dirs := IntersectLineage(local, local) + // Shared only what the peer verifiably holds too — see + // markResolvedLocal for what recording the whole local copy as shared + // did once the peer's pull did not finish. + var files, dirs []string + if remoteData, err := e.Transport.FetchManifest(ctx, peer, gameID, ManifestQuery{ + Name: game.Name, SavePath: game.SavePath, AppID: game.AppID, CoverURL: game.CoverURL, + }); err == nil { + if remoteRoot, has := remoteData.Manifest.Extra[root]; has { + files, dirs = IntersectLineage(local, delta.Manifest{Files: remoteRoot.Files, Dirs: remoteRoot.Dirs}) + } + } _ = e.Store.SetSyncStateForRoot(gameID, peerID, root, files, dirs) e.clearRootConflict(gameID, root) e.Transport.TriggerPeerPull(peer, gameID) diff --git a/internal/p2p/syncengine/neversynced_test.go b/internal/p2p/syncengine/neversynced_test.go new file mode 100644 index 00000000..33a6def7 --- /dev/null +++ b/internal/p2p/syncengine/neversynced_test.go @@ -0,0 +1,137 @@ +package syncengine + +import ( + "context" + "os" + "path/filepath" + "testing" + + "github.com/opensave/opensave/internal/delta" +) + +// Steam's remotecache.vdf is each device's own bookkeeping: never taken from +// a peer, never pushed, never deleted because the peer lacks it. +func TestNeverSynced_LeftAloneOnBothSides(t *testing.T) { + env := setupEngine(t) + write(t, env.localDir, "save.dat", "progress") + write(t, env.localDir, "remotecache.vdf", "this device's steam") + write(t, env.remoteDir, "save.dat", "progress") + write(t, env.remoteDir, "remotecache.vdf", "the other device's steam") + + res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err != nil { + t.Fatalf("SyncWithPeer error = %v", err) + } + if res.Status != "in_sync" { + t.Fatalf("status = %q, want in_sync: the saves are the same", res.Status) + } + got, _ := os.ReadFile(filepath.Join(env.localDir, "remotecache.vdf")) + if string(got) != "this device's steam" { + t.Errorf("this device's remotecache.vdf became %q", got) + } + if got, _ := os.ReadFile(filepath.Join(env.remoteDir, "remotecache.vdf")); string(got) != "the other device's steam" { + t.Errorf("the peer's remotecache.vdf became %q", got) + } +} + +// The served manifest keeps listing it: a device on an older build would read +// the gap as the file deleted, and delete its own. +func TestNeverSynced_StillServed(t *testing.T) { + dir := t.TempDir() + write(t, dir, "remotecache.vdf", "steam") + m, err := delta.BuildManifest(dir) + if err != nil { + t.Fatal(err) + } + if _, ok := m.Files["remotecache.vdf"]; !ok { + t.Fatal("the manifest a device serves left remotecache.vdf out") + } +} + +// A build with a new list of such files re-takes the recorded hash quietly +// when the save is the one it was recorded for, and the next change seen is +// not called a new version over it. A save that did change is left to count. +func TestNeverSynced_AdoptingTheViewIsNotAVersion(t *testing.T) { + env := setupEngine(t) + game, err := env.store.GetGame("game1") + if err != nil { + t.Fatal(err) + } + game.AutoSync = true + if err := env.store.UpdateGame(game); err != nil { + t.Fatal(err) + } + write(t, env.localDir, "save.dat", "progress") + write(t, env.localDir, "remotecache.vdf", "steam") + env.engine.NoteLocalChange("game1") // the first version: named by content + + m, err := delta.BuildManifest(env.localDir) + if err != nil { + t.Fatal(err) + } + // As a build before the list recorded it. + rec, err := env.store.GetGameVersion("game1") + if err != nil || rec.Vector == "" { + t.Fatalf("no version recorded: %v", err) + } + vector := rec.Vector + rec.Hash = versionHashBeforeNeverSynced("", m) + if err := env.store.SaveGameVersion(rec); err != nil { + t.Fatal(err) + } + + env.engine.AdoptNeverSyncedView("game1", m) + env.engine.NoteLocalChange("game1") + rec, _ = env.store.GetGameVersion("game1") + if rec.Vector != vector { + t.Fatalf("version moved from %s to %s over a save nobody changed", vector, rec.Vector) + } + if rec.Hash != env.engine.versionHashOf("game1", m) { + t.Error("the recorded hash was not re-taken") + } + + // Recorded for a save that has since changed: not adopted, so the change + // is a version. + write(t, env.localDir, "save.dat", "more progress") + delta.InvalidateRoot(env.localDir) + changed, _ := delta.BuildManifest(env.localDir) + rec.Hash = versionHashBeforeNeverSynced("", m) + _ = env.store.SaveGameVersion(rec) + env.engine.AdoptNeverSyncedView("game1", changed) + env.engine.NoteLocalChange("game1") + rec, _ = env.store.GetGameVersion("game1") + if rec.Vector == vector { + t.Error("a real change after the upgrade did not become a version") + } +} + +// Held only by the peer: not taken. Held only here: not pushed, and not +// deleted because the peer lacks it. +func TestNeverSynced_OneSidedIsNotADifference(t *testing.T) { + env := setupEngine(t) + write(t, env.localDir, "save.dat", "progress") + write(t, env.remoteDir, "save.dat", "progress") + write(t, env.remoteDir, "RemoteCache.VDF", "only there") + + res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err != nil || res.Status != "in_sync" { + t.Fatalf("status = %q, %v; want in_sync", res.Status, err) + } + if _, err := os.Stat(filepath.Join(env.localDir, "RemoteCache.VDF")); err == nil { + t.Error("the peer's remotecache.vdf was pulled") + } + + os.Remove(filepath.Join(env.remoteDir, "RemoteCache.VDF")) + write(t, env.localDir, "remotecache.vdf", "only here") + delta.InvalidateRoot(env.localDir) + res, err = env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err != nil || res.Status != "in_sync" { + t.Fatalf("status = %q, %v; want in_sync", res.Status, err) + } + if _, err := os.Stat(filepath.Join(env.remoteDir, "remotecache.vdf")); err == nil { + t.Error("this device's remotecache.vdf was pushed") + } + if _, err := os.Stat(filepath.Join(env.localDir, "remotecache.vdf")); err != nil { + t.Error("this device's remotecache.vdf was deleted") + } +} diff --git a/internal/p2p/syncengine/partial_safety_test.go b/internal/p2p/syncengine/partial_safety_test.go new file mode 100644 index 00000000..59588c39 --- /dev/null +++ b/internal/p2p/syncengine/partial_safety_test.go @@ -0,0 +1,113 @@ +package syncengine + +import ( + "context" + "fmt" + "os" + "path/filepath" + "sync/atomic" + "testing" +) + +// A transfer that stops part-way must never turn into deletions. +// +// Seen in 2.3.1: a device that had received part of a 240k-file save offered +// its partial copy back, the missing files were read as "deleted over there", +// and 23k files were deleted on the device that still had them. 2.4 records +// as shared only what both manifests showed (persistLineage), which should +// rule that out; these hold it there, for the per-file and the batched pull. + +func countFiles(t *testing.T, dir string) int { + t.Helper() + n := 0 + _ = filepath.Walk(dir, func(p string, info os.FileInfo, err error) error { + if err == nil && !info.IsDir() { + n++ + } + return nil + }) + return n +} + +func writeSmallFiles(t *testing.T, dir string, n int) { + t.Helper() + for i := 0; i < n; i++ { + write(t, dir, fmt.Sprintf("world/map_%04d.bin", i), fmt.Sprintf("chunk %d", i)) + } +} + +func interruptedThenResumed(t *testing.T, env *engineEnv, total, stopAfter int) { + t.Helper() + writeSmallFiles(t, env.remoteDir, total) + + ctx, cancel := context.WithCancel(context.Background()) + var fetched atomic.Int64 + env.transport.onFetchBlocks = func() { + if fetched.Add(1) == int64(stopAfter) { + cancel() + } + } + if _, err := env.engine.SyncWithPeer(ctx, "game1", env.peer); err == nil { + t.Fatal("setup: the first sync was meant to be interrupted") + } + env.transport.onFetchBlocks = nil + cancel() + + got := countFiles(t, env.localDir) + if got == 0 || got >= total { + t.Fatalf("setup: after the interruption this side has %d of %d files; want a partial copy", got, total) + } + + // Resume. Nothing may be deleted anywhere, and everything must arrive. + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatalf("resumed sync: %v", err) + } + if len(env.transport.deletedOnPeer) > 0 { + t.Fatalf("the partial copy was propagated as %d deletions on the peer, e.g. %s", + len(env.transport.deletedOnPeer), env.transport.deletedOnPeer[0]) + } + if n := countFiles(t, env.remoteDir); n != total { + t.Fatalf("peer has %d files, want all %d", n, total) + } + if n := countFiles(t, env.localDir); n != total { + t.Fatalf("this side has %d files after resuming, want %d", n, total) + } +} + +func TestPartialPull_InterruptedThenResumed(t *testing.T) { + env := setupEngine(t) + interruptedThenResumed(t, env, 200, 60) +} + +func TestPartialPull_InterruptedThenResumed_Batched(t *testing.T) { + // One batch in flight, so the first batch of 256 is written before the + // interruption lands in the second: a deterministic partial copy. + old := batchWorkers + batchWorkers = 1 + t.Cleanup(func() { batchWorkers = old }) + env, _ := setupBatchEngine(t, ProtoBatchFiles) + // Batches fetch through the same fake reader, one call per file; the + // 300th is inside the second batch. + interruptedThenResumed(t, env, 600, 300) +} + +// The other side of the same situation: this device has everything, the peer +// holds only part of it and never synced. The part it lacks is new to it, not +// deleted by it. +func TestPartialPeer_NeverReadAsDeletions(t *testing.T) { + env := setupEngine(t) + writeSmallFiles(t, env.localDir, 300) + for i := 0; i < 80; i++ { + write(t, env.remoteDir, fmt.Sprintf("world/map_%04d.bin", i), fmt.Sprintf("chunk %d", i)) + } + + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatalf("SyncWithPeer: %v", err) + } + if n := countFiles(t, env.localDir); n != 300 { + t.Fatalf("this side went from 300 to %d files — the peer's partial copy was read as deletions", n) + } + if len(env.transport.deletedOnPeer) > 0 { + t.Fatalf("deleted on peer: %v", env.transport.deletedOnPeer[:1]) + } +} diff --git a/internal/p2p/syncengine/quick.go b/internal/p2p/syncengine/quick.go new file mode 100644 index 00000000..cf0c3906 --- /dev/null +++ b/internal/p2p/syncengine/quick.go @@ -0,0 +1,83 @@ +package syncengine + +import ( + "context" + "fmt" + + "github.com/opensave/opensave/internal/store" +) + +// quickInSync settles the common case of a sync in one small exchange: this +// device's save still hashes to the base both devices last agreed on, and +// the peer, asked whether it does too (ManifestQuery.IfHash), says +// "unchanged". Then there is nothing to decide, and neither manifest needs +// to cross the wire — for a save of a quarter-million files that is tens of +// MB of JSON per peer, which the minute-long reconcile used to fetch every +// minute to learn that nothing had changed. +// +// It only ever answers "in sync" when both sides provably hold the agreed +// state. In every other case it reports not done and the full sync runs; a +// full manifest that came back anyway (a peer that predates the question) +// is handed over in prefetched so it is not fetched twice. +// +// Deliberately narrow: +// - no exclusion rules: the agreed base is taken over the filtered +// manifest, the peer hashes its unfiltered one; +// - no extra save locations: those carry bases of their own (RootHash); +// - no conflict waiting, and the same branch on both sides. +func (e *Engine) quickInSync(ctx context.Context, gameID string, game store.Game, peer Peer, q ManifestQuery) (res Result, done bool, prefetched *ManifestResponse) { + agreed := e.Store.GetAgreedHash(gameID, peer.ID) + if agreed == "" || !e.rulesFor(gameID).Empty() { + return Result{}, false, nil + } + if roots, err := e.Store.GameRootPaths(gameID); err != nil || len(roots) > 0 { + return Result{}, false, nil + } + e.mu.Lock() + conflicted := e.activeConflicts[gameID] != nil + e.mu.Unlock() + if conflicted { + return Result{}, false, nil + } + local, err := e.ReadManifest(ctx, gameID, game.SavePath) + if err != nil || len(local.Extra) > 0 || local.ManifestHash() != agreed { + return Result{}, false, nil + } + + q.IfHash = agreed + resp, err := e.Transport.FetchManifest(ctx, peer, gameID, q) + if err != nil { + // Let the full path fetch again and classify the error as it always has. + return Result{}, false, nil + } + if !resp.Unchanged { + return Result{}, false, &resp + } + if resp.ManifestHash != agreed || (resp.ActiveBranch != "" && resp.ActiveBranch != game.ActiveBranch) { + // Unchanged, but not in a way this shortcut may act on: the full + // exchange decides. + return Result{}, false, nil + } + // The same files; the versions still have to say the same (version.go). + if resp.Version != nil && game.AutoSync { + res, final := e.quickVersionCheck(game, peer, local, *resp.Version) + if !final { + return Result{}, false, nil + } + if res.Status == versionStatusWaiting { + return res, true, nil + } + } + + // Same as the in-sync outcome of a full sync, minus what is already + // recorded: the base is unchanged and the lineage with it. + e.Transport.ReportSyncEvent(peer, gameID, "in-sync", map[string]any{ + "peerName": e.deviceName(), + "manifestHash": agreed, + }) + if resp.LatestSnapshot != nil && len(local.Files) > 0 && !e.hasSnapshot(gameID, game) { + e.recordMirrorSnapshot(gameID, game, peer, *resp.LatestSnapshot, + fmt.Sprintf("Synced from peer: %s (%s)", peer.Name, resp.LatestSnapshot.Comment)) + } + return Result{Status: "in_sync", Direction: "none"}, true, nil +} diff --git a/internal/p2p/syncengine/quick_test.go b/internal/p2p/syncengine/quick_test.go new file mode 100644 index 00000000..3cfdf8d8 --- /dev/null +++ b/internal/p2p/syncengine/quick_test.go @@ -0,0 +1,117 @@ +package syncengine + +import ( + "context" + "testing" + + "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/snapshot" +) + +// ifHashTransport answers IfHash the way a current peer does, and counts +// how many full manifests it had to send. +type ifHashTransport struct { + *fakeTransport + full, unchanged int +} + +func (t *ifHashTransport) FetchManifest(ctx context.Context, peer Peer, gameID string, q ManifestQuery) (ManifestResponse, error) { + resp, err := t.fakeTransport.FetchManifest(ctx, peer, gameID, q) + if err != nil { + return resp, err + } + if q.IfHash != "" && resp.Manifest.ManifestHash() == q.IfHash { + t.unchanged++ + return ManifestResponse{ActiveBranch: resp.ActiveBranch, LatestSnapshot: resp.LatestSnapshot, + Unchanged: true, ManifestHash: q.IfHash, Manifest: delta.Manifest{}}, nil + } + t.full++ + return resp, nil +} + +func setupIfHashEngine(t *testing.T) (*engineEnv, *ifHashTransport) { + t.Helper() + env := setupEngine(t) + tr := &ifHashTransport{fakeTransport: env.transport} + env.engine = New(env.store, snapshot.New(env.store), tr) + return env, tr +} + +func TestQuickInSync_SkipsTheManifestWhenNothingChanged(t *testing.T) { + env, tr := setupIfHashEngine(t) + write(t, env.localDir, "save.dat", "same") + write(t, env.remoteDir, "save.dat", "same") + + // First sync converges and records the agreed base. + if res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil || res.Status != "in_sync" { + t.Fatalf("first sync = %+v, %v", res, err) + } + if tr.full != 1 { + t.Fatalf("setup: full manifests = %d, want 1", tr.full) + } + + // Nothing changed anywhere: settled without a full manifest. + res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err != nil || res.Status != "in_sync" { + t.Fatalf("second sync = %+v, %v", res, err) + } + if tr.full != 1 || tr.unchanged != 1 { + t.Errorf("second sync fetched %d full manifests and %d unchanged answers; want 0 more full, 1 unchanged", tr.full-1, tr.unchanged) + } +} + +// A change on either side must go the full way — the shortcut may only ever +// say "in sync" when both sides provably hold the agreed state. +func TestQuickInSync_ChangesStillSync(t *testing.T) { + env, tr := setupIfHashEngine(t) + write(t, env.localDir, "save.dat", "same") + write(t, env.remoteDir, "save.dat", "same") + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatal(err) + } + + // The peer changes. + write(t, env.remoteDir, "new.dat", "made on the other device") + res, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err != nil { + t.Fatal(err) + } + if res.Status != "updated" || res.Direction != "pull" { + t.Fatalf("peer change: result = %+v, want updated/pull", res) + } + + // This side changes. + write(t, env.localDir, "mine.dat", "made here") + before := env.transport.pullTriggers + res, err = env.engine.SyncWithPeer(context.Background(), "game1", env.peer) + if err != nil { + t.Fatal(err) + } + if res.Status == "in_sync" { + t.Fatalf("local change was reported in sync: %+v", res) + } + if env.transport.pullTriggers == before { + t.Error("the peer was not told to pull this side's change") + } + if tr.unchanged != 0 { + t.Errorf("unchanged answers = %d with a change on one side, want 0", tr.unchanged) + } +} + +// A peer that predates the question sends the full manifest; it is used, +// not fetched a second time. +func TestQuickInSync_OldPeerManifestIsNotFetchedTwice(t *testing.T) { + env := setupEngine(t) // plain fake: ignores IfHash + write(t, env.localDir, "save.dat", "same") + write(t, env.remoteDir, "save.dat", "same") + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatal(err) + } + before := env.transport.manifestCalls + if _, err := env.engine.SyncWithPeer(context.Background(), "game1", env.peer); err != nil { + t.Fatal(err) + } + if got := env.transport.manifestCalls - before; got != 1 { + t.Errorf("manifest requests for one sync = %d, want 1", got) + } +} diff --git a/internal/p2p/syncengine/settle.go b/internal/p2p/syncengine/settle.go index 34942e10..a66c1c8d 100644 --- a/internal/p2p/syncengine/settle.go +++ b/internal/p2p/syncengine/settle.go @@ -131,10 +131,44 @@ func (e *Engine) Writing(gameID string) (done func()) { once = true g := e.gateLocked(gameID) g.writers-- + if e.writtenAt == nil { + e.writtenAt = map[string]time.Time{} + } + e.writtenAt[gameID] = time.Now() e.wakeLocked(gameID, g) } } +// writeEcho is how long after a sync stops writing a game its file events may +// still be arriving. +const writeEcho = 10 * time.Second + +// BeingWritten reports whether a sync on this device is writing a game's save +// now, or was a moment ago — what the watcher asks before taking a burst it +// could not attribute (an overflow of file events) for a change of the game's. +// A pull of a save of many files overflows the event queue again and again, +// and each one read as the game saving: an automatic snapshot of a half-pulled +// save every few seconds, which pushed the real ones out of retention. +func (e *Engine) BeingWritten(gameID string) bool { + if e.WritingNow(gameID) { + return true + } + e.settleMu.Lock() + defer e.settleMu.Unlock() + return time.Since(e.writtenAt[gameID]) < writeEcho +} + +// WritingNow reports whether a sync on this device is writing a game's save at +// this moment — what the watcher waits out before looking at a burst, so it +// never judges a save half-way between two states. Not the moments after: +// a change the game makes then is the game's, and its snapshot is not held up. +func (e *Engine) WritingNow(gameID string) bool { + e.settleMu.Lock() + defer e.settleMu.Unlock() + g := e.gates[gameID] + return g != nil && (g.writers > 0 || g.waitingWriters > 0) +} + // Reading holds the game's save still for a sync to read it: it waits until // nothing is writing it, then keeps writers out until done is called. It // returns ErrSettling if ctx ends first. Hold it only for the reading itself — diff --git a/internal/p2p/syncengine/transport.go b/internal/p2p/syncengine/transport.go index c39e315c..c2afbf8d 100644 --- a/internal/p2p/syncengine/transport.go +++ b/internal/p2p/syncengine/transport.go @@ -2,6 +2,7 @@ package syncengine import ( "context" + "errors" "github.com/opensave/opensave/internal/delta" ) @@ -25,12 +26,54 @@ type SnapshotInfo struct { ID string `json:"id"` Timestamp string `json:"timestamp"` Comment string `json:"comment"` + // ContentHash names what the snapshot holds (snapshot.ContentKey): a + // device that already has a snapshot of exactly those files records this + // one's id as an alias of it rather than archiving the save again. + ContentHash string `json:"contentHash,omitempty"` } // ProtoMultiRoot is the protocol revision at which a peer understands games // with more than one save location. const ProtoMultiRoot = 1 +// ProtoBatchFiles is the revision at which a peer serves the blocks of many +// files in one request (FetchFileBatch). Before it, every file cost its own +// round trip — ~200ms each in practice, which for a save of a quarter-million +// small files (Project Zomboid map chunks) meant half a day for 50MB. +const ProtoBatchFiles = 2 + +// ProtoVersions is the revision at which a peer keeps save versions +// (version.go). A peer below it decides by the old file comparison, which +// reads whatever its copy lacks as deleted; nothing is taken from such a peer +// or deleted on its word (OldBuildRefused). +const ProtoVersions = 3 + +// FileBlocksRequest asks for some blocks of one file, as one entry of a batch. +type FileBlocksRequest struct { + RelPath string `json:"relPath"` + BlockIndices []int `json:"blockIndices"` + BlockSize int `json:"blockSize"` +} + +// FileBlocks is one file's answer inside a batch. Error is set, and Blocks +// empty, when the responder could not read that file. +type FileBlocks struct { + RelPath string `json:"relPath"` + Blocks []BlockData `json:"blocks"` + Error string `json:"error,omitempty"` +} + +// BatchFetcher is implemented by transports that can fetch the blocks of +// many files, all in one save location, in a single request. Optional: the +// engine checks for it, and for a peer that advertised ProtoBatchFiles. +type BatchFetcher interface { + FetchFileBatch(ctx context.Context, peer Peer, gameID, root string, files []FileBlocksRequest) ([]FileBlocks, error) +} + +// ErrBatchUnsupported is what a transport returns when it cannot batch for +// this peer (the relay path); the engine then pulls file by file. +var ErrBatchUnsupported = errors.New("batched file fetch is not supported for this peer") + // ManifestResponse is what a peer returns for a manifest request. type ManifestResponse struct { Manifest delta.Manifest `json:"manifest"` @@ -56,6 +99,26 @@ type ManifestResponse struct { // did (see hold.go). Without it an empty location is not taken as every // file in it deleted — see emptiedUnconfirmed. DeletionConfirmed bool `json:"deletionConfirmed,omitempty"` + + // Unchanged answers a request that carried ManifestQuery.IfHash: the + // responder's save still hashes to exactly that, so Manifest is left + // empty. Only ever set when asked; a peer that predates it ignores the + // question and sends the full manifest, which is handled as before. + Unchanged bool `json:"unchanged,omitempty"` + ManifestHash string `json:"manifestHash,omitempty"` + + // Version is the responder's version of the save (version.go), absent + // from builds that predate it and for games it does not version. + Version *VersionInfo `json:"version,omitempty"` + // Versions says the responder keeps save versions at all. The relay + // path sends it instead of a Proto, which there would also claim things + // that path does not serve. + Versions bool `json:"versions,omitempty"` +} + +// KeepsVersions reports whether the responder keeps save versions. +func (r ManifestResponse) KeepsVersions() bool { + return r.Versions || r.Proto >= ProtoVersions } // FileRef identifies one file inside one of a game's save locations. @@ -81,6 +144,9 @@ type ManifestQuery struct { // cover art, instead of a blank tile. AppID string CoverURL string + // IfHash asks the peer to answer "unchanged" instead of its manifest when + // its save still hashes to this — the base both devices last agreed on. + IfHash string } // BlockData is one fetched block. diff --git a/internal/p2p/syncengine/version.go b/internal/p2p/syncengine/version.go new file mode 100644 index 00000000..1183125a --- /dev/null +++ b/internal/p2p/syncengine/version.go @@ -0,0 +1,1257 @@ +package syncengine + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "io/fs" + "sort" + "strings" + "time" + + "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/ignore" + "github.com/opensave/opensave/internal/owntouch" + "github.com/opensave/opensave/internal/store" +) + +// Save versions. +// +// Every device knows which version of a game's save it holds, as a version +// vector: one counter per device that has ever changed the save, moved only +// when the save really changes there (the watcher saw the game or the user +// write it — never a sync). Comparing two vectors answers, exactly, the +// question the file-by-file comparison could only guess at from what each side +// happens to hold: +// +// - one is older than the other: that device is simply behind. It takes the +// newer save as a whole — every file, and every deletion — and only from a +// device that holds it completely. Nothing is ever taken from it, pushed +// by it, or deleted on its say-so, and it is a source for nobody. +// - equal: the same save. +// - neither: both changed it independently. Only then is anyone asked. +// +// A device also remembers the newest version it has heard of (the target) +// until it holds it. That is what keeps two outdated devices from syncing with +// each other: each knows something newer exists, so each waits for a device +// that holds it, instead of reading the other's half-finished copy as the save +// and passing on what it lacks as deletions. +// +// An empty folder holds no version at all, which is older than every version, +// so a new or emptied device fills itself from a complete one and never raises +// a conflict. A save that existed before this was introduced gets a version +// named after its content (genesisKey), so devices that already held the same +// save agree at once, and ones that did not are compared the way they always +// were, until they agree. + +// VersionVector is writer key → counter. +type VersionVector map[string]int64 + +type versionOrder int + +const ( + versionEqual versionOrder = iota + versionOlder + versionNewer + versionConcurrent +) + +// genesisAll is the entry "Use this save everywhere" adds (UseSaveEverywhere): +// a version holding it counts as having every save from before versions +// existed behind it, whatever its content was named. +const genesisAll = "g:*" + +// holds reports whether v includes entry k at counter n. +func holds(v VersionVector, k string, n int64) bool { + if v[k] >= n { + return true + } + return k != genesisAll && strings.HasPrefix(k, "g:") && v[genesisAll] >= 1 +} + +// compareVersions says how a relates to b. +func compareVersions(a, b VersionVector) versionOrder { + aBehind, bBehind := false, false + for k, bv := range b { + if !holds(a, k, bv) { + aBehind = true + break + } + } + for k, av := range a { + if !holds(b, k, av) { + bBehind = true + break + } + } + switch { + case !aBehind && !bBehind: + return versionEqual + case aBehind && !bBehind: + return versionOlder + case !aBehind && bBehind: + return versionNewer + default: + return versionConcurrent + } +} + +// covers reports whether a is b or newer. +func covers(a, b VersionVector) bool { + o := compareVersions(a, b) + return o == versionEqual || o == versionNewer +} + +func mergeVersions(a, b VersionVector) VersionVector { + out := make(VersionVector, len(a)+len(b)) + for k, v := range a { + out[k] = v + } + for k, v := range b { + if v > out[k] { + out[k] = v + } + } + return out +} + +func (v VersionVector) clone() VersionVector { return mergeVersions(v, nil) } + +// genesisOnly reports whether every entry names a save that predates version +// tracking (genesisKey), i.e. nobody has changed it since. +func (v VersionVector) genesisOnly() bool { + if len(v) == 0 { + return false + } + for k := range v { + if !strings.HasPrefix(k, "g:") { + return false + } + } + return true +} + +// String is a short, stable form for the log. +func (v VersionVector) String() string { + if len(v) == 0 { + return "none" + } + keys := make([]string, 0, len(v)) + for k := range v { + keys = append(keys, k) + } + sort.Strings(keys) + parts := make([]string, len(keys)) + for i, k := range keys { + short := k + if len(short) > 8 { + short = short[:8] + } + parts[i] = fmt.Sprintf("%s=%d", short, v[k]) + } + return strings.Join(parts, ",") +} + +// genesisKey names the version of a save that existed before versions did, +// after its content: devices that hold the same files get the same name. +// Directories are left out — they are created by any sync, and are not what +// makes two saves the same. +func genesisKey(m delta.Manifest) string { + return "g:" + filesKey(m, ignore.Rules{})[:16] +} + +// filesKey hashes which files a save holds and what is in them: paths and +// content hashes, nothing else — not folders, not times. Excluded paths are +// left out. +func filesKey(m delta.Manifest, rules ignore.Rules) string { + return filesKeyOf(m, func(p string) bool { return excluded(rules, p) }) +} + +func filesKeyOf(m delta.Manifest, leaveOut func(string) bool) string { + paths := make([]string, 0, len(m.Files)) + for p := range m.Files { + if !leaveOut(p) { + paths = append(paths, p) + } + } + sort.Strings(paths) + h := sha256.New() + for _, p := range paths { + h.Write([]byte(p)) + h.Write([]byte{0}) + h.Write([]byte(m.Files[p].Hash)) + h.Write([]byte{0}) + } + return hex.EncodeToString(h.Sum(nil)) +} + +// versionHashOf is what is recorded with a version: the save's files as this +// device syncs them, tagged with the exclusion rules it was taken under. A +// save that no longer hashes to it was changed here since, whether or not the +// watcher said so. +// +// Filtered, because a file nobody syncs — a device's own config — changing +// is not a new version of the save. Tagged, because writing a rule is not one +// either: it changes the filtered hash on every device at once, and each took +// that as a change of its own, so the next sync asked about a divergence +// nobody made. A hash taken under other rules is re-taken, not compared +// (sameSaveAs). The files no save is made of (delta.NeverSyncedList) are in +// the tag for the same reason. +func (e *Engine) versionHashOf(gameID string, primary delta.Manifest) string { + text := "" + if game, err := e.Store.GetGame(gameID); err == nil { + text = game.SyncIgnore + } + tag := sha256.Sum256([]byte(text + "\x00" + delta.NeverSyncedList)) + return hex.EncodeToString(tag[:4]) + ":" + filesKey(primary, ignore.Parse(text)) +} + +// versionHashBeforeNeverSynced is versionHashOf as builds before +// delta.NeverSynced took it: the rules alone in the tag, those files counted. +func versionHashBeforeNeverSynced(rulesText string, primary delta.Manifest) string { + tag := sha256.Sum256([]byte(rulesText)) + rules := ignore.Parse(rulesText) + return hex.EncodeToString(tag[:4]) + ":" + filesKeyOf(primary, func(p string) bool { + return !rules.Empty() && rules.Match(p) + }) +} + +// AdoptNeverSyncedView re-takes the hash recorded with this device's version +// of a game in the terms of delta.NeverSynced, when the save is the one it was +// recorded for — the files no save is made of aside. Run once, when a build +// with a new list starts, before anything compares: otherwise the only +// difference, which files are counted, reads as the save changed here, and +// every device names the same save a new version of its own at once. +func (e *Engine) AdoptNeverSyncedView(gameID string, primary delta.Manifest) { + game, err := e.Store.GetGame(gameID) + if err != nil { + return + } + e.versionMu.Lock() + defer e.versionMu.Unlock() + gv, err := e.loadVersionLocked(gameID) + if err != nil || gv.rec.GameID == "" || gv.rec.Hash == "" { + return + } + if gv.rec.Hash == versionHashBeforeNeverSynced(game.SyncIgnore, primary) { + gv.rec.Hash = e.versionHashOf(gameID, primary) + _ = e.saveVersionLocked(gv) + } +} + +// sameSaveAs reports whether a hash recorded with a version still describes +// the save that now hashes to current: the same, or taken under different +// exclusion rules, which says nothing about the save. +func sameSaveAs(recorded, current string) bool { + if recorded == current { + return true + } + r, _, okR := strings.Cut(recorded, ":") + c, _, okC := strings.Cut(current, ":") + return okR && okC && r != c +} + +// VersionInfo is a device's version of one game, as it travels in a manifest +// response. Absent from devices that predate it, which are synced the way they +// always were. +type VersionInfo struct { + Vector VersionVector `json:"vector"` + // Target is the newer version the device knows of and does not hold yet. + Target VersionVector `json:"target,omitempty"` + // HasFiles: the device holds at least one save file for the game. + HasFiles bool `json:"hasFiles"` + // Answered is the version someone on the device kept its own save over + // in a conflict (Unix ms in AnsweredAt). Resolution is mutual: the device + // holding that version is asked in turn, not overwritten. + Answered VersionVector `json:"answered,omitempty"` + AnsweredAt int64 `json:"answeredAt,omitempty"` +} + +// Outdated reports whether the device knows its save is behind another's. +func (v VersionInfo) Outdated() bool { + return len(v.Target) > 0 && compareVersions(v.Vector, v.Target) == versionOlder +} + +// gameVersion is the stored record with its vectors decoded. +type gameVersion struct { + rec store.GameVersion + vec VersionVector + target VersionVector + answered VersionVector +} + +func (e *Engine) loadVersionLocked(gameID string) (gameVersion, error) { + rec, err := e.Store.GetGameVersion(gameID) + if err != nil { + return gameVersion{}, err + } + gv := gameVersion{rec: rec, vec: VersionVector{}} + _ = json.Unmarshal([]byte(rec.Vector), &gv.vec) + if gv.vec == nil { + gv.vec = VersionVector{} + } + if rec.Target != "" { + _ = json.Unmarshal([]byte(rec.Target), &gv.target) + } + if rec.Answered != "" { + _ = json.Unmarshal([]byte(rec.Answered), &gv.answered) + } + return gv, nil +} + +func (e *Engine) saveVersionLocked(gv gameVersion) error { + if gv.rec.GameID == "" { + return nil + } + raw, _ := json.Marshal(gv.vec) + gv.rec.Vector = string(raw) + gv.rec.Target = "" + if len(gv.target) > 0 { + t, _ := json.Marshal(gv.target) + gv.rec.Target = string(t) + } + gv.rec.Answered = "" + if len(gv.answered) > 0 { + a, _ := json.Marshal(gv.answered) + gv.rec.Answered = string(a) + } else { + gv.rec.AnsweredAt = 0 + } + return e.Store.SaveGameVersion(gv.rec) +} + +// ensureGenesisLocked names a save that has files and no version yet. Not +// while the device is taking a newer version: its folder is then a part-way +// copy of someone else's, and naming that would make it a version of its own. +func (gv *gameVersion) ensureGenesis(local delta.Manifest, hash string) bool { + if len(gv.vec) > 0 || len(local.Files) == 0 || len(gv.target) > 0 || gv.rec.Pulling { + return false + } + gv.vec = VersionVector{genesisKey(local): 1} + gv.rec.Hash = hash + return true +} + +// refreshLocked brings this device's version up to date with its save before +// anyone compares it: a save that predates versions is named, and one that no +// longer hashes to what its version recorded was changed here — by the game, +// whether or not the watcher has said so yet (it waits for the game to finish +// writing, and while a game holds its files open that can be a long time). It +// is a version of its own from now. Without this, a device whose change was +// still unrecorded looked merely behind, and took the other's save over it. +// +// Not while a pull towards a newer version is unfinished: the folder is then a +// part-way copy of someone else's save, not a change made here. Reports +// whether the record changed. +func (e *Engine) refreshLocked(gv *gameVersion, game store.Game, local delta.Manifest) bool { + hash := e.versionHashOf(game.ID, local) + if gv.ensureGenesis(local, hash) { + return true + } + if gv.rec.Pulling || len(gv.vec) == 0 || gv.rec.Hash == "" || gv.rec.Hash == hash { + return false + } + if sameSaveAs(gv.rec.Hash, hash) { + // Taken under other exclusion rules, so it cannot say whether the + // save changed: re-taken under the new ones. A change made at the + // same time still becomes a version — the watcher reports it + // (NoteLocalChange). + gv.rec.Hash = hash + return true + } + e.bumpLocked(gv, game.Name, hash) + return true +} + +// learn takes in what a peer knows: a version newer than this device's +// becomes its target, unless it already knows of one newer still. +func (gv *gameVersion) learn(theirs VersionInfo) { + for _, c := range []VersionVector{theirs.Vector, theirs.Target} { + if len(c) == 0 || compareVersions(gv.vec, c) != versionOlder { + continue + } + if len(gv.target) == 0 || compareVersions(gv.target, c) == versionOlder { + gv.target = c.clone() + } + } + gv.dropReachedTarget() +} + +// dropReachedTarget forgets a target this device's version covers — or has +// gone past it sideways, by changing on its own: that is no longer "behind", +// it is a conflict, and the version comparison says so by itself. +func (gv *gameVersion) dropReachedTarget() { + if len(gv.target) > 0 && compareVersions(gv.vec, gv.target) != versionOlder { + gv.target = nil + } +} + +func (gv gameVersion) info(hasFiles bool) VersionInfo { + v := VersionInfo{Vector: gv.vec.clone(), Target: gv.target.clone(), HasFiles: hasFiles} + if len(gv.answered) > 0 { + v.Answered, v.AnsweredAt = gv.answered.clone(), gv.rec.AnsweredAt + } + return v +} + +// adopt makes this device hold theirs: their version, and their answer to a +// conflict if they carry one — the device holding the version they kept +// theirs over is still to be asked, whoever it meets. +func (gv *gameVersion) adopt(vec VersionVector, theirs VersionInfo) { + gv.vec = vec.clone() + gv.answered, gv.rec.AnsweredAt = nil, 0 + if len(theirs.Answered) > 0 { + gv.answered, gv.rec.AnsweredAt = theirs.Answered.clone(), theirs.AnsweredAt + } + gv.dropReachedTarget() +} + +// LocalVersion is this device's version of a game, for a manifest response. +// manifest is the save being served; a save that predates versions is named +// here if it has not been yet. Nil when the game takes no part in versions: +// one with automatic sync turned off is not watched, so nothing would ever +// move its version — the peer then syncs it the way it always has. +func (e *Engine) LocalVersion(game store.Game, manifest delta.Manifest) *VersionInfo { + if !game.AutoSync { + return nil + } + e.versionMu.Lock() + defer e.versionMu.Unlock() + gv, err := e.loadVersionLocked(game.ID) + if err != nil { + return nil + } + if e.refreshLocked(&gv, game, manifest) { + _ = e.saveVersionLocked(gv) + } + v := gv.info(len(manifest.Files) > 0) + return &v +} + +// NoteLocalChange moves this device's version on: the watcher saw the game +// or the user change the save (never a sync — those are told apart by +// internal/owntouch). Called before the change is offered to anyone. +func (e *Engine) NoteLocalChange(gameID string) { + game, err := e.Store.GetGame(gameID) + if err != nil || !game.AutoSync { + return + } + e.versionMu.Lock() + defer e.versionMu.Unlock() + gv, err := e.loadVersionLocked(gameID) + if err != nil || gv.rec.GameID == "" { + return + } + if gv.rec.Pulling { + // Files a pull towards a newer version wrote — this run's own, still + // arriving, or left by a run that stopped part-way (after a restart + // the watcher reports those as a change: it cannot know who wrote + // them). Not a version: the folder is a part-way copy of someone + // else's, and naming it would hand that copy to every other device. + // The pull finishing is what moves the version. Anything written here + // meanwhile is in the snapshot taken before the pull replaces it. + return + } + m, err := delta.BuildManifest(game.SavePath) + if err != nil { + return + } + hash := e.versionHashOf(gameID, m) + if len(gv.vec) == 0 { + // First change ever seen here, on a save that predates versions or a + // new one. Named after its content, so a device holding the same files + // agrees with it rather than conflicting. + if gv.ensureGenesis(m, hash) { + _ = e.saveVersionLocked(gv) + } + return + } + if hash == gv.rec.Hash { + // Already a version: a sync noticed it first (refreshLocked), or the + // change was only to files nobody syncs. + return + } + // The watcher saw the game or the user write the save: a version, even if + // the exclusion rules changed since the last one (writing a rule raises + // no file event, so that alone never arrives here). + e.bumpLocked(&gv, game.Name, hash) +} + +// bumpLocked gives this device a new version of its own, after hash. +func (e *Engine) bumpLocked(gv *gameVersion, name, hash string) { + gv.rec.Counter++ + gv.vec = gv.vec.clone() + gv.vec[gv.rec.Key] = gv.rec.Counter + gv.rec.Hash = hash + gv.dropReachedTarget() + if err := e.saveVersionLocked(*gv); err == nil { + e.Log("info", fmt.Sprintf("%s changed on this device — version %s", name, gv.vec)) + } +} + +// versionAfterResolution records a conflict's answer as a version. +// +// Keeping the peer's save makes this device hold what both held: newer than +// both, so the peer finds nothing to take and devices behind either take it. +// +// Keeping this device's own is a new version of its own — but deliberately +// not one that includes the peer's: that would make the peer simply behind, +// and its save would be replaced without anyone there being asked. Resolution +// is mutual. The answer is recorded instead (Answered), so this device is not +// asked again about the same version, and the peer, still independent of it, +// asks in turn. Whoever answers last decides (syncByVersion). +func (e *Engine) versionAfterResolution(gameID string, theirs *VersionInfo, keptLocal bool) { + if theirs == nil { + return + } + e.versionMu.Lock() + defer e.versionMu.Unlock() + gv, err := e.loadVersionLocked(gameID) + if err != nil || gv.rec.GameID == "" { + return + } + if keptLocal { + gv.rec.Counter++ + gv.vec = gv.vec.clone() + gv.vec[gv.rec.Key] = gv.rec.Counter + gv.answered, gv.rec.AnsweredAt = theirs.Vector.clone(), time.Now().UnixMilli() + } else { + gv.vec = mergeVersions(gv.vec, theirs.Vector) + gv.answered, gv.rec.AnsweredAt = nil, 0 + } + gv.rec.Pulling = false // whatever was being taken, the answer replaces it + gv.rec.Hash = "" + if game, err := e.Store.GetGame(gameID); err == nil { + if m, err := delta.BuildManifest(game.SavePath); err == nil { + gv.rec.Hash = e.versionHashOf(gameID, m) + } + } + gv.dropReachedTarget() + _ = e.saveVersionLocked(gv) +} + +// discardLocalVersion is for a "keep theirs" that did not finish: this +// device's own version is given up, and it takes the peer's like any outdated +// device would — whole, and only from a device that holds it. +func (e *Engine) discardLocalVersion(gameID string, theirs *VersionInfo) { + if theirs == nil { + return + } + e.versionMu.Lock() + defer e.versionMu.Unlock() + gv, err := e.loadVersionLocked(gameID) + if err != nil || gv.rec.GameID == "" { + return + } + gv.vec = VersionVector{} + gv.target = theirs.Vector.clone() + gv.rec.Hash = "" + _ = e.saveVersionLocked(gv) +} + +// versionStatusWaiting: this device or the peer is behind, and the other +// cannot give it the newer version either. Nothing happens; both wait for a +// device that holds it. +const versionStatusWaiting = "waiting" + +// syncByVersion decides a sync from the two devices' versions. handled false +// leaves it to the file comparison: the two hold the same version and the +// same files, or both saves predate versions and have not met since. +// +// local and remote are filtered by the exclusion rules; unfilteredLocal is +// this device's save as it is on disk, which names a version. +func (e *Engine) syncByVersion(ctx context.Context, game store.Game, peer Peer, + local, unfilteredLocal delta.Manifest, remote ManifestResponse) (Result, bool, error) { + + gameID := game.ID + theirs := *remote.Version + + e.versionMu.Lock() + gv, err := e.loadVersionLocked(gameID) + if err != nil || gv.rec.GameID == "" { + e.versionMu.Unlock() + return Result{}, false, nil + } + before := gv.target.clone() + changed := e.refreshLocked(&gv, game, unfilteredLocal) + gv.learn(theirs) + if changed || compareVersions(before, gv.target) != versionEqual || len(before) != len(gv.target) { + _ = e.saveVersionLocked(gv) + } + mine := gv.info(len(local.Files) > 0) + e.versionMu.Unlock() + + waiting := func(why string) (Result, bool, error) { + e.logOnce(gameID+"|"+peer.ID, why) + return Result{Status: versionStatusWaiting, PeerID: peer.ID, PeerName: peer.Name}, true, nil + } + pushTo := func() (Result, bool, error) { + // Once per state: a peer still taking it is asked again on every + // pass until it has, and that is not news each time. + e.logOnce(gameID+"|"+peer.ID, fmt.Sprintf("%q: %s has an older version (%s, this device %s) — asking it to take this one", + game.Name, peer.Name, theirs.Vector, mine.Vector)) + e.Transport.TriggerPeerPull(peer, gameID) + return Result{Status: "triggered_peer_pull", Direction: "push", PeerID: peer.ID, PeerName: peer.Name}, true, nil + } + + switch { + case mine.Outdated(): + if theirs.Outdated() || !covers(theirs.Vector, mine.Target) { + return waiting(fmt.Sprintf("%q: this device is behind (holds %s, newest known %s); %s does not hold that version either — waiting for a device that does", + game.Name, mine.Vector, mine.Target, peer.Name)) + } + return e.pullVersion(ctx, game, peer, local, remote, theirs.Vector) + case theirs.Outdated(): + if covers(mine.Vector, theirs.Target) { + return pushTo() + } + return waiting(fmt.Sprintf("%q: %s is behind and this device does not hold the version it is waiting for either", + game.Name, peer.Name)) + } + + switch compareVersions(mine.Vector, theirs.Vector) { + case versionNewer: + return pushTo() + case versionOlder: + // learn made this device outdated above, so this is only reached if + // the record could not be read back; the peer holds the newer version + // completely, so take it. + return e.pullVersion(ctx, game, peer, local, remote, theirs.Vector) + case versionEqual: + e.forgetLogOnce(gameID + "|" + peer.ID) + if sameFiles(local, remote.Manifest) { + return Result{}, false, nil + } + // One version, different files. Each side checks its own save against + // its version before answering (refreshLocked), so this is a change + // that landed after the peer answered; its sync carries it. + return waiting(fmt.Sprintf("%q: same version as %s but the files differ — waiting for the change to be recorded", + game.Name, peer.Name)) + } + + // Concurrent: both changed independently. + e.forgetLogOnce(gameID + "|" + peer.ID) + if sameFiles(local, remote.Manifest) { + // …into the same save. Both record the union, which each computes the + // same way, and are equal from then on. + e.versionMu.Lock() + if gv, err := e.loadVersionLocked(gameID); err == nil { + gv.vec = mergeVersions(gv.vec, theirs.Vector) + gv.rec.Hash = e.versionHashOf(gameID, unfilteredLocal) + gv.dropReachedTarget() + _ = e.saveVersionLocked(gv) + } + e.versionMu.Unlock() + return Result{}, false, nil + } + if !mine.HasFiles { + // An empty folder is never one side of a conflict: it takes the save. + return e.pullVersion(ctx, game, peer, local, remote, mergeVersions(mine.Vector, theirs.Vector)) + } + if !theirs.HasFiles { + return pushTo() + } + if mine.Vector.genesisOnly() || theirs.Vector.genesisOnly() { + // A save from before versions existed, never compared with this one + // since: nothing recorded says which is newer, so the files decide + // (transitionNewer) and the older device takes the newer save whole, + // as any device behind does — never file by file, which is how devices + // ended up holding mixtures of two saves that no game ever wrote. Only + // an exact tie is asked about. + switch transitionNewer(local, remote.Manifest, e.Store.GetAgreedHash(gameID, peer.ID)) { + case 1: + return pushTo() + case -1: + e.Log("info", fmt.Sprintf("%q: %s's save holds the more recent work — taking it whole (this device's is kept in a snapshot)", + game.Name, peer.Name)) + return e.pullVersion(ctx, game, peer, local, remote, theirs.Vector) + } + e.Log("warn", fmt.Sprintf("%q differs from %s's and neither holds more recent work — asking", game.Name, peer.Name)) + e.registerConflict(gameID, peer, local, remote) + return Result{Status: "conflict", PeerID: peer.ID, PeerName: peer.Name}, true, nil + } + + // Someone already answered: here, keeping this device's save over the + // peer's version (or one it has since moved on from), or there, keeping + // theirs over this one's. + iAnswered := len(mine.Answered) > 0 && covers(theirs.Vector, mine.Answered) + theyAnswered := len(theirs.Answered) > 0 && covers(mine.Vector, theirs.Answered) + switch { + case iAnswered && theyAnswered: + // Both were answered, each keeping its own. The later answer stands, + // as the person who gave it saw the other one too; the earlier one's + // save is in the snapshot taken before it is replaced. + // Answers in the same millisecond are settled the same way on both + // devices, or each would keep asking the other to take its own. + if mine.AnsweredAt > theirs.AnsweredAt || + (mine.AnsweredAt == theirs.AnsweredAt && mine.Vector.String() > theirs.Vector.String()) { + return pushTo() + } + e.Log("info", fmt.Sprintf("%q: %s's answer to the conflict came later — taking its version", game.Name, peer.Name)) + return e.pullVersion(ctx, game, peer, local, remote, mergeVersions(mine.Vector, theirs.Vector)) + case iAnswered: + // Answered here; the peer is asked in turn when it syncs. + return pushTo() + } + e.Log("warn", fmt.Sprintf("%q was changed independently on this device (%s) and on %s (%s)", + game.Name, mine.Vector, peer.Name, theirs.Vector)) + e.registerConflict(gameID, peer, local, remote) + return Result{Status: "conflict", PeerID: peer.ID, PeerName: peer.Name}, true, nil +} + +// pullVersion makes this device's save exactly the peer's — every file it +// holds, and none it does not — and only then records the peer's version as +// this device's. Until that, the device stays behind and keeps waiting. +// +// local and remote are filtered by the exclusion rules, so excluded files are +// neither taken nor removed. +func (e *Engine) pullVersion(ctx context.Context, game store.Game, peer Peer, + local delta.Manifest, remote ManifestResponse, adopt VersionVector) (Result, bool, error) { + + gameID := game.ID + if until, blocked := e.unwritableUntil(gameID); blocked { + // Said once when it happened (below); waiting is the answer until then. + e.logOnce(gameID+"|unwritable", fmt.Sprintf("%q: waiting until %s before writing to its save folder again", + game.Name, until.Format("15:04"))) + return Result{Status: "unwritable", PeerID: peer.ID, PeerName: peer.Name}, true, nil + } + if len(remote.Manifest.Files) == 0 && len(local.Files) > 0 && !remote.DeletionConfirmed { + // The newer version is an empty folder nobody has confirmed emptying. + e.Log("info", fmt.Sprintf("%q holds none of %q's save files now, and has not confirmed deleting them — keeping this device's copies", + peer.Name, game.Name)) + return Result{Status: "peer_holding", PeerID: peer.ID, PeerName: peer.Name}, true, nil + } + + var d Decision + for p, lf := range local.Files { + rf, ok := remote.Manifest.Files[p] + switch { + case !ok: + d.FilesToDeleteLocally = append(d.FilesToDeleteLocally, p) + case rf.Hash != lf.Hash: + d.FilesToPull = append(d.FilesToPull, p) + } + } + for p := range remote.Manifest.Files { + if _, ok := local.Files[p]; !ok { + d.FilesToPull = append(d.FilesToPull, p) + } + } + localDirs, remoteDirs := toSet(local.Dirs), toSet(remote.Manifest.Dirs) + for dir := range remoteDirs { + if _, ok := localDirs[dir]; !ok { + d.DirsToPull = append(d.DirsToPull, dir) + } + } + for dir := range localDirs { + if _, ok := remoteDirs[dir]; !ok { + d.DirsToDeleteLocally = append(d.DirsToDeleteLocally, dir) + } + } + sort.Strings(d.FilesToPull) + + e.Log("info", fmt.Sprintf("%q: taking %s's newer version (%s): %d file(s) to fetch, %d to remove", + game.Name, peer.Name, adopt, len(d.FilesToPull), len(d.FilesToDeleteLocally))) + + // Whatever this device holds is kept in a snapshot before any of it is + // replaced or removed. + replacing := len(d.FilesToDeleteLocally) + for _, p := range d.FilesToPull { + if _, ok := local.Files[p]; ok { + replacing++ + } + } + if replacing > 0 { + if _, err := e.Snapshots.CreateBeforeReplacing(gameID, fmt.Sprintf("Before taking %s's newer version", peer.Name)); err != nil { + return Result{}, true, fmt.Errorf("refusing to replace %q's files: they could not be snapshotted first: %w", game.Name, err) + } + } + + e.setPulling(gameID, true) + defer e.setPulling(gameID, false) + + applied := e.Writing(gameID) + started := time.Now() + e.applyLocalDeletions(gameID, primaryRootOf(game), d) + if n := len(d.FilesToDeleteLocally); n > 0 { + e.RecordActivity(store.ActivityEvent{GameID: gameID, Kind: store.ActivityDeleted, Device: peer.Name, Files: n}) + e.noteEmptiedByPeer(gameID, started) + } + e.createPulledDirs(gameID, game, d.DirsToPull) + if len(d.FilesToPull) > 0 { + if err := e.pullFiles(ctx, peer, gameID, game, primaryRootOf(game), local, remote, d.FilesToPull); err != nil { + applied() + if errors.Is(err, fs.ErrPermission) { + // Not something the next pass fixes: the folder needs other + // rights (one under Program Files does). Said once, plainly, + // and tried again later rather than on every pass. + e.markUnwritable(gameID) + e.Log("error", fmt.Sprintf("%q: the save folder %s cannot be written (%v). Give OpenSave write access to it, or track the game at another folder. Trying again in %s.", + game.Name, game.SavePath, err, unwritableRetry)) + } + return Result{}, true, fmt.Errorf("taking %s's version of %q did not finish (it is resumed on the next sync): %w", peer.Name, game.Name, err) + } + } + applied() + + // Only a save that is now exactly the peer's takes its version. Read + // straight from disk, not through the gate (settle.go): this sync is the + // one that wrote it, and after following a peer onto a branch it still + // holds the save — waiting for that would wait for itself until the + // sync's time ran out. + fresh, err := delta.BuildManifest(game.SavePath) + if err != nil { + return Result{}, true, err + } + freshHash := e.versionHashOf(gameID, fresh) + fresh = filterManifest(fresh, e.rulesFor(gameID)) + changedSince := false + if !sameFiles(fresh, remote.Manifest) { + if !changedSincePulled(primaryRootOf(game).Path, fresh, remote.Manifest, d) { + return Result{}, true, fmt.Errorf("taking %s's version of %q did not finish: the save still differs; it is resumed on the next sync", peer.Name, game.Name) + } + changedSince = true + } + + e.versionMu.Lock() + if gv, err := e.loadVersionLocked(gameID); err == nil { + theirs := VersionInfo{} + if remote.Version != nil { + theirs = *remote.Version + } + gv.adopt(adopt, theirs) + gv.rec.Hash = freshHash + gv.rec.Pulling = false + e.Log("success", fmt.Sprintf("%q now holds version %s from %s", game.Name, gv.vec, peer.Name)) + if changedSince { + // The peer's version arrived whole, and was then changed here - + // the game saving straight after the pull. That is a version of + // this device's own, made from theirs, and goes back to them. + e.bumpLocked(&gv, game.Name, freshHash) + } + _ = e.saveVersionLocked(gv) + } + e.versionMu.Unlock() + if changedSince { + e.Transport.TriggerPeerPull(peer, gameID) + } + + // The file comparison's own records, so it agrees if it is ever asked. + e.persistLineage(gameID, peer.ID, fresh, remote.Manifest) + _ = e.Store.SetAgreedHash(gameID, peer.ID, remote.Manifest.ManifestHash()) + + e.syncExtraRoots(ctx, gameID, game, peer, remote) + if !d.HasChanges() { + return Result{Status: "in_sync", Direction: "none", PeerID: peer.ID, PeerName: peer.Name}, true, nil + } + return Result{Status: "updated", Direction: "pull", PeerID: peer.ID, PeerName: peer.Name}, true, nil +} + +// unwritableRetry is how long a game whose save folder could not be written +// waits before a pull into it is tried again. +const unwritableRetry = 30 * time.Minute + +func (e *Engine) markUnwritable(gameID string) { + e.versionMu.Lock() + defer e.versionMu.Unlock() + if e.unwritable == nil { + e.unwritable = map[string]time.Time{} + } + e.unwritable[gameID] = time.Now().Add(unwritableRetry) +} + +func (e *Engine) unwritableUntil(gameID string) (time.Time, bool) { + e.versionMu.Lock() + defer e.versionMu.Unlock() + until, ok := e.unwritable[gameID] + if !ok || time.Now().After(until) { + delete(e.unwritable, gameID) + return time.Time{}, false + } + return until, true +} + +// setPulling marks a pull towards a newer version as running, so the files it +// writes are never taken for a change made here. In memory for this run; in +// the record from the start until the pull has completed, which only adopting +// the version clears — a pull that stopped part-way leaves a folder that is +// neither this device's version nor the peer's, and it must not be read as a +// change made here (refreshLocked, NoteLocalChange) while it waits to resume. +func (e *Engine) setPulling(gameID string, on bool) { + e.versionMu.Lock() + defer e.versionMu.Unlock() + if e.pullingNow == nil { + e.pullingNow = map[string]bool{} + } + if !on { + delete(e.pullingNow, gameID) + return + } + e.pullingNow[gameID] = true + if gv, err := e.loadVersionLocked(gameID); err == nil && gv.rec.GameID != "" && !gv.rec.Pulling { + gv.rec.Pulling = true + _ = e.saveVersionLocked(gv) + } +} + +// quickVersionCheck is the version half of quickInSync: both saves are the +// agreed one, so the question is only whether the versions say the same. +// done reports a final answer; otherwise the full sync decides. +func (e *Engine) quickVersionCheck(game store.Game, peer Peer, local delta.Manifest, theirs VersionInfo) (res Result, done bool) { + e.versionMu.Lock() + gv, err := e.loadVersionLocked(game.ID) + if err != nil || gv.rec.GameID == "" { + e.versionMu.Unlock() + return Result{}, false + } + before := gv.target.clone() + changed := e.refreshLocked(&gv, game, local) + gv.learn(theirs) + if changed || compareVersions(before, gv.target) != versionEqual || len(before) != len(gv.target) { + _ = e.saveVersionLocked(gv) + } + mine := gv.info(len(local.Files) > 0) + e.versionMu.Unlock() + + if compareVersions(mine.Vector, theirs.Vector) != versionEqual { + return Result{}, false + } + if mine.Outdated() || theirs.Outdated() { + // The same save, and at least one of the two knows a newer one exists + // that neither holds: nothing to do between these two. + return Result{Status: versionStatusWaiting, PeerID: peer.ID, PeerName: peer.Name}, true + } + return Result{}, true +} + +// logOnce logs a line the first time it is the answer for key, and not again +// while it stays the answer: waiting is re-decided every reconcile. +func (e *Engine) logOnce(key, msg string) { + e.versionMu.Lock() + if e.loggedOnce == nil { + e.loggedOnce = map[string]string{} + } + same := e.loggedOnce[key] == msg + e.loggedOnce[key] = msg + e.versionMu.Unlock() + if !same { + e.Log("info", msg) + } +} + +func (e *Engine) forgetLogOnce(key string) { + e.versionMu.Lock() + delete(e.loggedOnce, key) + e.versionMu.Unlock() +} + +// transitionNewer says which of two saves from before versions is the newer, +// when that is beyond doubt: 1 for local, -1 for remote, 0 when it is not. +// +// Beyond doubt means one of two things. The devices last agreed on a state +// and one of them still holds exactly it: that one has not changed since, so +// the other is newer. Or one save is newer in every file that differs — each +// is newer there or missing on the other side — and the other holds no file +// of its own written after the first's newest. That is what a save someone +// played on looks like next to one nobody touched. +// +// It is deliberately not "the newer file wins, file by file": that makes a +// mixture of two saves no game wrote. And not "the newest file wins", which a +// mixture passes as easily as a real save. Here the device that gives way was +// older in every file that differs, so it never loses anything newer than +// what it takes — at worst it takes a copy that arrived only half-way, and +// then the complete save, newer than both, replaces that too when it is +// seen. States newer each in different files fail the test both ways and are +// left to a person. +func transitionNewer(local, remote delta.Manifest, agreed string) int { + if agreed != "" { + lh, rh := local.ManifestHash(), remote.ManifestHash() + switch { + case lh == agreed && rh != agreed: + return -1 + case rh == agreed && lh != agreed: + return 1 + } + } + // Two saves with no file in common are not two versions of the same + // files — a save kept as one file named differently on each device, say + // — and nothing about their times says one should replace the other. + if len(local.Files) > 0 && len(remote.Files) > 0 && !shareAPath(local, remote) { + return 0 + } + l, r := newerInEveryFile(local, remote), newerInEveryFile(remote, local) + switch { + case l && !r: + return 1 + case r && !l: + return -1 + } + return newestWork(local, remote) +} + +// newestWork settles two saves from before versions that neither rule above +// could: the one holding the most recent work — the latest-written of the +// files where they differ — is the save, and if that is a tie, the one newer +// in more of those files. The other takes it whole, keeping a snapshot of its +// own first. +// +// Before versions, devices that never ran a game held copies relayed from the +// others at different times, and syncs file by file had left some holding +// mixtures; those were asked about, about states nobody made. Taking the one +// with the most recent work is what syncing is for — an older state is never +// what anyone wants handed on — and whatever the other held stays in its +// snapshot. Changes made since versions existed are not decided this way: +// those are versions of their own, and a real conflict between them is asked. +// +// 0 only for an exact tie, which leaves it to a person. +func newestWork(a, b delta.Manifest) int { + var aNewest, bNewest delta.Milli + aCount, bCount := 0, 0 + consider := func(p string) { + af, inA := a.Files[p] + bf, inB := b.Files[p] + if inA && inB && af.Hash == bf.Hash { + return + } + if inA && af.MtimeMs > aNewest { + aNewest = af.MtimeMs + } + if inB && bf.MtimeMs > bNewest { + bNewest = bf.MtimeMs + } + if inA && inB { + switch { + case af.MtimeMs > bf.MtimeMs: + aCount++ + case bf.MtimeMs > af.MtimeMs: + bCount++ + } + } + } + for p := range a.Files { + consider(p) + } + for p := range b.Files { + if _, inA := a.Files[p]; !inA { + consider(p) + } + } + switch { + case aNewest > bNewest: + return 1 + case bNewest > aNewest: + return -1 + case aCount > bCount: + return 1 + case bCount > aCount: + return -1 + } + return 0 +} + +func shareAPath(a, b delta.Manifest) bool { + for p := range a.Files { + if _, ok := b.Files[p]; ok { + return true + } + } + return false +} + +// newerInEveryFile reports whether a is newer than b wherever they differ, +// and they do differ. +func newerInEveryFile(a, b delta.Manifest) bool { + differ := false + var aNewest delta.Milli + for p, af := range a.Files { + if af.MtimeMs > aNewest { + aNewest = af.MtimeMs + } + bf, ok := b.Files[p] + if !ok { + differ = true + continue + } + if af.Hash != bf.Hash { + differ = true + if af.MtimeMs <= bf.MtimeMs { + return false + } + } + } + for p, bf := range b.Files { + if _, ok := a.Files[p]; ok { + continue + } + differ = true + if bf.MtimeMs >= aNewest { + return false + } + } + return differ +} + +// StatusPeerOutdatedApp: the peer runs an OpenSave without save versions. +// Nothing is taken from it until it is updated; it can still take from here. +const StatusPeerOutdatedApp = "peer_outdated_app" + +// OldBuildDeleteMessage is the answer to a deletion asked for by a device +// whose OpenSave does not keep save versions. +const OldBuildDeleteMessage = "this device keeps save versions and takes no deletions from an OpenSave that does not — update OpenSave on the asking device" + +// notePeerApp records whether a peer's OpenSave keeps save versions, as its +// last manifest answer said. +func (e *Engine) notePeerApp(peerID string, versions bool) { + e.versionMu.Lock() + defer e.versionMu.Unlock() + if e.peerVersions == nil { + e.peerVersions = map[string]bool{} + } + e.peerVersions[peerID] = versions +} + +// PeerNeedsUpdate reports whether a peer is known to run an OpenSave without +// save versions — one this device takes nothing from — so the person can be +// told to update it. +func (e *Engine) PeerNeedsUpdate(peerID string) bool { + e.versionMu.Lock() + defer e.versionMu.Unlock() + v, known := e.peerVersions[peerID] + return known && !v +} + +// RefusedOldBuildDelete logs, once per game and device, that a deletion asked +// for by an OpenSave without save versions was refused. +func (e *Engine) RefusedOldBuildDelete(gameID, from string) { + e.logOnce("refused|"+gameID+"|"+from, fmt.Sprintf( + "%s asked to delete files of %q; refused — its OpenSave does not keep save versions and decides by the old file comparison. Update it there.", + from, gameID)) +} + +// supersedes reports whether theirs is newer than this device's version and +// than the version the open conflict is with: both sides of the question are +// behind it. +func (e *Engine) supersedes(gameID string, theirs VersionInfo) bool { + e.mu.Lock() + c := e.activeConflicts[gameID] + e.mu.Unlock() + if c == nil || c.remoteVersion == nil { + return false + } + e.versionMu.Lock() + gv, err := e.loadVersionLocked(gameID) + e.versionMu.Unlock() + if err != nil || len(gv.vec) == 0 { + return false + } + return compareVersions(gv.vec, theirs.Vector) == versionOlder && + compareVersions(c.remoteVersion.Vector, theirs.Vector) == versionOlder +} + +// UseSaveEverywhere makes this device's save of a game the one every other +// device takes: its version becomes newer than every save from before +// versions existed, on any device, so each of those takes this one whole, +// keeping a snapshot of its own first. For when several devices hold states +// nobody can rank — copies that arrived half-way, mixtures of two saves — +// and a person knows which device has the right one. +// +// A change made on another device since versions existed is not overruled: +// that device and this one have each changed the save independently, and +// the person there is asked, as for any conflict. +func (e *Engine) UseSaveEverywhere(gameID string) (VersionVector, error) { + game, err := e.Store.GetGame(gameID) + if err != nil { + return nil, err + } + if !game.AutoSync { + return nil, errors.New("automatic sync is off for this game, so it does not take part in versions") + } + m, err := delta.BuildManifest(game.SavePath) + if err != nil { + return nil, err + } + if len(m.Files) == 0 { + return nil, errors.New("this device holds no save files for this game") + } + e.versionMu.Lock() + gv, err := e.loadVersionLocked(gameID) + if err != nil || gv.rec.GameID == "" { + e.versionMu.Unlock() + if err == nil { + err = errors.New("the game is not tracked") + } + return nil, err + } + gv.vec = mergeVersions(gv.vec, VersionVector{genesisAll: 1}) + gv.rec.Counter++ + gv.vec[gv.rec.Key] = gv.rec.Counter + gv.rec.Hash = e.versionHashOf(gameID, m) + gv.target, gv.answered, gv.rec.AnsweredAt = nil, nil, 0 + gv.rec.Pulling = false + saveErr := e.saveVersionLocked(gv) + vec := gv.vec.clone() + e.versionMu.Unlock() + if saveErr != nil { + return nil, saveErr + } + e.mu.Lock() + _, conflicted := e.activeConflicts[gameID] + e.mu.Unlock() + if conflicted { + e.clearConflict(gameID) + } + e.Log("info", fmt.Sprintf("%q: this device's save is to be used everywhere — version %s", game.Name, vec)) + return vec, nil +} + +// changedSincePulled reports a pull that finished and whose save was then +// changed here: every way the save now differs from the peer's is a change +// someone other than OpenSave made since the pull was planned. A game saving, +// or a person deleting a file, the moment a pull is done leaves exactly that. +// +// - A file the pull did not touch was the same on both sides when the pull +// was planned (anything that differed was to be fetched or removed). If +// it differs now, it was changed here meanwhile. +// - A file the pull wrote or removed differs because of someone else only +// if someone other than OpenSave has written it, or put it back, since +// (owntouch). Still as OpenSave left it, it is the pull's own unfinished +// part: a pull that stopped part-way, resumed on the next sync as before. +// +// Without this, a save the game had just written, or a file just deleted, +// was taken for the unfinished part of the pull and replaced with the peer's +// on the next sync; only the snapshot taken before replacing it still held it. +func changedSincePulled(root string, fresh, remote delta.Manifest, d Decision) bool { + touched := make(map[string]bool, len(d.FilesToPull)+len(d.FilesToDeleteLocally)) + for _, p := range d.FilesToPull { + touched[p] = true + } + for _, p := range d.FilesToDeleteLocally { + touched[p] = true + } + changedByOthers := func(p string) bool { + return !touched[p] || !owntouch.Recent(delta.LocalNameFor(root, p)) + } + differs := 0 + for p, lf := range fresh.Files { + if rf, ok := remote.Files[p]; ok && rf.Hash == lf.Hash { + continue + } + differs++ + if !changedByOthers(p) { + return false + } + } + for p := range remote.Files { + if _, ok := fresh.Files[p]; ok { + continue + } + differs++ + if !changedByOthers(p) { + return false + } + } + return differs > 0 +} diff --git a/internal/p2p/syncengine/version_test.go b/internal/p2p/syncengine/version_test.go new file mode 100644 index 00000000..67e2d189 --- /dev/null +++ b/internal/p2p/syncengine/version_test.go @@ -0,0 +1,803 @@ +package syncengine + +import ( + "context" + "errors" + "fmt" + "os" + "path/filepath" + "sort" + "sync" + "testing" + "time" + + "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/snapshot" + "github.com/opensave/opensave/internal/store" +) + +func TestCompareVersions(t *testing.T) { + cases := []struct { + a, b VersionVector + want versionOrder + }{ + {VersionVector{}, VersionVector{}, versionEqual}, + {VersionVector{}, VersionVector{"x": 1}, versionOlder}, + {VersionVector{"x": 2}, VersionVector{"x": 1}, versionNewer}, + {VersionVector{"x": 1, "y": 1}, VersionVector{"x": 1, "y": 1}, versionEqual}, + {VersionVector{"x": 2}, VersionVector{"x": 1, "y": 1}, versionConcurrent}, + {VersionVector{"x": 1}, VersionVector{"x": 1, "y": 1}, versionOlder}, + } + for _, c := range cases { + if got := compareVersions(c.a, c.b); got != c.want { + t.Errorf("compare(%v, %v) = %d, want %d", c.a, c.b, got, c.want) + } + } +} + +// "Use this save everywhere" covers every save from before versions, not +// changes made since. +func TestCompareVersions_UseEverywhereCoversEveryOldSave(t *testing.T) { + everywhere := VersionVector{"g:aaaa": 1, genesisAll: 1, "x": 1} + if got := compareVersions(VersionVector{"g:bbbb": 1}, everywhere); got != versionOlder { + t.Errorf("an old save of other content vs everywhere = %d, want older", got) + } + if got := compareVersions(VersionVector{}, everywhere); got != versionOlder { + t.Errorf("an empty folder vs everywhere = %d, want older", got) + } + if got := compareVersions(VersionVector{"g:bbbb": 1, "y": 1}, everywhere); got != versionConcurrent { + t.Errorf("a save changed since versions vs everywhere = %d, want concurrent", got) + } +} + +func setTimes(t *testing.T, dir, rel string, at time.Time) { + t.Helper() + if err := os.Chtimes(filepath.Join(dir, filepath.FromSlash(rel)), at, at); err != nil { + t.Fatal(err) + } +} + +// Two saves from before versions, one played on since they last matched: +// that one is newer in every file that differs, and the other takes it whole +// — no question, and no mixture. +func TestVersions_TransitionTakesTheClearlyNewerSave(t *testing.T) { + m, n := newMesh(t, "played", "old") + played, old := n["played"], n["old"] + t0 := time.Now().Add(-48 * time.Hour) + t1 := time.Now().Add(-1 * time.Hour) + for _, d := range []*meshNode{played, old} { + write(t, d.dir, "slot1.sav", "day one") + write(t, d.dir, "meta.dat", "m1") + write(t, d.dir, "stale.tmp", "gone later") + setTimes(t, d.dir, "slot1.sav", t0) + setTimes(t, d.dir, "meta.dat", t0) + setTimes(t, d.dir, "stale.tmp", t0) + } + // Played on, before the update: changed, added, removed. + write(t, played.dir, "slot1.sav", "day five") + write(t, played.dir, "slot2.sav", "new slot") + _ = os.Remove(filepath.Join(played.dir, "stale.tmp")) + setTimes(t, played.dir, "slot1.sav", t1) + setTimes(t, played.dir, "slot2.sav", t1) + + for _, pair := range [][2]*meshNode{{old, played}, {played, old}} { + res, err := syncPair(t, pair[0], pair[1]) + if err != nil || res.Status == "conflict" { + t.Fatalf("%s↔%s: %+v, %v", pair[0].name, pair[1].name, res, err) + } + } + settleAll(t, played, old) + if !sameSave(old.files(t), played.files(t)) { + t.Errorf("the old device did not take the played save whole: %s", describe(old.files(t))) + } + if m.deletes() != 0 { + t.Errorf("%d deletion(s) were sent", m.deletes()) + } +} + +// Two copies relayed from other devices at different times, each newer in a +// different file — the state syncing file by file used to leave. Nobody is +// asked: the one holding the more recent work is the save, and the other +// takes it whole, keeping its own in a snapshot. Never a mixture. +func TestVersions_TransitionTakesTheMostRecentWorkWithoutAsking(t *testing.T) { + m, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + t0 := time.Now().Add(-48 * time.Hour) + t1 := time.Now().Add(-2 * time.Hour) + t2 := time.Now().Add(-1 * time.Hour) + write(t, a.dir, "one.sav", "a-new") + write(t, a.dir, "two.sav", "old") + write(t, b.dir, "one.sav", "old") + write(t, b.dir, "two.sav", "b-newest") + setTimes(t, a.dir, "one.sav", t1) + setTimes(t, a.dir, "two.sav", t0) + setTimes(t, b.dir, "one.sav", t0) + setTimes(t, b.dir, "two.sav", t2) + want := b.files(t) + + for _, pair := range [][2]*meshNode{{a, b}, {b, a}} { + if res, err := syncPair(t, pair[0], pair[1]); err != nil || res.Status == "conflict" { + t.Fatalf("%s↔%s: %+v, %v — want no question", pair[0].name, pair[1].name, res, err) + } + } + settleAll(t, a, b) + if !sameSave(a.files(t), want) || !sameSave(b.files(t), want) { + t.Errorf("not both on the save with the most recent work: a %s, b %s", describe(a.files(t)), describe(b.files(t))) + } + if m.deletes() != 0 { + t.Errorf("%d deletion(s) were sent", m.deletes()) + } +} + +// Two saves with no file in common — one file, named differently on each +// device — are not two versions of the same files: asked about, never one +// replacing the other because it is newer. +func TestVersions_TransitionAsksWhenNoFileIsShared(t *testing.T) { + _, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + write(t, a.dir, "alice.sav", "A's world") + write(t, b.dir, "bob.sav", "B's world") + setTimes(t, a.dir, "alice.sav", time.Now().Add(-time.Hour)) + setTimes(t, b.dir, "bob.sav", time.Now().Add(-48*time.Hour)) + bBefore := b.files(t) + for _, pair := range [][2]*meshNode{{a, b}, {b, a}} { + if res, err := syncPair(t, pair[0], pair[1]); err != nil || res.Status != "conflict" { + t.Fatalf("%s↔%s: %+v, %v — want a question", pair[0].name, pair[1].name, res, err) + } + } + if !sameSave(b.files(t), bBefore) { + t.Error("b's save was replaced by one with no file in common") + } +} + +// Only an exact tie — the same latest time, newer in as many files — is asked +// about: there is nothing to tell the two apart. +func TestVersions_TransitionAsksOnlyOnAnExactTie(t *testing.T) { + _, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + t1 := time.Now().Add(-1 * time.Hour) + write(t, a.dir, "slot.sav", "a-content") + write(t, b.dir, "slot.sav", "b-content") + setTimes(t, a.dir, "slot.sav", t1) + setTimes(t, b.dir, "slot.sav", t1) + aBefore, bBefore := a.files(t), b.files(t) + + res, err := syncPair(t, a, b) + if err != nil || res.Status != "conflict" { + t.Fatalf("an exact tie: %+v, %v — want a question", res, err) + } + if !sameSave(a.files(t), aBefore) || !sameSave(b.files(t), bBefore) { + t.Error("a save changed before anyone answered") + } +} + +// "Use this save everywhere" on the device with the right save: every other +// device takes it whole — even one already waiting on a question about two +// mixed states — and nobody is asked. +func TestVersions_UseThisSaveEverywhere(t *testing.T) { + m, n := newMesh(t, "right", "mixed1", "mixed2") + right, mixed1, mixed2 := n["right"], n["mixed1"], n["mixed2"] + t0 := time.Now().Add(-48 * time.Hour) + t1 := time.Now().Add(-1 * time.Hour) + writeMany(t, right.dir, "map", 12) + write(t, right.dir, "players.db", "current") + // Two states nothing tells apart: one file each, the same time. + write(t, mixed1.dir, "players.db", "old-1") + write(t, mixed2.dir, "players.db", "old-2") + setTimes(t, mixed1.dir, "players.db", t1) + setTimes(t, mixed2.dir, "players.db", t1) + _ = t0 + + if res, _ := syncPair(t, mixed1, mixed2); res.Status != "conflict" { + t.Fatalf("setup: an exact tie should be asked about, got %+v", res) + } + if _, err := right.eng.UseSaveEverywhere("game1"); err != nil { + t.Fatal(err) + } + settleAll(t, right, mixed1, mixed2) + want := right.files(t) + for _, d := range []*meshNode{mixed1, mixed2} { + if !sameSave(d.files(t), want) { + t.Errorf("%s did not take the save used everywhere: %s", d.name, describe(d.files(t))) + } + if len(d.eng.ActiveConflicts()) != 0 { + t.Errorf("%s still has a conflict open", d.name) + } + } + if m.deletes() != 0 { + t.Errorf("%d deletion(s) were sent", m.deletes()) + } +} + +// A change made on another device after versions existed is not overruled by +// "Use this save everywhere": that is a conflict like any other. +func TestVersions_UseEverywhereDoesNotOverruleALaterChange(t *testing.T) { + _, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + write(t, a.dir, "slot.sav", "start") + settleAll(t, a, b) + b.change(t, func(dir string) { write(t, dir, "slot.sav", "b-played") }) + if _, err := a.eng.UseSaveEverywhere("game1"); err != nil { + t.Fatal(err) + } + bBefore := b.files(t) + settleAll(t, a, b) + if !sameSave(b.files(t), bBefore) { + t.Error("b's change since the update was replaced without anyone being asked") + } + if len(a.eng.ActiveConflicts())+len(b.eng.ActiveConflicts()) == 0 { + t.Error("nobody was asked") + } +} + +// --- a network of real engines ------------------------------------------- + +// meshNode is one device: its own store, save folder and engine. +type meshNode struct { + name string + dir string + st *store.Store + eng *Engine + // blockBudget, when >= 0, is how many more files this device serves + // before every further block request fails: a pull that stops part-way. + blockBudget int + // proto is the protocol revision it answers with; below ProtoVersions it + // stands for an OpenSave that keeps no save versions. + proto int +} + +func (n *meshNode) peer() Peer { + return Peer{ID: "node_" + n.name, Name: n.name, Address: "127.0.0.1", Port: 1} +} + +type mesh struct { + mu sync.Mutex + nodes map[string]*meshNode // by peer id + offline map[string]bool + // deletesSent counts DeleteRemote calls: the version path never makes one. + deletesSent int +} + +type meshTransport struct { + m *mesh + self *meshNode +} + +var errUnreachable = errors.New("peer unreachable") + +func (t *meshTransport) target(peer Peer) (*meshNode, error) { + t.m.mu.Lock() + defer t.m.mu.Unlock() + n := t.m.nodes[peer.ID] + if n == nil || t.m.offline[peer.ID] { + return nil, errUnreachable + } + return n, nil +} + +// FetchManifest answers the way the real route does: the save, and the +// device's version of it. +func (t *meshTransport) FetchManifest(ctx context.Context, peer Peer, gameID string, q ManifestQuery) (ManifestResponse, error) { + n, err := t.target(peer) + if err != nil { + return ManifestResponse{}, err + } + game, err := n.st.GetGame(gameID) + if err != nil { + return ManifestResponse{}, errors.New("Game not found.") + } + m, err := delta.BuildManifest(n.dir) + if err != nil { + return ManifestResponse{}, err + } + return ManifestResponse{ + Manifest: m, + ActiveBranch: game.ActiveBranch, + DeletionConfirmed: n.eng.DeletionConfirmed(gameID), + Version: n.eng.LocalVersion(game, m), + Proto: n.proto, + }, nil +} + +func (t *meshTransport) FetchBlocks(ctx context.Context, peer Peer, ref FileRef, blockIndices []int, blockSize int) ([]BlockData, error) { + n, err := t.target(peer) + if err != nil { + return nil, err + } + t.m.mu.Lock() + if n.blockBudget == 0 { + t.m.mu.Unlock() + return nil, errors.New("connection reset") + } + if n.blockBudget > 0 { + n.blockBudget-- + } + t.m.mu.Unlock() + full := filepath.Join(n.dir, filepath.FromSlash(ref.RelPath)) + entry, err := delta.HashFile(full) + if err != nil { + return nil, err + } + raw, err := os.ReadFile(full) + if err != nil { + return nil, err + } + var out []BlockData + for _, idx := range blockIndices { + if idx >= len(entry.Blocks) { + continue + } + start := idx * entry.BlockSize + end := start + entry.Blocks[idx].Length + if end > len(raw) { + end = len(raw) + } + out = append(out, BlockData{Index: idx, Data: raw[start:end], Length: end - start}) + } + return out, nil +} + +func (t *meshTransport) DeleteRemote(ctx context.Context, peer Peer, ref FileRef) error { + n, err := t.target(peer) + if err != nil { + return err + } + t.m.mu.Lock() + t.m.deletesSent++ + t.m.mu.Unlock() + return os.RemoveAll(filepath.Join(n.dir, filepath.FromSlash(ref.RelPath))) +} + +func (t *meshTransport) TriggerPeerPull(peer Peer, gameID string) {} +func (t *meshTransport) ReportSyncEvent(peer Peer, gameID, eventType string, data map[string]any) {} + +func newMesh(t *testing.T, names ...string) (*mesh, map[string]*meshNode) { + t.Helper() + m := &mesh{nodes: map[string]*meshNode{}, offline: map[string]bool{}} + byName := map[string]*meshNode{} + for _, name := range names { + root := t.TempDir() + dir := filepath.Join(root, "saves") + if err := os.MkdirAll(dir, 0o777); err != nil { + t.Fatal(err) + } + st, err := store.Open(filepath.Join(root, "opensave.db")) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { st.Close() }) + if err := st.EnsureDefaultSettings(root, filepath.Join(root, "backups")); err != nil { + t.Fatal(err) + } + if err := st.CreateGame(store.Game{ID: "game1", Name: "Game One", SavePath: dir, AutoSync: true, MaxSnapshots: 50}); err != nil { + t.Fatal(err) + } + n := &meshNode{name: name, dir: dir, st: st, blockBudget: -1, proto: ProtoVersions} + n.eng = New(st, snapshot.New(st), &meshTransport{m: m, self: n}) + m.nodes[n.peer().ID] = n + byName[name] = n + } + for _, a := range byName { + for _, b := range byName { + if a != b { + p := b.peer() + if err := a.st.UpsertPeer(store.Peer{ID: p.ID, Name: p.Name, Address: p.Address, Port: p.Port, Status: "online"}); err != nil { + t.Fatal(err) + } + } + } + } + return m, byName +} + +func (m *mesh) setOffline(n *meshNode, off bool) { + m.mu.Lock() + m.offline[n.peer().ID] = off + m.mu.Unlock() +} + +func (m *mesh) deletes() int { + m.mu.Lock() + defer m.mu.Unlock() + return m.deletesSent +} + +// syncPair runs one sync of a with b, as a's engine would. +func syncPair(t *testing.T, a, b *meshNode) (Result, error) { + t.Helper() + return a.eng.SyncWithPeer(context.Background(), "game1", b.peer()) +} + +// change edits the save the way a game does, then tells the engine — the +// watcher's part. +func (n *meshNode) change(t *testing.T, edit func(dir string)) { + t.Helper() + edit(n.dir) + n.eng.NoteLocalChange("game1") +} + +func (n *meshNode) files(t *testing.T) map[string]string { + t.Helper() + m, err := delta.BuildManifest(n.dir) + if err != nil { + t.Fatal(err) + } + out := map[string]string{} + for p, f := range m.Files { + out[p] = f.Hash + } + return out +} + +func (n *meshNode) version(t *testing.T) VersionInfo { + t.Helper() + game, err := n.st.GetGame("game1") + if err != nil { + t.Fatal(err) + } + m, err := delta.BuildManifest(n.dir) + if err != nil { + t.Fatal(err) + } + return *n.eng.LocalVersion(game, m) +} + +func sameSave(a, b map[string]string) bool { + if len(a) != len(b) { + return false + } + for p, h := range a { + if b[p] != h { + return false + } + } + return true +} + +func describe(files map[string]string) string { + paths := make([]string, 0, len(files)) + for p := range files { + paths = append(paths, p) + } + sort.Strings(paths) + if len(paths) > 6 { + return fmt.Sprintf("%d files (%v …)", len(paths), paths[:6]) + } + return fmt.Sprintf("%v", paths) +} + +func writeMany(t *testing.T, dir, prefix string, n int) { + t.Helper() + for i := 0; i < n; i++ { + write(t, dir, fmt.Sprintf("%s/chunk_%03d.bin", prefix, i), fmt.Sprintf("%s-%d", prefix, i)) + } +} + +// settleAll syncs every reachable node with every other, a few rounds, the +// way devices that are all online settle. +func settleAll(t *testing.T, nodes ...*meshNode) { + t.Helper() + for round := 0; round < 4; round++ { + for _, a := range nodes { + for _, b := range nodes { + if a != b { + _, _ = syncPair(t, a, b) + } + } + } + } +} + +// --- scenarios --------------------------------------------------------------- + +// The case this exists for. Three devices share a save; the server changes it +// a lot. One laptop starts taking the new version and is cut off part-way; the +// server then goes offline. The two laptops must not sync with each other in +// the meantime — not the half-finished copy, and not its missing files read as +// deletions — and when the server is back, both end up with its save exactly. +func TestVersions_IncompleteDevicesWaitForTheNewestOne(t *testing.T) { + m, n := newMesh(t, "server", "laptop", "desk") + server, laptop, desk := n["server"], n["laptop"], n["desk"] + + writeMany(t, server.dir, "map", 40) + write(t, server.dir, "players.db", "v1") + settleAll(t, server, laptop, desk) + original := server.files(t) + for _, d := range []*meshNode{laptop, desk} { + if !sameSave(d.files(t), original) { + t.Fatalf("%s did not take the initial save: %s", d.name, describe(d.files(t))) + } + } + + // The desk is switched off for a while; the server plays and the laptop + // takes that. The desk is now two versions behind, the laptop one — + // which makes the laptop newer than the desk, the comparison a device + // part-way through a pull must never win. + m.setOffline(desk, true) + server.change(t, func(dir string) { write(t, dir, "players.db", "v1.5") }) + settleAll(t, server, laptop) + m.setOffline(desk, false) + + // The server plays on: new chunks, some removed, one changed. + server.change(t, func(dir string) { + writeMany(t, dir, "map2", 30) + for i := 0; i < 5; i++ { + _ = os.Remove(filepath.Join(dir, "map", fmt.Sprintf("chunk_%03d.bin", i))) + } + write(t, dir, "players.db", "v2") + }) + newest := server.files(t) + + // The laptop starts taking it and is cut off after 10 files. + server.blockBudget = 10 + if _, err := syncPair(t, laptop, server); err == nil { + t.Fatal("a pull cut off part-way reported success") + } + server.blockBudget = -1 + if !laptop.version(t).Outdated() { + t.Fatalf("the laptop does not know it is behind: %+v", laptop.version(t)) + } + m.setOffline(server, true) + + // Server gone. The laptop and desk meet, repeatedly, both ways. + deskBefore := desk.files(t) + laptopBefore := laptop.files(t) + for i := 0; i < 3; i++ { + for _, pair := range [][2]*meshNode{{desk, laptop}, {laptop, desk}} { + res, err := syncPair(t, pair[0], pair[1]) + if err != nil { + t.Fatalf("%s↔%s: %v", pair[0].name, pair[1].name, err) + } + if res.Status == "conflict" { + t.Fatalf("%s↔%s raised a conflict between two copies of the same history", pair[0].name, pair[1].name) + } + } + } + if !sameSave(desk.files(t), deskBefore) { + t.Errorf("the desk changed while the newest device was away: was %s, now %s", + describe(deskBefore), describe(desk.files(t))) + } + if !sameSave(laptop.files(t), laptopBefore) { + t.Errorf("the laptop's partial copy changed while the newest device was away: was %s, now %s", + describe(laptopBefore), describe(laptop.files(t))) + } + if !desk.version(t).Outdated() { + t.Error("the desk never learned from the laptop that a newer version exists") + } + if m.deletes() != 0 { + t.Errorf("%d deletion(s) were sent to another device", m.deletes()) + } + + // The server returns; everyone ends with exactly its save. + m.setOffline(server, false) + settleAll(t, server, laptop, desk) + for _, d := range []*meshNode{server, laptop, desk} { + if !sameSave(d.files(t), newest) { + t.Errorf("%s does not hold the newest save: %s", d.name, describe(d.files(t))) + } + if v := d.version(t); v.Outdated() { + t.Errorf("%s still thinks it is behind: %+v", d.name, v) + } + } + if m.deletes() != 0 { + t.Errorf("%d deletion(s) were sent to another device", m.deletes()) + } +} + +// An empty folder — a new device, or one whose copy never arrived — is never +// one side of a conflict, and never makes the device with the save lose files. +func TestVersions_AnEmptyDeviceTakesTheSaveWithoutAsking(t *testing.T) { + m, n := newMesh(t, "full", "empty") + full, empty := n["full"], n["empty"] + writeMany(t, full.dir, "map", 100) + want := full.files(t) + + // The device with the save syncs first: nothing of its own may go. + res, err := syncPair(t, full, empty) + if err != nil { + t.Fatal(err) + } + if res.Status == "conflict" { + t.Fatal("a save against an empty folder raised a conflict") + } + if !sameSave(full.files(t), want) { + t.Fatalf("the device with the save lost files to an empty one: %s", describe(full.files(t))) + } + if res, err := syncPair(t, empty, full); err != nil || res.Status == "conflict" { + t.Fatalf("the empty device's sync: %+v, %v", res, err) + } + if !sameSave(empty.files(t), want) { + t.Errorf("the empty device did not take the save: %s", describe(empty.files(t))) + } + if m.deletes() != 0 { + t.Errorf("%d deletion(s) were sent", m.deletes()) + } +} + +// A save that existed before versions did, identical on two devices, is one +// version at once: no conflict, nothing moved. +func TestVersions_IdenticalSavesFromBeforeVersionsAgree(t *testing.T) { + _, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + writeMany(t, a.dir, "s", 5) + writeMany(t, b.dir, "s", 5) + res, err := syncPair(t, a, b) + if err != nil || res.Status != "in_sync" { + t.Fatalf("identical saves: %+v, %v", res, err) + } + if compareVersions(a.version(t).Vector, b.version(t).Vector) != versionEqual { + t.Errorf("identical saves got different versions: %v vs %v", a.version(t).Vector, b.version(t).Vector) + } +} + +// A deletion made in the game is part of the new version, and reaches the +// other device as that — whole — however many files it is. +func TestVersions_DeletionsTravelWithTheVersion(t *testing.T) { + m, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + writeMany(t, a.dir, "map", 20) + settleAll(t, a, b) + a.change(t, func(dir string) { + _ = os.RemoveAll(filepath.Join(dir, "map")) + write(t, dir, "fresh.sav", "new world") + }) + settleAll(t, a, b) + if !sameSave(b.files(t), a.files(t)) { + t.Errorf("b = %s, want %s", describe(b.files(t)), describe(a.files(t))) + } + if m.deletes() != 0 { + t.Errorf("the newer device sent %d deletion(s) itself; the older one takes them", m.deletes()) + } +} + +// Both devices changed the save while apart: that, and only that, is asked +// about. Resolution is mutual — keeping A's on A asks B in turn rather than +// overwriting it, and A is not asked again. Once B agrees, the answer spreads, +// and a third device that had changed nothing takes it without being asked. +func TestVersions_IndependentChangesAskEachSideOnceAndTheAnswerSpreads(t *testing.T) { + _, n := newMesh(t, "a", "b", "c") + a, b, c := n["a"], n["b"], n["c"] + write(t, a.dir, "slot.sav", "start") + settleAll(t, a, b, c) + + a.change(t, func(dir string) { write(t, dir, "slot.sav", "a-played") }) + b.change(t, func(dir string) { write(t, dir, "slot.sav", "b-played") }) + + res, err := syncPair(t, a, b) + if err != nil || res.Status != "conflict" { + t.Fatalf("independent changes: %+v, %v — want a conflict", res, err) + } + if _, err := a.eng.ResolveConflict(context.Background(), "game1", b.peer().ID, "keep-local"); err != nil { + t.Fatal(err) + } + bBefore := b.files(t) + settleAll(t, a, b, c) + if len(a.eng.ActiveConflicts()) != 0 { + t.Error("A was asked again after answering") + } + if len(b.eng.ActiveConflicts()) == 0 { + t.Fatal("B was never asked: A's answer must not decide B's save on its own") + } + if !sameSave(b.files(t), bBefore) { + t.Error("B's save changed before anyone there answered") + } + if _, err := b.eng.ResolveConflict(context.Background(), "game1", a.peer().ID, "keep-remote"); err != nil { + t.Fatal(err) + } + settleAll(t, a, b, c) + for _, d := range []*meshNode{b, c} { + if got := d.files(t)["slot.sav"]; got != a.files(t)["slot.sav"] { + t.Errorf("%s did not take the kept version", d.name) + } + if len(d.eng.ActiveConflicts()) != 0 { + t.Errorf("%s still has a conflict after both answered", d.name) + } + } +} + +// Both sides answer "keep mine" (or "keep both"): the later answer stands and +// both end on one save, instead of asking each other forever. +func TestVersions_TwoAnswersSettleOnTheLater(t *testing.T) { + _, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + write(t, a.dir, "slot.sav", "start") + settleAll(t, a, b) + a.change(t, func(dir string) { write(t, dir, "slot.sav", "a-played") }) + b.change(t, func(dir string) { write(t, dir, "slot.sav", "b-played") }) + if res, _ := syncPair(t, a, b); res.Status != "conflict" { + t.Fatalf("A: %+v, want a conflict", res) + } + if res, _ := syncPair(t, b, a); res.Status != "conflict" { + t.Fatalf("B: %+v, want a conflict", res) + } + if _, err := a.eng.ResolveConflict(context.Background(), "game1", b.peer().ID, "keep-local"); err != nil { + t.Fatal(err) + } + time.Sleep(5 * time.Millisecond) + if _, err := b.eng.ResolveConflict(context.Background(), "game1", a.peer().ID, "keep-local"); err != nil { + t.Fatal(err) + } + settleAll(t, a, b) + if got := a.files(t)["slot.sav"]; got != b.files(t)["slot.sav"] { + t.Fatal("the two never settled on one save") + } + if len(a.eng.ActiveConflicts())+len(b.eng.ActiveConflicts()) != 0 { + t.Error("a conflict is still open after both answered") + } + if got, _ := os.ReadFile(filepath.Join(a.dir, "slot.sav")); string(got) != "b-played" { + t.Errorf("A holds %q; B answered last, so its save should stand", got) + } +} + +// A change the watcher never reported — made while the app was closed — still +// becomes a version once the two devices meet, instead of leaving them on one +// version with different files, each waiting for the other. +func TestVersions_AnUnnoticedChangeIsFoundWhenTheDevicesMeet(t *testing.T) { + _, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + write(t, a.dir, "slot.sav", "start") + settleAll(t, a, b) + write(t, a.dir, "slot.sav", "played offline") // no NoteLocalChange + settleAll(t, b, a) + if got := b.files(t)["slot.sav"]; got != a.files(t)["slot.sav"] { + t.Errorf("the unnoticed change never reached b") + } +} + +// A device whose OpenSave keeps no save versions is never a source: its half +// of a save is not taken, and nothing here changes on its account. It is +// reported as needing an update. +func TestVersions_NothingIsTakenFromAnOldBuild(t *testing.T) { + _, n := newMesh(t, "current", "old") + current, old := n["current"], n["old"] + writeMany(t, current.dir, "map", 10) + writeMany(t, old.dir, "map", 3) + write(t, old.dir, "only-old.bin", "x") + old.proto = ProtoBatchFiles + before := current.files(t) + + res, err := syncPair(t, current, old) + if err != nil || res.Status != StatusPeerOutdatedApp { + t.Fatalf("sync with an old build: %+v, %v — want %s", res, err, StatusPeerOutdatedApp) + } + if !sameSave(current.files(t), before) { + t.Error("the save changed on account of a device without save versions") + } + if !current.eng.PeerNeedsUpdate(old.peer().ID) { + t.Error("the old build is not reported as needing an update") + } + old.proto = ProtoVersions + _, _ = syncPair(t, current, old) + if current.eng.PeerNeedsUpdate(old.peer().ID) { + t.Error("still reported as needing an update after it was updated") + } +} + +// Files a sync writes are not a change made here: the pull-in-progress record +// keeps them from becoming a version, even across a restart. +func TestVersions_PulledFilesAreNotANewVersion(t *testing.T) { + _, n := newMesh(t, "a", "b") + a, b := n["a"], n["b"] + writeMany(t, a.dir, "map", 10) + settleAll(t, a, b) + before := b.version(t).Vector + + a.change(t, func(dir string) { writeMany(t, dir, "map3", 10) }) + a.blockBudget = 3 + _, _ = syncPair(t, b, a) + a.blockBudget = -1 + + // What a restart leaves: the record says a pull was running, and the + // watcher reports the half-written folder as a change. + rec, _ := b.st.GetGameVersion("game1") + rec.Pulling = true + _ = b.st.SaveGameVersion(rec) + b.eng.NoteLocalChange("game1") + if got := b.version(t).Vector; compareVersions(got, before) != versionEqual { + t.Fatalf("a half-finished pull became a version of its own: %v (was %v)", got, before) + } + settleAll(t, a, b) + if !sameSave(b.files(t), a.files(t)) { + t.Errorf("b did not finish taking a's version: %s", describe(b.files(t))) + } +} diff --git a/internal/p2p/transport_lan.go b/internal/p2p/transport_lan.go index 1955b640..44dcf3d6 100644 --- a/internal/p2p/transport_lan.go +++ b/internal/p2p/transport_lan.go @@ -35,6 +35,23 @@ type lanTransport struct { var lanClient = &http.Client{Timeout: 30 * time.Second} +// lanBulkClient is for the two requests whose size grows with the save: a +// manifest, and a batch of files. Thirty seconds was fine for a few hundred +// files and not for a few hundred thousand over a VPN; the manifest then never +// arrived, every sync of that game failed, and the devices never converged. +// The sync's own context still bounds it (perPeerSyncTimeout), and the +// transport still gives up on a peer that does not answer at all. +var lanBulkClient = &http.Client{ + Timeout: 10 * time.Minute, + Transport: func() http.RoundTripper { + t := http.DefaultTransport.(*http.Transport).Clone() + // Headers arrive once the peer has built its manifest, which for a big + // save not hashed recently takes a while — but not this long. + t.ResponseHeaderTimeout = 3 * time.Minute + return t + }(), +} + func peerURL(peer syncengine.Peer, route string) string { return fmt.Sprintf("http://%s:%d/api/p2p%s", peer.Address, peer.Port, route) } @@ -54,9 +71,17 @@ func (t *lanTransport) FetchManifest(ctx context.Context, peer syncengine.Peer, if q.CoverURL != "" { params.Set("coverUrl", q.CoverURL) } + if q.IfHash != "" { + params.Set("ifHash", q.IfHash) + } var resp syncengine.ManifestResponse - err := t.getJSON(ctx, peer, peerURL(peer, "/manifest/"+gameID)+"?"+params.Encode(), &resp) + req, err := http.NewRequestWithContext(ctx, http.MethodGet, peerURL(peer, "/manifest/"+gameID)+"?"+params.Encode(), nil) + if err != nil { + return resp, err + } + t.sign(req, peer, nil) + err = doJSONWith(lanBulkClient, req, &resp) return resp, err } @@ -67,17 +92,95 @@ func (t *lanTransport) FetchBlocks(ctx context.Context, peer syncengine.Peer, re // No encodings advertised: on a LAN the wire is typically faster than the // compressor, so the bytes saved cost more than they're worth. Responses // are still decoded, so a peer that compresses anyway is handled. - err := t.postJSON(ctx, peer, peerURL(peer, "/blocks/"+ref.GameID), map[string]any{ + // + // Given time for what it asks for. "LAN" here is any direct address, + // a VPN one (Tailscale) included, and a few MB of blocks at a VPN relay's + // hundred-odd KB a second outlasted the fixed 30 seconds every request + // used to get: the request failed, and with it the whole pull, over and + // over. Sized like the relay's (slowLinkBytesPerSec), within the bulk + // client's ceiling. + expected := int64(len(blockIndices)) * int64(max(blockSize, 1)) + ctx, cancel := context.WithTimeout(ctx, 30*time.Second+time.Duration(expected/(slowLinkBytesPerSec/2))*time.Second) + defer cancel() + raw, err := json.Marshal(map[string]any{ "relPath": ref.RelPath, "root": ref.Root, "blockIndices": blockIndices, "blockSize": blockSize, - }, &resp) + }) + if err != nil { + return nil, err + } + req, err := http.NewRequestWithContext(ctx, http.MethodPost, peerURL(peer, "/blocks/"+ref.GameID), bytes.NewReader(raw)) if err != nil { return nil, err } + req.Header.Set("Content-Type", "application/json") + t.sign(req, peer, raw) + if err := doJSONWith(lanBulkClient, req, &resp); err != nil { + return nil, err + } return decodeBlocks(resp.Blocks) } +// ProbeSpeed times fetching n bytes from the peer's speed test +// (syncengine.SpeedProber). The clock starts once the answer begins to +// arrive, so the figure is the link's rate rather than its round trip. +func (t *lanTransport) ProbeSpeed(ctx context.Context, peer syncengine.Peer, n int) (int64, time.Duration, error) { + req, err := http.NewRequestWithContext(ctx, http.MethodGet, peerURL(peer, fmt.Sprintf("/speedtest?n=%d", n)), nil) + if err != nil { + return 0, 0, err + } + t.sign(req, peer, nil) + resp, err := lanBulkClient.Do(req) + if err != nil { + return 0, 0, err + } + defer resp.Body.Close() + if resp.StatusCode != http.StatusOK { + return 0, 0, fmt.Errorf("peer returned %d", resp.StatusCode) + } + start := time.Now() + got, err := io.Copy(io.Discard, resp.Body) + if err != nil { + return got, time.Since(start), err + } + return got, time.Since(start), nil +} + +// FetchFileBatch fetches the blocks of many files in one request. Only asked +// of a peer that advertised syncengine.ProtoBatchFiles. +func (t *lanTransport) FetchFileBatch(ctx context.Context, peer syncengine.Peer, gameID, root string, files []syncengine.FileBlocksRequest) ([]syncengine.FileBlocks, error) { + var resp struct { + Files []syncengine.FileBlocks `json:"files"` + } + raw, err := json.Marshal(map[string]any{"root": root, "files": files}) + if err != nil { + return nil, err + } + req, err := http.NewRequestWithContext(ctx, http.MethodPost, peerURL(peer, "/files/"+gameID), bytes.NewReader(raw)) + if err != nil { + return nil, err + } + req.Header.Set("Content-Type", "application/json") + t.sign(req, peer, raw) + if err := doJSONWith(lanBulkClient, req, &resp); err != nil { + return nil, err + } + for i := range resp.Files { + if resp.Files[i].Error != "" { + continue + } + blocks, err := decodeBlocks(resp.Files[i].Blocks) + if err != nil { + return nil, fmt.Errorf("%s: %w", resp.Files[i].RelPath, err) + } + resp.Files[i].Blocks = blocks + } + return resp.Files, nil +} + func (t *lanTransport) DeleteRemote(ctx context.Context, peer syncengine.Peer, ref syncengine.FileRef) error { - return t.postJSON(ctx, peer, peerURL(peer, "/delete-file/"+ref.GameID), map[string]any{"relPath": ref.RelPath, "root": ref.Root}, nil) + // "versioned": this device keeps save versions; a peer that does too + // refuses deletions from one that does not (handleDeleteFile). + return t.postJSON(ctx, peer, peerURL(peer, "/delete-file/"+ref.GameID), map[string]any{"relPath": ref.RelPath, "root": ref.Root, "versioned": true}, nil) } // TriggerPeerPull tells a peer that this device holds newer content. @@ -224,7 +327,11 @@ func (t *lanTransport) sign(req *http.Request, peer syncengine.Peer, body []byte } func doJSON(req *http.Request, out any) error { - resp, err := lanClient.Do(req) + return doJSONWith(lanClient, req, out) +} + +func doJSONWith(client *http.Client, req *http.Request, out any) error { + resp, err := client.Do(req) if err != nil { return err } diff --git a/internal/p2p/transport_wan.go b/internal/p2p/transport_wan.go index 859e91ef..b2b5d90d 100644 --- a/internal/p2p/transport_wan.go +++ b/internal/p2p/transport_wan.go @@ -3,6 +3,7 @@ package p2p import ( "context" "encoding/json" + "errors" "fmt" "time" @@ -90,7 +91,7 @@ func (t *wanTransport) FetchBlocks(ctx context.Context, peer syncengine.Peer, re } func (t *wanTransport) DeleteRemote(ctx context.Context, peer syncengine.Peer, ref syncengine.FileRef) error { - _, err := t.wan.Request(ctx, peer.ID, "/delete-file/"+ref.GameID, "POST", map[string]any{"relPath": ref.RelPath, "root": ref.Root}) + _, err := t.wan.Request(ctx, peer.ID, "/delete-file/"+ref.GameID, "POST", map[string]any{"relPath": ref.RelPath, "root": ref.Root, "versioned": true}) return err } @@ -154,6 +155,30 @@ func (t *routingTransport) FetchBlocks(ctx context.Context, peer syncengine.Peer return t.pick(peer).FetchBlocks(ctx, peer, ref, blockIndices, blockSize) } +// FetchFileBatch batches over the LAN only. Relay messages are sealed and +// size-limited per client (see BatchIndices), so the relay keeps pulling file +// by file. +func (t *routingTransport) FetchFileBatch(ctx context.Context, peer syncengine.Peer, gameID, root string, files []syncengine.FileBlocksRequest) ([]syncengine.FileBlocks, error) { + if peer.Wan() { + return nil, syncengine.ErrBatchUnsupported + } + b, ok := t.lan.(syncengine.BatchFetcher) + if !ok { + return nil, syncengine.ErrBatchUnsupported + } + return b.FetchFileBatch(ctx, peer, gameID, root, files) +} + +// ProbeSpeed tests LAN links only (syncengine.SpeedProber); the relay is the +// slowest link whatever a test would say. +func (t *routingTransport) ProbeSpeed(ctx context.Context, peer syncengine.Peer, n int) (int64, time.Duration, error) { + p, ok := t.lan.(syncengine.SpeedProber) + if peer.Wan() || !ok { + return 0, 0, errors.New("no speed test over the relay") + } + return p.ProbeSpeed(ctx, peer, n) +} + func (t *routingTransport) DeleteRemote(ctx context.Context, peer syncengine.Peer, ref syncengine.FileRef) error { return t.pick(peer).DeleteRemote(ctx, peer, ref) } diff --git a/internal/p2p/wanclient_handlers.go b/internal/p2p/wanclient_handlers.go index 95203dd3..448f3894 100644 --- a/internal/p2p/wanclient_handlers.go +++ b/internal/p2p/wanclient_handlers.go @@ -172,6 +172,7 @@ func (w *WanClient) handleMessage(ctx context.Context, msg RelayMessage) { if w.engine.Sync.Progress.OnSyncError != nil { w.engine.Sync.Progress.OnSyncError(msg.GameID, ev) } + w.engine.Sync.NotePeerPulled(msg.GameID, msg.From) } case "request": @@ -617,12 +618,20 @@ func (w *WanClient) serveManifest(route string, body json.RawMessage, peerID str return 500, map[string]string{"error": err.Error()} } w.engine.Sync.NoteServed(game.ID, peerID, manifest) // syncengine/served.go + // "versions" says this device keeps save versions, and "version" is its + // version of this save (syncengine/version.go). Not "proto": over the + // relay that would also claim extra save locations and batched blocks, + // which this path does not serve. resp := map[string]any{ "gameId": gameID, "activeBranch": game.ActiveBranch, "manifest": manifest, "deletionConfirmed": w.engine.Sync.DeletionConfirmed(game.ID), + "versions": true, + } + if v := w.engine.Sync.LocalVersion(game, manifest); v != nil { + resp["version"] = v } if latest, err := w.engine.Snapshots.LatestSnapshot(gameID, ""); err == nil { - resp["latestSnapshot"] = syncengine.SnapshotInfo{ID: latest.ID, Timestamp: latest.Timestamp, Comment: latest.Comment} + resp["latestSnapshot"] = syncengine.SnapshotInfo{ID: latest.ID, Timestamp: latest.Timestamp, Comment: latest.Comment, ContentHash: latest.ContentHash} } else { resp["latestSnapshot"] = nil } @@ -678,6 +687,9 @@ func (w *WanClient) serveDeleteFile(route string, rawBody json.RawMessage, fromP gameID := route[strings.LastIndex(route, "/")+1:] var body struct { RelPath string `json:"relPath"` + // Versioned: the asking device keeps save versions (see the LAN + // route, handleDeleteFile). + Versioned bool `json:"versioned"` } if err := json.Unmarshal(rawBody, &body); err != nil || body.RelPath == "" { return 400, map[string]string{"error": "relPath is required."} @@ -686,6 +698,14 @@ func (w *WanClient) serveDeleteFile(route string, rawBody json.RawMessage, fromP if err != nil { return 404, map[string]string{"error": "Game not found."} } + if game.AutoSync && !body.Versioned { + from := fromPeerID + if p, err := w.engine.Store.GetPeer(fromPeerID); err == nil && p.Name != "" { + from = p.Name + } + w.engine.Sync.RefusedOldBuildDelete(game.Name, from) + return 409, map[string]string{"error": syncengine.OldBuildDeleteMessage} + } if !delta.IsSafePath(game.SavePath, body.RelPath) { return 403, map[string]string{"error": "invalid path"} } diff --git a/internal/snapshot/content.go b/internal/snapshot/content.go new file mode 100644 index 00000000..ccb7fb73 --- /dev/null +++ b/internal/snapshot/content.go @@ -0,0 +1,397 @@ +package snapshot + +import ( + "archive/zip" + "crypto/sha256" + "encoding/hex" + "fmt" + "io" + "sort" + "strings" + + "github.com/opensave/opensave/internal/delta" + "github.com/opensave/opensave/internal/store" +) + +// Snapshot content. +// +// A snapshot is named by what it holds: every file's name in the archive and +// the SHA-256 of its content, hashed together. The file hashes are the ones the +// sync compares (the hash cache both use), so the same save gives the same +// value on every device — which is how a device knows the snapshot another +// device took is one it already has, and keeps the save archived once. +// +// Before this, a save of a quarter-million files was archived again for every +// device it was compared with, and again on every new snapshot over there: +// gigabytes of identical archives for one 780 MB save. + +// contentEntry is one file of a snapshot: its name in the archive and its hash. +type contentEntry struct{ name, hash string } + +func contentKeyOf(entries []contentEntry) string { + sort.Slice(entries, func(i, j int) bool { return entries[i].name < entries[j].name }) + h := sha256.New() + for _, e := range entries { + h.Write([]byte(e.name)) + h.Write([]byte{0}) + h.Write([]byte(e.hash)) + h.Write([]byte{0}) + } + return hex.EncodeToString(h.Sum(nil)) +} + +// archiveName is the name a captured file has in the archive. +func archiveName(c store.CapturedFile) string { + if c.Root == "" { + return c.Path + } + return RootPrefix + c.Root + "/" + c.Path +} + +// ContentKey names what a snapshot holds, from its recorded files. +func ContentKey(files []store.CapturedFile) string { + entries := make([]contentEntry, 0, len(files)) + for _, f := range files { + entries = append(entries, contentEntry{archiveName(f), f.Hash}) + } + return contentKeyOf(entries) +} + +// ContentKeyOfArchive names what a snapshot holds by reading its archive: for +// a snapshot recorded without its file list (one taken for a sync with a peer). +func ContentKeyOfArchive(zipPath string) (string, error) { + files, err := ArchiveFiles(zipPath) + if err != nil { + return "", err + } + return ContentKey(files), nil +} + +// ArchiveFiles reads a snapshot's archive and lists its files with their +// content hashes, as the snapshot would have recorded them. +func ArchiveFiles(zipPath string) ([]store.CapturedFile, error) { + path, done, err := OpenArchive(zipPath) + if err != nil { + return nil, err + } + defer done() + r, err := zip.OpenReader(path) + if err != nil { + return nil, err + } + defer r.Close() + out := make([]store.CapturedFile, 0, len(r.File)) + for _, f := range r.File { + if f.FileInfo().IsDir() || strings.HasSuffix(f.Name, "/") || strings.HasSuffix(f.Name, delta.TmpSuffix) { + continue + } + rc, err := f.Open() + if err != nil { + return nil, fmt.Errorf("%s: %w", f.Name, err) + } + h := sha256.New() + _, err = io.Copy(h, rc) + rc.Close() + if err != nil { + return nil, fmt.Errorf("%s: %w", f.Name, err) + } + c := store.CapturedFile{Path: f.Name, Hash: hex.EncodeToString(h.Sum(nil))} + if rest, ok := strings.CutPrefix(f.Name, RootPrefix); ok { + if root, p, ok := strings.Cut(rest, "/"); ok { + c.Root, c.Path = root, p + } + } + out = append(out, c) + } + return out, nil +} + +// filesOf is what a snapshot holds, from its recorded file list — read from +// its archive once, and recorded, when it has none. +func (m *Manager) filesOf(snap store.Snapshot) ([]store.CapturedFile, error) { + if files, err := m.Store.SnapshotFiles(snap.ID); err == nil && len(files) > 0 { + return files, nil + } + files, err := ArchiveFiles(snap.ZipPath) + if err != nil { + return nil, err + } + if len(files) > 0 { + _ = m.Store.RecordSnapshotFiles(snap.ID, files) + } + return files, nil +} + +// contentHashOf works out what a snapshot holds (filesOf). +func (m *Manager) contentHashOf(snap store.Snapshot) (string, error) { + files, err := m.filesOf(snap) + if err != nil { + return "", err + } + return ContentKey(files), nil +} + +// PruneContained removes automatic snapshots that a newer snapshot on the same +// branch holds entirely: every file of the older one is in the newer one, with +// the same content. Such a snapshot is only a step on the way to the newer +// one — a pull that stopped part-way, a save that only gained files — and +// restoring it would give nothing the newer one lacks but the absence of what +// was added since. The other way round is kept: a newer snapshot contained in +// an older one is a save something was deleted from, on purpose. Pinned +// snapshots and ones taken by hand are never removed. +// +// The largest snapshots are looked at first as the ones that may hold others — +// they hold the most — and passes repeat until one removes nothing. After +// that, each new snapshot is checked against the older ones as it is taken +// (PruneContainedBy), which is all it takes to stay that way. +func (m *Manager) PruneContained(gameID string) (removed int, freed int64, err error) { + for { + n, f, err := m.pruneContainedPass(gameID) + removed += n + freed += f + if err != nil || n == 0 { + return removed, freed, err + } + } +} + +func (m *Manager) pruneContainedPass(gameID string) (removed int, freed int64, err error) { + snaps, err := m.Store.AllSnapshotsOfGame(gameID) // oldest first + if err != nil { + return 0, 0, err + } + byBranch := map[string][]store.Snapshot{} + counts := map[string]int{} + for _, s := range snaps { + if !ArchiveExists(s.ZipPath) { + continue + } + files, ferr := m.filesOf(s) + if ferr != nil { + continue + } + counts[s.ID] = len(files) + byBranch[s.BranchName] = append(byBranch[s.BranchName], s) + } + for _, list := range byBranch { + order := make(map[string]int, len(list)) // position in time + for i, s := range list { + order[s.ID] = i + } + holders := append([]store.Snapshot(nil), list...) + sort.SliceStable(holders, func(i, j int) bool { return counts[holders[i].ID] > counts[holders[j].ID] }) + gone := map[string]bool{} + for _, holder := range holders { + if gone[holder.ID] || counts[holder.ID] == 0 { + continue + } + holderFiles, ferr := m.filesOf(holder) + if ferr != nil { + continue + } + have := make(map[string]string, len(holderFiles)) + for _, f := range holderFiles { + have[archiveName(f)] = f.Hash + } + for _, older := range list[:order[holder.ID]] { + if gone[older.ID] || older.Pinned || !older.IsSystemAuto || counts[older.ID] == 0 || counts[older.ID] > len(holderFiles) { + continue + } + olderFiles, ferr := m.filesOf(older) + if ferr != nil || !containedIn(olderFiles, have) { + continue + } + identical := len(olderFiles) == len(holderFiles) + if identical { + // The same files: its id still leads to them. Before the + // delete, so no moment exists where the id leads nowhere; + // while the row is there it still resolves to itself. + _ = m.Store.RepointSnapshotAliases(older.ID, holder.ID) + _ = m.Store.AddSnapshotAlias(older.ID, holder.ID) + m.carryName(older, &holder) + } + f, derr := m.DeleteSnapshot(gameID, older.ID) + if derr != nil { + continue + } + gone[older.ID] = true + removed++ + freed += f + if m.Log != nil { + m.Log("info", fmt.Sprintf("snapshot %s removed: every file of it is in the newer %s", older.ID, holder.ID)) + } + } + } + } + return removed, freed, nil +} + +func containedIn(files []store.CapturedFile, have map[string]string) bool { + for _, f := range files { + if have[archiveName(f)] != f.Hash { + return false + } + } + return true +} + +// PruneContainedBy checks the older snapshots of snap's branch against snap, +// just taken, and removes those it holds entirely (PruneContained). Only when +// PruneOnCreate is set; in the background, counted for shutdown. +func (m *Manager) PruneContainedBy(snap store.Snapshot, files []store.CapturedFile) { + if !m.PruneOnCreate || len(files) == 0 { + return + } + m.inFlight.Add() + go func() { + defer m.inFlight.Done() + have := make(map[string]string, len(files)) + for _, f := range files { + have[archiveName(f)] = f.Hash + } + snaps, err := m.Store.ListSnapshots(snap.GameID, snap.BranchName) + if err != nil { + return + } + for _, older := range snaps { + if older.ID == snap.ID || older.Timestamp >= snap.Timestamp || older.Pinned || !older.IsSystemAuto || + !ArchiveExists(older.ZipPath) { + continue + } + olderFiles, ferr := m.filesOf(older) + if ferr != nil || len(olderFiles) == 0 || len(olderFiles) > len(files) || !containedIn(olderFiles, have) { + continue + } + identical := len(olderFiles) == len(files) + if identical { + // Aliased before the delete (see PruneContained). + _ = m.Store.RepointSnapshotAliases(older.ID, snap.ID) + _ = m.Store.AddSnapshotAlias(older.ID, snap.ID) + m.carryName(older, &snap) + } + _, derr := m.DeleteSnapshot(snap.GameID, older.ID) + if derr == nil && m.Log != nil { + m.Log("info", fmt.Sprintf("snapshot %s removed: every file of it is in the newer %s", older.ID, snap.ID)) + } + } + }() +} + +// ArchiveSaveTo archives a game's save — every location — into zipPath the way +// a snapshot here is archived: files unchanged since the branch's newest +// snapshot are copied from it rather than compressed again (reuse.go). For a +// snapshot recorded under another device's id after a sync; compressing a save +// of a quarter-million files from scratch kept a finished pull at 100% for +// minutes. +func (m *Manager) ArchiveSaveTo(gameID, branch, zipPath string) (skipped []string, captured []store.CapturedFile, err error) { + m.inFlight.Add() + defer m.inFlight.Done() + game, err := m.Store.GetGame(gameID) + if err != nil { + return nil, nil, err + } + extra, rootsErr := m.Store.GameRootPaths(gameID) + if rootsErr != nil { + extra = nil + } + reuse := m.openReuseSource(gameID, branch) + defer reuse.Close() + return zipRootsCapturing(game.SavePath, extra, zipPath, reuse) +} + +// isNewest reports whether id is the newest snapshot on a game's branch. +func (m *Manager) isNewest(gameID, branch, id string) bool { + snaps, err := m.Store.ListSnapshots(gameID, branch) + return err == nil && len(snaps) > 0 && snaps[0].ID == id +} + +// MergeDuplicates names every snapshot of a game by its content and keeps each +// content archived once per branch: of snapshots holding the same files, one +// keeps its archive — a pinned one, else one taken by hand, else the newest, +// which retention keeps longest — +// and the others are removed, their ids kept as aliases of it so a device that +// knows a snapshot by one of them still finds it. Pinned snapshots are never +// removed. Returns how many were merged and the bytes freed. +func (m *Manager) MergeDuplicates(gameID string) (merged int, freed int64, err error) { + missing, err := m.Store.SnapshotsWithoutContentHash(gameID) + if err != nil { + return 0, 0, err + } + for _, snap := range missing { + if !ArchiveExists(snap.ZipPath) { + continue + } + key, err := m.contentHashOf(snap) + if err != nil { + if m.Log != nil { + m.Log("warn", fmt.Sprintf("could not read snapshot %s to name its content: %v", snap.ID, err)) + } + continue + } + _ = m.Store.SetSnapshotContentHash(snap.ID, key) + } + + snaps, err := m.Store.AllSnapshotsOfGame(gameID) + if err != nil { + return 0, 0, err + } + groups := map[string][]store.Snapshot{} + for _, s := range snaps { + if s.ContentHash == "" { + continue + } + k := s.BranchName + "\x00" + s.ContentHash + groups[k] = append(groups[k], s) + } + for _, group := range groups { + if len(group) < 2 { + continue + } + keep := group[len(group)-1] // newest (oldest first) + for _, s := range group { + if s.Pinned && !keep.Pinned || s.Pinned == keep.Pinned && !s.IsSystemAuto && keep.IsSystemAuto { + keep = s + } + } + for _, s := range group { + if s.ID == keep.ID || s.Pinned { + continue + } + _ = m.Store.RepointSnapshotAliases(s.ID, keep.ID) + _ = m.Store.AddSnapshotAlias(s.ID, keep.ID) + f, err := m.DeleteSnapshot(gameID, s.ID) + if err != nil { + continue + } + merged++ + freed += f + if m.Log != nil { + m.Log("info", fmt.Sprintf("snapshot %s held the same files as %s; kept once", s.ID, keep.ID)) + } + } + } + return merged, freed, nil +} + +// defaultAutoComment is what an automatic snapshot taken for no particular +// reason is called. +const defaultAutoComment = "Auto backup" + +// carryName keeps what an older snapshot is called when a newer one holding +// exactly the same files takes its place: its reason ("After playing (1 h)", +// "Put back") if the newer one has only the default an automatic snapshot +// gets, and a note someone wrote if the newer one has none. A session's +// named snapshot and the watcher's copy of the same save, taken a moment +// later as the game's last write settled, otherwise left only "Auto backup". +func (m *Manager) carryName(older store.Snapshot, newer *store.Snapshot) { + if newer.Comment == defaultAutoComment && older.Comment != "" && older.Comment != defaultAutoComment { + if m.Store.SetSnapshotComment(newer.ID, older.Comment) == nil { + newer.Comment = older.Comment + } + } + if newer.Note == "" && older.Note != "" { + if m.Store.SetSnapshotNote(newer.ID, older.Note) == nil { + newer.Note = older.Note + } + } +} diff --git a/internal/snapshot/content_test.go b/internal/snapshot/content_test.go new file mode 100644 index 00000000..ab4db8ee --- /dev/null +++ b/internal/snapshot/content_test.go @@ -0,0 +1,259 @@ +package snapshot + +import ( + "os" + "path/filepath" + "testing" + "time" +) + +// The content a snapshot records and the content read back from its archive +// are named alike — which is what lets a device that only has another's +// archive, or only its file list, tell they hold the same save. +func TestContentKey_FromFilesAndFromArchiveAgree(t *testing.T) { + env := setup(t) + writeSave(t, env.saveDir, "slot1.sav", "one") + writeSave(t, env.saveDir, "sub/slot2.sav", "two") + snap, err := env.mgr.Create("game1", "", true) + if err != nil { + t.Fatal(err) + } + files, err := env.store.SnapshotFiles(snap.ID) + if err != nil || len(files) == 0 { + t.Fatalf("no file list recorded: %v", err) + } + fromArchive, err := ContentKeyOfArchive(snap.ZipPath) + if err != nil { + t.Fatal(err) + } + if got := ContentKey(files); got != fromArchive { + t.Errorf("from the file list %s, from the archive %s", got, fromArchive) + } + stored, _ := env.store.GetSnapshot(snap.ID) + if stored.ContentHash != fromArchive { + t.Errorf("recorded content %q, want %q", stored.ContentHash, fromArchive) + } +} + +// Steps on the way to a newer snapshot — every file of them in it, unchanged — +// go. A newer snapshot that lacks files (something was deleted) does not make +// the older one redundant; pinned and hand-taken snapshots are never removed. +func TestPruneContained(t *testing.T) { + env := setup(t) + game, err := env.store.GetGame("game1") + if err != nil { + t.Fatal(err) + } + game.MaxSnapshots = 50 + if err := env.store.UpdateGame(game); err != nil { + t.Fatal(err) + } + snap := func(auto bool) string { + t.Helper() + s, err := env.mgr.Create("game1", "", auto) + if err != nil { + t.Fatal(err) + } + return s.ID + } + writeSave(t, env.saveDir, "a.bin", "a") + step1 := snap(true) + writeSave(t, env.saveDir, "b.bin", "b") + step2 := snap(true) + pinnedStep := snap(false) // same files as step2, by hand, then pinned below + _ = env.store.SetSnapshotPinned(pinnedStep, true) + writeSave(t, env.saveDir, "c.bin", "c") + full := snap(true) + // Then a file is deleted on purpose. + if err := os.Remove(filepath.Join(env.saveDir, "c.bin")); err != nil { + t.Fatal(err) + } + writeSave(t, env.saveDir, "a.bin", "a, played on") + after := snap(true) + + removed, _, err := env.mgr.PruneContained("game1") + if err != nil { + t.Fatal(err) + } + gone := func(id string) bool { _, err := env.store.GetSnapshot(id); return err != nil } + if !gone(step1) || !gone(step2) { + t.Errorf("steps on the way to a newer snapshot kept: step1 gone=%v step2 gone=%v", gone(step1), gone(step2)) + } + if gone(pinnedStep) { + t.Error("a pinned snapshot was removed") + } + if gone(full) { + t.Error("removed a snapshot holding a file deleted later on purpose") + } + if gone(after) { + t.Error("removed the newest snapshot") + } + if removed != 2 { + t.Errorf("removed %d, want 2", removed) + } +} + +// With PruneOnCreate, a new snapshot takes the place of older automatic ones it +// holds entirely, as it is taken; one that lacks a file of them does not. +func TestPruneContainedBy_ANewSnapshotReplacesItsSteps(t *testing.T) { + env := setup(t) + env.mgr.PruneOnCreate = true + game, _ := env.store.GetGame("game1") + game.MaxSnapshots = 50 + _ = env.store.UpdateGame(game) + + writeSave(t, env.saveDir, "a.bin", "a") + first, err := env.mgr.Create("game1", "", true) + if err != nil { + t.Fatal(err) + } + writeSave(t, env.saveDir, "b.bin", "b") + if _, err := env.mgr.Create("game1", "", true); err != nil { + t.Fatal(err) + } + deadline := time.Now().Add(5 * time.Second) + for time.Now().Before(deadline) { + if _, err := env.store.GetSnapshot(first.ID); err != nil { + break + } + time.Sleep(50 * time.Millisecond) + } + if _, err := env.store.GetSnapshot(first.ID); err == nil { + t.Fatal("the step the new snapshot holds entirely was kept") + } + + // Now a file goes: the newer snapshot does not hold the older one. + before, _ := env.store.ListSnapshots("game1", "main") + if err := os.Remove(filepath.Join(env.saveDir, "b.bin")); err != nil { + t.Fatal(err) + } + if _, err := env.mgr.Create("game1", "", true); err != nil { + t.Fatal(err) + } + time.Sleep(500 * time.Millisecond) + if _, err := env.store.GetSnapshot(before[0].ID); err != nil { + t.Error("removed the snapshot holding a file deleted afterwards") + } +} + +// A copy taken before a sync replaces a save already archived is that archive, +// not a second one. +func TestBeforeReplacing_AnArchivedSaveIsNotArchivedAgain(t *testing.T) { + env := setup(t) + writeSave(t, env.saveDir, "slot1.sav", "progress") + first, err := env.mgr.Create("game1", "", true) + if err != nil { + t.Fatal(err) + } + copy, err := env.mgr.CreateBeforeReplacing("game1", "before sync") + if err != nil { + t.Fatal(err) + } + if copy.ID != first.ID { + t.Errorf("a second archive %s of the save %s already holds", copy.ID, first.ID) + } + snaps, _ := env.store.ListSnapshots("game1", "main") + if len(snaps) != 1 { + t.Errorf("%d snapshots, want 1", len(snaps)) + } +} + +// Several snapshots of the same files — one per device the save was compared +// with — are kept once: the newest unless one is pinned or taken by hand; +// pinned ones are never removed; the others' ids still lead to it. +func TestMergeDuplicates(t *testing.T) { + env := setup(t) + // Room for every snapshot below: this is about merging, not retention. + game, err := env.store.GetGame("game1") + if err != nil { + t.Fatal(err) + } + game.MaxSnapshots = 50 + if err := env.store.UpdateGame(game); err != nil { + t.Fatal(err) + } + writeSave(t, env.saveDir, "slot1.sav", "the save") + var same []string + for i := 0; i < 3; i++ { + s, err := env.mgr.Create("game1", "", true) + if err != nil { + t.Fatal(err) + } + same = append(same, s.ID) + } + pinned, err := env.mgr.Create("game1", "pinned copy", true) + if err != nil { + t.Fatal(err) + } + if err := env.store.SetSnapshotPinned(pinned.ID, true); err != nil { + t.Fatal(err) + } + writeSave(t, env.saveDir, "slot1.sav", "played on") + other, err := env.mgr.Create("game1", "", true) + if err != nil { + t.Fatal(err) + } + // As if taken before contents were recorded. + for _, id := range append(same, pinned.ID, other.ID) { + _ = env.store.SetSnapshotContentHash(id, "") + } + + merged, _, err := env.mgr.MergeDuplicates("game1") + if err != nil { + t.Fatal(err) + } + if merged != 3 { + t.Errorf("merged %d, want the 3 unpinned copies of the pinned one", merged) + } + if _, err := env.store.GetSnapshot(pinned.ID); err != nil { + t.Error("the pinned snapshot was removed") + } + if _, err := env.store.GetSnapshot(other.ID); err != nil { + t.Error("a snapshot of different content was removed") + } + for _, id := range same { + if got, ok := env.store.ResolveSnapshotID(id); !ok || got != pinned.ID { + t.Errorf("%s leads to %q (%v), want the snapshot kept, %s", id, got, ok, pinned.ID) + } + } + // Run again: nothing more to do. + if again, _, _ := env.mgr.MergeDuplicates("game1"); again != 0 { + t.Errorf("a second pass merged %d more", again) + } +} + +// A session's named snapshot, then the watcher's copy of the same save a +// moment later: the copy takes its place and keeps its name. +func TestPruneContainedBy_AnIdenticalCopyKeepsTheName(t *testing.T) { + env := setup(t) + env.mgr.PruneOnCreate = true + writeSave(t, env.saveDir, "slot1.sav", "after an evening") + named, err := env.mgr.Create("game1", "After playing (1 h)", true) + if err != nil { + t.Fatal(err) + } + if err := env.store.SetSnapshotNote(named.ID, "boss beaten"); err != nil { + t.Fatal(err) + } + copied, err := env.mgr.Create("game1", "", true) + if err != nil { + t.Fatal(err) + } + if copied.ID == named.ID { + t.Skip("the copy was not taken: nothing to replace") + } + deadline := time.Now().Add(5 * time.Second) + for time.Now().Before(deadline) { + if _, err := env.store.GetSnapshot(named.ID); err != nil { + break + } + time.Sleep(50 * time.Millisecond) + } + kept, err := env.store.GetSnapshot(copied.ID) + if err != nil { + t.Fatal(err) + } + if kept.Comment != "After playing (1 h)" || kept.Note != "boss beaten" { + t.Errorf("the snapshot kept is called %q (note %q), want the name and note of the one it replaced", kept.Comment, kept.Note) + } +} diff --git a/internal/snapshot/reuse.go b/internal/snapshot/reuse.go new file mode 100644 index 00000000..90da079e --- /dev/null +++ b/internal/snapshot/reuse.go @@ -0,0 +1,124 @@ +package snapshot + +import ( + "archive/zip" + "os" + "sync/atomic" + + "github.com/opensave/opensave/internal/delta" +) + +// Snapshots that re-pack only what changed. +// +// Every snapshot is a complete zip, restorable on its own — that stays. What +// changes is how it is written: a file whose content is exactly what the +// previous snapshot of the same branch holds is copied into the new archive +// as its already-compressed entry (zip.Writer.Copy), not read and deflated +// again. Only new and changed files are compressed. +// +// For a save of a quarter-million small files (Project Zomboid map chunks) +// where a play session touches a few hundred, that turns a snapshot from a +// minute of reading and deflating 0.76GB into writing mostly copied bytes. +// +// Content is identified by SHA-256 — the hash each snapshot already records +// per file (store.SnapshotFiles) and the one the manifest's hash cache holds, +// so an unchanged file is recognised without being read. Size is checked as +// well, cheaply, before the hash is asked for. +type reuseSource struct { + zr *zip.ReadCloser + files map[string]*zip.File // entry name -> previous entry + hashes map[string]string // entry name -> sha256 recorded for it + reused int // entries copied, for tests +} + +// lastReused is how many entries the most recent snapshot copied from its +// predecessor. Tests read it; nothing else does. +var lastReused atomic.Int64 + +// openReuseSource prepares the newest snapshot of a branch for reuse. Any +// problem — no snapshot, no recorded file list (an older snapshot, or a +// mirror), a zip that does not open — means no reuse, never a failed snapshot. +func (m *Manager) openReuseSource(gameID, branch string) *reuseSource { + prev, err := m.LatestSnapshot(gameID, branch) + if err != nil || prev.ZipPath == "" { + return nil + } + recorded, err := m.Store.SnapshotFiles(prev.ID) + if err != nil || len(recorded) == 0 { + return nil + } + zr, err := zip.OpenReader(prev.ZipPath) + if err != nil { + return nil + } + r := &reuseSource{ + zr: zr, + files: make(map[string]*zip.File, len(zr.File)), + hashes: make(map[string]string, len(recorded)), + } + for _, f := range zr.File { + r.files[f.Name] = f + } + for _, c := range recorded { + name := c.Path + if c.Root != "" { + name = RootPrefix + c.Root + "/" + c.Path + } + r.hashes[name] = c.Hash + } + return r +} + +func (r *reuseSource) Close() { + if r == nil { + lastReused.Store(0) + return + } + lastReused.Store(int64(r.reused)) + if r.zr != nil { + r.zr.Close() + } +} + +// addFileEntry adds filePath under entryName, copying the previous +// snapshot's entry when the content is the same, and archiving it afresh +// otherwise. Returns the file's sha256 either way. A nil receiver archives +// afresh, as every snapshot did before. +func (r *reuseSource) addFileEntry(w *zip.Writer, filePath, entryName string, info os.FileInfo) (string, error) { + if r != nil { + if hash, ok := r.reusable(filePath, entryName, info); ok { + if err := w.Copy(r.files[entryName]); err != nil { + return "", err + } + r.reused++ + return hash, nil + } + } + return addFileEntry(w, filePath, entryName) +} + +// reusable reports whether the previous snapshot holds exactly this file's +// current content under the same name. +func (r *reuseSource) reusable(filePath, entryName string, info os.FileInfo) (string, bool) { + prevHash, ok := r.hashes[entryName] + prev := r.files[entryName] + if !ok || prev == nil || prev.Mode().IsDir() { + return "", false + } + if info == nil { + var err error + if info, err = os.Stat(filePath); err != nil { + return "", false + } + } + if uint64(info.Size()) != prev.UncompressedSize64 { + return "", false + } + // From the hash cache when the file is unchanged since it was last + // hashed, which after the watcher's manifest build it nearly always is. + entry, err := delta.FileEntryForInfo(filePath, info) + if err != nil || entry.Hash != prevHash { + return "", false + } + return prevHash, true +} diff --git a/internal/snapshot/reuse_bench_test.go b/internal/snapshot/reuse_bench_test.go new file mode 100644 index 00000000..641a0335 --- /dev/null +++ b/internal/snapshot/reuse_bench_test.go @@ -0,0 +1,61 @@ +package snapshot + +import ( + "archive/zip" + "os" + "path/filepath" + "testing" + "time" + + "github.com/opensave/opensave/internal/delta" +) + +// TestBench_SnapshotReuse times a full snapshot of a real save folder and a +// second one that reuses it. Opt-in: OPENSAVE_BENCH_SAVE=. +// Reads the folder only; archives go to a temp dir. +func TestBench_SnapshotReuse(t *testing.T) { + src := os.Getenv("OPENSAVE_BENCH_SAVE") + if src == "" { + t.Skip("set OPENSAVE_BENCH_SAVE to run") + } + dir := t.TempDir() + + first := filepath.Join(dir, "first.zip") + start := time.Now() + skipped, captured, err := zipRootsCapturing(src, nil, first, nil) + if err != nil { + t.Fatal(err) + } + full := time.Since(start) + t.Logf("full snapshot skipped %d unreadable files", len(skipped)) + + // As in the app: the watcher's manifest build has hashed the save just + // before a snapshot is taken, so the cache is warm. + if _, err := delta.BuildManifest(src); err != nil { + t.Fatal(err) + } + + zr, err := zip.OpenReader(first) + if err != nil { + t.Fatal(err) + } + r := &reuseSource{zr: zr, files: map[string]*zip.File{}, hashes: map[string]string{}} + for _, f := range zr.File { + r.files[f.Name] = f + } + for _, c := range captured { + r.hashes[c.Path] = c.Hash + } + second := filepath.Join(dir, "second.zip") + start = time.Now() + if _, _, err := zipRootsCapturing(src, nil, second, r); err != nil { + t.Fatal(err) + } + incr := time.Since(start) + r.Close() + + fi1, _ := os.Stat(first) + fi2, _ := os.Stat(second) + t.Logf("%d files: full snapshot %s (%d MB), reusing snapshot %s (%d MB), %d entries copied", + len(captured), full.Round(time.Millisecond), fi1.Size()>>20, incr.Round(time.Millisecond), fi2.Size()>>20, lastReused.Load()) +} diff --git a/internal/snapshot/reuse_test.go b/internal/snapshot/reuse_test.go new file mode 100644 index 00000000..104a8efd --- /dev/null +++ b/internal/snapshot/reuse_test.go @@ -0,0 +1,69 @@ +package snapshot + +import ( + "fmt" + "testing" +) + +// A snapshot copies the files that did not change from the previous one, and +// still restores to exactly the save it was taken of. +func TestSnapshotReusesUnchangedFiles(t *testing.T) { + env := setup(t) + for i := 0; i < 50; i++ { + write(t, env.saveDir, fmt.Sprintf("map/chunk_%02d.bin", i), fmt.Sprintf("chunk %d %s", i, "................................................")) + } + first, err := env.mgr.Create("game1", "first", false) + if err != nil { + t.Fatal(err) + } + if got := lastReused.Load(); got != 0 { + t.Errorf("first snapshot reused %d entries from nothing", got) + } + + // A session changes two chunks and adds one. + write(t, env.saveDir, "map/chunk_03.bin", "changed three") + write(t, env.saveDir, "map/chunk_40.bin", "changed forty") + write(t, env.saveDir, "map/chunk_99.bin", "new") + second, err := env.mgr.Create("game1", "second", false) + if err != nil { + t.Fatal(err) + } + if got := lastReused.Load(); got != 48 { + t.Errorf("second snapshot copied %d unchanged entries, want 48", got) + } + + // Each snapshot restores its own state, independently of the other. + write(t, env.saveDir, "map/chunk_03.bin", "ruined") + if _, err := env.mgr.Restore("game1", second.ID); err != nil { + t.Fatal(err) + } + for name, want := range map[string]string{ + "map/chunk_03.bin": "changed three", + "map/chunk_40.bin": "changed forty", + "map/chunk_99.bin": "new", + "map/chunk_07.bin": "chunk 7 ................................................", + } { + if got := read(env.saveDir, name); got != want { + t.Errorf("after restoring the second snapshot, %s = %q, want %q", name, got, want) + } + } + if _, err := env.mgr.Restore("game1", first.ID); err != nil { + t.Fatal(err) + } + if got := read(env.saveDir, "map/chunk_03.bin"); got != "chunk 3 ................................................" { + t.Errorf("after restoring the first snapshot, chunk_03 = %q", got) + } + if got := read(env.saveDir, "map/chunk_99.bin"); got != "" { + t.Errorf("restoring the first snapshot left the later file: %q", got) + } + + // And the recorded contents of the second snapshot are right for the + // copied entries too. + files, err := env.store.SnapshotFiles(second.ID) + if err != nil { + t.Fatal(err) + } + if len(files) != 51 { + t.Errorf("second snapshot records %d files, want 51", len(files)) + } +} diff --git a/internal/snapshot/shared_test.go b/internal/snapshot/shared_test.go index f125591e..73049548 100644 --- a/internal/snapshot/shared_test.go +++ b/internal/snapshot/shared_test.go @@ -558,7 +558,9 @@ func TestSwitchingToABranchWhoseNewestIsCompacted(t *testing.T) { env := setup(t) snaps, originals := threeSnapshots(t, env) // Over to another branch: the switch keeps the save on main first, and - // that copy is main's newest. + // that copy is main's newest. Changed first, or the copy would hold the + // same files as the third snapshot, and be that snapshot (content.go). + writeSave(t, env.saveDir, "slot_switch.sav", "before the switch") if _, err := env.mgr.CreateBranch("game1", "other", true); err != nil { t.Fatal(err) } diff --git a/internal/snapshot/snapshot.go b/internal/snapshot/snapshot.go index 77d0cf6f..8a82e379 100644 --- a/internal/snapshot/snapshot.go +++ b/internal/snapshot/snapshot.go @@ -41,6 +41,18 @@ import ( // before starting one. type UploadHook func(zipPath, remoteFileName string) +// IncompleteError is a snapshot discarded because files of the save could not +// be read — in use by the game, or changing while it was archived. +type IncompleteError struct { + Game string + Skipped []string +} + +func (e *IncompleteError) Error() string { + return fmt.Sprintf("snapshot of %q discarded: %d file(s) could not be read (e.g. %s) — taken again at the next chance", + e.Game, len(e.Skipped), e.Skipped[0]) +} + // Manager performs snapshot/branch operations against the store and // filesystem. type Manager struct { @@ -80,6 +92,10 @@ type Manager struct { // running past shutdown it wrote into a directory that was being deleted // and recorded against a database that was already closed. inFlight drain.Group + // PruneOnCreate checks each new snapshot against the older ones of its + // branch and removes those it holds entirely (content.go, + // PruneContainedBy). Off unless set: the daemon sets it. + PruneOnCreate bool // sharedMu is held by a compaction and by the clean-up of shared files // (shared.go): a compaction names shared files before any list does, and // a clean-up running then would take them for unused. pendingRoots are @@ -236,20 +252,48 @@ func (m *Manager) createOnBranchFrom(gameID, branch, from, comment string, isSys } else { src = from } - skipped, captured, err := ZipRootsCapturing(src, extraRoots, stagingPath) + // The game's own save, onto its branch: unchanged files are copied from + // the branch's previous snapshot rather than compressed again (reuse.go). + var reuse *reuseSource + if from == "" { + reuse = m.openReuseSource(gameID, branch) + } + skipped, captured, err := zipRootsCapturing(src, extraRoots, stagingPath, reuse) + reuse.Close() if err != nil { os.Remove(stagingPath) return store.Snapshot{}, fmt.Errorf("zip save data: %w", err) } defer os.Remove(stagingPath) // no-op once renamed - if len(skipped) > 0 && m.Log != nil { - sample := skipped[0] - m.Log("warn", fmt.Sprintf("snapshot of %q skipped %d unreadable file(s), e.g. %s", game.Name, len(skipped), sample)) + // A snapshot missing files is not kept: it would stand in the history as + // a save that never existed — and, counted against retention, push out a + // complete one. The caller tries again at the next chance (the watcher + // leaves the change unrecorded, so its next pass takes it again; a sync + // that needed the copy before replacing files does not replace them). + if len(skipped) > 0 { + return store.Snapshot{}, &IncompleteError{Game: game.Name, Skipped: skipped} + } + + // A copy taken before a sync replaces the save, of files a snapshot on this + // branch already holds: that one is the copy, and the save is not archived + // again (content.go). Syncs took one of these each time, mostly of a save + // already archived. Only when that snapshot is the branch's newest: a + // branch switch puts back the newest, and retention removes older ones + // first — an older copy standing in would leave the save behind either way. Snapshots the + // game's changes or a person ask for are always taken: the watcher takes + // them only when something changed, and a person asked. + contentKey := ContentKey(captured) + if existing, ok, _ := m.Store.SnapshotByContent(gameID, branch, contentKey); !current && ok && ArchiveExists(existing.ZipPath) && + m.isNewest(gameID, branch, existing.ID) { + if m.Log != nil { + m.Log("info", fmt.Sprintf("%q holds the same files as snapshot %s; not archived again", game.Name, existing.ID)) + } + return existing, nil } if comment == "" { if isSystemAuto { - comment = "Auto backup" + comment = defaultAutoComment } else { comment = "Manual snapshot" } @@ -260,6 +304,11 @@ func (m *Manager) createOnBranchFrom(gameID, branch, from, comment string, isSys return store.Snapshot{}, err } snapshotID := snap.ID + _ = m.Store.SetSnapshotContentHash(snapshotID, contentKey) + snap.ContentHash = contentKey + // Older automatic snapshots this one holds entirely are steps on the way + // to it (content.go). + defer m.PruneContainedBy(snap, captured) // What this snapshot holds, file by file, recorded once and never // revisited. It is what lets a later caller ask whether some exact content @@ -730,7 +779,18 @@ func (m *Manager) Restore(gameID, snapshotID string) (store.Snapshot, error) { if err != nil { return store.Snapshot{}, err } + // An id kept as an alias of the snapshot holding the same files + // (content.go) restores that snapshot. + if resolved, ok := m.Store.ResolveSnapshotID(snapshotID); ok { + snapshotID = resolved + } snap, err := m.Store.GetSnapshot(snapshotID) + if err != nil { + // Removed in favour of the same files just now: its alias is there. + if resolved, ok := m.Store.ResolveSnapshotID(snapshotID); ok { + snap, err = m.Store.GetSnapshot(resolved) + } + } if err != nil || snap.GameID != gameID { return store.Snapshot{}, fmt.Errorf("snapshot %q not found for game %q", snapshotID, gameID) } diff --git a/internal/snapshot/zip.go b/internal/snapshot/zip.go index 1ea2920c..b60ea395 100644 --- a/internal/snapshot/zip.go +++ b/internal/snapshot/zip.go @@ -8,6 +8,7 @@ import ( "encoding/hex" "fmt" "io" + "io/fs" "os" "path/filepath" "strings" @@ -38,6 +39,10 @@ func ZipPath(sourcePath, outPath string) (skipped []string, err error) { // archived so the caller can record what the snapshot holds. The hashes come // free: the bytes pass through a hasher on their way into the archive. func ZipPathCapturing(sourcePath, outPath string) (skipped []string, captured []store.CapturedFile, err error) { + return zipPathCapturing(sourcePath, outPath, nil) +} + +func zipPathCapturing(sourcePath, outPath string, reuse *reuseSource) (skipped []string, captured []store.CapturedFile, err error) { info, err := os.Stat(sourcePath) if err != nil { return nil, nil, fmt.Errorf("source path does not exist: %w", err) @@ -54,7 +59,7 @@ func ZipPathCapturing(sourcePath, outPath string) (skipped []string, captured [] if !info.IsDir() { name := filepath.Base(sourcePath) - hash, addErr := addFileEntry(w, sourcePath, name) + hash, addErr := reuse.addFileEntry(w, sourcePath, name, info) if addErr != nil { return nil, nil, addErr } @@ -62,7 +67,10 @@ func ZipPathCapturing(sourcePath, outPath string) (skipped []string, captured [] } archived := 0 - walkErr := filepath.Walk(sourcePath, func(path string, walkInfo os.FileInfo, walkErr error) error { + // WalkDir, not Walk: Walk's extra Lstat per entry opens every file on + // Windows, which on a save of a quarter-million files cost more than the + // archiving itself once unchanged files are copied (reuse.go). + walkErr := filepath.WalkDir(sourcePath, func(path string, d fs.DirEntry, walkErr error) error { if walkErr != nil { // Unreadable subtree — record it and move on. if path != sourcePath { @@ -80,14 +88,24 @@ func ZipPathCapturing(sourcePath, outPath string) (skipped []string, captured [] } rel = filepath.ToSlash(rel) - if walkInfo.IsDir() { + if d.IsDir() { // Explicit directory entries keep empty dirs restorable. if _, err := w.CreateHeader(&zip.FileHeader{Name: rel + "/", Method: zip.Store}); err != nil { return err } return nil } - hash, addErr := addFileEntry(w, path, rel) + // OpenSave's own temporary file for a file being pulled: never part + // of the save, and gone by the time it is read. + if strings.HasSuffix(rel, delta.TmpSuffix) || delta.NeverSynced(rel) { + return nil + } + info, infoErr := d.Info() + if infoErr != nil { + skipped = append(skipped, path) + return nil + } + hash, addErr := reuse.addFileEntry(w, path, rel, info) if addErr != nil { skipped = append(skipped, path) return nil diff --git a/internal/snapshot/zip_roots.go b/internal/snapshot/zip_roots.go index 8f70ff2f..5100f6ea 100644 --- a/internal/snapshot/zip_roots.go +++ b/internal/snapshot/zip_roots.go @@ -3,6 +3,7 @@ package snapshot import ( "archive/zip" "fmt" + "io/fs" "os" "path/filepath" "sort" @@ -44,8 +45,14 @@ func ZipRoots(primary string, extra map[string]string, outPath string) (skipped // instead of inferred from a single whole-save hash that nothing reliably // kept current. func ZipRootsCapturing(primary string, extra map[string]string, outPath string) (skipped []string, captured []store.CapturedFile, err error) { + return zipRootsCapturing(primary, extra, outPath, nil) +} + +// zipRootsCapturing is ZipRootsCapturing, copying unchanged files straight +// out of reuse (the previous snapshot) when it is given. See reuse.go. +func zipRootsCapturing(primary string, extra map[string]string, outPath string, reuse *reuseSource) (skipped []string, captured []store.CapturedFile, err error) { if len(extra) == 0 { - return ZipPathCapturing(primary, outPath) + return zipPathCapturing(primary, outPath, reuse) } // One archive, written in a single pass. @@ -69,7 +76,7 @@ func ZipRootsCapturing(primary string, extra map[string]string, outPath string) defer f.Close() w := newSnapshotWriter(f) - primarySkipped, primaryFiles, err := archiveInto(w, primary, "", "") + primarySkipped, primaryFiles, err := archiveInto(w, primary, "", "", reuse) skipped = append(skipped, primarySkipped...) captured = append(captured, primaryFiles...) if err != nil { @@ -82,7 +89,7 @@ func ZipRootsCapturing(primary string, extra map[string]string, outPath string) if strings.TrimSpace(path) == "" { continue } - sub, subFiles, subErr := archiveInto(w, path, RootPrefix+name+"/", name) + sub, subFiles, subErr := archiveInto(w, path, RootPrefix+name+"/", name, reuse) skipped = append(skipped, sub...) captured = append(captured, subFiles...) if subErr != nil { @@ -94,14 +101,14 @@ func ZipRootsCapturing(primary string, extra map[string]string, outPath string) } // archiveInto writes one directory tree into an open zip under prefix. -func archiveInto(w *zip.Writer, sourcePath, prefix, root string) (skipped []string, captured []store.CapturedFile, err error) { +func archiveInto(w *zip.Writer, sourcePath, prefix, root string, reuse *reuseSource) (skipped []string, captured []store.CapturedFile, err error) { info, err := os.Stat(sourcePath) if err != nil { return nil, nil, err } if !info.IsDir() { name := filepath.Base(sourcePath) - hash, addErr := addFileEntry(w, sourcePath, prefix+name) + hash, addErr := reuse.addFileEntry(w, sourcePath, prefix+name, info) if addErr != nil { return nil, nil, addErr } @@ -117,7 +124,7 @@ func archiveInto(w *zip.Writer, sourcePath, prefix, root string) (skipped []stri } } - walkErr := filepath.Walk(sourcePath, func(path string, walkInfo os.FileInfo, walkErr error) error { + walkErr := filepath.WalkDir(sourcePath, func(path string, d fs.DirEntry, walkErr error) error { if walkErr != nil { if path != sourcePath { skipped = append(skipped, path) @@ -133,11 +140,20 @@ func archiveInto(w *zip.Writer, sourcePath, prefix, root string) (skipped []stri return err } rel = filepath.ToSlash(rel) - if walkInfo.IsDir() { + if d.IsDir() { _, err := w.CreateHeader(&zip.FileHeader{Name: prefix + rel + "/", Method: zip.Store}) return err } - hash, addErr := addFileEntry(w, path, prefix+rel) + // OpenSave's own temporary file for a file being pulled (zip.go). + if strings.HasSuffix(rel, delta.TmpSuffix) || delta.NeverSynced(rel) { + return nil + } + info, infoErr := d.Info() + if infoErr != nil { + skipped = append(skipped, path) + return nil + } + hash, addErr := reuse.addFileEntry(w, path, prefix+rel, info) if addErr != nil { skipped = append(skipped, path) return nil diff --git a/internal/store/marks.go b/internal/store/marks.go new file mode 100644 index 00000000..5928757a --- /dev/null +++ b/internal/store/marks.go @@ -0,0 +1,22 @@ +package store + +import "fmt" + +// Mark returns the value a one-time step was taken for, empty if never. +// See migrations/0040_marks.sql. +func (s *Store) Mark(name string) string { + var v string + if err := s.db.Get(&v, `SELECT value FROM marks WHERE name = ?`, name); err != nil { + return "" + } + return v +} + +// SetMark records that a one-time step was taken for value. +func (s *Store) SetMark(name, value string) error { + if _, err := s.db.Exec(`INSERT INTO marks (name, value) VALUES (?, ?) + ON CONFLICT(name) DO UPDATE SET value = excluded.value`, name, value); err != nil { + return fmt.Errorf("set mark %s: %w", name, err) + } + return nil +} diff --git a/internal/store/migrations/0038_peer_links.sql b/internal/store/migrations/0038_peer_links.sql new file mode 100644 index 00000000..4cff6cfb --- /dev/null +++ b/internal/store/migrations/0038_peer_links.sql @@ -0,0 +1,15 @@ +-- How fast this device has measured its connection to each paired device: +-- from transfers it made, and from a short speed test when there is nothing +-- recent (see internal/p2p/syncengine/linkspeed.go). Syncs go to the fastest +-- connections first, so a save reaches the devices close to it before it +-- crawls over a slow relay — and those can then hand it on. +-- +-- bytes_per_sec is a running average; measured_ms when it last changed; +-- probed_ms when a speed test last ran, so a slow link is not tested over and +-- over. +CREATE TABLE peer_links ( + peer_id TEXT PRIMARY KEY, + bytes_per_sec REAL NOT NULL DEFAULT 0, + measured_ms INTEGER NOT NULL DEFAULT 0, + probed_ms INTEGER NOT NULL DEFAULT 0 +); diff --git a/internal/store/migrations/0039_game_versions.sql b/internal/store/migrations/0039_game_versions.sql new file mode 100644 index 00000000..3c0c2ce1 --- /dev/null +++ b/internal/store/migrations/0039_game_versions.sql @@ -0,0 +1,42 @@ +-- Which version of a game's save this device holds, as a version vector, and +-- whether it knows of a newer one it has not finished taking yet (see +-- internal/p2p/syncengine/version.go). +-- +-- vector is a JSON object of writer key to counter. A writer key belongs to +-- one device for one game (key below), and its counter moves only when the +-- save really changes there — the game or the user wrote it, never a sync. +-- Comparing two vectors says, without guessing from files or clocks, whether +-- one device is simply behind the other or both changed independently. +-- +-- target is the newest version this device has heard of that is newer than +-- its own, '' when none. While it is set this device is outdated: it takes +-- the save only from a device holding that version, and is a source for +-- nobody — its files are not handed out, and what it lacks is not read as +-- deleted. +-- +-- counter is this device's own counter for the game. It never goes back, even +-- when the device adopts another's version, so the same key and number never +-- name two different saves. +-- +-- pulling is set while a pull towards target is under way, so the files it +-- writes are not mistaken for a new local version if the app stops half-way. +-- +-- hash is the save's content hash when vector was last set ('' unknown). Two +-- devices on one version holding different files means a change went +-- unnoticed on one of them; this says which. +-- +-- answered is the other device's version this one kept its own save over when +-- someone answered a conflict here ('' none), and answered_at when (Unix ms). +-- It travels with the version, so the other device is asked in turn rather +-- than overwritten, and this one is not asked again. +CREATE TABLE game_versions ( + game_id TEXT PRIMARY KEY REFERENCES games(id) ON DELETE CASCADE, + key TEXT NOT NULL, + counter INTEGER NOT NULL DEFAULT 0, + vector TEXT NOT NULL DEFAULT '{}', + target TEXT NOT NULL DEFAULT '', + pulling INTEGER NOT NULL DEFAULT 0, + hash TEXT NOT NULL DEFAULT '', + answered TEXT NOT NULL DEFAULT '', + answered_at INTEGER NOT NULL DEFAULT 0 +); diff --git a/internal/store/migrations/0040_snapshot_content.sql b/internal/store/migrations/0040_snapshot_content.sql new file mode 100644 index 00000000..46e0a102 --- /dev/null +++ b/internal/store/migrations/0040_snapshot_content.sql @@ -0,0 +1,16 @@ +-- What a snapshot holds, as one value: a hash over every file's name in the +-- archive and its content hash (snapshot.ContentKey). The same files give the +-- same value on every device, so two snapshots of one save — taken here, or +-- taken on another device and recorded here as "Synced from peer" — are known +-- to be the same, and the archive is kept once. '' until computed. +ALTER TABLE snapshots ADD COLUMN content_hash TEXT NOT NULL DEFAULT ''; + +CREATE INDEX IF NOT EXISTS idx_snapshots_content ON snapshots(game_id, branch_name, content_hash); + +-- Another snapshot id that names the same content as a snapshot kept here: a +-- peer's snapshot whose save this device already has archived. Known, so it is +-- not archived again, and not asked about again on every sync. +CREATE TABLE snapshot_aliases ( + alias_id TEXT PRIMARY KEY, + snapshot_id TEXT NOT NULL REFERENCES snapshots(id) ON DELETE CASCADE +); diff --git a/internal/store/migrations/0041_marks.sql b/internal/store/migrations/0041_marks.sql new file mode 100644 index 00000000..e9992819 --- /dev/null +++ b/internal/store/migrations/0041_marks.sql @@ -0,0 +1,8 @@ +-- One-time steps this device has already taken, by name, with the value they +-- were taken for. A step whose value changes in a later build runs again. +-- Used for delta.NeverSyncedList: when the files no save is made of change, +-- the hashes recorded under the old list are re-taken once (daemon.go). +CREATE TABLE marks ( + name TEXT PRIMARY KEY, + value TEXT NOT NULL +); diff --git a/internal/store/peer_links.go b/internal/store/peer_links.go new file mode 100644 index 00000000..d65b3a37 --- /dev/null +++ b/internal/store/peer_links.go @@ -0,0 +1,37 @@ +package store + +import "fmt" + +// PeerLink is how fast this device has measured its connection to a peer. +// See migrations/0038_peer_links.sql. +type PeerLink struct { + PeerID string `db:"peer_id"` + BytesPerSec float64 `db:"bytes_per_sec"` + MeasuredMs int64 `db:"measured_ms"` + ProbedMs int64 `db:"probed_ms"` +} + +// PeerLinks returns every recorded link, by peer. +func (s *Store) PeerLinks() (map[string]PeerLink, error) { + var rows []PeerLink + if err := s.db.Select(&rows, `SELECT * FROM peer_links`); err != nil { + return nil, fmt.Errorf("list peer links: %w", err) + } + out := make(map[string]PeerLink, len(rows)) + for _, r := range rows { + out[r.PeerID] = r + } + return out, nil +} + +// SavePeerLink records a peer's link, replacing what was there. +func (s *Store) SavePeerLink(l PeerLink) error { + _, err := s.db.Exec(`INSERT INTO peer_links (peer_id, bytes_per_sec, measured_ms, probed_ms) VALUES (?, ?, ?, ?) + ON CONFLICT(peer_id) DO UPDATE SET bytes_per_sec = excluded.bytes_per_sec, + measured_ms = excluded.measured_ms, probed_ms = excluded.probed_ms`, + l.PeerID, l.BytesPerSec, l.MeasuredMs, l.ProbedMs) + if err != nil { + return fmt.Errorf("save peer link %s: %w", l.PeerID, err) + } + return nil +} diff --git a/internal/store/snapshot_content.go b/internal/store/snapshot_content.go new file mode 100644 index 00000000..f9df1be2 --- /dev/null +++ b/internal/store/snapshot_content.go @@ -0,0 +1,92 @@ +package store + +import ( + "database/sql" + "errors" + "fmt" +) + +// Snapshot content: which snapshots hold the same files, so a save is +// archived once however many times — and on however many devices — it is +// snapshotted. See migrations/0039_snapshot_content.sql. + +// SetSnapshotContentHash records what a snapshot holds. +func (s *Store) SetSnapshotContentHash(id, hash string) error { + if _, err := s.db.Exec(`UPDATE snapshots SET content_hash = ? WHERE id = ?`, hash, id); err != nil { + return fmt.Errorf("set content hash of %s: %w", id, err) + } + return nil +} + +// SnapshotByContent finds a snapshot of a game's branch holding exactly the +// content hash names; ok is false when there is none. The newest, so a copy +// kept in its place is the one retention keeps longest. +func (s *Store) SnapshotByContent(gameID, branch, hash string) (snap Snapshot, ok bool, err error) { + if hash == "" { + return Snapshot{}, false, nil + } + err = s.db.Get(&snap, `SELECT * FROM snapshots WHERE game_id = ? AND branch_name = ? AND content_hash = ? + ORDER BY timestamp DESC LIMIT 1`, gameID, branch, hash) + if errors.Is(err, sql.ErrNoRows) { + return Snapshot{}, false, nil + } + if err != nil { + return Snapshot{}, false, fmt.Errorf("find snapshot by content: %w", err) + } + return snap, true, nil +} + +// AddSnapshotAlias records that alias (another device's snapshot id) names the +// same content as snapshotID here. +func (s *Store) AddSnapshotAlias(alias, snapshotID string) error { + if alias == "" || alias == snapshotID { + return nil + } + if _, err := s.db.Exec(`INSERT INTO snapshot_aliases (alias_id, snapshot_id) VALUES (?, ?) + ON CONFLICT(alias_id) DO UPDATE SET snapshot_id = excluded.snapshot_id`, alias, snapshotID); err != nil { + return fmt.Errorf("add snapshot alias %s: %w", alias, err) + } + return nil +} + +// RepointSnapshotAliases moves every alias of from onto to, before from is +// removed in favour of to. +func (s *Store) RepointSnapshotAliases(from, to string) error { + if _, err := s.db.Exec(`UPDATE snapshot_aliases SET snapshot_id = ? WHERE snapshot_id = ?`, to, from); err != nil { + return fmt.Errorf("repoint aliases of %s: %w", from, err) + } + return nil +} + +// ResolveSnapshotID says which snapshot here an id names: itself when it is +// one, the snapshot it is an alias of when it is that; ok false when neither. +func (s *Store) ResolveSnapshotID(id string) (string, bool) { + var found string + if err := s.db.Get(&found, `SELECT id FROM snapshots WHERE id = ?`, id); err == nil { + return found, true + } + if err := s.db.Get(&found, `SELECT snapshot_id FROM snapshot_aliases WHERE alias_id = ?`, id); err == nil { + return found, true + } + return "", false +} + +// SnapshotsWithoutContentHash lists a game's snapshots whose content has not +// been computed yet. +func (s *Store) SnapshotsWithoutContentHash(gameID string) ([]Snapshot, error) { + var out []Snapshot + if err := s.db.Select(&out, `SELECT * FROM snapshots WHERE game_id = ? AND content_hash = '' ORDER BY timestamp`, gameID); err != nil { + return nil, fmt.Errorf("list snapshots without content hash: %w", err) + } + return out, nil +} + +// AllSnapshotsOfGame lists every snapshot of a game, on every branch, oldest +// first. +func (s *Store) AllSnapshotsOfGame(gameID string) ([]Snapshot, error) { + var out []Snapshot + if err := s.db.Select(&out, `SELECT * FROM snapshots WHERE game_id = ? ORDER BY timestamp`, gameID); err != nil { + return nil, fmt.Errorf("list snapshots of %s: %w", gameID, err) + } + return out, nil +} diff --git a/internal/store/snapshots.go b/internal/store/snapshots.go index bf0efc23..19ed403c 100644 --- a/internal/store/snapshots.go +++ b/internal/store/snapshots.go @@ -26,13 +26,16 @@ type Snapshot struct { // if it was not (migration 0032; internal/snapshot/verify.go). CheckedMs int64 `db:"checked_ms" json:"checkedMs"` Problem string `db:"problem" json:"problem"` + // ContentHash names what the snapshot holds (snapshot.ContentKey), the + // same on every device for the same files; '' until computed. + ContentHash string `db:"content_hash" json:"contentHash"` } // CreateSnapshot inserts a new snapshot record. func (s *Store) CreateSnapshot(snap Snapshot) error { _, err := s.db.NamedExec(` - INSERT INTO snapshots (id, game_id, branch_name, timestamp, comment, is_system_auto, zip_path, size_bytes) - VALUES (:id, :game_id, :branch_name, :timestamp, :comment, :is_system_auto, :zip_path, :size_bytes)`, + INSERT INTO snapshots (id, game_id, branch_name, timestamp, comment, is_system_auto, zip_path, size_bytes, content_hash) + VALUES (:id, :game_id, :branch_name, :timestamp, :comment, :is_system_auto, :zip_path, :size_bytes, :content_hash)`, snap) if err != nil { return fmt.Errorf("create snapshot %s: %w", snap.ID, err) diff --git a/internal/store/versions.go b/internal/store/versions.go new file mode 100644 index 00000000..3803db67 --- /dev/null +++ b/internal/store/versions.go @@ -0,0 +1,71 @@ +package store + +import ( + "crypto/rand" + "database/sql" + "encoding/hex" + "errors" + "fmt" +) + +// Save versions: which version of a game's save this device holds. See +// migrations/0037_game_versions.sql for the columns and +// internal/p2p/syncengine/version.go for the rules. The vectors are kept as +// the JSON the sync engine writes; this layer does not interpret them. + +// GameVersion is one game's version record on this device. +type GameVersion struct { + GameID string `db:"game_id"` + Key string `db:"key"` + Counter int64 `db:"counter"` + Vector string `db:"vector"` + Target string `db:"target"` + Pulling bool `db:"pulling"` + Hash string `db:"hash"` + // Answered is the JSON vector this device kept its save over in a + // conflict, AnsweredAt when. + Answered string `db:"answered"` + AnsweredAt int64 `db:"answered_at"` +} + +// GetGameVersion returns a game's version record, creating an empty one — +// with a fresh writer key — the first time it is asked for. A game that is not +// tracked gets a record that is not stored. +func (s *Store) GetGameVersion(gameID string) (GameVersion, error) { + var v GameVersion + err := s.db.Get(&v, `SELECT * FROM game_versions WHERE game_id = ?`, gameID) + if err == nil { + return v, nil + } + if !errors.Is(err, sql.ErrNoRows) { + return GameVersion{}, fmt.Errorf("get game version %s: %w", gameID, err) + } + var b [8]byte + if _, err := rand.Read(b[:]); err != nil { + return GameVersion{}, err + } + v = GameVersion{GameID: gameID, Key: hex.EncodeToString(b[:]), Vector: "{}"} + // OR IGNORE: two callers creating it at once keep the first key. + if _, err := s.db.Exec(`INSERT OR IGNORE INTO game_versions (game_id, key, vector) + SELECT ?, ?, '{}' WHERE EXISTS (SELECT 1 FROM games WHERE id = ?)`, + gameID, v.Key, gameID); err != nil { + return GameVersion{}, fmt.Errorf("create game version %s: %w", gameID, err) + } + if err := s.db.Get(&v, `SELECT * FROM game_versions WHERE game_id = ?`, gameID); err != nil && + !errors.Is(err, sql.ErrNoRows) { + return GameVersion{}, fmt.Errorf("get game version %s: %w", gameID, err) + } + return v, nil +} + +// SaveGameVersion writes a game's version record. The writer key is never +// changed by this: it is the record's identity. +func (s *Store) SaveGameVersion(v GameVersion) error { + _, err := s.db.Exec(`UPDATE game_versions SET counter = ?, vector = ?, target = ?, pulling = ?, hash = ?, + answered = ?, answered_at = ? WHERE game_id = ?`, + v.Counter, v.Vector, v.Target, v.Pulling, v.Hash, v.Answered, v.AnsweredAt, v.GameID) + if err != nil { + return fmt.Errorf("save game version %s: %w", v.GameID, err) + } + return nil +} diff --git a/internal/watcher/neversynced_test.go b/internal/watcher/neversynced_test.go new file mode 100644 index 00000000..f3d19a75 --- /dev/null +++ b/internal/watcher/neversynced_test.go @@ -0,0 +1,26 @@ +package watcher + +import ( + "testing" + + "github.com/opensave/opensave/internal/delta" +) + +// Steam rewriting remotecache.vdf is not a change of the save; a value +// recorded by an older build, which counted it, is still recognisable. +func TestContentHashLeavesOutNeverSynced(t *testing.T) { + save := delta.Manifest{Files: map[string]delta.FileEntry{"save.sav": {Hash: "a"}}} + withCache := delta.Manifest{Files: map[string]delta.FileEntry{ + "save.sav": {Hash: "a"}, + "remotecache.vdf": {Hash: "steam"}, + }} + if ContentHash(withCache, "") != ContentHash(save, "") { + t.Error("remotecache.vdf counts toward the content hash") + } + if ContentHash(save, "") != save.ContentHash() { + t.Error("a save without such files hashes differently than before") + } + if ContentHashBeforeNeverSynced(withCache, "") != withCache.ContentHash() { + t.Error("the hash an older build recorded cannot be reproduced") + } +} diff --git a/internal/watcher/owntouch_test.go b/internal/watcher/owntouch_test.go new file mode 100644 index 00000000..d984bc29 --- /dev/null +++ b/internal/watcher/owntouch_test.go @@ -0,0 +1,157 @@ +package watcher + +import ( + "fmt" + "os" + "path/filepath" + "sync/atomic" + "testing" + "time" + + "github.com/opensave/opensave/internal/owntouch" +) + +// Files OpenSave writes or deletes itself — a pull, a peer's deletions — are +// not a new save. Only the recorded hash moves; no auto-snapshot, and no +// OnChanged echoing the sync back out. +func TestOwnChanges_AreNotSnapshotted(t *testing.T) { + saveDir := t.TempDir() + col := newCollector() + eng := New(col.callbacks()) + defer eng.Stop() + if err := eng.Watch("game1", saveDir); err != nil { + t.Fatal(err) + } + + // A peer's deletions arriving one at a time, as they do, plus pulled files. + for i := 0; i < 5; i++ { + p := filepath.Join(saveDir, fmt.Sprintf("map_%d.bin", i)) + owntouch.Mark(p) + if err := os.WriteFile(p, []byte(fmt.Sprint(i)), 0o666); err != nil { + t.Fatal(err) + } + time.Sleep(300 * time.Millisecond) + } + // Past the debounce, with room to spare. + if !waitFor(t, 8*time.Second, func() bool { + col.mu.Lock() + defer col.mu.Unlock() + return col.manifestHashes["game1"] != "" + }) { + t.Fatal("the recorded hash was not moved to the new content") + } + time.Sleep(3 * time.Second) + if n := col.snapshotCount(); n != 0 { + t.Errorf("%d auto-snapshot(s) for changes OpenSave made itself, want 0", n) + } + if n := col.changedCount(); n != 0 { + t.Errorf("OnChanged fired %d time(s) for changes OpenSave made itself", n) + } + + // The game then saves: that is snapshotted, as always. + if err := os.WriteFile(filepath.Join(saveDir, "slot1.sav"), []byte("played"), 0o666); err != nil { + t.Fatal(err) + } + if !waitFor(t, 10*time.Second, func() bool { return col.snapshotCount() >= 1 }) { + t.Fatal("a change the game made was not snapshotted") + } +} + +// A pull that creates folders and writes into them before the watcher has put +// them under watch is still the pull: the rescan that follows registering +// them is not someone else's change. +func TestOwnNewFolders_AreNotSnapshotted(t *testing.T) { + saveDir := t.TempDir() + col := newCollector() + eng := New(col.callbacks()) + defer eng.Stop() + if err := eng.Watch("game1", saveDir); err != nil { + t.Fatal(err) + } + for i := 0; i < 20; i++ { + dir := filepath.Join(saveDir, fmt.Sprintf("slot%02d", i)) + owntouch.Mark(dir) + if err := os.MkdirAll(dir, 0o777); err != nil { + t.Fatal(err) + } + p := filepath.Join(dir, "data.sav") + owntouch.Mark(p) + if err := os.WriteFile(p, []byte(fmt.Sprint(i)), 0o666); err != nil { + t.Fatal(err) + } + } + if !waitFor(t, 10*time.Second, func() bool { + col.mu.Lock() + defer col.mu.Unlock() + return col.manifestHashes["game1"] != "" + }) { + t.Fatal("the recorded hash was not moved to the new content") + } + time.Sleep(4 * time.Second) + if n := col.snapshotCount(); n != 0 { + t.Errorf("%d auto-snapshot(s) for folders and files a pull created", n) + } + + // A folder the game creates is a change, as before. + if err := os.MkdirAll(filepath.Join(saveDir, "newslot"), 0o777); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(saveDir, "newslot", "data.sav"), []byte("played"), 0o666); err != nil { + t.Fatal(err) + } + if !waitFor(t, 10*time.Second, func() bool { return col.snapshotCount() >= 1 }) { + t.Fatal("a folder the game created was not snapshotted") + } +} + +// While a sync is writing the game, the watcher looks at what changed only +// once it has finished — never at a save half-way between two states — and a +// change of the game's is still snapshotted then. +func TestWhileASyncWrites_TheWatcherWaits(t *testing.T) { + saveDir := t.TempDir() + col := newCollector() + var writing atomic.Bool + cb := col.callbacks() + cb.SyncWriting = func(string) bool { return writing.Load() } + cb.SyncWritingNow = cb.SyncWriting + eng := New(cb) + defer eng.Stop() + if err := eng.Watch("game1", saveDir); err != nil { + t.Fatal(err) + } + writing.Store(true) + if err := os.WriteFile(filepath.Join(saveDir, "slot1.sav"), []byte("played"), 0o666); err != nil { + t.Fatal(err) + } + time.Sleep(5 * time.Second) + if n := col.snapshotCount(); n != 0 { + t.Fatalf("%d snapshot(s) while a sync was writing the game", n) + } + writing.Store(false) + if !waitFor(t, 10*time.Second, func() bool { return col.snapshotCount() >= 1 }) { + t.Fatal("the game's change was not snapshotted once the sync had finished") + } +} + +// A burst that mixes OpenSave's own changes with one from the game is a real +// change and is snapshotted. +func TestMixedChanges_AreSnapshotted(t *testing.T) { + saveDir := t.TempDir() + col := newCollector() + eng := New(col.callbacks()) + defer eng.Stop() + if err := eng.Watch("game1", saveDir); err != nil { + t.Fatal(err) + } + own := filepath.Join(saveDir, "pulled.bin") + owntouch.Mark(own) + if err := os.WriteFile(own, []byte("from a peer"), 0o666); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(saveDir, "slot1.sav"), []byte("played"), 0o666); err != nil { + t.Fatal(err) + } + if !waitFor(t, 10*time.Second, func() bool { return col.snapshotCount() >= 1 }) { + t.Fatal("a burst including the game's own change was not snapshotted") + } +} diff --git a/internal/watcher/watcher.go b/internal/watcher/watcher.go index 612ebfb6..38d6d439 100644 --- a/internal/watcher/watcher.go +++ b/internal/watcher/watcher.go @@ -32,6 +32,7 @@ import ( "github.com/opensave/opensave/internal/delta" "github.com/opensave/opensave/internal/ignore" "github.com/opensave/opensave/internal/logging" + "github.com/opensave/opensave/internal/owntouch" ) const ( @@ -118,6 +119,25 @@ type Callbacks struct { OnChanged func(gameID string) // Log receives human-readable watcher activity. May be nil. Log func(level, msg string) + // SyncWriting reports whether a sync on this device is writing the game + // now, or was a moment ago: a burst the watcher cannot attribute — an + // overflow of file events — is then taken for the sync's. May be nil. + SyncWriting func(gameID string) bool + // SyncWritingNow reports whether a sync is writing the game at this + // moment: the watcher looks at the burst once it has finished. May be nil. + SyncWritingNow func(gameID string) bool +} + +// syncWriting reports whether a sync is or was just writing a game +// (Callbacks.SyncWriting). +func (e *Engine) syncWriting(gameID string) bool { + return e.cb.SyncWriting != nil && e.cb.SyncWriting(gameID) +} + +// syncWritingNow reports whether a sync is writing a game at this moment +// (Callbacks.SyncWritingNow). +func (e *Engine) syncWritingNow(gameID string) bool { + return e.cb.SyncWritingNow != nil && e.cb.SyncWritingNow(gameID) } // Engine owns one watch goroutine per tracked game. @@ -188,6 +208,17 @@ type gameWatch struct { // registerFolders when an Add fails. rewatch atomic.Bool + // What the burst since the last snapshot check was made of: changes + // OpenSave applied itself (owntouch), anything else, or both. Read and + // written only by the event loop. + ownEvents, otherEvents bool + // foreignFolders: a folder queued for watching since the last rescan was + // not one OpenSave created, or the watch started on a busy folder. The + // rescan that follows registration counts as someone else's change only + // then — a pull creates folders and writes into them before they are + // watched, and that is still the pull. Event loop only. + foreignFolders bool + // folders carries directories to put under watch to registerFolders. See // there for why the event loop does not register them itself. folders chan string @@ -450,6 +481,11 @@ func (e *Engine) WatchWithLocations(gameID, savePath string, extra map[string]st } if busy { gw.rewatch.Store(true) + // What the folder holds is nobody's change OpenSave knows of: this + // rescan must not record it as its own, or a change made while the + // app was closed would become the baseline before the catch-up + // below could see it. Set before the event loop starts. + gw.foreignFolders = true gw.rescan <- struct{}{} // buffered and empty: cannot block } e.games[gameID] = gw @@ -676,10 +712,19 @@ func (e *Engine) run(ctx context.Context, gw *gameWatch) { if !gw.eventRelevant(event) { continue } + own := owntouch.Recent(event.Name) + if own { + gw.ownEvents = true + } else { + gw.otherEvents = true + } // New subdirectory in directory mode: extend the watch — on // registerFolders, never here. See registerFolders. if !gw.isFile && event.Has(fsnotify.Create) { if info, err := os.Stat(event.Name); err == nil && info.IsDir() { + if !own { + gw.foreignFolders = true + } gw.queueFolder(event.Name) } } @@ -689,7 +734,14 @@ func (e *Engine) run(ctx context.Context, gw *gameWatch) { // Folders just came under watch, or the watch started on a folder // that was busy. Anything written into a folder before its watch // existed raised no event of its own, so read the tree again once - // things are quiet. + // things are quiet. Whose change that is follows from whose + // folders they were: what OpenSave created, it also filled. + if gw.foreignFolders { + gw.otherEvents = true + } else { + gw.ownEvents = true + } + gw.foreignFolders = false resetDebounce() case err, ok := <-gw.fsw.Errors: @@ -718,6 +770,16 @@ func (e *Engine) run(ctx context.Context, gw *gameWatch) { // directory and allocates no second buffer for one it already // has. gw.rewatch.Store(true) + // The missed events could be anyone's — unless a sync is + // writing this game, which is what overflows the queue: a + // pull of a save of many files does, again and again, and + // each was taken for the game saving, an automatic snapshot + // of a half-pulled save every few seconds. + if e.syncWriting(gw.gameID) { + gw.ownEvents = true + } else { + gw.otherEvents = true + } } else { e.log("warn", fmt.Sprintf("watching %q: %v", gw.gameID, err)) continue @@ -727,6 +789,12 @@ func (e *Engine) run(ctx context.Context, gw *gameWatch) { case <-debounceC: debounce = nil debounceC = nil + // A sync still writing this game: look at the burst once it has + // finished, not at a save half-way between two states. + if e.syncWritingNow(gw.gameID) { + resetDebounce() + continue + } // Re-register, so a folder that went unwatched is watched from // here on, and read the tree below, so it is found now. Doing only // one of those leaves it correct today and silent tomorrow. @@ -758,7 +826,9 @@ func (e *Engine) run(ctx context.Context, gw *gameWatch) { for _, path := range gw.extra { delta.InvalidateRoot(path) } - e.handleChange(ctx, gw) + ownOnly := gw.ownEvents && !gw.otherEvents + gw.ownEvents, gw.otherEvents = false, false + e.handleChangeFrom(ctx, gw, ownOnly) } } } @@ -775,6 +845,12 @@ func (gw *gameWatch) eventRelevant(event fsnotify.Event) bool { if strings.HasSuffix(name, ".opensave.tmp") { return false } + // A file the launcher rewrites on its own (Steam's remotecache.vdf, on + // every start) is not a save: it is never synced or snapshotted, so it + // does not wake the watcher either. + if delta.NeverSynced(name) { + return false + } if gw.isFile { return strings.EqualFold( filepath.Clean(event.Name), @@ -788,6 +864,22 @@ func (gw *gameWatch) eventRelevant(event fsnotify.Event) bool { // (gameplay guard), skip if content is unchanged since the last // auto-snapshot, then snapshot with retries and notify. func (e *Engine) handleChange(ctx context.Context, gw *gameWatch) { + e.handleChangeFrom(ctx, gw, false) +} + +// handleChangeFrom is handleChange told whether every change in the burst was +// one OpenSave applied itself — a pull, a deletion a peer asked for, mtimes a +// conflict resolution touched (owntouch). +// +// Those are not a new save. They are a sync, which keeps its own history +// (the mirror snapshot of a pull, the safety snapshot before files are +// replaced). Snapshotting them as well took a full auto-snapshot every couple +// of seconds while a peer deleted files one request at a time, each of a save +// half-way between two states, until they had filled the retention budget and +// pushed out the snapshots worth keeping. So the burst only moves the recorded +// hash, which keeps the next real change measured against what is actually on +// disk; anything not known to be ours is snapshotted as before. +func (e *Engine) handleChangeFrom(ctx context.Context, gw *gameWatch, ownOnly bool) { // Gameplay guard: the game may still be mid-write. for anyFileLocked(gw.savePath) { e.log("info", fmt.Sprintf("save files for %q are in use; waiting (gameplay guard)", gw.gameID)) @@ -821,6 +913,12 @@ func (e *Engine) handleChange(ctx context.Context, gw *gameWatch) { e.log("info", fmt.Sprintf("no content change for %q; skipping auto-snapshot", gw.gameID)) return } + if ownOnly { + if err := e.cb.SetLastManifestHash(gw.gameID, currentHash); err != nil { + e.log("warn", fmt.Sprintf("failed to record manifest hash for %q: %v", gw.gameID, err)) + } + return + } for attempt := 1; attempt <= snapshotMaxRetries; attempt++ { err = e.cb.CreateSnapshot(gw.gameID) @@ -1032,23 +1130,54 @@ func (e *Engine) log(level, msg string) { // against that recorded value, so it has to compute it this way — one // definition, or the question gets two answers. func ContentHash(m delta.Manifest, ignoreRules string) string { - if rules := ignore.Parse(ignoreRules); !rules.Empty() { - m = filterForHash(m, rules) + rules := ignore.Parse(ignoreRules) + leaveOut := func(p string) bool { + return delta.NeverSynced(p) || (!rules.Empty() && rules.Match(p)) + } + if rules.Empty() && !leavesAnyOut(m, leaveOut) { + return m.ContentHash() // the same value, without the copy + } + return filterForHash(m, leaveOut).ContentHash() +} + +func leavesAnyOut(m delta.Manifest, leaveOut func(string) bool) bool { + for p := range m.Files { + if leaveOut(p) { + return true + } + } + for _, root := range m.Extra { + for p := range root.Files { + if leaveOut(p) { + return true + } + } + } + return false +} + +// ContentHashBeforeNeverSynced is ContentHash as builds before +// delta.NeverSynced took it, with those files counted: what a value recorded +// by one of them is compared with, to tell whether the save changed since. +func ContentHashBeforeNeverSynced(m delta.Manifest, ignoreRules string) string { + rules := ignore.Parse(ignoreRules) + if rules.Empty() { + return m.ContentHash() } - return m.ContentHash() + return filterForHash(m, rules.Match).ContentHash() } // filterForHash drops excluded paths before the content hash is taken, so the // value recorded here means the same thing the sync engine means by it. -func filterForHash(m delta.Manifest, rules ignore.Rules) delta.Manifest { +func filterForHash(m delta.Manifest, leaveOut func(string) bool) delta.Manifest { out := delta.Manifest{Files: make(map[string]delta.FileEntry, len(m.Files))} for p, entry := range m.Files { - if !rules.Match(p) { + if !leaveOut(p) { out.Files[p] = entry } } for _, d := range m.Dirs { - if !rules.Match(d) { + if !leaveOut(d) { out.Dirs = append(out.Dirs, d) } } @@ -1057,12 +1186,12 @@ func filterForHash(m delta.Manifest, rules ignore.Rules) delta.Manifest { for name, root := range m.Extra { sub := delta.RootManifest{Files: make(map[string]delta.FileEntry, len(root.Files))} for p, entry := range root.Files { - if !rules.Match(p) { + if !leaveOut(p) { sub.Files[p] = entry } } for _, d := range root.Dirs { - if !rules.Match(d) { + if !leaveOut(d) { sub.Dirs = append(sub.Dirs, d) } }